Compare commits

...
203 Commits
Author SHA1 Message Date
Matthew Noto 3f818d0fc5 Remove explicit_package_bases from mypy settings
Remove explicit_package_bases setting from mypy configuration
2026-04-10 22:38:50 -07:00
Matthew Noto 567dd87f53 Update .gitignore 2026-04-10 19:40:29 -07:00
RandNMR73 42a292e4ed fix 2026-04-10 06:08:56 +00:00
RandNMR73 52e3484f6e fix 2026-04-10 05:59:36 +00:00
RandNMR73 6b42166f77 fix 2026-04-10 05:48:29 +00:00
RandNMR73 12a993e81c fix 2026-04-09 20:02:29 +00:00
RandNMR73 aafa257c28 fix 2026-04-09 19:46:22 +00:00
RandNMR73 ca963495f7 fix 2026-04-09 19:39:29 +00:00
RandNMR73 b978a92ef3 fix 2026-04-09 19:36:53 +00:00
RandNMR73 15bda1970f fix 2026-04-09 18:59:41 +00:00
RandNMR73 16d8874eef all tests passing 2026-04-09 18:45:18 +00:00
RandNMR73 88f9ac1db4 fix 2026-04-09 09:32:53 +00:00
RandNMR73 a3a4e0494a fix 2026-04-09 08:58:20 +00:00
RandNMR73 cb5eb77cd7 fix 2026-04-09 08:49:24 +00:00
RandNMR73 ee775e87b9 fix 2026-04-09 08:46:50 +00:00
RandNMR73 b46ce2cc4f fix 2026-04-09 08:42:55 +00:00
RandNMR73 2619107686 fix 2026-04-09 08:40:53 +00:00
RandNMR73 952b70268b fix 2026-04-09 08:39:16 +00:00
RandNMR73 bf39c83493 fix 2026-04-09 08:37:47 +00:00
RandNMR73 10d99ea37e fix 2026-04-09 08:30:34 +00:00
RandNMR73 eccf1a91d1 fix 2026-04-09 08:17:29 +00:00
RandNMR73 5fab762bd1 fix 2026-04-09 08:02:11 +00:00
RandNMR73 6ccc603489 fix 2026-04-09 08:00:32 +00:00
RandNMR73 6e1b285c0d fix 2026-04-09 07:59:09 +00:00
RandNMR73 5230aaa636 fix 2026-04-09 07:56:46 +00:00
RandNMR73 a6a1efd21c fix 2026-04-09 07:56:11 +00:00
RandNMR73 24d3f97d04 fix 2026-04-09 07:54:49 +00:00
RandNMR73 1d3ff7392b fix 2026-04-09 07:32:11 +00:00
RandNMR73 e4a70747bc fix 2026-04-09 07:16:50 +00:00
RandNMR73 2e2899f77c fix 2026-04-09 07:11:43 +00:00
RandNMR73 4636a8a097 fix 2026-04-09 07:05:29 +00:00
RandNMR73 f1e7b2a2e4 fix 2026-04-09 07:03:14 +00:00
RandNMR73 f313af09e9 fix 2026-04-09 07:01:39 +00:00
RandNMR73 1fa17399fb fix 2026-04-09 06:58:28 +00:00
RandNMR73 54b2182c05 fix 2026-04-09 06:56:43 +00:00
RandNMR73 6e7d1a9258 fix 2026-04-09 06:50:34 +00:00
RandNMR73 bf010e8479 fix 2026-04-09 06:45:24 +00:00
RandNMR73 9a6e327578 fix 2026-04-09 06:43:07 +00:00
RandNMR73 a06fb48184 fix 2026-04-09 06:39:58 +00:00
RandNMR73 159376ff68 fix 2026-04-09 05:42:01 +00:00
Matthew Notoandgemini-code-assist[bot] 8a70dd0a86 Apply suggestion from @gemini-code-assist[bot]
Co-authored-by: gemini-code-assist[bot] <176961590+gemini-code-assist[bot]@users.noreply.github.com>
2026-04-08 22:38:24 -07:00
RandNMR73 3fd5502d90 fix 2026-04-09 05:34:20 +00:00
RandNMR73 43d277e13c fix precommit 2026-04-09 01:40:26 +00:00
RandNMR73 4ed79b908e fix 2026-04-09 00:45:58 +00:00
RandNMR73 5832ad9270 clean 2026-04-09 00:30:35 +00:00
RandNMR73 d96de5f433 clean 2026-04-09 00:29:59 +00:00
RandNMR73 027c324d27 clean 2026-04-09 00:16:26 +00:00
RandNMR73 8af2a5b497 clean 2026-04-09 00:14:29 +00:00
RandNMR73 12639e1100 clean 2026-04-08 23:57:44 +00:00
RandNMR73 026ca65412 clean 2026-04-08 23:57:01 +00:00
RandNMR73 0232183cbc clean 2026-04-08 23:52:40 +00:00
RandNMR73 84f615be09 clean 2026-04-08 10:58:50 +00:00
RandNMR73 6278c9b65e clean 2026-04-08 10:52:06 +00:00
RandNMR73 71268798f4 clean 2026-04-08 10:42:06 +00:00
RandNMR73 764fa94a04 clean 2026-04-08 10:33:13 +00:00
RandNMR73 12829a6f5b clean 2026-04-08 10:08:32 +00:00
RandNMR73 e3e075a3b5 clean 2026-04-08 10:05:10 +00:00
RandNMR73 088cc8db23 clean 2026-04-08 09:55:33 +00:00
RandNMR73 46cb669fa5 clean 2026-04-08 09:44:51 +00:00
RandNMR73 bda89a5789 fix 2026-04-08 09:15:29 +00:00
RandNMR73 ad099ca18b clean 2026-04-08 09:10:11 +00:00
RandNMR73 1cdec40290 Merge branch 'matthew/clean' into sync-branch 2026-04-08 09:07:02 +00:00
RandNMR73 9c3d7e33fd clean 2026-04-08 06:16:57 +00:00
RandNMR73 061c205457 vsa with qat training kernel 2026-02-14 11:23:34 +00:00
RandNMR73 68002a0ae7 update default num heads 2026-01-29 01:07:02 +00:00
RandNMR73 c1a6d7f21f update benchmark scripts 2026-01-29 00:56:58 +00:00
RandNMR73 011b9e707d add combined benchmarks 2026-01-29 00:48:24 +00:00
RandNMR73 af03817af0 add fa2 benchmarking 2026-01-29 00:38:09 +00:00
RandNMR73 a8661636d0 fp4 1.3B inference 2026-01-29 00:19:08 +00:00
RandNMR73 96609c05d0 sage3 inference 1.3B baseline 2026-01-28 14:57:40 +00:00
RandNMR73 9361c72c31 remove incorrect two level qkv and revert 2026-01-28 11:49:22 +00:00
RandNMR73 3ee13575c1 use the product of the two sfs 2026-01-28 11:14:58 +00:00
RandNMR73 58cb2f48f8 test 2 level quant q, k, v 2026-01-28 10:40:34 +00:00
RandNMR73 098ee30e37 smooth k ablation: 2026-01-28 10:14:18 +00:00
RandNMR73 0543dbd689 two level P ablation 2026-01-28 08:38:15 +00:00
RandNMR73 3aa5b1fc0e update 2026-01-28 02:36:39 +00:00
RandNMR73 fc104818de modified sage3 with cfg=3 inference 2026-01-28 02:17:56 +00:00
RandNMR73 33a2d512b7 update sage3 to use per_block_mean=True 2026-01-27 13:07:40 +00:00
RandNMR73 643c8e9cd1 fix 14B validation videos 2026-01-27 12:57:08 +00:00
RandNMR73 6f532a183b update benchmarking script attention flop counts 2026-01-27 04:45:30 +00:00
RandNMR73 75d6cf1a31 update benchmark scripts 2026-01-27 04:38:11 +00:00
RandNMR73 72b61ab1fe add sage3 fwd + bf16 bwd script 2026-01-27 04:08:56 +00:00
RandNMR73 0c128c8536 ignore fastvideo_kernel 2026-01-27 04:07:19 +00:00
RandNMR73 dd2096abab fix sage3 api bug 2026-01-27 04:02:37 +00:00
RandNMR73 e57ffd2b3f update batch inference script 2026-01-27 02:44:14 +00:00
RandNMR73 1f39e5c790 ignore fastvideo_kernel 2026-01-27 02:19:16 +00:00
RandNMR73 afd6769a2f update pyrpoject.toml again 2026-01-27 00:54:15 +00:00
RandNMR73 2e7b3f787d update pyproject.toml 2026-01-27 00:51:30 +00:00
RandNMR73 d127e513e9 updates 2026-01-27 00:07:01 +00:00
Matthew Noto c8d3a059db update .gitmodules 2026-01-26 09:03:24 +00:00
RandNMR73 c00fac99b9 fix random seed in training pipeline bug 2026-01-25 22:10:16 +00:00
RandNMR73 aa49fb9dd8 add a bunch of scripts + modify sage3 to support turning off two level quant P 2026-01-25 04:24:34 +00:00
RandNMR73 a7098ad378 rebase 2026-01-22 13:02:32 +00:00
RandNMR73 900cc8d707 checkpoint (qat attn in progress) 2026-01-22 12:32:14 +00:00
Matthew Noto ce95729140 fix DeepGEMM path 2026-01-22 12:15:29 +00:00
Matthew Noto 5121718004 add inference repo 2026-01-22 12:15:26 +00:00
XOR-op aec9c4d313 fix: SP for hunyuanvideo 1.5 (#1026) 2026-01-22 12:09:56 +00:00
Shao DuanandWill Lin 51bef40779 Added LTX-2 Distilled T2V Generation (#1016)
Co-authored-by: Will Lin <wlsaidhi@gmail.com>
2026-01-22 12:09:56 +00:00
alexzmsandWilliam Lin e2979d6d56 [kernel] [bugfix] [ci] bump v0.2.4. Fix STA output handling, TurboDiffusion CUDA norm dtypes for fastvideo-kernel unit tests. (#1020)
Co-authored-by: William Lin <SolitaryThinker@users.noreply.github.com>
2026-01-22 12:09:55 +00:00
William Lin 9ba9a3f2bd [kernel] Fix fastvideo-kernel release workflow (#1019) 2026-01-22 12:09:55 +00:00
XOR-op bee1a5af98 [feat] Hooks API and layerwise offloading for all DiTs (#1006) 2026-01-22 12:09:55 +00:00
William Lin 5ec0e0947c [chore] release fastvideo-kernel 0.2.3 (#1018) 2026-01-22 12:09:55 +00:00
alexzms 0224ea9a14 [Bug Fix] Add autograd wrapper for block-sparse attention in fastvideo-kernel + fix CMake extension linking (#1015) 2026-01-22 12:09:55 +00:00
William Lin f78843ca3f [CI] Fix OOM issues in ssim tests (#1011) 2026-01-22 12:09:50 +00:00
alexzmsandWill Lin 8c9940713a [CI] SSIM tests optimization: load all model weights from Modal persistent Volume (#958)
Co-authored-by: Will Lin <wlsaidhi@gmail.com>
2026-01-22 12:09:05 +00:00
KyleShao ae439b1290 [feat] Introduce Cosmos 2.5 Text2World pipeline (#974) 2026-01-22 12:09:04 +00:00
William Lin 0d6c9a5033 [misc] [bugfix] unpin 'av' in pyproject (#1009) 2026-01-22 12:09:04 +00:00
XOR-op d8b9d7c6ab [feat!] Disable FSDP inference by default (#1001) 2026-01-22 12:09:04 +00:00
Loay Rashid 989f1d5462 [CI] Fixed Turbodiffusion I2V CI (#1002) 2026-01-22 12:09:04 +00:00
William Linandgemini-code-assist[bot] 5753d9273f [ci] temporarily disable turbodiffusion ssim test (#1000)
Co-authored-by: gemini-code-assist[bot] <176961590+gemini-code-assist[bot]@users.noreply.github.com>
2026-01-22 12:09:03 +00:00
Will Lin 6b9626c0f1 Revert "dit"
This reverts commit a6a9c9ca07.
2026-01-22 12:09:03 +00:00
Will Lin c6dc7e8181 dit 2026-01-22 12:09:03 +00:00
3e4b8b278e [bugfix] Add configs for TurboDiffusion T2V/I2V models (#993)
Co-authored-by: gemini-code-assist[bot] <176961590+gemini-code-assist[bot]@users.noreply.github.com>
Co-authored-by: Will Lin <wlsaidhi@gmail.com>
2026-01-22 12:09:03 +00:00
Shreejith SGandWill Lin 7381e19bda [docs]: add LoRA extraction utilities documentation (#992)
Co-authored-by: Will Lin <wlsaidhi@gmail.com>
2026-01-22 12:09:03 +00:00
Shao Duan 44d32ad772 [examples] Added longcat-video python api examples (#994) 2026-01-22 12:09:03 +00:00
William Lin 6b30205e3b [chore] release 0.1.7 (real) (#980) 2026-01-22 12:09:02 +00:00
William Lin cd5c0172a5 [misc] add pin_cpu_memory false for RTX 4090 (#990) 2026-01-22 12:09:02 +00:00
Loay Rashid 490ba382b3 [feat] add Turbodiffusion I2V pipeline (#984) 2026-01-22 12:09:02 +00:00
William Lin f795892514 [misc] pin fastvideo-kernel in .toml file (#989) 2026-01-22 12:09:02 +00:00
Shao Duan ae0316971d Add LongCat-Video I2V and Video Continuation (Base, Distillation and Refinement) Support to FastVideo (#953) 2026-01-22 12:09:02 +00:00
William Lin 98452d99dc [chore] update wechat QR code (#988) 2026-01-22 12:09:01 +00:00
William Lin 9d1cb80fd2 [chore] release fastvideo-kernel 0.2.2 (#986) 2026-01-22 12:09:01 +00:00
William Lin 5dc8625d78 [ci] increase ssim and lora inference test timeout (#985) 2026-01-22 12:09:01 +00:00
William Lin 0db7e3c1e9 [docs] Update docs and README (#975) 2026-01-22 12:09:01 +00:00
Ohm-Rishabh f72ced7611 Layer offloading (#966) 2026-01-22 12:09:01 +00:00
XOR-op a6b738fe42 [feat] Support text encoder weight override and quantization (#983) 2026-01-22 12:09:01 +00:00
Kaiqin Kong ac013acff5 [feat] support Matrix-Game 2.0 streaming generation (#957) 2026-01-22 12:09:00 +00:00
Loay Rashid f74d6e7aff [New Model] Turbodiffusion (#971) 2026-01-22 12:09:00 +00:00
XOR-op b894a8224e [feat] Support absmax style quantization for FP8 (#981) 2026-01-22 12:08:56 +00:00
Qi Jia e24ea17a67 [docs]: fix various broken links across the documentation (#979) 2026-01-22 12:08:00 +00:00
William Lin 6c18dda16c [kernel] add turbodiffusion kernels (#972) 2026-01-22 12:07:55 +00:00
William Lin 1afbf8c7c8 [misc] Add util script to create diffuser HF repo from custom component weights (#970) 2026-01-22 12:06:41 +00:00
RoyWangandroywang 9c3b8acc6f [fix]: fix STA trition kernel for AMD RDNA archs (#969)
Co-authored-by: roywang <roywang@amd.com>
2026-01-22 12:06:41 +00:00
RoyWangandroywang bf1892b8a8 [fix]: fix fastvideo-kernel Rocm build and Dockerfile for Rocm (#968)
Co-authored-by: roywang <roywang@amd.com>
2026-01-22 12:06:41 +00:00
6eedc0acb0 [fix]: fix sliding_tile_attn with sdpa(without flash_attn) (#967)
Co-authored-by: roywang <roywang@amd.com>
Co-authored-by: Will Lin <wlsaidhi@gmail.com>
2026-01-22 12:06:41 +00:00
Ketaki Tank 05214b5aef [feat] Add new feature extractors for fvd (#954) 2026-01-22 12:06:40 +00:00
William Lin 9eabb5da78 [chore] release v0.1.7 (#955) 2026-01-22 12:06:40 +00:00
William Lin 1a637e7a75 [kernel] Fix docker release build for kernel (#965) 2026-01-22 12:06:40 +00:00
William Lin 9ba352fcf3 [docs] refactor attention docs (#964) 2026-01-22 12:06:40 +00:00
William Lin 042c20fd9d [kernel] Release fastvideo-kernel v0.2.1 (#963) 2026-01-22 12:06:40 +00:00
William LinandShreejithSG 254b49cdb7 [kernel] Reorg and fix fastvideo-kernel (#962)
Co-authored-by: ShreejithSG <shreejithsg@gmail.com>
2026-01-22 12:06:40 +00:00
Shreejith SGandWilliam Lin 8f4cd10b49 feat: consolidate attention kernels into unified fastvideo-kernel package (#946)
Co-authored-by: William Lin <SolitaryThinker@users.noreply.github.com>
2026-01-22 12:06:39 +00:00
alexzmsandShao Duan 9211fa84fc Add LongCat T2V (Base, Distillation and Refinement) Support to FastVideo (#883)
Co-authored-by: Shao Duan <shaoxiongduan@gmail.com>
2026-01-22 12:06:39 +00:00
William Lin f2cced218d [rocm] Add rocm fastvideo docker image (#952) 2026-01-22 12:06:38 +00:00
RoyWang 2f49e50150 [feat] add sliding_tile attention triton kernel and ROCM support (#916) 2026-01-22 12:06:38 +00:00
Matthew Noto b824f9536b [docs] small fixes (#947) 2026-01-22 12:06:38 +00:00
Wei Zhou fdca8fdbaa [New Model] Hunyuan1.5 (#943) 2026-01-22 12:06:38 +00:00
William Lin 6c4cb78192 [misc] Allow manual override of Pipeline class through override_pipeline_cls_name (#945) 2026-01-22 12:06:38 +00:00
Loay Rashid 2b8c937abe [bugfix] Added VSA Padding logic (#944) 2026-01-22 12:06:37 +00:00
Kaiqin Kong bdbf327aa5 [feat] Add Matrix-Game 2.0 (#938) 2026-01-22 12:06:37 +00:00
RandNMR73 e332dbb857 more scripts 2026-01-22 11:51:12 +00:00
RandNMR73 e21848a655 add benchmarking script + script changes + fix distillation pipeline 2026-01-22 11:43:34 +00:00
RandNMR73 4cd1297078 a bunch of changes 2026-01-22 06:38:41 +00:00
RandNMR73 8d43721c86 5090 example 2026-01-13 09:39:55 +00:00
RandNMR73 71ecf4a481 fix qat_attn kwargs 2026-01-13 09:35:00 +00:00
RandNMR73 654cc4990a change block size to 128x128 2026-01-13 09:14:28 +00:00
RandNMR73 d76dfa0df3 add inference script with custom weights 2026-01-13 08:41:56 +00:00
RandNMR73 3adfb07465 try sage3 finetuning with aligned backward pass 2026-01-07 01:37:09 +00:00
RandNMR73 30eeec0411 try sage3 finetuning again 2026-01-05 23:55:01 +00:00
RandNMR73 dbd5bd8357 disable delta s 2025-12-26 05:50:03 +00:00
RandNMR73 388f6f1ab8 set per_block_mean=True and modify api 2025-12-26 05:49:09 +00:00
RandNMR73 6c3d8781b7 test sage3 inference with smoothing q but no delta s 2025-12-26 05:05:37 +00:00
RandNMR73 eb160a2e23 modify scripts + launch qat run 2025-12-25 09:30:27 +00:00
RandNMR73 a34a6df844 update training script 2025-12-25 06:33:30 +00:00
RandNMR73 ca3b3985fb 6 gpus 2025-12-25 03:35:15 +00:00
RandNMR73 efcb3c4058 8 gpus 2025-12-25 02:00:45 +00:00
RandNMR73 36cdddfa9a bigger batch size + no grad around wrapped_flash_attn 2025-12-25 01:36:17 +00:00
RandNMR73 56faaa37bd re-enable delta_s since inference is terrible without it 2025-12-25 00:09:36 +00:00
RandNMR73 d0e44170b2 disable delta_s 2025-12-24 23:55:11 +00:00
RandNMR73 f2eb855c2a change block size back to 64x64 2025-12-24 23:36:37 +00:00
RandNMR73 f6080f7cee add sageattn3 inference example 2025-12-24 22:59:02 +00:00
RandNMR73 8678798e58 add delta_s back 2025-12-24 22:58:10 +00:00
RandNMR73 346bbfaf6d revert block size to 128x128 2025-12-24 22:42:52 +00:00
RandNMR73 52480b2514 adjus quant kernel block size 2025-12-24 21:59:04 +00:00
RandNMR73 460df09e20 adjust sage3 block size to 64x64 2025-12-24 21:49:43 +00:00
RandNMR73 3641487374 print sageattn file 2025-12-24 20:16:07 +00:00
RandNMR73 8d68a4fd46 add SageAttn3 with QAT 2025-12-24 09:28:22 +00:00
RandNMR73 7afe836f28 5090 testing 2025-12-24 06:04:36 +00:00
RandNMR73 3b70702af9 fix masking + causal and non-causal logic in qat attn 2025-12-23 01:36:49 +00:00
RandNMR73 9bcc41164a checkpoint (qat attn in progress) 2025-12-23 01:36:46 +00:00
Matthew Noto d41912cd7f fix import 2025-12-23 01:34:18 +00:00
Matthew Noto b06c03845b qat attn in progress + refactor nvfp4 utils 2025-12-23 01:34:18 +00:00
Matthew Noto 6a7f08e73b fix DeepGEMM path 2025-12-23 01:34:18 +00:00
Matthew Noto fe9e94ed0c fix DeepGEMM path 2025-12-23 01:34:18 +00:00
Matthew Noto 2fa6fc102b add inference repo 2025-12-23 01:34:15 +00:00
Matthew Noto 9b125520f0 nvfp4 utils in progress 2025-12-23 01:31:21 +00:00
Matthew Noto ebdb160af9 checkpoint 2025-12-23 01:31:21 +00:00
Peiyuan Zhang b628b5fc7e save 2025-12-23 01:31:21 +00:00
Peiyuan Zhang 3a9e4d3b89 fake quant done 2025-12-23 01:31:21 +00:00
Matthew Noto c137c9f9c5 add real and fake quant precision tests 2025-12-23 01:31:21 +00:00
Peiyuan Zhang 18aebfadc7 update 2025-12-23 01:31:17 +00:00
Peiyuan Zhang 84ed3ceead 1005 morning 2025-12-23 01:30:18 +00:00
Peiyuan Zhang 130b46634a save 2025-12-23 01:30:16 +00:00
Peiyuan Zhang fa91001127 update 2025-12-23 01:28:09 +00:00
Peiyuan Zhang c47be5c795 update 2025-12-23 01:28:08 +00:00
Peiyuan Zhang 032016f7dd update 2025-12-23 01:28:08 +00:00
Peiyuan Zhang 2a28c081be update 2025-12-23 01:28:08 +00:00
Peiyuan Zhang a1ab4a7eeb stash 2025-12-23 01:28:05 +00:00
Peiyuan Zhang 8abfe234af update 2025-12-23 01:27:22 +00:00
Peiyuan Zhang 9161ef60da update 2025-12-23 01:25:05 +00:00
Peiyuan Zhang 6a58c3aa64 + generator sage 3 2025-12-23 01:21:59 +00:00
Peiyuan Zhang 85553c2717 + fp4 linear + fp4 attn, all with 16-bit bwd 2025-12-20 11:09:33 +00:00
Peiyuan Zhang da3f43c7fa +baseline 2025-12-20 11:09:32 +00:00
157 changed files with 12737 additions and 362 deletions
View File
+318
View File
@@ -0,0 +1,318 @@
# Attention QAT
Attention QAT in FastVideo covers two related, but different, backends:
- `ATTN_QAT_INFER`: the inference-oriented CUDA kernel path
- `ATTN_QAT_TRAIN`: the training-oriented Triton attention path
Both are selected with `FASTVIDEO_ATTENTION_BACKEND`, but they are not
interchangeable. The main practical split is:
- use `ATTN_QAT_INFER` for standalone inference with the dedicated inference
kernel
- use `ATTN_QAT_TRAIN` for finetuning, validation during training, or when you
specifically want to reproduce the training-side attention path
## Quick Start
If your goal is "run Wan 2.1 14B with Attention QAT inference weights", this is
the shortest path:
1. Build the in-repo kernel package so FastVideo can import `attn_qat_infer`.
2. Download the Wan 2.1 14B QAT checkpoint.
3. Edit the provided inference example to point at the 14B base model and the
downloaded QAT safetensors.
4. Run the example with `ATTN_QAT_INFER`.
### Step 1. Build the kernel package
Before using either Attention QAT backend, build the in-repo
`fastvideo-kernel` package from source:
```bash
git submodule update --init --recursive
cd fastvideo-kernel
./build.sh
```
After a successful build:
- `ATTN_QAT_TRAIN` should be able to import `fastvideo_kernel`
- `ATTN_QAT_INFER` should be able to import `attn_qat_infer`
`ATTN_QAT_INFER` currently targets the Blackwell CUDA path under
`fastvideo-kernel/attn_qat_infer/` and requires CUDA 12.8+.
### Step 2. Download the Wan 2.1 14B QAT checkpoint
FastVideo includes a helper script:
- `examples/inference/optimizations/download_14B_qat.sh`
By default it downloads:
- Hugging Face repo: `FastVideo/14B_qat_400`
- local directory: `checkpoints/14B_qat_400`
Prerequisites:
- `huggingface_hub` installed, for example:
`uv pip install huggingface_hub`
- access to the model repo if it is private or gated:
`huggingface-cli login`
Run the downloader:
```bash
bash examples/inference/optimizations/download_14B_qat.sh
```
To download into a custom directory:
```bash
bash examples/inference/optimizations/download_14B_qat.sh /path/to/14B_qat_400
```
The script prints a ready-to-copy `init_weights_from_safetensors=...` value at
the end.
### Step 3. Edit the provided inference example
The example to start from is:
- `examples/inference/optimizations/attn_qat_inference_example.py`
Open that file and update these two values:
1. Change the base model from `Wan-AI/Wan2.1-T2V-1.3B-Diffusers` to
`Wan-AI/Wan2.1-T2V-14B-Diffusers`
2. Replace
`init_weights_from_safetensors="safetensors_path"` with the directory that
contains the downloaded `.safetensors` files
Example:
```python
import os
from fastvideo import VideoGenerator
os.environ["FASTVIDEO_ATTENTION_BACKEND"] = "ATTN_QAT_INFER"
generator = VideoGenerator.from_pretrained(
"Wan-AI/Wan2.1-T2V-14B-Diffusers",
num_gpus=1,
use_fsdp_inference=True,
dit_cpu_offload=False,
vae_cpu_offload=False,
text_encoder_cpu_offload=True,
pin_cpu_memory=False,
init_weights_from_safetensors="checkpoints/14B_qat_400",
)
```
Important:
- the checked-in example currently uses the `1.3B` base model until you edit it
- do not load the 14B QAT weights on top of the `1.3B` base model; the weights
and model config will not match
### Step 4. Run the inference example
```bash
python examples/inference/optimizations/attn_qat_inference_example.py
```
Generated videos are written to `video_samples/` by default.
## Backend Overview
| Backend | Best for | Package requirement | Primary kernel location |
|---------|----------|---------------------|-------------------------|
| `ATTN_QAT_TRAIN` | finetuning, training-time validation, reproducing the training path | `fastvideo_kernel` | `fastvideo-kernel/python/fastvideo_kernel/triton_kernels/attn_qat_train.py` |
| `ATTN_QAT_INFER` | standalone inference with the dedicated CUDA kernel | `attn_qat_infer` from the in-repo `fastvideo-kernel` checkout | `fastvideo-kernel/attn_qat_infer/` |
FastVideo routes backend selection through:
- `fastvideo/envs.py`
- `fastvideo/platforms/cuda.py`
- `fastvideo/attention/backends/attn_qat_train.py`
- `fastvideo/attention/backends/attn_qat_infer.py`
The legacy training pipeline also contains explicit Attention QAT integration:
- `fastvideo/training/training_pipeline.py`
That pipeline forces generator loading through `ATTN_QAT_TRAIN` when
`FASTVIDEO_ATTENTION_BACKEND=ATTN_QAT_TRAIN` or `--generator_4bit_attn` is
enabled.
## Inference Workflows
For standalone inference, prefer `ATTN_QAT_INFER` when the CUDA kernel is
available. Use `ATTN_QAT_TRAIN` for inference only if you intentionally want to
exercise the training-side attention path for debugging or parity checks.
### Minimal Python example
```python
import os
from fastvideo import VideoGenerator
os.environ["FASTVIDEO_ATTENTION_BACKEND"] = "ATTN_QAT_INFER"
generator = VideoGenerator.from_pretrained(
"Wan-AI/Wan2.1-T2V-1.3B-Diffusers",
num_gpus=1,
)
generator.generate_video(
"A cinematic close-up of rain on a neon street at night.",
output_path="video_samples",
save_video=True,
)
```
### Loading custom safetensors during inference
FastVideo supports loading custom transformer weights through
`init_weights_from_safetensors`.
This value can point to either:
- a directory containing one or more `.safetensors` files
- a single `.safetensors` file
For Wan 2.1 14B QAT inference, the common pattern is:
```python
generator = VideoGenerator.from_pretrained(
"Wan-AI/Wan2.1-T2V-14B-Diffusers",
num_gpus=1,
use_fsdp_inference=True,
init_weights_from_safetensors="checkpoints/14B_qat_400",
)
```
### CLI example
You can also force the backend from the command line:
```bash
FASTVIDEO_ATTENTION_BACKEND=ATTN_QAT_INFER \
fastvideo generate \
--model-path Wan-AI/Wan2.1-T2V-14B-Diffusers \
--num-gpus 1 \
--sp-size 1 \
--tp-size 1 \
--height 480 \
--width 832 \
--num-frames 77 \
--num-inference-steps 50 \
--guidance-scale 6.0 \
--prompt "A cinematic close-up of rain on a neon street at night." \
--output-path outputs_video/
```
If you want to use custom QAT transformer weights from the CLI, pass the same
custom weight override that the Python API uses:
```bash
FASTVIDEO_ATTENTION_BACKEND=ATTN_QAT_INFER \
fastvideo generate \
--model-path Wan-AI/Wan2.1-T2V-14B-Diffusers \
--init-weights-from-safetensors checkpoints/14B_qat_400 \
--num-gpus 1 \
--output-path outputs_video/ \
--prompt "A cinematic close-up of rain on a neon street at night."
```
## Training Workflows
Today the checked-in Attention QAT training launchers use the legacy training
pipeline in `fastvideo/training/wan_training_pipeline.py`.
### Ready-made launchers
Use the provided SLURM scripts directly:
```bash
sbatch examples/training/finetune/wan_t2v_1.3B/crush_smol/finetune_t2v_qat_attn.sh
sbatch examples/training/finetune/wan_t2v_14B/finetune_t2v_qat_attn.sh
```
Both scripts already set:
```bash
export FASTVIDEO_ATTENTION_BACKEND=ATTN_QAT_TRAIN
```
Before launching, update the script-local values that depend on your
environment:
- `WANDB_API_KEY`
- `MODEL_PATH`
- `DATA_DIR`
- `VALIDATION_DATASET_FILE`
- output directory and SLURM resource requests
### What the launchers run
The training scripts eventually invoke:
```bash
torchrun fastvideo/training/wan_training_pipeline.py ...
```
If you are adapting the workflow to your own cluster or running outside SLURM,
the main Attention QAT requirement is still:
```bash
export FASTVIDEO_ATTENTION_BACKEND=ATTN_QAT_TRAIN
```
Then launch the normal Wan training pipeline with your preferred `torchrun`
arguments and training flags.
## Where The Code Lives
Use these paths when you want to trace or modify the Attention QAT flow:
| Location | Purpose |
|----------|---------|
| `fastvideo/attention/backends/attn_qat_train.py` | FastVideo wrapper that imports and calls the Triton training kernel |
| `fastvideo/attention/backends/attn_qat_infer.py` | FastVideo wrapper that imports and calls the inference kernel |
| `fastvideo-kernel/CMakeLists.txt` | Kernel build definition that compiles the `attn_qat_infer` inference extensions |
| `fastvideo/platforms/cuda.py` | Chooses the concrete attention backend at runtime |
| `fastvideo/envs.py` | Documents supported `FASTVIDEO_ATTENTION_BACKEND` values |
| `fastvideo/training/training_pipeline.py` | Training-time forcing logic for the generator attention backend |
| `fastvideo-kernel/python/fastvideo_kernel/triton_kernels/attn_qat_train.py` | Triton implementation for `ATTN_QAT_TRAIN` |
| `fastvideo-kernel/attn_qat_infer/api.py` | Python API entrypoint for the inference kernel |
| `fastvideo-kernel/benchmarks/benchmark_*.py` | Kernel-side benchmark scripts for FlashAttn2, SageAttention3, FP4, and comparison plots |
| `fastvideo-kernel/attn_qat_infer/blackwell/api.cu` | CUDA implementation behind `ATTN_QAT_INFER` |
| `fastvideo-kernel/tests/test_attn_qat_train.py` | Kernel-level test coverage for the training path |
| `examples/inference/optimizations/attn_qat_inference_example.py` | Ready-to-edit inference example for custom Attention QAT weights |
| `examples/inference/optimizations/download_14B_qat.sh` | Helper script for downloading the Wan 2.1 14B QAT checkpoint |
| `examples/training/finetune/wan_t2v_1.3B/crush_smol/finetune_t2v_qat_attn.sh` | Ready-to-run Wan 1.3B Attention QAT finetune launcher |
| `examples/training/finetune/wan_t2v_14B/finetune_t2v_qat_attn.sh` | Ready-to-run Wan 14B Attention QAT finetune launcher |
## Troubleshooting
- If `ATTN_QAT_TRAIN` fails to import, verify that `fastvideo-kernel` built
successfully and exposes `fastvideo_kernel`.
- If `ATTN_QAT_INFER` fails to import, verify that the local build exposes the
`attn_qat_infer` package.
- If the Wan 2.1 14B example fails after you changed only the checkpoint path,
make sure you also changed the base model to
`Wan-AI/Wan2.1-T2V-14B-Diffusers`.
- If you hit issues with CPU memory pressure or obscure CUDA argument errors in
the example script, try setting `pin_cpu_memory=False`.
- If you want a known-safe fallback for debugging, use
`FASTVIDEO_ATTENTION_BACKEND=TORCH_SDPA`.
## Related Pages
- [Attention Overview](../index.md)
- [Inference Optimizations](../../inference/optimizations.md)
- [Debugging](../../utilities/debugging.md)
+2
View File
@@ -5,6 +5,8 @@ FastVideo provides highly optimized custom attention kernels to accelerate video
## Supported Kernels
* **[Video Sparse Attention (VSA)](vsa/index.md)**: Sparse attention mechanism selecting top-k blocks.
* **[Attention QAT](attn_qat/index.md)**: Dedicated guide for Attention QAT
inference, training, checkpoint loading, and troubleshooting.
* **[Sliding Tile Attention (STA)](sta/index.md)**: STA kernel support is kept in
`fastvideo-kernel`; full FastVideo STA pipeline workflow is archived in
`sta_do_not_delete`.
@@ -35,6 +35,7 @@ surfaces:
prompt_txt: request.inputs.prompt_path
override_text_encoder_safetensors: generator.pipeline.components.text_encoder_weights
override_text_encoder_quant: generator.engine.quantization.text_encoder_quant
transformer_quant: generator.engine.quantization.transformer_quant
override_transformer_cls_name: generator.pipeline.components.override_transformer_cls_name
init_weights_from_safetensors: generator.pipeline.components.transformer_weights
init_weights_from_safetensors_2: generator.pipeline.components.transformer_2_weights
@@ -345,6 +346,7 @@ surfaces:
num_inference_steps: request.sampling.num_inference_steps
num_inference_steps_sr: request.sampling.num_inference_steps_sr
guidance_scale: request.sampling.guidance_scale
guidance_scale_2: request.sampling.guidance_scale_2
guidance_rescale: request.sampling.guidance_rescale
boundary_ratio: request.sampling.boundary_ratio
sigmas: request.sampling.sigmas
@@ -364,15 +366,7 @@ surfaces:
data_type: "Derived from the request shape and not a public input."
sampling_param_extensions:
moved:
guidance_scale_2:
target: request.sampling.guidance_scale_2
sources:
- fastvideo.configs.sample.lingbotworld.LingBotWorld_SamplingParam
- fastvideo.configs.sample.lingbotworld.Wan2_2_I2V_A14B_SamplingParam
- fastvideo.configs.sample.wan.SelfForcingWan2_2_T2V_A14B_480P_SamplingParam
- fastvideo.configs.sample.wan.Wan2_2_I2V_A14B_SamplingParam
- fastvideo.configs.sample.wan.Wan2_2_T2V_A14B_SamplingParam
moved: {}
profile_owned:
action_list:
target: request.extensions.hunyuangamecraft.action_list
@@ -597,6 +591,7 @@ cli:
- text_encoder_cpu_offload
- text_encoder_precisions
- torch_compile_kwargs
- transformer_quant
- tp_size
- trust_remote_code
- use_fsdp_inference
@@ -702,6 +697,7 @@ cli:
- text_encoder_cpu_offload
- text_encoder_precisions
- torch_compile_kwargs
- transformer_quant
- tp_size
- trust_remote_code
- use_fsdp_inference
+5
View File
@@ -167,6 +167,11 @@ How this maps to FastVideo:
- Attention backends live in `fastvideo/attention/` and can be selected via
`FASTVIDEO_ATTENTION_BACKEND`.
- SageAttention3 is split into two selectable backends:
`SAGE_ATTN_THREE` for the regular upstream package and
`ATTN_QAT_INFER` for the FastVideoKernel-backed inference variant.
- `ATTN_QAT_TRAIN` is a separate FastVideoKernel Triton backend for the QAT attention
path.
- `LocalAttention` is used for cross-attention and most attention layers.
- `DistributedAttention` is used for full-sequence self-attention in the DiT.
- Tensor-parallel layers live in `fastvideo/layers/`.
+2
View File
@@ -107,6 +107,8 @@ If you encounter CUDA out of memory errors:
(single GPU) or `use_fsdp_inference=True` (multi-GPU)
- Try a smaller model or use distilled versions
- Use `num_gpus` > 1 if multiple GPUs are available
- Try enabling FSDP inference with `use_fsdp_inference=True` (may slow down generation)
- Try enabling DiT layerwise offload with `dit_layerwise_offload=True` (now only a few models support this, but may introduce less overhead than FSDP)
### Slow Generation
+57
View File
@@ -21,6 +21,8 @@ This page describes the various options for speeding up generation times in Fast
- Video Sparse Attention: `FASTVIDEO_ATTENTION_BACKEND=VIDEO_SPARSE_ATTN`
- Sage Attention: `FASTVIDEO_ATTENTION_BACKEND=SAGE_ATTN`
- Sage Attention 3: `FASTVIDEO_ATTENTION_BACKEND=SAGE_ATTN_THREE`
- Attn QAT Infer: `FASTVIDEO_ATTENTION_BACKEND=ATTN_QAT_INFER`
- Attn QAT Train: `FASTVIDEO_ATTENTION_BACKEND=ATTN_QAT_TRAIN`
- Video MoBA Attention: `FASTVIDEO_ATTENTION_BACKEND=VMOBA_ATTN`
- Sparse Linear Attention: `FASTVIDEO_ATTENTION_BACKEND=SLA_ATTN`
- SageSLA Attention: `FASTVIDEO_ATTENTION_BACKEND=SAGE_SLA_ATTN`
@@ -103,6 +105,14 @@ python setup.py install # or pip install -e .
### Sage Attention 3
FastVideo now exposes two SageAttention3-compatible backends with distinct
environment variable values:
- `SAGE_ATTN_THREE`: the regular upstream SageAttention3 backend imported from
the `sageattn3` package.
- `ATTN_QAT_INFER`: the inference CUDA-kernel backend imported from the
in-repo `attn_qat_infer` package.
**`SAGE_ATTN_THREE`**
[SageAttention 3](https://github.com/thu-ml/SageAttention/tree/main/sageattention3_blackwell) is an advanced attention mechanism that leverages FP4 quantization and Blackwell GPU Tensor Cores for significant performance improvements.
@@ -117,6 +127,53 @@ Note that Sage Attention 3 requires `python>=3.13`, `torch>=2.8.0`, `CUDA >=12.8
To use Sage Attention 3 in FastVideo, follow the `README.md` in the linked repository to install the package from source.
### Attn QAT Infer
**`ATTN_QAT_INFER`**
This backend uses the `attn_qat_infer` implementation that lives in the
`fastvideo-kernel` repository alongside the `fastvideo_kernel` Triton kernels.
Use this backend when you want to run the dedicated FP4 inference CUDA kernel
directly during inference.
For the full Attention QAT guide, including Wan 2.1 14B checkpoint download,
example editing steps, training launchers, and troubleshooting, see
[Attention QAT](../attention/attn_qat/index.md).
This backend currently assumes access to the in-repo `fastvideo-kernel`
checkout or an equivalent editable/source install that exposes:
- `attn_qat_infer`
Example:
```python
os.environ["FASTVIDEO_ATTENTION_BACKEND"] = "ATTN_QAT_INFER"
```
### QAT Attention
**`ATTN_QAT_TRAIN`**
This backend uses the FastVideoKernel Triton attention implementation from
`fastvideo_kernel.triton_kernels.attn_qat_train`. Use it when you specifically
want the training-oriented Triton attention path rather than the
`attn_qat_infer` CUDA kernel path.
The dedicated [Attention QAT](../attention/attn_qat/index.md) page covers when
to use `ATTN_QAT_TRAIN` versus `ATTN_QAT_INFER`, the ready-made training
launchers, and the end-to-end Wan 2.1 14B inference workflow.
This backend currently assumes access to an install that exposes:
- `fastvideo_kernel`
Example:
```python
os.environ["FASTVIDEO_ATTENTION_BACKEND"] = "ATTN_QAT_TRAIN"
```
### V-MoBA / SLA / SageSLA
These backends are model-specific and require the corresponding kernels and
+7 -2
View File
@@ -27,7 +27,8 @@ Useful variables:
- `FASTVIDEO_LOGGING_LEVEL`: `DEBUG`, `INFO`, `WARNING`, `ERROR`
- `FASTVIDEO_STAGE_LOGGING`: print per-stage timings during pipeline execution
- `FASTVIDEO_ATTENTION_BACKEND`: force an attention backend (for example
`TORCH_SDPA` or `FLASH_ATTN`)
`TORCH_SDPA`, `FLASH_ATTN`, `SAGE_ATTN_THREE`, or
`ATTN_QAT_INFER`, or `ATTN_QAT_TRAIN`)
## Common Failure Modes
@@ -52,7 +53,11 @@ If forcing a backend fails, verify optional dependencies are installed:
- `VIDEO_SPARSE_ATTN`: `fastvideo-kernel`
- `SLIDING_TILE_ATTN`: STA legacy workflow in
`sta_do_not_delete` + `fastvideo-kernel`
- `SAGE_ATTN` / `SAGE_ATTN_THREE`: SageAttention packages
- `SAGE_ATTN`: SageAttention package
- `SAGE_ATTN_THREE`: upstream `sageattn3` package
- `ATTN_QAT_INFER`: `fastvideo-kernel` checkout/source install that exposes
`attn_qat_infer`
- `ATTN_QAT_TRAIN`: `fastvideo-kernel` install exposing `fastvideo_kernel`
As a fallback, use:
@@ -22,7 +22,7 @@ export NODE_RANK=$SLURM_PROCID
nodes=( $(scontrol show hostnames $SLURM_JOB_NODELIST) )
export MASTER_ADDR=${nodes[0]}
export TOKENIZERS_PARALLELISM=false
export WANDB_API_KEY="2f25ad37933894dbf0966c838c0b8494987f9f2f"
export WANDB_API_KEY=YOUR_WANDB_API_KEY
# export WANDB_API_KEY='your_wandb_api_key_here'
export WANDB_BASE_URL="https://api.wandb.ai"
export WANDB_MODE=online
@@ -9,7 +9,7 @@ pip install vsa
### 1. Download dataset:
```bash
bash examples/distill/Wan-Syn-480P/download_dataset.sh
bash examples/distill/Wan2.1-T2V/Wan-Syn-Data-480P/download_dataset.sh
```
### 2. Configure and run distillation:
@@ -1,3 +1,3 @@
#!/bin/bash
python scripts/huggingface/download_hf.py --repo_id "FastVideo/Wan-Syn_77x448x832_600k" --local_dir "FastVideo/Wan-Syn_77x448x832_600k" --repo_type "dataset"
mkdir -p data
python scripts/huggingface/download_hf.py --repo_id "FastVideo/Wan-Syn_77x448x832_600k" --local_dir "data/Wan-Syn_77x448x832_600k" --repo_type "dataset"
+75 -1
View File
@@ -1,5 +1,79 @@
# Optimization Examples
## Wan 2.1 QAT Attention 14B Inference
Use these files for Wan 2.1 14B inference with the `ATTN_QAT_INFER` backend:
- `examples/inference/optimizations/download_14B_qat.sh`
- `examples/inference/optimizations/attn_qat_inference_example.py`
### 1. Download the 14B QAT checkpoint
The helper script downloads the QAT safetensors from
`FastVideo/14B_qat_400` into `checkpoints/14B_qat_400` by default.
Prerequisites:
- `huggingface_hub` installed, for example: `uv pip install huggingface_hub`
- access to the model repo if it is private or gated: `huggingface-cli login`
Run:
```bash
python examples/inference/optimizations/attention_example.py
bash examples/inference/optimizations/download_14B_qat.sh
```
To download into a custom directory, pass it as the first argument:
```bash
bash examples/inference/optimizations/download_14B_qat.sh /path/to/14B_qat_400
```
### 2. Edit the inference example for Wan 2.1 14B
Open `examples/inference/optimizations/attn_qat_inference_example.py` and
update these two values:
1. Change the base model from `Wan-AI/Wan2.1-T2V-1.3B-Diffusers` to
`Wan-AI/Wan2.1-T2V-14B-Diffusers`.
2. Replace the placeholder
`init_weights_from_safetensors="safetensors_path"` with the directory that
contains the downloaded `.safetensors` files.
Example:
```python
generator = VideoGenerator.from_pretrained(
"Wan-AI/Wan2.1-T2V-14B-Diffusers",
num_gpus=1,
use_fsdp_inference=True,
dit_cpu_offload=False,
vae_cpu_offload=False,
text_encoder_cpu_offload=True,
pin_cpu_memory=False,
init_weights_from_safetensors="checkpoints/14B_qat_400",
)
```
The script already sets:
```python
os.environ["FASTVIDEO_ATTENTION_BACKEND"] = "ATTN_QAT_INFER"
```
### 3. Run the example
```bash
python examples/inference/optimizations/attn_qat_inference_example.py
```
The generated videos are written to `video_samples/` by default.
### Notes
- `ATTN_QAT_INFER` requires the in-repo `fastvideo-kernel` build to expose the
`attn_qat_infer` package.
- If you have not built the kernel yet, run `cd fastvideo-kernel && ./build.sh`
first.
- If you keep the example on the `1.3B` base model while loading the 14B QAT
weights, the model/config will not match.
@@ -0,0 +1,54 @@
from fastvideo import VideoGenerator
import os
from pathlib import Path
# from fastvideo.configs.sample import SamplingParam
OUTPUT_PATH = "video_samples"
os.environ["FASTVIDEO_ATTENTION_BACKEND"] = "ATTN_QAT_INFER"
CHECKPOINT_PATH = Path(__file__).parent.parent.parent
def main():
# FastVideo will automatically use the optimal default arguments for the
# model.
# If a local path is provided, FastVideo will make a best effort
# attempt to identify the optimal arguments.
generator = VideoGenerator.from_pretrained(
"Wan-AI/Wan2.1-T2V-1.3B-Diffusers",
# FastVideo will automatically handle distributed setup
num_gpus=1,
use_fsdp_inference=True,
dit_cpu_offload=False,
vae_cpu_offload=False,
text_encoder_cpu_offload=True,
pin_cpu_memory=False, # set to false if low CPU RAM or hit obscure "CUDA error: Invalid argument"
# image_encoder_cpu_offload=False,
# Load custom weights from checkpoint
init_weights_from_safetensors="safetensors_path"
)
# sampling_param = SamplingParam.from_pretrained("Wan-AI/Wan2.1-T2V-1.3B-Diffusers")
# sampling_param.num_frames = 45
# sampling_param.image_path = "https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/diffusers/astronaut.jpg"
# Generate videos with the same simple API, regardless of GPU count
prompt = (
"A curious raccoon peers through a vibrant field of yellow sunflowers, its eyes "
"wide with interest. The playful yet serene atmosphere is complemented by soft "
"natural light filtering through the petals. Mid-shot, warm and cheerful tones."
)
video = generator.generate_video(prompt, output_path=OUTPUT_PATH, save_video=True)
# video = generator.generate_video(prompt, sampling_param=sampling_param, output_path="wan_t2v_videos/")
# Generate another video with a different prompt, without reloading the
# model!
prompt2 = (
"A majestic lion strides across the golden savanna, its powerful frame "
"glistening under the warm afternoon sun. The tall grass ripples gently in "
"the breeze, enhancing the lion's commanding presence. The tone is vibrant, "
"embodying the raw energy of the wild. Low angle, steady tracking shot, "
"cinematic.")
video2 = generator.generate_video(prompt2, output_path=OUTPUT_PATH, save_video=True)
if __name__ == "__main__":
main()
+58
View File
@@ -0,0 +1,58 @@
#!/usr/bin/env bash
set -euo pipefail
SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" && pwd)"
REPO_ROOT="$(cd -- "${SCRIPT_DIR}/../../.." && pwd)"
HF_REPO_ID="${HF_REPO_ID:-FastVideo/14B_qat_400}"
HF_REVISION="${HF_REVISION:-main}"
LOCAL_DIR="${1:-${REPO_ROOT}/checkpoints/14B_qat_400}"
PYTHON_BIN="${PYTHON:-python}"
if ! command -v "${PYTHON_BIN}" >/dev/null 2>&1; then
echo "Python executable not found: ${PYTHON_BIN}" >&2
exit 1
fi
if ! "${PYTHON_BIN}" -c "import huggingface_hub" >/dev/null 2>&1; then
echo "Missing dependency: huggingface_hub" >&2
echo "Install it with: uv pip install huggingface_hub" >&2
exit 1
fi
mkdir -p "${LOCAL_DIR}"
echo "Downloading ${HF_REPO_ID}@${HF_REVISION}"
echo "Local directory: ${LOCAL_DIR}"
"${PYTHON_BIN}" -c '
import argparse
from huggingface_hub import snapshot_download
parser = argparse.ArgumentParser()
parser.add_argument("--repo-id", required=True)
parser.add_argument("--revision", required=True)
parser.add_argument("--local-dir", required=True)
args = parser.parse_args()
snapshot_download(
repo_id=args.repo_id,
revision=args.revision,
repo_type="model",
local_dir=args.local_dir,
local_dir_use_symlinks=False,
resume_download=True,
)
' \
--repo-id "${HF_REPO_ID}" \
--revision "${HF_REVISION}" \
--local-dir "${LOCAL_DIR}"
echo
echo "Download complete."
echo "Use this in your inference script:"
echo "init_weights_from_safetensors=\"${LOCAL_DIR}\""
echo
echo "If the repo is private or gated, make sure you are logged in with:"
echo "huggingface-cli login"
@@ -0,0 +1,88 @@
import torch
from fastvideo import VideoGenerator
from fastvideo.configs.pipelines.base import PipelineConfig
OUTPUT_PATH = "video_samples"
def main():
print("=== FP4 Quantization Video Generation Example ===")
if not torch.cuda.is_available():
print("Warning: CUDA not available. FP4 quantization requires GPU.")
return
gpu_capability = torch.cuda.get_device_capability()
if gpu_capability[0] < 9: # H100 and newer
print(f"Warning: GPU capability {gpu_capability} may not support FP4. Recommended: 9.0+")
print(f"GPU: {torch.cuda.get_device_name()}")
print(f"GPU Capability: {gpu_capability}")
model_id = "Wan-AI/Wan2.1-T2V-1.3B-Diffusers"
# model_id = "Wan-AI/Wan2.1-T2V-14B-Diffusers"
pipeline_config = PipelineConfig.from_pretrained(model_id)
pipeline_config.dit_precision = "bf16"
print("\nLoading model with FP4 quantization...")
generator = VideoGenerator.from_pretrained(
model_id,
pipeline_config=pipeline_config,
num_gpus=1,
use_fsdp_inference=True,
transformer_quant="fp4",
dit_cpu_offload=False,
vae_cpu_offload=False,
text_encoder_cpu_offload=True,
pin_cpu_memory=False,
)
print("FP4 configuration applied. Generating videos...")
print("\n=== Generating Video with FP4 Quantization ===")
prompt1 = (
"A curious raccoon peers through a vibrant field of yellow sunflowers, its eyes "
"wide with interest. The playful yet serene atmosphere is complemented by soft "
"natural light filtering through the petals. Mid-shot, warm and cheerful tones."
)
print(f"Prompt: {prompt1}")
print("Generating video...")
try:
video1 = generator.generate_video(
prompt1,
output_path=OUTPUT_PATH,
save_video=True,
)
print("✓ First video generated successfully with FP4 quantization!")
# # Generate a second video to show the model can be reused
prompt2 = (
"A majestic lion strides across the golden savanna, its powerful frame "
"glistening under the warm afternoon sun. The tall grass ripples gently in "
"the breeze, enhancing the lion's commanding presence. The tone is vibrant, "
"embodying the raw energy of the wild. Low angle, steady tracking shot, "
"cinematic."
)
print(f"\nGenerating second video...")
print(f"Prompt: {prompt2}")
video2 = generator.generate_video(
prompt2,
output_path=OUTPUT_PATH,
save_video=True,
)
print("✓ Second video generated successfully with FP4 quantization!")
except Exception as e:
print(f"Error during video generation: {e}")
return
print(f"Videos saved to: {OUTPUT_PATH}")
if __name__ == "__main__":
main()
@@ -1,27 +1,47 @@
#!/bin/bash
#SBATCH --job-name=wan_t2v_1.3B_finetune
#SBATCH --partition=all
#SBATCH --nodes=1
#SBATCH --gres=gpu:4
#SBATCH --ntasks-per-node=1
#SBATCH --output=logs/wan_t2v_1.3B_finetune.out
#SBATCH --error=logs/wan_t2v_1.3B_finetune.err
source .venv/bin/activate
export WANDB_BASE_URL="https://api.wandb.ai"
export WANDB_MODE=online
export TOKENIZERS_PARALLELISM=false
# export FASTVIDEO_ATTENTION_BACKEND=TORCH_SDPA
# export TRITON_PRINT_AUTOTUNING=1 # to print the best config
export WANDB_API_KEY=YOUR_WANDB_API_KEY
MODEL_PATH="Wan-AI/Wan2.1-T2V-1.3B-Diffusers"
DATA_DIR="data/crush-smol_processed_t2v/combined_parquet_dataset/"
VALIDATION_DATASET_FILE="$(dirname "$0")/validation.json"
NUM_GPUS=4
DATA_DIR=data/Wan-Syn_77x448x832_600k
VALIDATION_DATASET_FILE="examples/training/finetune/wan_t2v_1.3B/crush_smol/validation.json"
NUM_GPUS=1
# export CUDA_VISIBLE_DEVICES=4,5
set -euo pipefail
# ---- torchrun rendezvous (multi-node) ----
# Launch ONE torchrun per node (via srun) and let torchrun spawn 4 workers per node.
MASTER_ADDR="$(scontrol show hostnames "$SLURM_JOB_NODELIST" | head -n 1)"
MASTER_PORT="${MASTER_PORT:-29500}"
export MASTER_ADDR MASTER_PORT
# Training arguments
training_args=(
--tracker_project_name "wan_t2v_finetune"
--output_dir "checkpoints/wan_t2v_finetune"
--max_train_steps 5000
--tracker_project_name "wan_t2v_finetune_qat"
--output_dir "checkpoints/wan_t2v_finetune_1.3B_77"
--max_train_steps 4000
--train_batch_size 1
--train_sp_batch_size 1
--gradient_accumulation_steps 8
--gradient_accumulation_steps 1
--num_latent_t 20
--num_height 480
--num_height 448
--num_width 832
--num_frames 77
--enable_gradient_checkpointing_type "full"
@@ -30,7 +50,7 @@ training_args=(
# Parallel arguments
parallel_args=(
--num_gpus $NUM_GPUS
--sp_size $NUM_GPUS
--sp_size 1
--tp_size 1
--hsdp_replicate_dim 1
--hsdp_shard_dim $NUM_GPUS
@@ -45,7 +65,7 @@ model_args=(
# Dataset arguments
dataset_args=(
--data_path $DATA_DIR
--dataloader_num_workers 1
--dataloader_num_workers 4
)
# Validation arguments
@@ -54,16 +74,16 @@ validation_args=(
--validation_dataset_file $VALIDATION_DATASET_FILE
--validation_steps 200
--validation_sampling_steps "50"
--validation_guidance_scale "3.0"
--validation_guidance_scale "5.0"
)
# Optimizer arguments
optimizer_args=(
--learning_rate 5e-5
--learning_rate 1e-6
--mixed_precision "bf16"
--weight_only_checkpointing_steps 1000
--training_state_checkpointing_steps 1000
--weight_decay 1e-4
--weight_decay 0.01
--max_grad_norm 1.0
)
@@ -72,23 +92,24 @@ miscellaneous_args=(
--inference_mode False
--checkpoints_total_limit 3
--training_cfg_rate 0.1
--multi_phased_distill_schedule "4000-1"
--not_apply_cfg_solver
--dit_precision "fp32"
--num_euler_timesteps 50
--ema_start_step 0
--enable_gradient_checkpointing_type "full"
# --resume_from_checkpoint "checkpoints/wan_t2v_finetune/checkpoint-2500"
--flow_shift 5
--seed 1000
)
torchrun \
--nnodes 1 \
--nproc_per_node $NUM_GPUS \
fastvideo/training/wan_training_pipeline.py \
"${parallel_args[@]}" \
"${model_args[@]}" \
"${dataset_args[@]}" \
"${training_args[@]}" \
"${optimizer_args[@]}" \
"${validation_args[@]}" \
"${miscellaneous_args[@]}"
srun --nodes="$SLURM_NNODES" --ntasks="$SLURM_NNODES" --ntasks-per-node=1 \
torchrun \
--nnodes "$SLURM_NNODES" \
--nproc_per_node 4 \
--rdzv_backend c10d \
--rdzv_endpoint "${MASTER_ADDR}:${MASTER_PORT}" \
--rdzv_id "$SLURM_JOB_ID" \
fastvideo/training/wan_training_pipeline.py \
"${parallel_args[@]}" \
"${model_args[@]}" \
"${dataset_args[@]}" \
"${training_args[@]}" \
"${optimizer_args[@]}" \
"${validation_args[@]}" \
"${miscellaneous_args[@]}"
@@ -0,0 +1,124 @@
#!/bin/bash
#SBATCH --job-name=wan_t2v_1.3B_finetune_qat_16
#SBATCH --partition=all
#SBATCH --nodes=4
#SBATCH --gres=gpu:4
#SBATCH --ntasks-per-node=1
#SBATCH --output=logs/wan_t2v_1.3B_finetune_qat_16.out
#SBATCH --error=logs/wan_t2v_1.3B_finetune_qat_16.err
source .venv/bin/activate
export WANDB_BASE_URL="https://api.wandb.ai"
export WANDB_MODE=online
export TOKENIZERS_PARALLELISM=false
export FASTVIDEO_ATTENTION_BACKEND=ATTN_QAT_TRAIN
# export TRITON_PRINT_AUTOTUNING=1 # to print the best config
export WANDB_API_KEY=YOUR_WANDB_API_KEY
# Use node-local Triton cache to avoid stale file handle errors on shared filesystems
export TRITON_CACHE_DIR="/tmp/triton_cache_${SLURM_JOB_ID}_${SLURM_NODEID}"
MODEL_PATH="Wan-AI/Wan2.1-T2V-1.3B-Diffusers"
DATA_DIR=YOUR_DATA_DIR
VALIDATION_DATASET_FILE="examples/training/finetune/wan_t2v_1.3B/crush_smol/validation.json"
NUM_GPUS=16
# export CUDA_VISIBLE_DEVICES=4,5
set -euo pipefail
# ---- torchrun rendezvous (multi-node) ----
# 1. Get the hostname of the first node (Master)
nodes=( $( scontrol show hostnames $SLURM_JOB_NODELIST ) )
nodes_array=($nodes)
head_node=${nodes_array[0]}
MASTER_ADDR=$(srun --nodes=1 --ntasks=1 -w "$head_node" hostname --ip-address)
MASTER_PORT=29500
# 2. Get the node count automatically
NNODES=$SLURM_NNODES
GPUS_PER_NODE=$SLURM_GPUS_ON_NODE
NUM_GPUS=$((NNODES * GPUS_PER_NODE))
echo "MASTER_ADDR=$MASTER_ADDR MASTER_PORT=$MASTER_PORT NNODES=$NNODES"
# Training arguments
training_args=(
--tracker_project_name "wan_t2v_finetune_qat"
--output_dir "checkpoints/wan_1.3B_t2v_finetune_qat"
--max_train_steps 4000
--train_batch_size 1
--train_sp_batch_size 1
--gradient_accumulation_steps 1
--num_latent_t 20
--num_height 448
--num_width 832
--num_frames 77
--enable_gradient_checkpointing_type "full" # if OOM enable this
)
# Parallel arguments
parallel_args=(
--num_gpus $NUM_GPUS
--sp_size 1
--tp_size 1
--hsdp_replicate_dim $NUM_GPUS
--hsdp_shard_dim 1
)
# Model arguments
model_args=(
--model_path $MODEL_PATH
--pretrained_model_name_or_path $MODEL_PATH
)
# Dataset arguments
dataset_args=(
--data_path "$DATA_DIR"
--dataloader_num_workers 4
)
# Validation arguments
validation_args=(
--log_validation
--validation_dataset_file $VALIDATION_DATASET_FILE
--validation_steps 200
--validation_sampling_steps "50"
--validation_guidance_scale "5.0"
)
# Optimizer arguments
optimizer_args=(
--learning_rate 1e-6
--mixed_precision "bf16"
--weight_only_checkpointing_steps 200
--training_state_checkpointing_steps 200
--weight_decay 0.01
--max_grad_norm 1.0
)
# Miscellaneous arguments
miscellaneous_args=(
--inference_mode False
--checkpoints_total_limit 3
--training_cfg_rate 0.1
--dit_precision "fp32"
--ema_start_step 0
--flow_shift 1
--seed 1000
)
srun torchrun \
--nnodes $NNODES \
--nproc_per_node $GPUS_PER_NODE \
--node_rank $SLURM_PROCID \
--rdzv_backend c10d \
--rdzv_endpoint $MASTER_ADDR:$MASTER_PORT \
fastvideo/training/wan_training_pipeline.py \
"${parallel_args[@]}" \
"${model_args[@]}" \
"${dataset_args[@]}" \
"${training_args[@]}" \
"${optimizer_args[@]}" \
"${validation_args[@]}" \
"${miscellaneous_args[@]}"
@@ -1,31 +1,131 @@
{
"data": [
{
"caption": "A large metal cylinder is seen pressing down on a pile of Oreo cookies, flattening them as if they were under a hydraulic press.",
"image_path": null,
"video_path": null,
"num_inference_steps": 50,
"height": 480,
"caption": "In the video, a woman is elegantly showcasing her earrings, bringing attention to their intricate design with a gentle touch of her fingers. She is bathed in ambient purple and pink lighting, which casts a soft glow on her delicate features and enhances the vivid tones of her lipstick and eye makeup. Her hair is styled to frame her face smoothly, emphasizing the contours of her jawline and cheekbones. The background features a blurred neon light, adding an artistic and modern touch to the overall aesthetic.",
"video_path": "Fashion/mixkit-face-of-an-elegant-and-captivating-woman-41914_clip_1.mp4",
"num_inference_steps": 50,
"height": 448,
"width": 832,
"num_frames": 77
},
{
"caption": "A large metal cylinder is seen compressing colorful clay into a compact shape, demonstrating the power of a hydraulic press.",
"image_path": null,
"video_path": null,
"num_inference_steps": 50,
"height": 480,
"caption": "In the video, a lone rider guides a majestic horse across an expansive, open field as the sun sets in the background. The rider, dressed in a classic blue shirt and wide-brimmed hat, sits confidently in the saddle, silhouetted against the warm glow of the evening sky. The horse moves gracefully, its mane and tail flowing with each step, creating a sense of harmony between horse and rider. Surrounding the pair, towering trees form a natural border, their leaves gently rustling in the breeze. The shadows lengthen on the ground, accentuating the serene and timeless feel of the scene. The distant hills and wooden fences frame the horizon, adding depth to the tranquil landscape. A few horses graze peacefully in the background, blending into the pastoral setting. The overall ambiance evokes a sense of calmness and quietude, capturing a perfect moment in the golden light of dusk.",
"video_path": "Man/mixkit-a-rancher-riding-a-horse-at-sunset-1143_clip_1.mp4",
"num_inference_steps": 50,
"height": 448,
"width": 832,
"num_frames": 77
},
{
"caption": "A large metal cylinder is seen pressing down on a pile of colorful candies, flattening them as if they were under a hydraulic press. The candies are crushed and broken into small pieces, creating a mess on the table.",
"image_path": null,
"video_path": null,
"num_inference_steps": 50,
"height": 480,
"caption": "In a dimly lit, eerie setting, a mysterious pink bottle labeled \"Authentic 100% organic POISON\" sits prominently in the foreground, casting a menacing aura. The bottle is accentuated by green fog, which swirls lightly around it, enhancing its sinister allure. Behind it, a shadowy golden bottle adorned with a spider emblem subtly emerges, adding an extra layer of mystery to the scene. Dim candles provide faint, flickering light, which complements the dark atmosphere, making the setting ideal for an illusion of hidden dangers.",
"video_path": "smoke/mixkit-poison-in-halloween-ritual-33879_clip_1.mp4",
"num_inference_steps": 50,
"height": 448,
"width": 832,
"num_frames": 77
}
]
},
{
"caption": "The video opens with a tranquil scene in the heart of a dense forest, emphasizing two large, textured tree trunks in the foreground framing the view. Sunlight filters through the canopy above, casting intricate patterns of light and shadow on the trees and the ground. Between the tree trunks, a clear view of a calm, muddy river unfolds, its surface shimmering under the gentle sunlight. The riverbank is decorated with a variety of small bushes and vibrant foliage, subtly transitioning into the deep greens of tall, leafy plants. In the background, the dense forest looms, filled with dark, towering trees, their branches intertwining to form an intricate canopy. The scene is bathed in the soft glow of the sun, creating a serene and picturesque setting. Occasional sunbeams pierce through the foliage, adding a magical aura to the landscape. The vibrant reds and oranges of the smaller plants add contrast, bringing warmth to the earthy tones of the scenery. Overall, this harmonious blend of natural elements creates a peaceful and idyllic forest setting.",
"video_path": "forest/mixkit-view-of-a-river-between-two-old-trees-560_clip_1.mp4",
"num_inference_steps": 50,
"height": 448,
"width": 832,
"num_frames": 77
},
{
"caption": "In the video, a martial artist dressed in a traditional white uniform with a black belt demonstrates a series of precise movements against a stark black background. The individual gracefully transitions between stances, embodying a sense of focused discipline and control. Each motion is executed with a deliberate pace, showcasing the fluidity of martial arts techniques. The soft lighting creates subtle highlights on the uniform, adding depth to the figure as it moves. The practitioner begins with an open-hand pose, feet firmly grounded, gradually shifting to a powerful forward punch. The fluidity of the sequence displays a mastery of balance and poise. Every trajectory of the limbs is precise and deliberate, capturing the elegance and strength of martial arts. The serene, isolated setting enhances the intensity and concentration of the practitioner. This visual presentation is an elegant interplay of motion and stillness, displaying the art form's discipline and grace.",
"video_path": "Man/mixkit-a-young-man-practicing-his-karate-moves-49635_clip_1.mp4",
"num_inference_steps": 50,
"height": 448,
"width": 832,
"num_frames": 77
},
{
"caption": "A tranquil coastal scene unfolds with a drone's aerial view capturing a serene beach landscape. The camera glides over a quiet stretch of sandy shoreline, where gentle waves kiss the shore under a clear blue sky. Nestled amidst lush palm trees are a series of traditional thatched-roof huts, their earthy tones blending harmoniously with the natural surroundings. The sandy beach stretches endlessly, bordered by the rhythmic dance of ocean waves on one side and verdant greenery on the other. A pair of white umbrellas is set up on the sand, suggesting a place to relax and enjoy the sun. In the distance, two small human figures can be seen walking leisurely along the water's edge, leaving faint footprints behind them. The scene exudes a calm and inviting atmosphere, with the soft rustle of palm leaves and the whisper of the ocean breeze almost audible. The overall composition is a captivating blend of nature's tranquility and architectural simplicity. This picturesque setting invites viewers to imagine themselves steps away from this idyllic coastal escape.",
"video_path": "beach/mixkit-sunny-beach-in-a-dynamic-shot-from-a-drone-44383_clip_1.mp4",
"num_inference_steps": 50,
"height": 448,
"width": 832,
"num_frames": 77
},
{
"caption": "A lone figure stands on a large, moss-covered rock, surrounded by the soft rush of a nearby stream. The figure is wearing white sneakers and shorts, with a plaid shirt that hangs loosely in the breeze. The lighting creates dramatic shadows, enhancing the textures of the rock and the subtle movement of the water below. In the background, a waterfall cascades into the stream, completing this tranquil and serene nature scene.",
"video_path": "forest/mixkit-woman-standing-in-front-of-waterfall-559_clip_1.mp4",
"num_inference_steps": 50,
"height": 448,
"width": 832,
"num_frames": 77
},
{
"caption": "In an industrial setting, a person leans casually against a railing, exuding a sense of confidence and composure. They are wearing a striking outfit, consisting of a vibrant, patterned jacket over a simple white crop top, creating a bold contrast. The atmosphere is infused with warm, ambient lighting that casts soft shadows on the concrete walls and metallic surfaces. Intricate wiring and pipes form an intricate backdrop, enhancing the urban aesthetic. Their relaxed posture and direct, engaging gaze suggest a sense of ease in this industrial environment. This scene encapsulates a blend of modern fashion and gritty, urban architecture, creating a visually compelling narrative.",
"video_path": "Fashion/mixkit-portrait-of-a-hipster-woman-walking-down-a-stairs-1297_clip_1.mp4",
"num_inference_steps": 50,
"height": 448,
"width": 832,
"num_frames": 77
},
{
"caption": "A man is energetically stretching in an open-air setting, surrounded by rows of vibrant red seats that suggest an amphitheater or outdoor venue. He wears a sleeveless black shirt layered with a hooded vest, emphasizing his athletic build as he engages in a warm-up routine. Behind him, the striking modern architecture of the building features geometric panels, with large sections of glass and overlapping metallic beams creating a dynamic backdrop. The scene captures the contrast between his focused movements and the static, bold design of the structure, while the surrounding greenery adds a touch of nature to the environment. The overall atmosphere is one of preparation and anticipation, with the man appearing determined and ready for an upcoming event or performance.",
"video_path": "Sport/mixkit-man-doing-arm-stretches-595_clip_1.mp4",
"num_inference_steps": 50,
"height": 448,
"width": 832,
"num_frames": 77
},
{
"caption": "A young woman is seated on the floor in front of a plush, beige tufted couch, fully engrossed in sorting through a stack of papers. Her dark hair falls loosely past her shoulders, and she wears a green plaid shirt, contributing to the casual yet focused atmosphere. She gently places the papers onto a small round white table, occasionally lifting individual sheets to examine them more closely. Her expression shifts subtly, reflecting concentration and contemplation as she processes the information on the pages. Two small, round nested tables hold her documents, along with a small plant in a gray pot, adding a touch of greenery to the scene. The background features a dark paneled wall, creating a contrasting backdrop for the light-colored furniture. The setting is tranquil and organized, the couch and tables arranged symmetrically, conveying a sense of harmony. A calculator rests on the smaller table, hinting at a task involving calculations or budgeting.",
"video_path": "Woman/mixkit-frustrated-woman-throws-paperwork-on-the-floor-4526_clip_1.mp4",
"num_inference_steps": 50,
"height": 448,
"width": 832,
"num_frames": 77
},
{
"caption": "A heavily rusted metal gate stands firmly locked, with two vertical bars joined by a thick, old chain that loops elegantly around them. The chain's texture is coarse and rugged, its surface reflecting varying shades of orange and brown, indicative of years exposed to the elements. At the heart of the chain, a black iron padlock, slightly worn yet imposing, secures the gate, its curves and edges smooth against the aged links. The gate's metalwork is outlined by a backdrop of soft, blurred greenery, suggesting a serene and isolated location beyond the barrier. Tall trees rise in the distance, their trunks and leaves creating a lush, forest-like setting that contrasts with the gate's severe rust. A pathway leads away from the gate, its surface uneven with patches of moss and weathered stone visible in the soft focus, inviting yet inaccessible. The ambiance is quiet and mysterious, with a sense of abandonment hanging subtly in the air, evoking curiosity about what lies beyond. Shadows play across the gate, cast by branches swaying gently in the breeze, adding to the dynamic interaction of light and texture. This scene, rich in detail and atmosphere, captures the viewer's imagination, evoking both the allure of the forbidden and the beauty of decay.",
"video_path": "forest/mixkit-rusty-fence-with-a-chain-of-a-property-in-nature-5294_clip_1.mp4",
"num_inference_steps": 50,
"height": 448,
"width": 832,
"num_frames": 77
},
{
"caption": "In a serene and softly lit yoga studio, three individuals engage in a yoga session, each performing an upward-facing stretch. The central figure is a woman with shoulder-length brown hair, dressed in a light cropped top and green leggings, her posture reflecting grace and concentration. To her right, another participant, a woman in a purple outfit, mirrors the pose with equal poise. On her left, a person with a bun focuses intently, supported slightly by yoga blocks beneath their hands. The warm-colored wooden floor contrasts soothingly with the soft pastel mural on the back wall, featuring an abstract design and partial visage of a serene face. Natural light floods the space from a large window on the right, where lush greens peek through, adding an element of tranquility. In the corner of the room, a collection of meditation instruments, including a gong and a Buddha statue, subtly frame the peaceful setting. The mood is calm yet focused, as all three participants are deeply engaged in their practice. The scene combines elements of balance, harmony, and a shared journey towards mindfulness. This depiction captures the essence of a yoga session that blends personal growth with collective experience.",
"video_path": "People/mixkit-small-group-of-people-doing-yoga-together-43730_clip_1.mp4",
"num_inference_steps": 50,
"height": 448,
"width": 832,
"num_frames": 77
},
{
"caption": "In the deep blue expanse of the ocean, two dolphins glide effortlessly, their sleek bodies reflecting the sunlight filtering through the water. The prominent shadows and caustics create a shimmering effect on their skin, capturing the beauty of their natural habitat. Each dolphin moves with a fluid grace, occasionally interacting with gentle nudges, showcasing their playful and social nature. The scene is vibrant and dynamic, with the clear blue background accentuating the dolphins' movements, making it an ideal subject for AI recreation.",
"video_path": "sea/mixkit-dolphins-underwater-4133_clip_1.mp4",
"num_inference_steps": 50,
"height": 448,
"width": 832,
"num_frames": 77
},
{
"caption": "In the video, a young woman stands against a vibrant graffiti-covered wall, deeply engrossed in her smartphone. Her expression reflects a mix of focus and subtle satisfaction as she interacts with the screen. She wears a black floral-patterned top, which contrasts with the bright, abstract shapes and bold colors of the mural behind her. As she continues to engage with her phone, a series of like count notifications appear on the screen, indicating a growing online appreciation. The wall behind her features a striking mix of geometric and organic shapes, including swirls of teal, orange, and black, with large humanoid figures in a pop-art style. Her long, light-brown hair frames her face, adding a calm, composed aura amidst the lively backdrop. The video captures a blend of contemporary digital interaction and expressive urban art, creating a dynamic yet harmonious scene.",
"video_path": "Girl/mixkit-girl-looking-at-the-likes-in-her-post-4914_clip_1.mp4",
"num_inference_steps": 50,
"height": 448,
"width": 832,
"num_frames": 77
},
{
"caption": "A young mother and her baby sit comfortably on a bed, surrounded by an inviting, cozy atmosphere. The woman, wearing a sleeveless top and jeans, is gently engaging with the baby, who is dressed in an adorable animal-print onesie. The child is seated on the bed with colorful toys scattered around, including a plush toy and a board book. The warm glow from a hanging lamp casts a soft light on them, enhancing the serene environment. Pillows are propped up against the headboard, providing a cushioned backdrop as the mother leans slightly over to interact with the baby. A small bottle is visible beside her, suggesting a nurturing setting. Her hand gestures animatedly as she holds up a soft, white cushion with red and blue accents, likely stimulating the baby\u2019s curiosity. Their shared moment is filled with affection and joy, a perfect snapshot of familial bonding.",
"video_path": "Baby/mixkit-loving-mother-and-her-baby-playing-with-soft-toys-49966_clip_1.mp4",
"num_inference_steps": 50,
"height": 448,
"width": 832,
"num_frames": 77
},
{
"caption": "A young girl with long brown hair sits at a round wooden table, engrossed in working on her laptop. The laptop screen is a vivid green, suggesting a green screen effect is in use. To her left, a doll dressed in a yellow and white outfit is casually laid on top of some books, adding a playful and innocent touch to the scene. The setting is cozy, with sheer curtains in the background allowing soft natural light to spill into the room. The girl's posture and focused attention on the laptop suggest she is either playing a game or learning something new. This serene and domestic atmosphere is complemented by the slight blur of a dark couch in the foreground, framing the focused activity of the child.",
"video_path": "Girl/mixkit-little-girl-doing-homework-on-a-laptop-4757_clip_1.mp4",
"num_inference_steps": 50,
"height": 448,
"width": 832,
"num_frames": 77
} ]
}
@@ -0,0 +1,123 @@
#!/bin/bash
#SBATCH --job-name=wan_t2v_1.3B_finetune_qat_16
#SBATCH --partition=main
#SBATCH --nodes=4
#SBATCH --ntasks-per-node=1
#SBATCH --gres=gpu:8
#SBATCH --cpus-per-task=128
#SBATCH --mem=1440G
#SBATCH --output=logs/wan_t2v_1.3B_finetune_qat_16.out
#SBATCH --error=logs/wan_t2v_1.3B_finetune_qat_16.err
#SBATCH --exclusive
source ~/conda/miniconda/bin/activate
conda activate matthew-fv
# Basic Info
export WANDB_MODE="online"
export NCCL_P2P_DISABLE=1
export TORCH_NCCL_ENABLE_MONITORING=0
# different cache dir for different processes
export TRITON_CACHE_DIR=/tmp/triton_cache_${SLURM_PROCID}
export MASTER_PORT=29500
export NODE_RANK=$SLURM_PROCID
nodes=( $(scontrol show hostnames $SLURM_JOB_NODELIST) )
export MASTER_ADDR=${nodes[0]}
export CUDA_VISIBLE_DEVICES=$SLURM_LOCALID
export TOKENIZERS_PARALLELISM=false
export WANDB_BASE_URL="https://api.wandb.ai"
export WANDB_MODE=online
export FASTVIDEO_ATTENTION_BACKEND=ATTN_QAT_TRAIN
echo "MASTER_ADDR: $MASTER_ADDR"
echo "NODE_RANK: $NODE_RANK"
# export TRITON_PRINT_AUTOTUNING=1 # to print the best config
export WANDB_API_KEY=YOUR_WANDB_API_KEY
MODEL_PATH="Wan-AI/Wan2.1-T2V-14B-Diffusers"
DATA_DIR=YOUR_DATA_DIR
VALIDATION_DATASET_FILE="examples/training/finetune/wan_t2v_1.3B/crush_smol/validation.json"
NUM_GPUS_PER_NODE=8
TOTAL_GPUS=$((NUM_GPUS_PER_NODE * SLURM_JOB_NUM_NODES))
# export CUDA_VISIBLE_DEVICES=4,5
# Training arguments
training_args=(
--tracker_project_name "wan_t2v_finetune_qat"
--output_dir "checkpoints/wan_14B_t2v_finetune_qat"
--max_train_steps 4000
--train_batch_size 1
--train_sp_batch_size 1
--gradient_accumulation_steps 1
--num_latent_t 20
--num_height 768
--num_width 1280
--num_frames 77
--enable_gradient_checkpointing_type "full" # if OOM enable this
)
# Parallel arguments
parallel_args=(
--num_gpus $TOTAL_GPUS
--sp_size 4
--tp_size 1
--hsdp_replicate_dim 4
--hsdp_shard_dim 8
)
# Model arguments
model_args=(
--model_path $MODEL_PATH
--pretrained_model_name_or_path $MODEL_PATH
)
# Dataset arguments
dataset_args=(
--data_path "$DATA_DIR"
--dataloader_num_workers 4
)
# Validation arguments
validation_args=(
# --log_validation
--validation_dataset_file $VALIDATION_DATASET_FILE
--validation_steps 200
--validation_sampling_steps "50"
--validation_guidance_scale "5.0"
)
# Optimizer arguments
optimizer_args=(
--learning_rate 1e-6
--mixed_precision "bf16"
--weight_only_checkpointing_steps 200
--training_state_checkpointing_steps 200
--weight_decay 0.01
--max_grad_norm 1.0
)
# Miscellaneous arguments
miscellaneous_args=(
--inference_mode False
--checkpoints_total_limit 3
--training_cfg_rate 0.1
--dit_precision "fp32"
--ema_start_step 0
--flow_shift 5
--seed 1000
)
srun torchrun \
--nnodes $SLURM_JOB_NUM_NODES \
--nproc_per_node $NUM_GPUS_PER_NODE \
--node_rank $SLURM_PROCID \
--rdzv_backend=c10d \
--rdzv_endpoint="$MASTER_ADDR:$MASTER_PORT" \
fastvideo/training/wan_training_pipeline.py \
"${parallel_args[@]}" \
"${model_args[@]}" \
"${dataset_args[@]}" \
"${training_args[@]}" \
"${optimizer_args[@]}" \
"${validation_args[@]}" \
"${miscellaneous_args[@]}"
@@ -0,0 +1,131 @@
{
"data": [
{
"caption": "In the video, a woman is elegantly showcasing her earrings, bringing attention to their intricate design with a gentle touch of her fingers. She is bathed in ambient purple and pink lighting, which casts a soft glow on her delicate features and enhances the vivid tones of her lipstick and eye makeup. Her hair is styled to frame her face smoothly, emphasizing the contours of her jawline and cheekbones. The background features a blurred neon light, adding an artistic and modern touch to the overall aesthetic.",
"video_path": "Fashion/mixkit-face-of-an-elegant-and-captivating-woman-41914_clip_1.mp4",
"num_inference_steps": 50,
"height": 768,
"width": 1280,
"num_frames": 77
},
{
"caption": "In the video, a lone rider guides a majestic horse across an expansive, open field as the sun sets in the background. The rider, dressed in a classic blue shirt and wide-brimmed hat, sits confidently in the saddle, silhouetted against the warm glow of the evening sky. The horse moves gracefully, its mane and tail flowing with each step, creating a sense of harmony between horse and rider. Surrounding the pair, towering trees form a natural border, their leaves gently rustling in the breeze. The shadows lengthen on the ground, accentuating the serene and timeless feel of the scene. The distant hills and wooden fences frame the horizon, adding depth to the tranquil landscape. A few horses graze peacefully in the background, blending into the pastoral setting. The overall ambiance evokes a sense of calmness and quietude, capturing a perfect moment in the golden light of dusk.",
"video_path": "Man/mixkit-a-rancher-riding-a-horse-at-sunset-1143_clip_1.mp4",
"num_inference_steps": 50,
"height": 768,
"width": 1280,
"num_frames": 77
},
{
"caption": "In a dimly lit, eerie setting, a mysterious pink bottle labeled \"Authentic 100% organic POISON\" sits prominently in the foreground, casting a menacing aura. The bottle is accentuated by green fog, which swirls lightly around it, enhancing its sinister allure. Behind it, a shadowy golden bottle adorned with a spider emblem subtly emerges, adding an extra layer of mystery to the scene. Dim candles provide faint, flickering light, which complements the dark atmosphere, making the setting ideal for an illusion of hidden dangers.",
"video_path": "smoke/mixkit-poison-in-halloween-ritual-33879_clip_1.mp4",
"num_inference_steps": 50,
"height": 768,
"width": 1280,
"num_frames": 77
},
{
"caption": "The video opens with a tranquil scene in the heart of a dense forest, emphasizing two large, textured tree trunks in the foreground framing the view. Sunlight filters through the canopy above, casting intricate patterns of light and shadow on the trees and the ground. Between the tree trunks, a clear view of a calm, muddy river unfolds, its surface shimmering under the gentle sunlight. The riverbank is decorated with a variety of small bushes and vibrant foliage, subtly transitioning into the deep greens of tall, leafy plants. In the background, the dense forest looms, filled with dark, towering trees, their branches intertwining to form an intricate canopy. The scene is bathed in the soft glow of the sun, creating a serene and picturesque setting. Occasional sunbeams pierce through the foliage, adding a magical aura to the landscape. The vibrant reds and oranges of the smaller plants add contrast, bringing warmth to the earthy tones of the scenery. Overall, this harmonious blend of natural elements creates a peaceful and idyllic forest setting.",
"video_path": "forest/mixkit-view-of-a-river-between-two-old-trees-560_clip_1.mp4",
"num_inference_steps": 50,
"height": 768,
"width": 1280,
"num_frames": 77
},
{
"caption": "In the video, a martial artist dressed in a traditional white uniform with a black belt demonstrates a series of precise movements against a stark black background. The individual gracefully transitions between stances, embodying a sense of focused discipline and control. Each motion is executed with a deliberate pace, showcasing the fluidity of martial arts techniques. The soft lighting creates subtle highlights on the uniform, adding depth to the figure as it moves. The practitioner begins with an open-hand pose, feet firmly grounded, gradually shifting to a powerful forward punch. The fluidity of the sequence displays a mastery of balance and poise. Every trajectory of the limbs is precise and deliberate, capturing the elegance and strength of martial arts. The serene, isolated setting enhances the intensity and concentration of the practitioner. This visual presentation is an elegant interplay of motion and stillness, displaying the art form's discipline and grace.",
"video_path": "Man/mixkit-a-young-man-practicing-his-karate-moves-49635_clip_1.mp4",
"num_inference_steps": 50,
"height": 768,
"width": 1280,
"num_frames": 77
},
{
"caption": "A tranquil coastal scene unfolds with a drone's aerial view capturing a serene beach landscape. The camera glides over a quiet stretch of sandy shoreline, where gentle waves kiss the shore under a clear blue sky. Nestled amidst lush palm trees are a series of traditional thatched-roof huts, their earthy tones blending harmoniously with the natural surroundings. The sandy beach stretches endlessly, bordered by the rhythmic dance of ocean waves on one side and verdant greenery on the other. A pair of white umbrellas is set up on the sand, suggesting a place to relax and enjoy the sun. In the distance, two small human figures can be seen walking leisurely along the water's edge, leaving faint footprints behind them. The scene exudes a calm and inviting atmosphere, with the soft rustle of palm leaves and the whisper of the ocean breeze almost audible. The overall composition is a captivating blend of nature's tranquility and architectural simplicity. This picturesque setting invites viewers to imagine themselves steps away from this idyllic coastal escape.",
"video_path": "beach/mixkit-sunny-beach-in-a-dynamic-shot-from-a-drone-44383_clip_1.mp4",
"num_inference_steps": 50,
"height": 768,
"width": 1280,
"num_frames": 77
},
{
"caption": "A lone figure stands on a large, moss-covered rock, surrounded by the soft rush of a nearby stream. The figure is wearing white sneakers and shorts, with a plaid shirt that hangs loosely in the breeze. The lighting creates dramatic shadows, enhancing the textures of the rock and the subtle movement of the water below. In the background, a waterfall cascades into the stream, completing this tranquil and serene nature scene.",
"video_path": "forest/mixkit-woman-standing-in-front-of-waterfall-559_clip_1.mp4",
"num_inference_steps": 50,
"height": 768,
"width": 1280,
"num_frames": 77
},
{
"caption": "In an industrial setting, a person leans casually against a railing, exuding a sense of confidence and composure. They are wearing a striking outfit, consisting of a vibrant, patterned jacket over a simple white crop top, creating a bold contrast. The atmosphere is infused with warm, ambient lighting that casts soft shadows on the concrete walls and metallic surfaces. Intricate wiring and pipes form an intricate backdrop, enhancing the urban aesthetic. Their relaxed posture and direct, engaging gaze suggest a sense of ease in this industrial environment. This scene encapsulates a blend of modern fashion and gritty, urban architecture, creating a visually compelling narrative.",
"video_path": "Fashion/mixkit-portrait-of-a-hipster-woman-walking-down-a-stairs-1297_clip_1.mp4",
"num_inference_steps": 50,
"height": 768,
"width": 1280,
"num_frames": 77
},
{
"caption": "A man is energetically stretching in an open-air setting, surrounded by rows of vibrant red seats that suggest an amphitheater or outdoor venue. He wears a sleeveless black shirt layered with a hooded vest, emphasizing his athletic build as he engages in a warm-up routine. Behind him, the striking modern architecture of the building features geometric panels, with large sections of glass and overlapping metallic beams creating a dynamic backdrop. The scene captures the contrast between his focused movements and the static, bold design of the structure, while the surrounding greenery adds a touch of nature to the environment. The overall atmosphere is one of preparation and anticipation, with the man appearing determined and ready for an upcoming event or performance.",
"video_path": "Sport/mixkit-man-doing-arm-stretches-595_clip_1.mp4",
"num_inference_steps": 50,
"height": 768,
"width": 1280,
"num_frames": 77
},
{
"caption": "A young woman is seated on the floor in front of a plush, beige tufted couch, fully engrossed in sorting through a stack of papers. Her dark hair falls loosely past her shoulders, and she wears a green plaid shirt, contributing to the casual yet focused atmosphere. She gently places the papers onto a small round white table, occasionally lifting individual sheets to examine them more closely. Her expression shifts subtly, reflecting concentration and contemplation as she processes the information on the pages. Two small, round nested tables hold her documents, along with a small plant in a gray pot, adding a touch of greenery to the scene. The background features a dark paneled wall, creating a contrasting backdrop for the light-colored furniture. The setting is tranquil and organized, the couch and tables arranged symmetrically, conveying a sense of harmony. A calculator rests on the smaller table, hinting at a task involving calculations or budgeting.",
"video_path": "Woman/mixkit-frustrated-woman-throws-paperwork-on-the-floor-4526_clip_1.mp4",
"num_inference_steps": 50,
"height": 768,
"width": 1280,
"num_frames": 77
},
{
"caption": "A heavily rusted metal gate stands firmly locked, with two vertical bars joined by a thick, old chain that loops elegantly around them. The chain's texture is coarse and rugged, its surface reflecting varying shades of orange and brown, indicative of years exposed to the elements. At the heart of the chain, a black iron padlock, slightly worn yet imposing, secures the gate, its curves and edges smooth against the aged links. The gate's metalwork is outlined by a backdrop of soft, blurred greenery, suggesting a serene and isolated location beyond the barrier. Tall trees rise in the distance, their trunks and leaves creating a lush, forest-like setting that contrasts with the gate's severe rust. A pathway leads away from the gate, its surface uneven with patches of moss and weathered stone visible in the soft focus, inviting yet inaccessible. The ambiance is quiet and mysterious, with a sense of abandonment hanging subtly in the air, evoking curiosity about what lies beyond. Shadows play across the gate, cast by branches swaying gently in the breeze, adding to the dynamic interaction of light and texture. This scene, rich in detail and atmosphere, captures the viewer's imagination, evoking both the allure of the forbidden and the beauty of decay.",
"video_path": "forest/mixkit-rusty-fence-with-a-chain-of-a-property-in-nature-5294_clip_1.mp4",
"num_inference_steps": 50,
"height": 768,
"width": 1280,
"num_frames": 77
},
{
"caption": "In a serene and softly lit yoga studio, three individuals engage in a yoga session, each performing an upward-facing stretch. The central figure is a woman with shoulder-length brown hair, dressed in a light cropped top and green leggings, her posture reflecting grace and concentration. To her right, another participant, a woman in a purple outfit, mirrors the pose with equal poise. On her left, a person with a bun focuses intently, supported slightly by yoga blocks beneath their hands. The warm-colored wooden floor contrasts soothingly with the soft pastel mural on the back wall, featuring an abstract design and partial visage of a serene face. Natural light floods the space from a large window on the right, where lush greens peek through, adding an element of tranquility. In the corner of the room, a collection of meditation instruments, including a gong and a Buddha statue, subtly frame the peaceful setting. The mood is calm yet focused, as all three participants are deeply engaged in their practice. The scene combines elements of balance, harmony, and a shared journey towards mindfulness. This depiction captures the essence of a yoga session that blends personal growth with collective experience.",
"video_path": "People/mixkit-small-group-of-people-doing-yoga-together-43730_clip_1.mp4",
"num_inference_steps": 50,
"height": 768,
"width": 1280,
"num_frames": 77
},
{
"caption": "In the deep blue expanse of the ocean, two dolphins glide effortlessly, their sleek bodies reflecting the sunlight filtering through the water. The prominent shadows and caustics create a shimmering effect on their skin, capturing the beauty of their natural habitat. Each dolphin moves with a fluid grace, occasionally interacting with gentle nudges, showcasing their playful and social nature. The scene is vibrant and dynamic, with the clear blue background accentuating the dolphins' movements, making it an ideal subject for AI recreation.",
"video_path": "sea/mixkit-dolphins-underwater-4133_clip_1.mp4",
"num_inference_steps": 50,
"height": 768,
"width": 1280,
"num_frames": 77
},
{
"caption": "In the video, a young woman stands against a vibrant graffiti-covered wall, deeply engrossed in her smartphone. Her expression reflects a mix of focus and subtle satisfaction as she interacts with the screen. She wears a black floral-patterned top, which contrasts with the bright, abstract shapes and bold colors of the mural behind her. As she continues to engage with her phone, a series of like count notifications appear on the screen, indicating a growing online appreciation. The wall behind her features a striking mix of geometric and organic shapes, including swirls of teal, orange, and black, with large humanoid figures in a pop-art style. Her long, light-brown hair frames her face, adding a calm, composed aura amidst the lively backdrop. The video captures a blend of contemporary digital interaction and expressive urban art, creating a dynamic yet harmonious scene.",
"video_path": "Girl/mixkit-girl-looking-at-the-likes-in-her-post-4914_clip_1.mp4",
"num_inference_steps": 50,
"height": 768,
"width": 1280,
"num_frames": 77
},
{
"caption": "A young mother and her baby sit comfortably on a bed, surrounded by an inviting, cozy atmosphere. The woman, wearing a sleeveless top and jeans, is gently engaging with the baby, who is dressed in an adorable animal-print onesie. The child is seated on the bed with colorful toys scattered around, including a plush toy and a board book. The warm glow from a hanging lamp casts a soft light on them, enhancing the serene environment. Pillows are propped up against the headboard, providing a cushioned backdrop as the mother leans slightly over to interact with the baby. A small bottle is visible beside her, suggesting a nurturing setting. Her hand gestures animatedly as she holds up a soft, white cushion with red and blue accents, likely stimulating the baby\u2019s curiosity. Their shared moment is filled with affection and joy, a perfect snapshot of familial bonding.",
"video_path": "Baby/mixkit-loving-mother-and-her-baby-playing-with-soft-toys-49966_clip_1.mp4",
"num_inference_steps": 50,
"height": 768,
"width": 1280,
"num_frames": 77
},
{
"caption": "A young girl with long brown hair sits at a round wooden table, engrossed in working on her laptop. The laptop screen is a vivid green, suggesting a green screen effect is in use. To her left, a doll dressed in a yellow and white outfit is casually laid on top of some books, adding a playful and innocent touch to the scene. The setting is cozy, with sheer curtains in the background allowing soft natural light to spill into the room. The girl's posture and focused attention on the laptop suggest she is either playing a game or learning something new. This serene and domestic atmosphere is complemented by the slight blur of a dark couch in the foreground, framing the focused activity of the child.",
"video_path": "Girl/mixkit-little-girl-doing-homework-on-a-laptop-4757_clip_1.mp4",
"num_inference_steps": 50,
"height": 768,
"width": 1280,
"num_frames": 77
} ]
}
+191 -9
View File
@@ -12,6 +12,18 @@ else()
enable_language(CUDA)
# Ensure CUDA toolkit targets (CUDA::cudart, CUDA::cuda_driver, etc.) are available.
find_package(CUDAToolkit REQUIRED)
if(NOT DEFINED CUDA_TOOLKIT_ROOT_DIR)
if(DEFINED CUDAToolkit_ROOT)
set(CUDA_TOOLKIT_ROOT_DIR "${CUDAToolkit_ROOT}" CACHE PATH
"CUDA toolkit root directory" FORCE)
elseif(DEFINED ENV{CUDAToolkit_ROOT})
set(CUDA_TOOLKIT_ROOT_DIR "$ENV{CUDAToolkit_ROOT}" CACHE PATH
"CUDA toolkit root directory" FORCE)
elseif(DEFINED ENV{CUDA_HOME})
set(CUDA_TOOLKIT_ROOT_DIR "$ENV{CUDA_HOME}" CACHE PATH
"CUDA toolkit root directory" FORCE)
endif()
endif()
endif()
# Import common utils if needed, but we keep it simple for now
@@ -19,13 +31,46 @@ endif()
# Find Python and Torch
find_package(Python COMPONENTS Interpreter Development.Module REQUIRED)
# Robustly find Torch include paths using Python
# Locate the installed torch package without importing it. This keeps CMake
# configure working even on nodes where CUDA runtime libraries are not yet on
# the dynamic loader path.
execute_process(
COMMAND "${Python_EXECUTABLE}" -c "import torch; from torch.utils.cpp_extension import include_paths; print(';'.join(include_paths()))"
OUTPUT_VARIABLE TORCH_INCLUDE_PATHS
COMMAND "${Python_EXECUTABLE}" -c "import sysconfig; print(sysconfig.get_path('platlib'))"
OUTPUT_VARIABLE PYTHON_PLATLIB
OUTPUT_STRIP_TRAILING_WHITESPACE
)
list(APPEND TORCH_INCLUDE_DIRS ${TORCH_INCLUDE_PATHS})
execute_process(
COMMAND "${Python_EXECUTABLE}" -c "import sysconfig; print(sysconfig.get_path('purelib'))"
OUTPUT_VARIABLE PYTHON_PURELIB
OUTPUT_STRIP_TRAILING_WHITESPACE
)
set(TORCH_PYTHON_PACKAGE_DIR "")
foreach(_candidate
"${PYTHON_PLATLIB}/torch"
"${PYTHON_PURELIB}/torch"
)
if(EXISTS "${_candidate}")
set(TORCH_PYTHON_PACKAGE_DIR "${_candidate}")
break()
endif()
endforeach()
if(NOT TORCH_PYTHON_PACKAGE_DIR)
message(FATAL_ERROR "Could not locate the installed torch Python package.")
endif()
list(APPEND TORCH_INCLUDE_DIRS
"${TORCH_PYTHON_PACKAGE_DIR}/include"
"${TORCH_PYTHON_PACKAGE_DIR}/include/torch/csrc/api/include"
)
if(NOT Torch_DIR)
set(_TORCH_CONFIG_DIR "${TORCH_PYTHON_PACKAGE_DIR}/share/cmake/Torch")
if(EXISTS "${_TORCH_CONFIG_DIR}/TorchConfig.cmake")
set(Torch_DIR "${_TORCH_CONFIG_DIR}" CACHE PATH "Path to Torch CMake config" FORCE)
endif()
endif()
# Find Torch package (still useful for libraries)
find_package(Torch REQUIRED)
@@ -50,6 +95,21 @@ include_directories(
set(FASTVIDEO_KERNEL_BUILD_TK "AUTO" CACHE STRING "Build ThunderKittens kernels: AUTO/ON/OFF")
set_property(CACHE FASTVIDEO_KERNEL_BUILD_TK PROPERTY STRINGS AUTO ON OFF)
set(_FASTVIDEO_KERNEL_BUILD_ATTN_QAT_INFER_DEFAULT "AUTO")
if(DEFINED FASTVIDEO_KERNEL_BUILD_MODIFIED_SAGE3 AND NOT DEFINED CACHE{FASTVIDEO_KERNEL_BUILD_ATTN_QAT_INFER})
set(_FASTVIDEO_KERNEL_BUILD_ATTN_QAT_INFER_DEFAULT "${FASTVIDEO_KERNEL_BUILD_MODIFIED_SAGE3}")
endif()
set(FASTVIDEO_KERNEL_BUILD_ATTN_QAT_INFER "${_FASTVIDEO_KERNEL_BUILD_ATTN_QAT_INFER_DEFAULT}" CACHE STRING
"Build attn_qat_infer Blackwell inference kernels: AUTO/ON/OFF")
set_property(CACHE FASTVIDEO_KERNEL_BUILD_ATTN_QAT_INFER PROPERTY STRINGS AUTO ON OFF)
if(DEFINED FASTVIDEO_KERNEL_BUILD_MODIFIED_SAGE3)
message(DEPRECATION
"FASTVIDEO_KERNEL_BUILD_MODIFIED_SAGE3 is deprecated. "
"Use FASTVIDEO_KERNEL_BUILD_ATTN_QAT_INFER instead.")
endif()
# Prefer environment variable (used by CI) if CMake var is not explicitly set.
if(NOT DEFINED TORCH_CUDA_ARCH_LIST AND DEFINED ENV{TORCH_CUDA_ARCH_LIST})
set(TORCH_CUDA_ARCH_LIST "$ENV{TORCH_CUDA_ARCH_LIST}")
@@ -57,6 +117,7 @@ endif()
message(STATUS "TORCH_CUDA_ARCH_LIST (cmake/env): ${TORCH_CUDA_ARCH_LIST}")
message(STATUS "FASTVIDEO_KERNEL_BUILD_TK: ${FASTVIDEO_KERNEL_BUILD_TK}")
message(STATUS "FASTVIDEO_KERNEL_BUILD_ATTN_QAT_INFER: ${FASTVIDEO_KERNEL_BUILD_ATTN_QAT_INFER}")
set(ENABLE_TK_KERNELS OFF)
if(FASTVIDEO_KERNEL_BUILD_TK STREQUAL "ON")
@@ -91,6 +152,54 @@ else()
message(STATUS "ThunderKittens kernels: DISABLED (will use Triton fallbacks at runtime)")
endif()
set(ENABLE_ATTN_QAT_INFER OFF)
if(GPU_BACKEND STREQUAL "ROCM")
message(STATUS "attn_qat_infer kernels: DISABLED (ROCm build)")
else()
set(_WANTS_ATTN_QAT_INFER OFF)
if(FASTVIDEO_KERNEL_BUILD_ATTN_QAT_INFER STREQUAL "ON")
set(_WANTS_ATTN_QAT_INFER ON)
elseif(FASTVIDEO_KERNEL_BUILD_ATTN_QAT_INFER STREQUAL "AUTO")
if(TORCH_CUDA_ARCH_LIST)
string(REGEX MATCH
"(^|[; ,])((12\\.0a)|(120a)|(sm_120a))([; ,]|$)"
_HAS_120A "${TORCH_CUDA_ARCH_LIST}")
if(_HAS_120A)
set(_WANTS_ATTN_QAT_INFER ON)
endif()
else()
execute_process(
COMMAND "${Python_EXECUTABLE}" -c
"import torch; print('1' if (torch.cuda.is_available() and torch.version.cuda and torch.cuda.get_device_capability()[0] >= 12) else '0')"
OUTPUT_VARIABLE _LOCAL_HAS_BLACKWELL
OUTPUT_STRIP_TRAILING_WHITESPACE
ERROR_QUIET
)
if(_LOCAL_HAS_BLACKWELL STREQUAL "1")
set(_WANTS_ATTN_QAT_INFER ON)
endif()
endif()
endif()
if(_WANTS_ATTN_QAT_INFER)
if(CUDAToolkit_VERSION VERSION_LESS 12.8)
message(WARNING
"attn_qat_infer kernels require CUDA Toolkit 12.8+. "
"Skipping because CUDAToolkit_VERSION=${CUDAToolkit_VERSION}.")
else()
set(ENABLE_ATTN_QAT_INFER ON)
endif()
endif()
if(ENABLE_ATTN_QAT_INFER)
message(STATUS "attn_qat_infer kernels: ENABLED")
else()
message(STATUS
"attn_qat_infer kernels: DISABLED "
"(requires CUDA 12.8+ and Blackwell sm_120a)")
endif()
endif()
# Always try to build the extension if CUDA is available, but conditionally add sources/flags
set(BUILD_CXX_KERNELS ON)
@@ -161,12 +270,15 @@ if(BUILD_CXX_KERNELS)
# Also link against libtorch_python to satisfy Python-binding symbols
# (e.g., torch::PyWarningHandler) required by torch/extension.h.
execute_process(
COMMAND "${Python_EXECUTABLE}" -c "import torch; from pathlib import Path; p=Path(torch.__file__).parent/'lib'; m=sorted(p.glob('libtorch_python*')); print(str(m[0]) if m else '')"
OUTPUT_VARIABLE TORCH_PYTHON_LIBRARY_PATH
OUTPUT_STRIP_TRAILING_WHITESPACE
ERROR_QUIET
file(GLOB TORCH_PYTHON_LIBRARY_CANDIDATES
"${TORCH_PYTHON_PACKAGE_DIR}/lib/libtorch_python*"
)
list(LENGTH TORCH_PYTHON_LIBRARY_CANDIDATES _TORCH_PYTHON_LIBRARY_COUNT)
if(_TORCH_PYTHON_LIBRARY_COUNT GREATER 0)
list(GET TORCH_PYTHON_LIBRARY_CANDIDATES 0 TORCH_PYTHON_LIBRARY_PATH)
else()
set(TORCH_PYTHON_LIBRARY_PATH "")
endif()
if(TORCH_PYTHON_LIBRARY_PATH)
message(STATUS "TORCH_PYTHON_LIBRARY_PATH: ${TORCH_PYTHON_LIBRARY_PATH}")
target_link_libraries(fastvideo_kernel_ops PRIVATE "${TORCH_PYTHON_LIBRARY_PATH}")
@@ -183,3 +295,73 @@ if(BUILD_CXX_KERNELS)
install(TARGETS fastvideo_kernel_ops LIBRARY DESTINATION fastvideo_kernel/_C)
endif()
if(ENABLE_ATTN_QAT_INFER)
set(ATTN_QAT_INFER_DIR ${CMAKE_SOURCE_DIR}/attn_qat_infer)
set(ATTN_QAT_INFER_INCLUDE_DIRS
${ATTN_QAT_INFER_DIR}
${CMAKE_SOURCE_DIR}/include/cutlass/include
${CMAKE_SOURCE_DIR}/include/cutlass/tools/util/include
${TORCH_INCLUDE_DIRS}
)
set(ATTN_QAT_INFER_CUDA_FLAGS
"-O3"
"-std=c++17"
"-U__CUDA_NO_HALF_OPERATORS__"
"-U__CUDA_NO_HALF_CONVERSIONS__"
"-U__CUDA_NO_BFLOAT16_OPERATORS__"
"-U__CUDA_NO_BFLOAT16_CONVERSIONS__"
"-U__CUDA_NO_BFLOAT162_OPERATORS__"
"-U__CUDA_NO_BFLOAT162_CONVERSIONS__"
"--expt-relaxed-constexpr"
"--expt-extended-lambda"
"--use_fast_math"
"--ptxas-options=--verbose,--warn-on-local-memory-usage"
"-lineinfo"
"-DCUTLASS_DEBUG_TRACE_LEVEL=0"
"-DNDEBUG"
"-DQBLKSIZE=128"
"-DKBLKSIZE=128"
"-DCTA256"
"-DDQINRMEM"
)
Python_add_library(fp4attn_cuda MODULE WITH_SOABI
attn_qat_infer/blackwell/api.cu
)
target_include_directories(fp4attn_cuda PRIVATE ${ATTN_QAT_INFER_INCLUDE_DIRS})
target_compile_definitions(fp4attn_cuda PRIVATE TORCH_EXTENSION_NAME=fp4attn_cuda)
target_compile_options(fp4attn_cuda PRIVATE
$<$<COMPILE_LANGUAGE:CXX>:-O3 -std=c++17>
$<$<COMPILE_LANGUAGE:CUDA>:${ATTN_QAT_INFER_CUDA_FLAGS}>
)
set_target_properties(fp4attn_cuda PROPERTIES
CUDA_ARCHITECTURES "120a"
CXX_STANDARD 17
CUDA_STANDARD 17
)
target_link_libraries(fp4attn_cuda PRIVATE ${TORCH_LIBRARIES} CUDA::cudart CUDA::cuda_driver)
Python_add_library(fp4quant_cuda MODULE WITH_SOABI
attn_qat_infer/quantization/fp4_quantization_4d.cu
)
target_include_directories(fp4quant_cuda PRIVATE ${ATTN_QAT_INFER_INCLUDE_DIRS})
target_compile_definitions(fp4quant_cuda PRIVATE TORCH_EXTENSION_NAME=fp4quant_cuda)
target_compile_options(fp4quant_cuda PRIVATE
$<$<COMPILE_LANGUAGE:CXX>:-O3 -std=c++17>
$<$<COMPILE_LANGUAGE:CUDA>:${ATTN_QAT_INFER_CUDA_FLAGS}>
)
set_target_properties(fp4quant_cuda PROPERTIES
CUDA_ARCHITECTURES "120a"
CXX_STANDARD 17
CUDA_STANDARD 17
)
target_link_libraries(fp4quant_cuda PRIVATE ${TORCH_LIBRARIES} CUDA::cudart CUDA::cuda_driver)
if(TORCH_PYTHON_LIBRARY_PATH)
target_link_libraries(fp4attn_cuda PRIVATE "${TORCH_PYTHON_LIBRARY_PATH}")
target_link_libraries(fp4quant_cuda PRIVATE "${TORCH_PYTHON_LIBRARY_PATH}")
endif()
install(TARGETS fp4attn_cuda LIBRARY DESTINATION .)
install(TARGETS fp4quant_cuda LIBRARY DESTINATION .)
endif()
+1
View File
@@ -2,5 +2,6 @@ include LICENSE
include README.md
include pyproject.toml
recursive-include python/fastvideo_kernel *.py
recursive-include attn_qat_infer *.py *.cu *.cuh *.cpp *.h
recursive-include csrc *.cu *.cuh *.cpp *.h
recursive-include include/tk *.cu *.cuh *.cpp *.h *.src
+17
View File
@@ -20,6 +20,11 @@ cd fastvideo-kernel
./build.sh
```
On supported Blackwell environments, the same install also packages
`attn_qat_infer` and builds its `fp4attn_cuda` / `fp4quant_cuda`
extensions directly from `fastvideo-kernel/attn_qat_infer/`. This path
requires CUDA Toolkit 12.8+ and targets `sm_120a`.
### Rocm Build
If you are in a rocm environment without the compilation toolchaine of CUDA.
@@ -58,6 +63,18 @@ cd fastvideo-kernel
python benchmarks/bench_vsa.py --batch_size 1 --num_heads 16 --head_dim 128 --q_seq_lens 49152 --topk 64
```
### Attn QAT Attention Benchmarks
The Attn QAT microbenchmarks now live alongside the kernel package:
```bash
cd fastvideo-kernel
python benchmarks/benchmark_flashattn2.py --batch-size 1 --num-heads 16 --seq-len 4096 --head-dim 128
python benchmarks/benchmark_sageattn3.py --batch-size 1 --num-heads 16 --seq-len 4096 --head-dim 128
python benchmarks/benchmark_blockscaled_fp4_attn.py --batch-size 1 --num-heads 16 --seq-len 4096 --head-dim 128
python benchmarks/benchmark_combined.py --output benchmark_attention.png
```
### TurboDiffusion Kernels
This package also includes kernels from [TurboDiffusion](https://github.com/thu-ml/TurboDiffusion), including INT8 GEMM, Quantization, RMSNorm and LayerNorm.
@@ -0,0 +1,16 @@
"""
Copyright (c) 2025 by SageAttention team.
Licensed under the Apache License, Version 2.0 (the "License");
you may not use this file except in compliance with the License.
You may obtain a copy of the License at
http://www.apache.org/licenses/LICENSE-2.0
Unless required by applicable law or agreed to in writing, software
distributed under the License is distributed on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
See the License for the specific language governing permissions and
limitations under the License.
"""
from .api import sageattn_blackwell
+185
View File
@@ -0,0 +1,185 @@
# Modified from the original SageATtention3 code
"""
Copyright (c) 2025 by SageAttention team.
Licensed under the Apache License, Version 2.0 (the "License");
you may not use this file except in compliance with the License.
You may obtain a copy of the License at
http://www.apache.org/licenses/LICENSE-2.0
Unless required by applicable law or agreed to in writing, software
distributed under the License is distributed on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
See the License for the specific language governing permissions and
limitations under the License.
"""
import torch
import triton
import triton.language as tl
import torch.nn.functional as F
from typing import Tuple
from torch.nn.functional import scaled_dot_product_attention as sdpa
import fp4attn_cuda
import fp4quant_cuda
# Centralized block size configuration for sageattn_blackwell kernels
# These should match the values in fastvideo/attention/backends/sageattn/blackwell/block_config.h
BLOCK_M = 128 # Block size for M dimension (query sequence length)
BLOCK_N = 128 # Block size for N dimension (key/value sequence length)
@triton.jit
def group_mean_kernel(
q_ptr,
q_out_ptr,
qm_out_ptr,
B, H, L, D: tl.constexpr,
stride_qb, stride_qh, stride_ql, stride_qd,
stride_qmb, stride_qmh, stride_qml, stride_qmd,
GROUP_SIZE: tl.constexpr
):
pid_b = tl.program_id(0)
pid_h = tl.program_id(1)
pid_group = tl.program_id(2)
group_start = pid_group * GROUP_SIZE
offsets = group_start + tl.arange(0, GROUP_SIZE)
q_offsets = pid_b * stride_qb + pid_h * stride_qh + offsets[:, None] * stride_ql + tl.arange(0, D)[None, :] * stride_qd
q_group = tl.load(q_ptr + q_offsets)
qm_group = tl.sum(q_group, axis=0) / GROUP_SIZE
q_group = q_group - qm_group
tl.store(q_out_ptr + q_offsets, q_group)
qm_offset = pid_b * stride_qmb + pid_h * stride_qmh + pid_group * stride_qml + tl.arange(0, D) * stride_qmd
tl.store(qm_out_ptr + qm_offset, qm_group)
def triton_group_mean(q: torch.Tensor):
B, H, L, D = q.shape
GROUP_SIZE = BLOCK_M
num_groups = L // GROUP_SIZE
q_out = torch.empty_like(q) # [B, H, L, D]
qm = torch.empty(B, H, num_groups, D, device=q.device, dtype=q.dtype)
grid = (B, H, num_groups)
group_mean_kernel[grid](
q, q_out, qm,
B, H, L, D,
q.stride(0), q.stride(1), q.stride(2), q.stride(3),
qm.stride(0), qm.stride(1), qm.stride(2), qm.stride(3),
GROUP_SIZE=GROUP_SIZE
)
return q_out, qm
def preprocess_qkv(q: torch.Tensor, k: torch.Tensor, v: torch.Tensor, per_block_mean: bool = True, enable_smoothing_q: bool = False, enable_smoothing_k: bool = False):
def pad_to_block_size(x):
L = x.size(2)
pad_len = (BLOCK_M - L % BLOCK_M) % BLOCK_M
if pad_len == 0:
return x.contiguous()
return F.pad(x, (0, 0, 0, pad_len), value=0).contiguous()
if enable_smoothing_k:
k -= k.mean(dim=-2, keepdim=True)
q, k, v = map(lambda x: pad_to_block_size(x), [q, k, v])
if per_block_mean and enable_smoothing_q:
q, qm = triton_group_mean(q)
elif enable_smoothing_q:
qm = q.mean(dim=-2, keepdim=True)
q = q - qm
if enable_smoothing_q:
delta_s = torch.matmul(qm, k.transpose(-2, -1)).to(torch.float32).contiguous()
else: # used to disable q smoothing
B, H, L, D = q.shape
delta_s = torch.zeros((B, H, L // BLOCK_M, k.shape[2]), device=q.device, dtype=torch.float32)
return q, k, v, delta_s
def scale_and_quant_fp4(x: torch.Tensor) -> Tuple[torch.Tensor, torch.Tensor]:
assert x.ndim == 4
B, H, N, D = x.shape
packed_fp4 = torch.empty((B, H, N, D // 2), device=x.device, dtype=torch.uint8)
fp8_scale = torch.empty((B, H, N, D // 16), device=x.device, dtype=torch.float8_e4m3fn)
fp4quant_cuda.scaled_fp4_quant(x, packed_fp4, fp8_scale, 1)
return packed_fp4, fp8_scale
def scale_and_quant_fp4_permute(x: torch.Tensor) -> Tuple[torch.Tensor, torch.Tensor]:
assert x.ndim == 4
B, H, N, D = x.shape
packed_fp4 = torch.empty((B, H, N, D // 2), device=x.device, dtype=torch.uint8)
fp8_scale = torch.empty((B, H, N, D // 16), device=x.device, dtype=torch.float8_e4m3fn)
fp4quant_cuda.scaled_fp4_quant_permute(x, packed_fp4, fp8_scale, 1)
return packed_fp4, fp8_scale
def scale_and_quant_fp4_transpose(x: torch.Tensor) -> Tuple[torch.Tensor, torch.Tensor]:
assert x.ndim == 4
B, H, N, D = x.shape
packed_fp4 = torch.empty((B, H, D, N // 2), device=x.device, dtype=torch.uint8)
fp8_scale = torch.empty((B, H, D, N // 16), device=x.device, dtype=torch.float8_e4m3fn)
fp4quant_cuda.scaled_fp4_quant_trans(x, packed_fp4, fp8_scale, 1)
return packed_fp4, fp8_scale
def blockscaled_fp4_attn(qlist: Tuple,
klist: Tuple,
vlist: Tuple,
delta_s: torch.Tensor,
KL: int,
is_causal: bool = False,
per_block_mean: bool = True,
is_bf16: bool = True,
single_level_p_quant: bool = False
):
softmax_scale = (qlist[0].shape[-1] * 2) ** (-0.5)
return fp4attn_cuda.fwd(qlist[0], klist[0], vlist[0], qlist[1], klist[1], vlist[1], delta_s, KL, None, softmax_scale, is_causal, per_block_mean, is_bf16, single_level_p_quant)
def sageattn_blackwell(q, k, v, attn_mask = None, is_causal = False, per_block_mean = True, single_level_p_quant = True, **kwargs):
"""
SageAttention3 Blackwell kernel for FP4 attention.
Args:
q: Query tensor [B, H, L, D]
k: Key tensor [B, H, L, D]
v: Value tensor [B, H, L, D]
attn_mask: Attention mask (not used)
is_causal: Whether to use causal masking
per_block_mean: Whether to use per-block mean for Q smoothing
single_level_p_quant: If True, use single-level quantization: s_P2, P̂_2 = φ(P̃) directly
(standard per-block FP4 quantization like V, no s_P1).
If False (default), use two-level quantization:
s_P1 = rowmax(P̃)/(448×6), then s_P2, P̂_2 = φ(P̃/s_P1).
**kwargs: Additional arguments (ignored)
Returns:
Output tensor [B, H, L, D]
"""
if q.size(-1) >= 256:
print(f"Unsupported Headdim {q.size(-1)}")
return sdpa(q, k, v, is_causal = is_causal)
QL = q.size(2)
KL = k.size(2)
is_bf16 = q.dtype == torch.bfloat16
q, k, v, delta_s = preprocess_qkv(q, k, v, per_block_mean)
qlist_from_cuda = scale_and_quant_fp4(q)
klist_from_cuda = scale_and_quant_fp4_permute(k)
vlist_from_cuda = scale_and_quant_fp4_transpose(v)
o_fp4 = blockscaled_fp4_attn(
qlist_from_cuda,
klist_from_cuda,
vlist_from_cuda,
delta_s,
KL,
is_causal,
per_block_mean,
is_bf16,
single_level_p_quant
)[0][:, :, :QL, :].contiguous()
return o_fp4
@@ -0,0 +1 @@
__version__ = "3.0.0.b1"
@@ -0,0 +1,346 @@
// Modified from the original SageAttention3 code
/*
* Copyright (c) 2025 by SageAttention team.
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
// Include these 2 headers instead of torch/extension.h since we don't need all of the torch headers.
#include <torch/python.h>
#include <torch/nn/functional.h>
#include <ATen/cuda/CUDAContext.h>
#include <c10/cuda/CUDAGuard.h>
#include <cutlass/numeric_types.h>
#include "params.h"
#include "launch.h"
#include "static_switch.h"
#include "block_config.h"
#define CHECK_DEVICE(x) TORCH_CHECK(x.is_cuda(), #x " must be on CUDA")
#define CHECK_SHAPE(x, ...) TORCH_CHECK(x.sizes() == torch::IntArrayRef({__VA_ARGS__}), #x " must have shape (" #__VA_ARGS__ ")")
#define CHECK_CONTIGUOUS(x) TORCH_CHECK(x.is_contiguous(), #x " must be contiguous")
void set_params_fprop(Flash_fwd_params &params,
// sizes
const size_t b,
const size_t seqlen_q,
const size_t seqlen_k,
const size_t unpadded_seqlen_k,
const size_t seqlen_q_rounded,
const size_t seqlen_k_rounded,
const size_t h,
const size_t h_k,
const size_t d,
const size_t d_rounded,
// device pointers
const at::Tensor q,
const at::Tensor k,
const at::Tensor v,
const at::Tensor delta_s,
at::Tensor out,
const at::Tensor sfq,
const at::Tensor sfk,
const at::Tensor sfv,
void *cu_seqlens_q_d,
void *cu_seqlens_k_d,
void *seqused_k,
void *p_d,
void *softmax_lse_d,
float p_dropout,
float softmax_scale,
int window_size_left,
int window_size_right,
bool per_block_mean,
bool is_bf16,
bool single_level_p_quant=false,
bool seqlenq_ngroups_swapped=false) {
// Reset the parameters
params = {};
// Set the pointers and strides.
params.q_ptr = q.data_ptr();
params.k_ptr = k.data_ptr();
params.v_ptr = v.data_ptr();
params.delta_s_ptr = delta_s.data_ptr();
params.sfq_ptr = sfq.data_ptr();
params.sfk_ptr = sfk.data_ptr();
params.sfv_ptr = sfv.data_ptr();
// All stride are in elements, not bytes.
params.q_row_stride = q.stride(-2) * 2;
params.k_row_stride = k.stride(-2) * 2;
params.v_row_stride = v.stride(-2) * 2;;
params.q_head_stride = q.stride(-3) * 2;
params.k_head_stride = k.stride(-3) * 2;
params.v_head_stride = v.stride(-3) * 2; // for packed q k v
params.ds_row_stride = delta_s.stride(-2);
params.ds_head_stride = delta_s.stride(-3);
params.sfq_row_stride = sfq.stride(-2);
params.sfk_row_stride = sfk.stride(-2);
params.sfv_row_stride = sfv.stride(-2);
params.sfq_head_stride = sfq.stride(-3);
params.sfk_head_stride = sfk.stride(-3);
params.sfv_head_stride = sfv.stride(-3);
params.o_ptr = out.data_ptr();
params.o_row_stride = out.stride(-2);
params.o_head_stride = out.stride(-3);
if (cu_seqlens_q_d == nullptr) {
params.q_batch_stride = q.stride(0) * 2;
params.k_batch_stride = k.stride(0) * 2;
params.v_batch_stride = v.stride(0) * 2;
params.ds_batch_stride = delta_s.stride(0);
params.sfq_batch_stride = sfq.stride(0);
params.sfk_batch_stride = sfk.stride(0);
params.sfv_batch_stride = sfv.stride(0);
params.o_batch_stride = out.stride(0);
if (seqlenq_ngroups_swapped) {
params.q_batch_stride *= seqlen_q;
params.o_batch_stride *= seqlen_q;
}
}
params.cu_seqlens_q = static_cast<int *>(cu_seqlens_q_d);
params.cu_seqlens_k = static_cast<int *>(cu_seqlens_k_d);
params.seqused_k = static_cast<int *>(seqused_k);
// P = softmax(QK^T)
params.p_ptr = p_d;
// Softmax sum
params.softmax_lse_ptr = softmax_lse_d;
// Set the dimensions.
params.b = b;
params.h = h;
params.h_k = h_k;
params.h_h_k_ratio = h / h_k;
params.seqlen_q = seqlen_q;
params.seqlen_k = seqlen_k;
params.unpadded_seqlen_k = unpadded_seqlen_k;
params.seqlen_q_rounded = seqlen_q_rounded;
params.seqlen_k_rounded = seqlen_k_rounded;
params.d = d;
params.d_rounded = d_rounded;
params.head_divmod = cutlass::FastDivmod(int(h));
// Set the different scale values.
params.scale_softmax = softmax_scale;
params.scale_softmax_log2 = softmax_scale * M_LOG2E;
__half scale_softmax_log2_half = __float2half(params.scale_softmax_log2);
__half2 scale_softmax_log2_half2 = __half2(scale_softmax_log2_half, scale_softmax_log2_half);
params.scale_softmax_log2_half2 = reinterpret_cast<uint32_t&>(scale_softmax_log2_half2);
// Set this to probability of keeping an element to simplify things.
params.p_dropout = 1.f - p_dropout;
// Convert p from float to int so we don't have to convert the random uint to float to compare.
// [Minor] We want to round down since when we do the comparison we use <= instead of <
// params.p_dropout_in_uint = uint32_t(std::floor(params.p_dropout * 4294967295.0));
// params.p_dropout_in_uint16_t = uint16_t(std::floor(params.p_dropout * 65535.0));
params.p_dropout_in_uint8_t = uint8_t(std::floor(params.p_dropout * 255.0));
params.rp_dropout = 1.f / params.p_dropout;
params.scale_softmax_rp_dropout = params.rp_dropout * params.scale_softmax;
TORCH_CHECK(p_dropout < 1.f);
#ifdef FLASHATTENTION_DISABLE_DROPOUT
TORCH_CHECK(p_dropout == 0.0f, "This flash attention build does not support dropout.");
#endif
// Causal is the special case where window_size_right == 0 and window_size_left < 0.
// Local is the more general case where window_size_right >= 0 or window_size_left >= 0.
params.is_causal = window_size_left < 0 && window_size_right == 0;
params.per_block_mean = per_block_mean;
if (per_block_mean) {
params.seqlen_s = seqlen_q;
} else {
params.seqlen_s = flash::BLOCK_M; // size of BLOCK_M
}
if (window_size_left < 0 && window_size_right >= 0) { window_size_left = seqlen_k; }
if (window_size_left >= 0 && window_size_right < 0) { window_size_right = seqlen_k; }
params.window_size_left = window_size_left;
params.window_size_right = window_size_right;
#ifdef FLASHATTENTION_DISABLE_LOCAL
TORCH_CHECK(params.is_causal || (window_size_left < 0 && window_size_right < 0),
"This flash attention build does not support local attention.");
#endif
params.is_seqlens_k_cumulative = true;
params.is_bf16 = is_bf16;
params.single_level_p_quant = single_level_p_quant;
#ifdef FLASHATTENTION_DISABLE_UNEVEN_K
TORCH_CHECK(d == d_rounded, "This flash attention build does not support headdim not being a multiple of 32.");
#endif
}
template<bool IsBF16>
void run_mha_fwd_dispatch_dtype(Flash_fwd_params &params, cudaStream_t stream) {
using OType = std::conditional_t<IsBF16, cutlass::bfloat16_t, cutlass::half_t>;
if (params.d == 64) {
run_mha_fwd_<cutlass::nv_float4_t<cutlass::float_e2m1_t>, 64, OType>(params, stream);
} else if (params.d == 128) {
run_mha_fwd_<cutlass::nv_float4_t<cutlass::float_e2m1_t>, 128, OType>(params, stream);
}
}
void run_mha_fwd(Flash_fwd_params &params, cudaStream_t stream, bool force_split_kernel = false) {
BOOL_SWITCH(params.is_bf16, IsBF16, ([&] {
run_mha_fwd_dispatch_dtype<IsBF16>(params, stream);
}));
}
std::vector<at::Tensor>
mha_fwd(at::Tensor &q, // batch_size x seqlen_q x num_heads x (head_size // 2)
const at::Tensor &k, // batch_size x seqlen_k x num_heads_k x (head_size // 2)
const at::Tensor &v, // batch_size x seqlen_k x num_heads_k x (head_size // 2)
const at::Tensor &sfq,
const at::Tensor &sfk,
const at::Tensor &sfv,
const at::Tensor &delta_s,
int unpadded_k,
c10::optional<at::Tensor> &out_, // batch_size x seqlen_q x num_heads x head_size
const float softmax_scale,
bool is_causal,
bool per_block_mean,
bool is_bf16,
bool single_level_p_quant=false // If true, use only per-row scale s_P2 (no per-block s_P1)
) {
auto dprops = at::cuda::getCurrentDeviceProperties();
bool is_sm120 = dprops->major == 12 && dprops->minor == 0;
TORCH_CHECK(is_sm120, "only supports Blackwell GPUs or newer.");
auto q_dtype = q.dtype();
auto sfq_dtype = sfq.dtype();
TORCH_CHECK(q_dtype == torch::kUInt8, "q dtype must be uint8");
TORCH_CHECK(k.dtype() == q_dtype, "query and key must have the same dtype");
TORCH_CHECK(v.dtype() == q_dtype, "query and value must have the same dtype");
CHECK_DEVICE(q); CHECK_DEVICE(k); CHECK_DEVICE(v);
TORCH_CHECK(sfq_dtype == torch::kFloat8_e4m3fn, "q dtype must be uint8");
TORCH_CHECK(sfk.dtype() == sfq_dtype, "query and key must have the same dtype");
TORCH_CHECK(sfv.dtype() == sfq_dtype, "query and value must have the same dtype");
CHECK_DEVICE(sfq); CHECK_DEVICE(sfk); CHECK_DEVICE(sfv);
TORCH_CHECK(q.stride(-1) == 1, "Input tensor must have contiguous last dimension");
TORCH_CHECK(k.stride(-1) == 1, "Input tensor must have contiguous last dimension");
TORCH_CHECK(v.stride(-1) == 1, "Input tensor must have contiguous last dimension");
TORCH_CHECK(delta_s.stride(-1) == 1, "Input tensor must have contiguous last dimension");
TORCH_CHECK(q.is_contiguous(), "Input tensor must be contiguous");
TORCH_CHECK(k.is_contiguous(), "Input tensor must be contiguous");
TORCH_CHECK(v.is_contiguous(), "Input tensor must be contiguous");
const auto sizes = q.sizes();
auto opts = q.options();
const int batch_size = sizes[0];
int seqlen_q = sizes[2];
int num_heads = sizes[1];
const int head_size_og = sizes[3];
const int unpacked_head_size = head_size_og * 2;
const int seqlen_k = k.size(2);
const int num_heads_k = k.size(1);
TORCH_CHECK(batch_size > 0, "batch size must be postive");
TORCH_CHECK(unpacked_head_size <= 256, "FlashAttention forward only supports head dimension at most 256");
TORCH_CHECK(num_heads % num_heads_k == 0, "Number of heads in key/value must divide number of heads in query");
TORCH_CHECK(num_heads == num_heads_k, "We do not support MQA/GQA yet");
TORCH_CHECK(unpacked_head_size == 64 || unpacked_head_size == 128 || unpacked_head_size == 256, "Only support head size 64, 128, and 256 for now");
CHECK_SHAPE(q, batch_size, num_heads, seqlen_q, head_size_og);
CHECK_SHAPE(k, batch_size, num_heads_k, seqlen_k, head_size_og);
CHECK_SHAPE(v, batch_size, num_heads_k, unpacked_head_size, seqlen_k/2);
// CHECK_SHAPE(delta_s, batch_size, num_heads, seqlen_q / 128, seqlen_k);
// CHECK_SHAPE(sfq, batch_size, seqlen_q, num_heads, unpacked_head_size);
// CHECK_SHAPE(sfk, batch_size, seqlen_k, num_heads_k, unpacked_head_size);
// CHECK_SHAPE(sfv, batch_size, unpacked_head_size, num_heads_k, seqlen_k);
TORCH_CHECK(unpacked_head_size % 8 == 0, "head_size must be a multiple of 8");
auto dtype = is_bf16 ? at::ScalarType::BFloat16 : at::ScalarType::Half;
at::Tensor out = torch::empty({batch_size, num_heads, seqlen_q, unpacked_head_size}, opts.dtype(dtype));
auto round_multiple = [](int x, int m) { return (x + m - 1) / m * m; };
// const int head_size = round_multiple(head_size_og, 8);
// const int head_size_rounded = round_multiple(head_size, 32);
const int seqlen_q_rounded = round_multiple(seqlen_q, flash::BLOCK_M);
const int seqlen_k_rounded = round_multiple(seqlen_k, flash::BLOCK_N);
// Otherwise the kernel will be launched from cuda:0 device
// Cast to char to avoid compiler warning about narrowing
at::cuda::CUDAGuard device_guard{(char)q.get_device()};
auto softmax_lse = torch::empty({batch_size, num_heads, seqlen_q}, opts.dtype(at::kFloat));
at::Tensor p;
Flash_fwd_params params;
set_params_fprop(params,
batch_size,
seqlen_q, seqlen_k, unpadded_k,
seqlen_q_rounded, seqlen_k_rounded,
num_heads, num_heads_k,
unpacked_head_size, unpacked_head_size,
q, k, v, delta_s, out,
sfq, sfk, sfv,
/*cu_seqlens_q_d=*/nullptr,
/*cu_seqlens_k_d=*/nullptr,
/*seqused_k=*/nullptr,
nullptr,
softmax_lse.data_ptr(),
/*p_dropout=*/0.f,
softmax_scale,
/*window_size_left=*/-1,
/*window_size_right=*/is_causal ? 0 : -1,
per_block_mean,
is_bf16,
single_level_p_quant
);
// TODO: 132 sm count?
auto tile_count_semaphore = is_causal ? torch::full({1}, 132, opts.dtype(torch::kInt32)) : torch::empty({1}, opts.dtype(torch::kInt32));
params.tile_count_semaphore = tile_count_semaphore.data_ptr<int>();
if (seqlen_k > 0) {
auto stream = at::cuda::getCurrentCUDAStream().stream();
run_mha_fwd(params, stream);
} else {
// If seqlen_k == 0, then we have an empty tensor. We need to set the output to 0.
out.zero_();
softmax_lse.fill_(std::numeric_limits<float>::infinity());
}
// at::Tensor out_padded = out;
// if (head_size_og % 8 != 0) {
// out = out.index({"...", torch::indexing::Slice(torch::indexing::None, head_size_og)});
// if (out_.has_value()) { out_.value().copy_(out); }
// }
// return {out, q_padded, k_padded, v_padded, out_padded, softmax_lse, p};
// cudaDeviceSynchronize();
// auto err = cudaGetLastError();
// printf("%s\n", cudaGetErrorString(err));
return {out, softmax_lse};
}
PYBIND11_MODULE(TORCH_EXTENSION_NAME, m) {
m.doc() = "FlashAttention";
m.def("fwd", &mha_fwd, "Forward pass");
}
@@ -0,0 +1,28 @@
/*
* Copyright (c) 2025 by SageAttention team.
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
#pragma once
// Centralized block size configuration for sageattn_blackwell kernels
// Block sizes for M and N dimensions
namespace flash {
// Block size for M dimension (query sequence length)
static constexpr int BLOCK_M = 128;
// Block size for N dimension (key/value sequence length)
static constexpr int BLOCK_N = 128;
}
@@ -0,0 +1,60 @@
/*
* Copyright (c) 2025 by SageAttention team.
*
* This code is based on code from FlashAttention3, https://github.com/Dao-AILab/flash-attention
* Copyright (c) 2024, Jay Shah, Ganesh Bikshandi, Ying Zhang, Vijay Thakkar, Pradeep Ramani, Tri Dao.
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
#pragma once
namespace flash {
////////////////////////////////////////////////////////////////////////////////////////////////////
template<bool Varlen=true>
struct BlockInfo {
template<typename Params>
__device__ BlockInfo(const Params &params, const int bidb)
: sum_s_q(!Varlen || params.cu_seqlens_q == nullptr ? -1 : params.cu_seqlens_q[bidb])
, sum_s_k(!Varlen || params.cu_seqlens_k == nullptr || !params.is_seqlens_k_cumulative ? -1 : params.cu_seqlens_k[bidb])
, actual_seqlen_q(!Varlen || params.cu_seqlens_q == nullptr ? params.seqlen_q : params.cu_seqlens_q[bidb + 1] - sum_s_q)
// If is_seqlens_k_cumulative, then seqlen_k is cu_seqlens_k[bidb + 1] - cu_seqlens_k[bidb].
// Otherwise it's cu_seqlens_k[bidb], i.e., we use cu_seqlens_k to store the sequence lengths of K.
, seqlen_k_cache(!Varlen || params.cu_seqlens_k == nullptr ? params.seqlen_k : (params.is_seqlens_k_cumulative ? params.cu_seqlens_k[bidb + 1] - sum_s_k : params.cu_seqlens_k[bidb]))
, actual_seqlen_k(params.seqused_k ? params.seqused_k[bidb] : seqlen_k_cache + (params.knew_ptr == nullptr ? 0 : params.seqlen_knew))
{
}
template <typename index_t>
__forceinline__ __device__ index_t q_offset(const index_t batch_stride, const index_t row_stride, const int bidb) const {
return sum_s_q == -1 ? bidb * batch_stride : uint32_t(sum_s_q) * row_stride;
}
template <typename index_t>
__forceinline__ __device__ index_t k_offset(const index_t batch_stride, const index_t row_stride, const int bidb) const {
return sum_s_k == -1 ? bidb * batch_stride : uint32_t(sum_s_k) * row_stride;
}
const int sum_s_q;
const int sum_s_k;
const int actual_seqlen_q;
// We have to have seqlen_k_cache declared before actual_seqlen_k, otherwise actual_seqlen_k is set to 0.
const int seqlen_k_cache;
const int actual_seqlen_k;
};
////////////////////////////////////////////////////////////////////////////////////////////////////
} // namespace flash
@@ -0,0 +1,149 @@
/***************************************************************************************************
* Copyright (c) 2023 - 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
* SPDX-License-Identifier: BSD-3-Clause
*
* Redistribution and use in source and binary forms, with or without
* modification, are permitted provided that the following conditions are met:
*
* 1. Redistributions of source code must retain the above copyright notice, this
* list of conditions and the following disclaimer.
*
* 2. Redistributions in binary form must reproduce the above copyright notice,
* this list of conditions and the following disclaimer in the documentation
* and/or other materials provided with the distribution.
*
* 3. Neither the name of the copyright holder nor the names of its
* contributors may be used to endorse or promote products derived from
* this software without specific prior written permission.
*
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
* AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
* IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
* DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
* FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
* DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
* SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
* CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
* OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
* OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
*
**************************************************************************************************/
/*! \file
\brief Blocked Scale configs specific for SM100 BlockScaled MMA
*/
#pragma once
#include "cutlass/layout/matrix.h"
#include "cute/int_tuple.hpp"
#include "cute/atom/mma_traits_sm100.hpp"
namespace flash {
/////////////////////////////////////////////////////////////////////////////////////////////////
using namespace cute;
template<int SFVecSize, UMMA::Major major = UMMA::Major::K>
struct BlockScaledBasicChunk {
using Blk_MN = _64;
using Blk_SF = _4;
using SfAtom = Layout< Shape< Shape<_16,_4>, Shape<Int<SFVecSize>, _4>>,
Stride<Stride<_16,_4>, Stride< _0, _1>>>;
};
template<int SFVecSize_>
struct BlockScaledConfig {
// We are creating the SFA and SFB tensors' layouts in the collective since they always have the same layout.
// k-major order
static constexpr int SFVecSize = SFVecSize_;
static constexpr int MMA_NSF = 4; // SFVecSize, MMA_NSF
using BlkScaledChunk = BlockScaledBasicChunk<SFVecSize>;
using Blk_MN = _64;
using Blk_SF = _4;
using mnBasicBlockShape = Shape<_16,_4>;
using mnBasicBlockStride = Stride<_16,_4>;
using kBasicBlockShape = Shape<Int<SFVecSize>, Int<MMA_NSF>>; // SFVecSize, MMA_NSF
using kBasicBlockStride = Stride<_0, _1>;
using SfAtom = Layout< Shape< mnBasicBlockShape, kBasicBlockShape>,
Stride<mnBasicBlockStride, kBasicBlockStride>>;
using LayoutSF = decltype(blocked_product(SfAtom{},
make_layout(
make_shape(int32_t(0), int32_t(0), int32_t(0), int32_t(0)),
make_stride(int32_t(0), _1{}, int32_t(0), int32_t(0)))));
// A single indivisible block will hold 4 scale factors of 64 rows/columns (A/B matrix).
// 4 is chosen to make consecutive 32bits of data to have scale factors for only a single row (col). 32bits corresponds to the TMEM word size
using Blk_Elems = decltype(Blk_MN{} * Blk_SF{});
using sSF_strideMN = decltype(prepend(Blk_Elems{}, mnBasicBlockStride{}));
// The following function is provided for user fill dynamic problem size to the layout_SFA.
template < class ProblemShape>
CUTE_HOST_DEVICE
static constexpr auto
tile_atom_to_shape_SFQKV(ProblemShape problem_shape) {
auto [Seqlen, Dim, HeadNum, Batch] = problem_shape;
return tile_to_shape(SfAtom{}, make_shape(Seqlen, Dim, HeadNum, Batch), Step<_2,_1,_3,_4>{});
}
// The following function is provided for user fill dynamic problem size to the layout_SFB.
template <class ProblemShape>
CUTE_HOST_DEVICE
static constexpr auto
tile_atom_to_shape_SFVt(ProblemShape problem_shape) {
auto [Dim, Seqlen, HeadNum, Batch] = problem_shape;
return tile_to_shape(SfAtom{}, make_shape(Dim, Seqlen, HeadNum, Batch), Step<_2,_1,_3,_4>{});
}
template<class TiledMma, class TileShape_MNK>
CUTE_HOST_DEVICE
static constexpr auto
deduce_smem_layoutSFQ(TiledMma tiled_mma, TileShape_MNK tileshape_mnk) {
using sSFQ_shapeK = decltype(prepend(make_shape(Blk_SF{}/Int<MMA_NSF>{}, size<2>(TileShape_MNK{}) / Int<SFVecSize>{} / Blk_SF{}), kBasicBlockShape{}));
using sSFQ_shapeM = decltype(prepend(size<0>(TileShape_MNK{}) / Blk_MN{}, mnBasicBlockShape{}));
using sSFQ_strideM = sSF_strideMN;
using sSFQ_strideK = decltype(prepend(make_stride(Int<MMA_NSF>{}, size<0>(TileShape_MNK{}) / Blk_MN{} * Blk_Elems{}), kBasicBlockStride{}));
using sSFQ_shape = decltype(make_shape(sSFQ_shapeM{}, sSFQ_shapeK{}));
using sSFQ_stride = decltype(make_stride(sSFQ_strideM{}, sSFQ_strideK{}));
using SmemLayoutAtomSFQ = decltype(make_layout(sSFQ_shape{}, sSFQ_stride{}));
return SmemLayoutAtomSFQ{};
}
template<class TiledMma, class TileShape_MNK>
CUTE_HOST_DEVICE
static constexpr auto
deduce_smem_layoutSFKV(TiledMma tiled_mma, TileShape_MNK tileshape_mnk) {
using sSFK_shapeK = decltype(prepend(make_shape(Blk_SF{}/Int<MMA_NSF>{}, size<2>(TileShape_MNK{}) / Int<SFVecSize>{} / Blk_SF{}), kBasicBlockShape{}));
using sSFK_shapeN = decltype(prepend(size<1>(TileShape_MNK{}) / Blk_MN{}, mnBasicBlockShape{}));
using sSFK_strideN = sSF_strideMN;
using sSFK_strideK = decltype(prepend(make_stride(Int<MMA_NSF>{}, size<1>(TileShape_MNK{}) / Blk_MN{} * Blk_Elems{}), kBasicBlockStride{}));
using sSFK_shape = decltype(make_shape(sSFK_shapeN{}, sSFK_shapeK{}));
using sSFK_stride = decltype(make_stride(sSFK_strideN{}, sSFK_strideK{}));
using SmemLayoutAtomSFK = decltype(make_layout(sSFK_shape{}, sSFK_stride{}));
return SmemLayoutAtomSFK{};
}
template<class TiledMma, class TileShape_MNK>
CUTE_HOST_DEVICE
static constexpr auto
deduce_smem_layoutSFVt(TiledMma tiled_mma, TileShape_MNK tileshape_mnk) {
using sSFVt_shapeK = decltype(prepend(make_shape(Blk_SF{}/Int<MMA_NSF>{}, size<2>(TileShape_MNK{}) / Int<SFVecSize>{} / Blk_SF{}), kBasicBlockShape{}));
using sSFVt_shapeN = decltype(prepend(size<1>(TileShape_MNK{}) / Blk_MN{}, mnBasicBlockShape{}));
using sSFVt_strideN = sSF_strideMN;
using sSFVt_strideK = decltype(prepend(make_stride(Int<MMA_NSF>{}, size<1>(TileShape_MNK{}) / Blk_MN{} * Blk_Elems{}), kBasicBlockStride{}));
using sSFVt_shape = decltype(make_shape(sSFVt_shapeN{}, sSFVt_shapeK{}));
using sSFVt_stride = decltype(make_stride(sSFVt_strideN{}, sSFVt_strideK{}));
using SmemLayoutAtomSFVt = decltype(make_layout(sSFVt_shape{}, sSFVt_stride{}));
return SmemLayoutAtomSFVt{};
}
};
} // namespace flash
@@ -0,0 +1,327 @@
/*
* Copyright (c) 2025 by SageAttention team.
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
#include "cute/arch/mma_sm120.hpp"
#include "cute/atom/mma_traits_sm120.hpp"
#include "cute/atom/mma_atom.hpp"
#include "cutlass/cutlass.h"
#include "cutlass/float8.h"
#include "cutlass/float_subbyte.h"
namespace cute::SM120::BLOCKSCALED {
using cutlass::float_e2m1_t;
using cutlass::float_ue4m3_t;
// MMA.SF 16x32x64 TN E2M1 x E2M1 with SF E4M3
struct SM120_16x32x64_TN_VS_NVFP4 {
using DRegisters = float[16];
using ARegisters = uint32_t[4];
using BRegisters = uint32_t[8];
using CRegisters = float[16];
static constexpr int SFBits = 32;
using RegTypeSF = cute::uint_bit_t<SFBits>;
using SFARegisters = RegTypeSF[1];
using SFBRegisters = RegTypeSF[1];
CUTE_HOST_DEVICE static void
fma(float & d0 , float & d1 , float & d2 , float & d3 ,
float & d4 , float & d5 , float & d6 , float & d7 ,
float & d8 , float & d9 , float & d10, float & d11,
float & d12, float & d13, float & d14, float & d15,
uint32_t const& a0 , uint32_t const& a1 , uint32_t const& a2 , uint32_t const& a3 ,
uint32_t const& b0 , uint32_t const& b1 , uint32_t const& b2 , uint32_t const& b3 ,
uint32_t const& b4 , uint32_t const& b5 , uint32_t const& b6 , uint32_t const& b7 ,
float const & c0 , float const & c1 , float const & c2 , float const & c3 ,
float const & c4 , float const & c5 , float const & c6 , float const & c7 ,
float const & c8 , float const & c9 , float const & c10 , float const & c11,
float const & c12, float const & c13, float const & c14, float const & c15,
RegTypeSF const& sfa0,
RegTypeSF const& sfb0)
{
static constexpr uint16_t tidA = 0;
static constexpr uint16_t bidA = 0;
static constexpr uint16_t bidB = 0;
static constexpr uint16_t tidB0 = 0;
static constexpr uint16_t tidB1 = 1;
static constexpr uint16_t tidB2 = 2;
static constexpr uint16_t tidB3 = 3;
#if defined(CUTE_ARCH_MXF4NVF4_4X_UE4M3_MMA_ENABLED)
asm volatile(
"mma.sync.aligned.kind::mxf4nvf4.block_scale.scale_vec::4X.m16n8k64.row.col.f32.e2m1.e2m1.f32.ue4m3 "
"{%0, %1, %2, %3},"
"{%4, %5, %6, %7},"
"{%8, %9},"
"{%10, %11, %12, %13},"
"{%14},"
"{%15, %16},"
"{%17},"
"{%18, %19};\n"
: "=f"(d0), "=f"(d1), "=f"(d8), "=f"(d9)
: "r"(a0), "r"(a1), "r"(a2), "r"(a3),
"r"(b0), "r"(b1),
"f"(c0), "f"(c1), "f"(c8), "f"(c9),
"r"(uint32_t(sfa0)) , "h"(bidA), "h"(tidA),
"r"(uint32_t(sfb0)) , "h"(bidB), "h"(tidB0));
asm volatile(
"mma.sync.aligned.kind::mxf4nvf4.block_scale.scale_vec::4X.m16n8k64.row.col.f32.e2m1.e2m1.f32.ue4m3 "
"{%0, %1, %2, %3},"
"{%4, %5, %6, %7},"
"{%8, %9},"
"{%10, %11, %12, %13},"
"{%14},"
"{%15, %16},"
"{%17},"
"{%18, %19};\n"
: "=f"(d2), "=f"(d3), "=f"(d10), "=f"(d11)
: "r"(a0), "r"(a1), "r"(a2), "r"(a3),
"r"(b2), "r"(b3),
"f"(c2), "f"(c3), "f"(c10), "f"(c11),
"r"(uint32_t(sfa0)) , "h"(bidA), "h"(tidA),
"r"(uint32_t(sfb0)) , "h"(bidB), "h"(tidB1));
asm volatile(
"mma.sync.aligned.kind::mxf4nvf4.block_scale.scale_vec::4X.m16n8k64.row.col.f32.e2m1.e2m1.f32.ue4m3 "
"{%0, %1, %2, %3},"
"{%4, %5, %6, %7},"
"{%8, %9},"
"{%10, %11, %12, %13},"
"{%14},"
"{%15, %16},"
"{%17},"
"{%18, %19};\n"
: "=f"(d4), "=f"(d5), "=f"(d12), "=f"(d13)
: "r"(a0), "r"(a1), "r"(a2), "r"(a3),
"r"(b4), "r"(b5),
"f"(c4), "f"(c5), "f"(c12), "f"(c13),
"r"(uint32_t(sfa0)) , "h"(bidA), "h"(tidA),
"r"(uint32_t(sfb0)) , "h"(bidB), "h"(tidB2));
asm volatile(
"mma.sync.aligned.kind::mxf4nvf4.block_scale.scale_vec::4X.m16n8k64.row.col.f32.e2m1.e2m1.f32.ue4m3 "
"{%0, %1, %2, %3},"
"{%4, %5, %6, %7},"
"{%8, %9},"
"{%10, %11, %12, %13},"
"{%14},"
"{%15, %16},"
"{%17},"
"{%18, %19};\n"
: "=f"(d6), "=f"(d7), "=f"(d14), "=f"(d15)
: "r"(a0), "r"(a1), "r"(a2), "r"(a3),
"r"(b6), "r"(b7),
"f"(c6), "f"(c7), "f"(c14), "f"(c15),
"r"(uint32_t(sfa0)) , "h"(bidA), "h"(tidA),
"r"(uint32_t(sfb0)) , "h"(bidB), "h"(tidB3));
#else
CUTE_INVALID_CONTROL_PATH("Attempting to use SM120::BLOCKSCALED::SM120_16x8x64_TN_VS without CUTE_ARCH_MXF4NVF4_4X_UE4M3_MMA_ENABLED");
#endif
}
};
} // namespace cute::SM120::BLOCKSCALED
namespace cute {
// MMA NVFP4 16x32x64 TN
template <>
struct MMA_Traits<SM120::BLOCKSCALED::SM120_16x32x64_TN_VS_NVFP4>
{
// The MMA accepts 4-bit inputs regardless of the types for A and B
using ValTypeA = uint4_t;
using ValTypeB = uint4_t;
using ValTypeD = float;
using ValTypeC = float;
using ValTypeSF = cutlass::float_ue4m3_t;
constexpr static int SFVecSize = 16;
using Shape_MNK = Shape<_16,_32,_64>;
using ThrID = Layout<_32>;
// (T32,V32) -> (M16,K64)
using ALayout = Layout<Shape <Shape < _4,_8>,Shape < _8,_2, _2>>,
Stride<Stride<_128,_1>,Stride<_16,_8,_512>>>;
// (T32,V64) -> (N32,K64)
using BLayout = Layout<Shape <Shape < _4,_8>,Shape <_8, _2, _4>>,
Stride<Stride<_256,_1>,Stride<_32,_1024, _8>>>;
// (T32,V64) -> (M16,K64)
using SFALayout = Layout<Shape <Shape <_2,_2,_8>,_64>,
Stride<Stride<_8,_0,_1>,_16>>;
// (T32,V64) -> (N32,K64)
using SFBLayout = Layout<Shape <Shape <_4,_8>,_64>,
Stride<Stride<_8,_1>, _32>>;
// (T32,V16) -> (M16,N32)
using CLayout = Layout<Shape <Shape < _4,_8>,Shape < Shape<_2, _4>,_2>>,
Stride<Stride<_32,_1>,Stride<Stride<_16, _128>,_8>>>;
};
template <class SFATensor, class Atom, class TiledThr, class TiledPerm>
CUTE_HOST_DEVICE constexpr
auto
thrfrg_SFA(SFATensor&& sfatensor, TiledMMA<Atom, TiledThr, TiledPerm>& mma)
{
CUTE_STATIC_ASSERT_V(rank(sfatensor) >= Int<2>{});
using AtomShape_MNK = typename Atom::Shape_MNK;
using AtomLayoutSFA_TV = typename Atom::Traits::SFALayout;
auto permutation_mnk = TiledPerm{};
auto thr_layout_vmnk = mma.get_thr_layout_vmnk();
// Reorder the tensor for the TiledAtom
auto t_tile = make_tile(get<0>(permutation_mnk),
get<2>(permutation_mnk));
auto t_tensor = logical_divide(sfatensor, t_tile); // (PermM,PermK)
// Tile the tensor for the Atom
auto a_tile = make_tile(make_layout(size<0>(AtomShape_MNK{})),
make_layout(size<2>(AtomShape_MNK{})));
auto a_tensor = zipped_divide(t_tensor, a_tile); // ((AtomM,AtomK),(RestM,RestK))
// Transform the Atom mode from (M,K) to (Thr,Val)
auto tv_tensor = a_tensor.compose(AtomLayoutSFA_TV{},_); // ((ThrV,FrgV),(RestM,RestK))
// Tile the tensor for the Thread
auto thr_tile = make_tile(_,
make_tile(make_layout(size<1>(thr_layout_vmnk)),
make_layout(size<3>(thr_layout_vmnk))));
auto thr_tensor = zipped_divide(tv_tensor, thr_tile); // ((ThrV,(ThrM,ThrK)),(FrgV,(RestM,RestK)))
return thr_tensor;
}
template <class SFBTensor, class Atom, class TiledThr, class TiledPerm>
CUTE_HOST_DEVICE constexpr
auto
thrfrg_SFB(SFBTensor&& sfbtensor, TiledMMA<Atom, TiledThr, TiledPerm>& mma)
{
CUTE_STATIC_ASSERT_V(rank(sfbtensor) >= Int<2>{});
using AtomShape_MNK = typename Atom::Shape_MNK;
using AtomLayoutSFB_TV = typename Atom::Traits::SFBLayout;
auto permutation_mnk = TiledPerm{};
auto thr_layout_vmnk = mma.get_thr_layout_vmnk();
// Reorder the tensor for the TiledAtom
auto t_tile = make_tile(get<1>(permutation_mnk),
get<2>(permutation_mnk));
auto t_tensor = logical_divide(sfbtensor, t_tile); // (PermN,PermK)
// Tile the tensor for the Atom
auto a_tile = make_tile(make_layout(size<1>(AtomShape_MNK{})),
make_layout(size<2>(AtomShape_MNK{})));
auto a_tensor = zipped_divide(t_tensor, a_tile); // ((AtomN,AtomK),(RestN,RestK))
// Transform the Atom mode from (M,K) to (Thr,Val)
auto tv_tensor = a_tensor.compose(AtomLayoutSFB_TV{},_); // ((ThrV,FrgV),(RestN,RestK))
// Tile the tensor for the Thread
auto thr_tile = make_tile(_,
make_tile(make_layout(size<2>(thr_layout_vmnk)),
make_layout(size<3>(thr_layout_vmnk))));
auto thr_tensor = zipped_divide(tv_tensor, thr_tile); // ((ThrV,(ThrN,ThrK)),(FrgV,(RestN,RestK)))
return thr_tensor;
}
template <class SFATensor, class ThrMma>
CUTE_HOST_DEVICE constexpr
auto
partition_SFA(SFATensor&& sfatensor, ThrMma& thread_mma) {
auto thr_tensor = make_tensor(static_cast<SFATensor&&>(sfatensor).data(), thrfrg_SFA(sfatensor.layout(),thread_mma));
auto thr_vmnk = thread_mma.thr_vmnk_;
auto thr_vmk = make_coord(get<0>(thr_vmnk), make_coord(get<1>(thr_vmnk), get<3>(thr_vmnk)));
return thr_tensor(thr_vmk, make_coord(_, repeat<rank<1,1>(thr_tensor)>(_)));
}
template <class SFATensor, class ThrMma>
CUTE_HOST_DEVICE constexpr
auto
partition_fragment_SFA(SFATensor&& sfatensor, ThrMma& thread_mma) {
using ValTypeSF = typename ThrMma::Atom::Traits::ValTypeSF;
return make_fragment_like<ValTypeSF>(partition_SFA(sfatensor, thread_mma));
}
template <class SFBTensor, class ThrMma>
CUTE_HOST_DEVICE constexpr
auto
partition_SFB(SFBTensor&& sfbtensor, ThrMma& thread_mma) {
auto thr_tensor = make_tensor(static_cast<SFBTensor&&>(sfbtensor).data(), thrfrg_SFB(sfbtensor.layout(),thread_mma));
auto thr_vmnk = thread_mma.thr_vmnk_;
auto thr_vnk = make_coord(get<0>(thr_vmnk), make_coord(get<2>(thr_vmnk), get<3>(thr_vmnk)));
return thr_tensor(thr_vnk, make_coord(_, repeat<rank<1,1>(thr_tensor)>(_)));
}
template <class SFBTensor, class ThrMma>
CUTE_HOST_DEVICE constexpr
auto
partition_fragment_SFB(SFBTensor&& sfbtensor, ThrMma& thread_mma) {
using ValTypeSF = typename ThrMma::Atom::Traits::ValTypeSF;
return make_fragment_like<ValTypeSF>(partition_SFB(sfbtensor, thread_mma));
}
template<class TiledMma>
CUTE_HOST_DEVICE constexpr
auto
get_layoutSFA_TV(TiledMma& mma)
{
// (M,K) -> (M,K)
auto tile_shape_mnk = tile_shape(mma);
auto ref_A = make_layout(make_shape(size<0>(tile_shape_mnk), size<2>(tile_shape_mnk)));
auto thr_layout_vmnk = mma.get_thr_layout_vmnk();
// (ThrV,(ThrM,ThrK)) -> (ThrV,(ThrM,ThrN,ThrK))
auto atile = make_tile(_,
make_tile(make_layout(make_shape (size<1>(thr_layout_vmnk), size<2>(thr_layout_vmnk)),
make_stride( Int<1>{} , Int<0>{} )),
_));
// thr_idx -> (ThrV,ThrM,ThrN,ThrK)
auto thridx_2_thrid = right_inverse(thr_layout_vmnk);
// (thr_idx,val) -> (M,K)
return thrfrg_SFA(ref_A, mma).compose(atile, _).compose(thridx_2_thrid, _);
}
template<class TiledMma>
CUTE_HOST_DEVICE constexpr
auto
get_layoutSFB_TV(TiledMma& mma)
{
// (N,K) -> (N,K)
auto tile_shape_mnk = tile_shape(mma);
auto ref_B = make_layout(make_shape(size<1>(tile_shape_mnk), size<2>(tile_shape_mnk)));
auto thr_layout_vmnk = mma.get_thr_layout_vmnk();
// (ThrV,(ThrM,ThrK)) -> (ThrV,(ThrM,ThrN,ThrK))
auto btile = make_tile(_,
make_tile(make_layout(make_shape (size<1>(thr_layout_vmnk), size<2>(thr_layout_vmnk)),
make_stride( Int<0>{} , Int<1>{} )),
_));
// thr_idx -> (ThrV,ThrM,ThrN,ThrK)
auto thridx_2_thrid = right_inverse(thr_layout_vmnk);
// (thr_idx,val) -> (M,K)
return thrfrg_SFB(ref_B, mma).compose(btile, _).compose(thridx_2_thrid, _);
}
} // namespace cute
@@ -0,0 +1,222 @@
/*
* Copyright (c) 2025 by SageAttention team.
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
#pragma once
#include <cutlass/cutlass.h>
#include "cute/tensor.hpp"
#include "cutlass/gemm/collective/collective_builder.hpp"
#include "named_barrier.h"
#include "utils.h"
namespace flash {
using namespace cute;
template <typename Ktraits>
struct CollectiveEpilogueFwd{
using Element = typename Ktraits::ElementOut;
static constexpr int kBlockM = Ktraits::kBlockM;
static constexpr int kBlockN = Ktraits::kBlockN;
static constexpr int kHeadDim = Ktraits::kHeadDim;
using TileShape_MNK = Shape<Int<kBlockM>, Int<kBlockN>, Int<kHeadDim>>;
static constexpr int kNWarps = Ktraits::kNWarps;
static constexpr int kNThreads = kNWarps * cutlass::NumThreadsPerWarp;
static constexpr int NumMmaThreads = kNThreads - cutlass::NumThreadsPerWarpGroup;
using GmemTiledCopyOTMA = cute::SM90_TMA_STORE;
// These are for storing the output tensor without TMA (e.g., for setting output to zero)
static constexpr int kGmemElemsPerLoad = sizeof(cute::uint128_t) / sizeof(Element);
static_assert(kHeadDim % kGmemElemsPerLoad == 0, "kHeadDim must be a multiple of kGmemElemsPerLoad");
static constexpr int kGmemThreadsPerRow = kHeadDim / kGmemElemsPerLoad;
static_assert(NumMmaThreads % kGmemThreadsPerRow == 0, "NumMmaThreads must be a multiple of kGmemThreadsPerRow");
using GmemLayoutAtom = Layout<Shape <Int<NumMmaThreads / kGmemThreadsPerRow>, Int<kGmemThreadsPerRow>>,
Stride<Int<kGmemThreadsPerRow>, _1>>;
using GmemTiledCopyO = decltype(
make_tiled_copy(Copy_Atom<DefaultCopy, Element>{},
GmemLayoutAtom{},
Layout<Shape<_1, Int<kGmemElemsPerLoad>>>{})); // Val layout, 8 or 16 vals per store
using SmemLayoutO = typename Ktraits::SmemLayoutO;
using SmemCopyAtomO = Copy_Atom<SM90_U32x2_STSM_N, Element>;
using SharedStorage = cute::array_aligned<Element, cute::cosize_v<SmemLayoutO>>;
using ShapeO = cute::Shape<int32_t, int32_t, int32_t, int32_t>; // (seqlen_q, d, head, batch)
using StrideO = cute::Stride<int64_t, _1, int64_t, int64_t>;
using StrideLSE = cute::Stride<_1, int64_t, int64_t>; // (seqlen_q, head, batch)
using TMA_O = decltype(make_tma_copy(
GmemTiledCopyOTMA{},
make_tensor(make_gmem_ptr(static_cast<Element*>(nullptr)), repeat_like(StrideO{}, int32_t(0)), StrideO{}),
SmemLayoutO{},
select<0, 2>(TileShape_MNK{}),
_1{})); // no mcast for O
// Host side kernel arguments
struct Arguments {
Element* ptr_O;
ShapeO const shape_O;
StrideO const stride_O;
float* ptr_LSE;
StrideLSE const stride_LSE;
};
// Device side kernel params
struct Params {
Element* ptr_O;
ShapeO const shape_O;
StrideO const stride_O;
float* ptr_LSE;
StrideLSE const stride_LSE;
TMA_O tma_store_O;
};
static Params
to_underlying_arguments(Arguments const& args) {
Tensor mO = make_tensor(make_gmem_ptr(args.ptr_O), args.shape_O, args.stride_O);
TMA_O tma_store_O = make_tma_copy(
GmemTiledCopyOTMA{},
mO,
SmemLayoutO{},
select<0, 2>(TileShape_MNK{}),
_1{}); // no mcast for O
return {args.ptr_O, args.shape_O, args.stride_O, args.ptr_LSE, args.stride_LSE, tma_store_O};
}
/// Issue Tma Descriptor Prefetch -- ideally from a single thread for best performance
CUTLASS_DEVICE
static void prefetch_tma_descriptors(Params const& epilogue_params) {
cute::prefetch_tma_descriptor(epilogue_params.tma_store_O.get_tma_descriptor());
}
template <typename SharedStorage, typename FrgTensorO, typename TiledMma>
CUTLASS_DEVICE void
mma_store(
SharedStorage& shared_storage,
TiledMma tiled_mma,
FrgTensorO const& tOrO,
int thread_idx
){
Tensor sO = cute::as_position_independent_swizzle_tensor(make_tensor(make_smem_ptr(shared_storage.smem_o.begin()), SmemLayoutO{}));
auto smem_tiled_copy_O = make_tiled_copy_C(SmemCopyAtomO{}, tiled_mma);
auto smem_thr_copy_O = smem_tiled_copy_O.get_thread_slice(thread_idx);
constexpr int numel = decltype(size(tOrO))::value;
cutlass::NumericArrayConverter<Element, float, numel> convert_op;
// HACK: this requires tensor to be "contiguous"
auto frag = convert_op(*reinterpret_cast<const cutlass::Array<float, numel> *>(tOrO.data()));
auto tOrO_out = make_tensor(make_rmem_ptr<Element>(&frag), tOrO.layout());
Tensor taccOrO = smem_thr_copy_O.retile_S(tOrO_out); // ((Atom,AtomNum), MMA_M, MMA_N)
Tensor taccOsO = smem_thr_copy_O.partition_D(sO); // ((Atom,AtomNum),PIPE_M,PIPE_N)
cute::copy(smem_tiled_copy_O, taccOrO, taccOsO);
cutlass::arch::fence_view_async_shared(); // ensure smem writes are visible to TMA
}
template<typename SharedStorage, typename Params, typename WorkTileInfo, typename SchedulerParams>
CUTLASS_DEVICE void
tma_store(
SharedStorage& shared_storage,
Params const& epilogue_params,
WorkTileInfo work_tile_info,
SchedulerParams const& scheduler_params,
int thread_idx
) {
auto [m_block, bidh, bidb] = work_tile_info.get_block_coord(scheduler_params);
Tensor sO = cute::as_position_independent_swizzle_tensor(make_tensor(make_smem_ptr(shared_storage.smem_o.begin()), SmemLayoutO{}));
Tensor mO = epilogue_params.tma_store_O.get_tma_tensor(epilogue_params.shape_O);
Tensor gO = local_tile(mO(_, _, bidh, bidb), select<0, 2>(TileShape_MNK{}), make_coord(m_block, _0{})); // (M, K)
auto block_tma_O = epilogue_params.tma_store_O.get_slice(_0{});
Tensor tOgO = block_tma_O.partition_D(gO); // (TMA, TMA_M, TMA_K)
Tensor tOsO = block_tma_O.partition_S(sO); // (TMA, TMA_M, TMA_K)
// auto shape_LSE = select<0, 2, 3>(epilogue_params.shape_O);
// Tensor mLSE = make_tensor(make_gmem_ptr(epilogue_params.ptr_LSE), shape_LSE, epilogue_params.stride_LSE);
// Tensor gLSE = local_tile(mLSE(_, bidh, bidb), Shape<Int<kBlockM>>{}, make_coord(m_block));
// Tensor caccO = cute::make_identity_tensor(select<0, 2>(TileShape_MNK{}));
// auto thread_mma = tiled_mma.get_thread_slice(thread_idx);
// Tensor taccOcO = thread_mma.partition_C(caccO); // (MMA,MMA_M,MMA_K)
// static_assert(decltype(size<0, 0>(taccOcO))::value == 2);
// static_assert(decltype(size<0, 1>(taccOcO))::value == 2);
// // // // taccOcO has shape ((2, 2, V), MMA_M, MMA_K), we only take only the row indices.
// Tensor taccOcO_row = taccOcO(make_coord(_0{}, _), _, _0{});
// CUTE_STATIC_ASSERT_V(size(lse) == size(taccOcO_row)); // MMA_M
// if (get<1>(taccOcO_row(_0{})) == 0) {
// #pragma unroll
// for (int mi = 0; mi < size(lse); ++mi) {
// const int row = get<0>(taccOcO_row(mi));
// if (row < get<0>(shape_LSE) - m_block * kBlockM) { gLSE(row) = lse(mi); }
// }
// }
// if (cutlass::canonical_warp_idx_sync() == kNWarps - 1) {
// cutlass::arch::NamedBarrier::sync(NumMmaThreads + cutlass::NumThreadsPerWarp,
// static_cast<uint32_t>(FP4NamedBarriers::EpilogueBarrier));
// int const lane_predicate = cute::elect_one_sync();
// if (lane_predicate) {
// cute::copy(epilogue_params.tma_store_O, tOsO, tOgO);
// tma_store_arrive();
// }
// }
cute::copy(epilogue_params.tma_store_O, tOsO, tOgO);
tma_store_arrive();
}
CUTLASS_DEVICE void
store_tail() {
tma_store_wait<0>();
}
// Write 0 to output and -inf to LSE
CUTLASS_DEVICE void
store_zero(
Params const& epilogue_params,
int thread_idx,
cute::tuple<int32_t, int32_t, int32_t> const& block_coord
) {
auto [m_block, bidh, bidb] = block_coord;
Tensor mO = make_tensor(make_gmem_ptr(epilogue_params.ptr_O), epilogue_params.shape_O, epilogue_params.stride_O);
Tensor gO = local_tile(mO(_, _, bidh, bidb), select<0, 2>(TileShape_MNK{}), make_coord(m_block, _0{})); // (M, K)
auto shape_LSE = select<0, 2, 3>(epilogue_params.shape_O);
Tensor mLSE = make_tensor(make_gmem_ptr(epilogue_params.ptr_LSE), shape_LSE, epilogue_params.stride_LSE);
Tensor gLSE = local_tile(mLSE(_, bidh, bidb), Shape<Int<kBlockM>>{}, make_coord(m_block));
GmemTiledCopyO gmem_tiled_copy_O;
auto gmem_thr_copy_O = gmem_tiled_copy_O.get_thread_slice(thread_idx);
Tensor tOgO = gmem_thr_copy_O.partition_D(gO);
Tensor tOrO = make_fragment_like(tOgO);
clear(tOrO);
// Construct identity layout for sO
Tensor cO = cute::make_identity_tensor(select<0, 2>(TileShape_MNK{})); // (BLK_M,BLK_K) -> (blk_m,blk_k)
// Repeat the partitioning with identity layouts
Tensor tOcO = gmem_thr_copy_O.partition_D(cO);
Tensor tOpO = make_tensor<bool>(make_shape(size<2>(tOgO)));
#pragma unroll
for (int k = 0; k < size(tOpO); ++k) { tOpO(k) = get<1>(tOcO(_0{}, _0{}, k)) < get<1>(epilogue_params.shape_O); }
// Clear_OOB_K must be false since we don't want to write zeros to gmem
flash::copy</*Is_even_MN=*/false, /*Is_even_K=*/false, /*Clear_OOB_MN=*/false, /*Clear_OOB_K=*/false>(
gmem_tiled_copy_O, tOrO, tOgO, tOcO, tOpO, get<0>(epilogue_params.shape_O) - m_block * kBlockM
);
static_assert(kBlockM <= NumMmaThreads);
if (thread_idx < get<0>(shape_LSE) - m_block * kBlockM) { gLSE(thread_idx) = INFINITY; }
}
};
} // namespace flash
@@ -0,0 +1,202 @@
/*
* Copyright (c) 2025 by SageAttention team.
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
#pragma once
#include "cute/algorithm/copy.hpp"
#include "cute/atom/mma_atom.hpp"
#include "cutlass/gemm/collective/collective_builder.hpp"
#include "cute/tensor.hpp"
#include "cutlass/cutlass.h"
#include "cutlass/layout/layout.h"
#include "cutlass/numeric_types.h"
#include "cutlass/pipeline/pipeline.hpp"
#include "blockscaled_layout.h"
#include "cute_extension.h"
#include "named_barrier.h"
using namespace cute;
template <
int kStages,
int EpiStages,
typename Element,
typename ElementSF,
typename OutputType,
typename SmemLayoutQ,
typename SmemLayoutK,
typename SmemLayoutV,
typename SmemLayoutDS,
typename SmemLayoutO,
typename SmemLayoutSFQ,
typename SmemLayoutSFK,
typename SmemLayoutSFV
>
struct SharedStorageQKVOwithSF : cute::aligned_struct<128, _0>{
alignas(1024) cute::ArrayEngine<Element, cute::cosize_v<SmemLayoutQ>> smem_q;
alignas(1024) cute::ArrayEngine<Element, cute::cosize_v<SmemLayoutK>> smem_k;
cute::ArrayEngine<ElementSF, cute::cosize_v<SmemLayoutSFQ>> smem_SFQ;
cute::ArrayEngine<ElementSF, cute::cosize_v<SmemLayoutSFK>> smem_SFK;
cute::ArrayEngine<ElementSF, cute::cosize_v<SmemLayoutSFV>> smem_SFV;
alignas(1024) cute::ArrayEngine<float, cute::cosize_v<SmemLayoutDS>> smem_ds;
alignas(1024) cute::ArrayEngine<Element, cute::cosize_v<SmemLayoutV>> smem_v;
alignas(1024) cute::ArrayEngine<OutputType, cute::cosize_v<SmemLayoutO>> smem_o;
struct {
alignas(16) typename cutlass::PipelineTmaAsync<1>::SharedStorage pipeline_q;
alignas(16) typename cutlass::PipelineTmaAsync<kStages>::SharedStorage pipeline_k;
alignas(16) typename cutlass::PipelineTmaAsync<kStages>::SharedStorage pipeline_v;
alignas(16) typename flash::OrderedSequenceBarrierVarGroupSize<EpiStages, 2>::SharedStorage barrier_o;
int tile_count_semaphore;
};
};
template <
int kHeadDim_,
int kBlockM_,
int kBlockN_,
int kStages_,
int kClusterM_,
bool BlockMean_,
typename ElementPairType_ = cutlass::nv_float4_t<cutlass::float_e2m1_t>,
typename ElementOut_ = cutlass::bfloat16_t
>
struct Flash_fwd_kernel_traits {
static constexpr int kBlockM = kBlockM_;
static constexpr int kBlockN = kBlockN_;
static constexpr int kHeadDim = kHeadDim_;
static constexpr bool BlockMean = BlockMean_;
static constexpr bool SmoothQ = true;
static_assert(kHeadDim % 32 == 0);
static_assert(kBlockM == 64 || kBlockM == 128);
static constexpr int kNWarps = kBlockM == 128 ? 12 : 8;
static constexpr int kNThreads = kNWarps * cutlass::NumThreadsPerWarp;
static constexpr int kClusterM = kClusterM_;
static constexpr int kStages = kStages_;
static constexpr int EpiStages = 1;
static constexpr int NumSFQK = kHeadDim / 16;
static constexpr int NumSFPV = kBlockN / 16;
using ElementSF = cutlass::float_ue4m3_t;
using Element = cutlass::float_e2m1_t;
using ElementAccum = float;
using ElementOut = ElementOut_;
using index_t = int64_t;
static constexpr auto SFVectorSize = 16;
using TileShape_MNK = Shape<Int<kBlockM>, Int<kBlockN>, Int<kHeadDim>>;
using ClusterShape_MNK = Shape<_1, _1, _1>;
using PermTileM = decltype(cute::min(size<0>(TileShape_MNK{}), _128{}));
using PermTileN = _32;
using PermTileK = Int<kHeadDim>;
using ElementQMma = decltype(cutlass::gemm::collective::detail::sm1xx_kernel_input_element_to_mma_input_element<Element>());
using ElementKMma = decltype(cutlass::gemm::collective::detail::sm1xx_kernel_input_element_to_mma_input_element<Element>());
using AtomLayoutMNK = std::conditional_t<kBlockM == 128,
Layout<Shape<_8, _1, _1>>,
Layout<Shape<_4, _1, _1>>
>;
using TiledMmaQK = decltype(cute::make_tiled_mma(
cute::SM120::BLOCKSCALED::SM120_16x32x64_TN_VS_NVFP4{},
AtomLayoutMNK{},
Tile<PermTileM, PermTileN, PermTileK>{}
));
using TiledMmaPV = decltype(cute::make_tiled_mma(
cute::SM120::BLOCKSCALED::SM120_16x32x64_TN_VS_NVFP4{},
AtomLayoutMNK{},
Tile<PermTileM, _32, PermTileK>{}
));
static constexpr int MMA_NSF = size<2>(typename TiledMmaQK::AtomShape_MNK{}) / SFVectorSize;
using GmemTiledCopy = SM90_TMA_LOAD;
using GmemTiledCopySF = SM90_TMA_LOAD;
using SmemLayoutAtomQ = decltype(cutlass::gemm::collective::detail::sm120_rr_smem_selector<Element, decltype(size<2>(TileShape_MNK{}))>());
using SmemLayoutAtomK = decltype(cutlass::gemm::collective::detail::sm120_rr_smem_selector<Element, decltype(size<2>(TileShape_MNK{}))>());
using SmemLayoutAtomV = decltype(cutlass::gemm::collective::detail::sm120_rr_smem_selector<Element, decltype(size<2>(TileShape_MNK{}))>());
using SmemLayoutAtomVt = decltype(cutlass::gemm::collective::detail::sm120_rr_smem_selector<Element, decltype(size<1>(TileShape_MNK{}))>());
using SmemLayoutQ = decltype(tile_to_shape(SmemLayoutAtomQ{}, select<0, 2>(TileShape_MNK{})));
using SmemLayoutK =
decltype(tile_to_shape(SmemLayoutAtomK{},
make_shape(shape<1>(TileShape_MNK{}), shape<2>(TileShape_MNK{}), Int<kStages>{})));
using SmemLayoutV =
decltype(tile_to_shape(SmemLayoutAtomV{},
make_shape(shape<1>(TileShape_MNK{}), shape<2>(TileShape_MNK{}), Int<kStages>{})));
using SmemLayoutVt =
decltype(tile_to_shape(SmemLayoutAtomVt{},
make_shape(shape<2>(TileShape_MNK{}), shape<1>(TileShape_MNK{}), Int<kStages>{})));
using SmemLayoutAtomDS = Layout<Shape<Int<kBlockM>, Int<kBlockN>>, Stride<_0, _1>>;
using SmemLayoutDS =
decltype(tile_to_shape(SmemLayoutAtomDS{},
make_shape(shape<0>(TileShape_MNK{}), shape<1>(TileShape_MNK{}), Int<kStages>{})));
using SmemCopyAtomQ = Copy_Atom<SM75_U32x4_LDSM_N, Element>;
using SmemCopyAtomKV = Copy_Atom<SM75_U32x4_LDSM_N, Element>;
using SmemCopyAtomSF = Copy_Atom<UniversalCopy<ElementSF>, ElementSF>;
using SmemCopyAtomDS = Copy_Atom<UniversalCopy<float>, float>;
using BlkScaledConfig = flash::BlockScaledConfig<SFVectorSize>;
using LayoutSF = typename BlkScaledConfig::LayoutSF;
using SfAtom = typename BlkScaledConfig::SfAtom;
using SmemLayoutAtomSFQ = decltype(BlkScaledConfig::deduce_smem_layoutSFQ(TiledMmaQK{}, TileShape_MNK{}));
using SmemLayoutAtomSFK = decltype(BlkScaledConfig::deduce_smem_layoutSFKV(TiledMmaQK{}, TileShape_MNK{}));
using SmemLayoutAtomSFV = decltype(BlkScaledConfig::deduce_smem_layoutSFKV(TiledMmaPV{}, TileShape_MNK{}));
using SmemLayoutAtomSFVt = decltype(BlkScaledConfig::deduce_smem_layoutSFVt(TiledMmaPV{}, Shape<Int<kBlockM>, Int<kHeadDim>, Int<kBlockN>>{}));
using LayoutSFP = decltype(
make_layout(
make_shape(make_shape(_16{}, _4{}), _1{}, Int<kBlockN / 64>{}),
make_stride(make_stride(_0{}, _1{}), _0{}, _4{})
)
);
using LayoutP = decltype(
make_layout(
make_shape(make_shape(_8{}, _2{}, _2{}), _1{}, Int<kBlockN / 64>{}),
make_stride(make_stride(_1{}, _8{}, _16{}), _0{}, _32{})
)
);
using SmemLayoutSFQ = decltype(make_layout(
shape(SmemLayoutAtomSFQ{}),
stride(SmemLayoutAtomSFQ{})
));
using SmemLayoutSFK = decltype(make_layout(
append(shape(SmemLayoutAtomSFK{}), Int<kStages>{}),
append(stride(SmemLayoutAtomSFK{}), size(filter_zeros(SmemLayoutAtomSFK{})))
));
using SmemLayoutSFV = decltype(make_layout(
append(shape(SmemLayoutAtomSFV{}), Int<kStages>{}),
append(stride(SmemLayoutAtomSFV{}), size(filter_zeros(SmemLayoutAtomSFV{})))
));
using SmemLayoutSFVt = decltype(make_layout(
append(shape(SmemLayoutAtomSFVt{}), Int<kStages>{}),
append(stride(SmemLayoutAtomSFVt{}), size(filter_zeros(SmemLayoutAtomSFVt{})))
));
using SmemLayoutAtomO = decltype(cutlass::gemm::collective::detail::ss_smem_selector<GMMA::Major::K, ElementOut,
decltype(cute::get<0>(TileShape_MNK{})), decltype(cute::get<2>(TileShape_MNK{}))>());
using SmemLayoutO = decltype(tile_to_shape(SmemLayoutAtomO{}, select<0, 2>(TileShape_MNK{}), Step<_1, _2>{}));
using SharedStorage = SharedStorageQKVOwithSF<kStages, EpiStages, Element, ElementSF, ElementOut,
SmemLayoutQ, SmemLayoutK, SmemLayoutV, SmemLayoutDS,
SmemLayoutO, SmemLayoutSFQ, SmemLayoutSFK, SmemLayoutSFVt>;
using MainloopPipeline = typename cutlass::PipelineTmaAsync<kStages>;
using PipelineState = typename cutlass::PipelineState<kStages>;
using MainloopPipelineQ = cutlass::PipelineTmaAsync<1>;
using PipelineParamsQ = typename MainloopPipelineQ::Params;
using PipelineStateQ = typename cutlass::PipelineState<1>;
using EpilogueBarrier = typename flash::OrderedSequenceBarrierVarGroupSize<EpiStages, 2>;
};
@@ -0,0 +1,204 @@
// Modified from the original SageAttention3 code
/*
* Copyright (c) 2025 by SageAttention team.
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
#pragma once
#include "cute/tensor.hpp"
#include <cutlass/cutlass.h>
#include <cutlass/arch/reg_reconfig.h>
#include <cutlass/array.h>
#include <cutlass/numeric_types.h>
#include <cutlass/numeric_conversion.h>
#include "cutlass/pipeline/pipeline.hpp"
#include "params.h"
#include "utils.h"
#include "tile_scheduler.h"
#include "mainloop_tma_ws.h"
#include "epilogue_tma_ws.h"
#include "named_barrier.h"
#include "softmax_fused.h"
namespace flash {
using namespace cute;
template <typename Ktraits, bool Is_causal, typename TileScheduler>
__global__ void __launch_bounds__(Ktraits::kNWarps * cutlass::NumThreadsPerWarp, 1)
compute_attn_ws(CUTE_GRID_CONSTANT Flash_fwd_params const params,
CUTE_GRID_CONSTANT typename CollectiveMainloopFwd<Ktraits, Is_causal>::Params const mainloop_params,
CUTE_GRID_CONSTANT typename CollectiveEpilogueFwd<Ktraits>::Params const epilogue_params,
CUTE_GRID_CONSTANT typename TileScheduler::Params const scheduler_params
) {
using Element = typename Ktraits::Element;
using ElementAccum = typename Ktraits::ElementAccum;
using SoftType = ElementAccum;
using TileShape_MNK = typename Ktraits::TileShape_MNK;
using ClusterShape = typename Ktraits::ClusterShape_MNK;
static constexpr int NumMmaThreads = size(typename Ktraits::TiledMmaQK{});
static constexpr int NumCopyThreads = cutlass::NumThreadsPerWarpGroup;
static constexpr int kBlockM = Ktraits::kBlockM;
using CollectiveMainloop = CollectiveMainloopFwd<Ktraits, Is_causal>;
using CollectiveEpilogue = CollectiveEpilogueFwd<Ktraits>;
using MainloopPipeline = typename Ktraits::MainloopPipeline;
using PipelineParams = typename MainloopPipeline::Params;
using PipelineState = typename MainloopPipeline::PipelineState;
using MainloopPipelineQ = typename Ktraits::MainloopPipelineQ;
using PipelineParamsQ = typename Ktraits::PipelineParamsQ;
using PipelineStateQ = typename Ktraits::PipelineStateQ;
using EpilogueBarrier = typename Ktraits::EpilogueBarrier;
enum class WarpGroupRole {
Producer = 0,
Consumer0 = 1,
Consumer1 = 2
};
enum class ProducerWarpRole {
Mainloop = 0,
Epilogue = 1,
Warp2 = 2,
Warp3 = 3
};
extern __shared__ char shared_memory[];
auto &shared_storage = *reinterpret_cast<typename Ktraits::SharedStorage*>(shared_memory);
int const lane_predicate = cute::elect_one_sync();
int const warp_idx = cutlass::canonical_warp_idx_sync();
int warp_group_idx = cutlass::canonical_warp_group_idx();
int const warp_group_thread_idx = threadIdx.x % cutlass::NumThreadsPerWarpGroup;
int warp_idx_in_warp_group = warp_idx % cutlass::NumWarpsPerWarpGroup;
auto warp_group_role = WarpGroupRole(warp_group_idx);
auto producer_warp_role = ProducerWarpRole(warp_idx_in_warp_group);
// Issue Tma Descriptor Prefetch from a single thread
if (warp_idx == 0 && lane_predicate) {
CollectiveMainloop::prefetch_tma_descriptors(mainloop_params);
CollectiveEpilogue::prefetch_tma_descriptors(epilogue_params);
}
// Obtain warp index
PipelineParams pipeline_params_v;
pipeline_params_v.transaction_bytes = CollectiveMainloop::TmaTransactionBytesV;
pipeline_params_v.role = warp_group_role == WarpGroupRole::Producer
? MainloopPipeline::ThreadCategory::Producer
: MainloopPipeline::ThreadCategory::Consumer;
pipeline_params_v.is_leader = warp_group_thread_idx == 0;
pipeline_params_v.num_consumers = NumMmaThreads;
PipelineParams pipeline_params_k;
pipeline_params_k.transaction_bytes = CollectiveMainloop::TmaTransactionBytesK;
pipeline_params_k.role = warp_group_role == WarpGroupRole::Producer
? MainloopPipeline::ThreadCategory::Producer
: MainloopPipeline::ThreadCategory::Consumer;
pipeline_params_k.is_leader = warp_group_thread_idx == 0;
pipeline_params_k.num_consumers = NumMmaThreads;
PipelineParamsQ pipeline_params_q;
pipeline_params_q.transaction_bytes = CollectiveMainloop::TmaTransactionBytesQ;
pipeline_params_q.role = warp_group_role == WarpGroupRole::Producer
? MainloopPipelineQ::ThreadCategory::Producer
: MainloopPipelineQ::ThreadCategory::Consumer;
pipeline_params_q.is_leader = warp_group_thread_idx == 0;
pipeline_params_q.num_consumers = NumMmaThreads;
// We're counting on pipeline_k to call cutlass::arch::fence_barrier_init();
MainloopPipelineQ pipeline_q(shared_storage.pipeline_q, pipeline_params_q, ClusterShape{});
MainloopPipeline pipeline_k(shared_storage.pipeline_k, pipeline_params_k, ClusterShape{});
MainloopPipeline pipeline_v(shared_storage.pipeline_v, pipeline_params_v, ClusterShape{});
uint32_t epilogue_barrier_group_size_list[2] = {cutlass::NumThreadsPerWarp, NumMmaThreads};
typename EpilogueBarrier::Params params_epilogue_barrier;
params_epilogue_barrier.group_id = (warp_group_role == WarpGroupRole::Producer);
params_epilogue_barrier.group_size_list = epilogue_barrier_group_size_list;
EpilogueBarrier barrier_o(shared_storage.barrier_o, params_epilogue_barrier);
CollectiveMainloop collective_mainloop;
CollectiveEpilogue collective_epilogue;
__syncthreads();
if (warp_group_role == WarpGroupRole::Producer) {
cutlass::arch::warpgroup_reg_dealloc<24>();
TileScheduler scheduler;
if (producer_warp_role == ProducerWarpRole::Mainloop) { // Load Q, K, V
PipelineStateQ smem_pipe_write_q = cutlass::make_producer_start_state<MainloopPipelineQ>();
PipelineState smem_pipe_write_k = cutlass::make_producer_start_state<MainloopPipeline>();
PipelineState smem_pipe_write_v = cutlass::make_producer_start_state<MainloopPipeline>();
int work_idx = 0;
for (auto work_tile_info = scheduler.get_initial_work(); work_tile_info.is_valid(scheduler_params); work_tile_info = scheduler.get_next_work(scheduler_params, work_tile_info)) {
int tile_count_semaphore = 0;
collective_mainloop.load(mainloop_params, scheduler_params,
pipeline_q, pipeline_k, pipeline_v,
smem_pipe_write_q, smem_pipe_write_k, smem_pipe_write_v,
shared_storage, work_tile_info, work_idx, tile_count_semaphore);
}
collective_mainloop.load_tail(pipeline_q, pipeline_k, pipeline_v,
smem_pipe_write_q, smem_pipe_write_k, smem_pipe_write_v);
} else if (producer_warp_role == ProducerWarpRole::Epilogue) {
for (auto work_tile_info = scheduler.get_initial_work(); work_tile_info.is_valid(scheduler_params); work_tile_info = scheduler.get_next_work(scheduler_params, work_tile_info)) {
barrier_o.wait();
collective_epilogue.tma_store(shared_storage, epilogue_params, work_tile_info, scheduler_params, threadIdx.x);
collective_epilogue.store_tail();
barrier_o.arrive();
}
}
} else if (warp_group_role == WarpGroupRole::Consumer0 || warp_group_role == WarpGroupRole::Consumer1) {
cutlass::arch::warpgroup_reg_alloc<232>();
typename Ktraits::TiledMmaPV tiled_mma_pv;
TileScheduler scheduler{};
PipelineState smem_pipe_read_k, smem_pipe_read_v;
PipelineStateQ smem_pipe_read_q;
int work_idx = 0;
CUTLASS_PRAGMA_NO_UNROLL
for (auto work_tile_info = scheduler.get_initial_work(); work_tile_info.is_valid(scheduler_params); work_tile_info = scheduler.get_next_work(scheduler_params, work_tile_info)) {
// Attention output (GEMM-II) accumulator.
Tensor tOrO = partition_fragment_C(tiled_mma_pv, select<0, 2>(TileShape_MNK{}));
// flash::Softmax<2 * (2 * kBlockM / NumMmaThreads)> softmax;
// Pass single_level_p_quant flag to control P quantization mode
flash::SoftmaxFused<2 * (2 * kBlockM / NumMmaThreads)> softmax_fused(params.single_level_p_quant);
auto block_coord = work_tile_info.get_block_coord(scheduler_params);
auto [m_block, bidh, bidb] = block_coord;
int n_block_max = collective_mainloop.get_n_block_max(mainloop_params, m_block);
if (Is_causal && n_block_max <= 0) { // We exit early and write 0 to gO and -inf to gLSE.
collective_epilogue.store_zero(epilogue_params, threadIdx.x - NumCopyThreads, block_coord);
continue;
}
collective_mainloop.mma(mainloop_params, pipeline_q, pipeline_k, pipeline_v, smem_pipe_read_q, smem_pipe_read_k, smem_pipe_read_v,
tOrO, softmax_fused, n_block_max, threadIdx.x - NumCopyThreads, work_idx, m_block, shared_storage);
barrier_o.wait();
collective_epilogue.mma_store(shared_storage, tiled_mma_pv, tOrO, threadIdx.x - NumCopyThreads);
barrier_o.arrive();
++work_idx;
}
}
}
} // namespace flash
@@ -0,0 +1,114 @@
/*
* Copyright (c) 2025 by SageAttention team.
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
#pragma once
#include <ATen/cuda/CUDAContext.h>
#include "cute/tensor.hpp"
#include "cutlass/cluster_launch.hpp"
#include "static_switch.h"
#include "params.h"
#include "tile_scheduler.h"
#include "kernel_ws.h"
#include "kernel_traits.h"
#include "block_config.h"
template<typename Kernel_traits, bool Is_causal>
void run_flash_fwd(Flash_fwd_params &params, cudaStream_t stream) {
using Element = typename Kernel_traits::Element;
using ElementSF = typename Kernel_traits::ElementSF;
using ElementOut = typename Kernel_traits::ElementOut;
using TileShape_MNK = typename Kernel_traits::TileShape_MNK;
using ClusterShape = typename Kernel_traits::ClusterShape_MNK;
using CollectiveMainloop = flash::CollectiveMainloopFwd<Kernel_traits, Is_causal>;
using CollectiveEpilogue = flash::CollectiveEpilogueFwd<Kernel_traits>;
// using Scheduler = flash::SingleTileScheduler;
using Scheduler = flash::StaticPersistentTileScheduler;
typename CollectiveMainloop::Params mainloop_params =
CollectiveMainloop::to_underlying_arguments({
static_cast<Element const*>(params.q_ptr),
{params.seqlen_q, params.d, params.h, params.b}, // shape_Q
{params.q_row_stride, _1{}, params.q_head_stride, params.q_batch_stride}, // stride_Q
static_cast<Element const*>(params.k_ptr),
{params.seqlen_k, params.d, params.h_k, params.b}, // shape_K
{params.k_row_stride, _1{}, params.k_head_stride, params.k_batch_stride}, // stride_K
{params.unpadded_seqlen_k, params.d, params.h_k, params.b}, // shape_K
static_cast<Element const*>(params.v_ptr),
{params.d, params.seqlen_k, params.h_k, params.b}, // shape_Vt
{params.v_row_stride, _1{}, params.v_head_stride, params.v_batch_stride}, // stride_Vt
static_cast<ElementSF const*>(params.sfq_ptr),
{params.seqlen_q, params.d, params.h, params.b}, // shape_SFQ
static_cast<ElementSF const*>(params.sfk_ptr),
{params.seqlen_k, params.d, params.h_k, params.b}, // shape_SFK
static_cast<ElementSF const*>(params.sfv_ptr),
{params.d, params.seqlen_k, params.h_k, params.b}, // shape_SFVt
static_cast<float const*>(params.delta_s_ptr),
{params.seqlen_s, params.seqlen_k, params.h_k, params.b},
{params.ds_row_stride, _1{}, params.ds_head_stride, params.ds_batch_stride},
params.scale_softmax_log2
});
typename CollectiveEpilogue::Params epilogue_params =
CollectiveEpilogue::to_underlying_arguments({
static_cast<ElementOut*>(params.o_ptr),
{params.seqlen_q, params.d, params.h, params.b}, // shape_O
{params.o_row_stride, _1{}, params.o_head_stride, params.o_batch_stride}, // stride_O
static_cast<float*>(params.softmax_lse_ptr),
{_1{}, params.seqlen_q, params.h * params.seqlen_q}, // stride_LSE
});
int num_blocks_m = cutlass::ceil_div(params.seqlen_q, Kernel_traits::kBlockM);
num_blocks_m = cutlass::ceil_div(num_blocks_m, size<0>(ClusterShape{})) * size<0>(ClusterShape{});
typename Scheduler::Arguments scheduler_args = {num_blocks_m, params.h, params.b};
typename Scheduler::Params scheduler_params = Scheduler::to_underlying_arguments(scheduler_args);
// Get the ptr to kernel function.
void *kernel;
kernel = (void *)flash::compute_attn_ws<Kernel_traits, Is_causal, Scheduler>;
int smem_size = sizeof(typename Kernel_traits::SharedStorage);
if (smem_size >= 48 * 1024) {
C10_CUDA_CHECK(cudaFuncSetAttribute(kernel, cudaFuncAttributeMaxDynamicSharedMemorySize, smem_size));
}
static constexpr int ctaSize = Kernel_traits::kNWarps * 32;
params.m_block_divmod = cutlass::FastDivmod(num_blocks_m);
params.total_blocks = num_blocks_m * params.h * params.b;
dim3 grid_dims = Scheduler::get_grid_dim(scheduler_args, 170);
dim3 block_dims(ctaSize);
dim3 cluster_dims(size<0>(ClusterShape{}), size<1>(ClusterShape{}), size<2>(ClusterShape{}));
cutlass::ClusterLaunchParams launch_params{grid_dims, block_dims, cluster_dims, smem_size, stream};
cutlass::launch_kernel_on_cluster(launch_params, kernel, params, mainloop_params, epilogue_params, scheduler_params);
C10_CUDA_KERNEL_LAUNCH_CHECK();
}
template<typename T, int Headdim, typename O = cutlass::bfloat16_t>
void run_mha_fwd_(Flash_fwd_params &params, cudaStream_t stream) {
BOOL_SWITCH(params.is_causal, Is_causal, [&] {
BOOL_SWITCH(params.per_block_mean, per_block, [&] {
if constexpr (Headdim == 64 || Headdim == 128) {
run_flash_fwd<
Flash_fwd_kernel_traits<Headdim, flash::BLOCK_M, flash::BLOCK_N, 3, 1, per_block, T, O>,
Is_causal
>(params, stream);
} else {
static_assert(Headdim == 64 || Headdim == 128, "Unsupported Headdim");
}
});
});
}
@@ -0,0 +1,908 @@
/*
* Copyright (c) 2025 by SageAttention team.
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
#pragma once
#include <cutlass/cutlass.h>
#include <cutlass/array.h>
#include <cutlass/numeric_types.h>
#include <cutlass/numeric_conversion.h>
#include "cutlass/pipeline/pipeline.hpp"
#include "cute/tensor.hpp"
#include "cutlass/gemm/collective/collective_builder.hpp"
#include "utils.h"
#include "named_barrier.h"
namespace flash {
using namespace cute;
template <typename Ktraits, bool Is_causal>
struct CollectiveMainloopFwd {
using Element = typename Ktraits::Element;
using ElementSF = typename Ktraits::ElementSF;
// using TMAElement = Element;
// using TMAElementSF = typename Ktraits::ElementSF;
using TileShape_MNK = typename Ktraits::TileShape_MNK;
using ClusterShape = typename Ktraits::ClusterShape_MNK;
static constexpr int kStages = Ktraits::kStages;
static constexpr int kHeadDim = Ktraits::kHeadDim;
static constexpr int BlockMean = Ktraits::BlockMean;
using GmemTiledCopy = typename Ktraits::GmemTiledCopy;
using SmemLayoutQ = typename Ktraits::SmemLayoutQ;
using SmemLayoutK = typename Ktraits::SmemLayoutK;
using SmemLayoutV = typename Ktraits::SmemLayoutV;
using SmemLayoutVt = typename Ktraits::SmemLayoutVt;
using SmemLayoutDS = typename Ktraits::SmemLayoutDS;
using SmemLayoutAtomDS = typename Ktraits::SmemLayoutAtomDS;
using LayoutDS = decltype(
blocked_product(
SmemLayoutAtomDS{},
make_layout(
make_shape(int32_t(0), int32_t(0), int32_t(0), int32_t(0)),
make_stride(int32_t(0), _1{}, int32_t(0), int32_t(0)))
)
);
using ShapeQKV = cute::Shape<int32_t, int32_t, int32_t, int32_t>; // (seqlen, d, head, batch)
using StrideQKV = cute::Stride<int64_t, _1, int64_t, int64_t>;
using ShapeSF = cute::Shape<int32_t, int32_t, int32_t, int32_t>; // (seqlen, d // 16, head, batch)
using LayoutSF = typename Ktraits::LayoutSF;
using LayoutP = typename Ktraits::LayoutP;
using LayoutSFP = typename Ktraits::LayoutSFP;
using SfAtom = typename Ktraits::SfAtom;
using TMA_Q = decltype(make_tma_copy(
GmemTiledCopy{},
make_tensor(make_gmem_ptr(static_cast<Element const*>(nullptr)), repeat_like(StrideQKV{}, int32_t(0)), StrideQKV{}),
SmemLayoutQ{},
select<0, 2>(TileShape_MNK{}),
_1{}));
using TMA_KV = decltype(make_tma_copy(
GmemTiledCopy{},
make_tensor(make_gmem_ptr(static_cast<Element const*>(nullptr)), repeat_like(StrideQKV{}, int32_t(0)), StrideQKV{}),
take<0, 2>(SmemLayoutK{}),
select<1, 2>(TileShape_MNK{}),
_1{}));
using TMA_Vt = decltype(make_tma_copy(
GmemTiledCopy{},
make_tensor(make_gmem_ptr(static_cast<Element const*>(nullptr)), repeat_like(StrideQKV{}, int32_t(0)), StrideQKV{}),
take<0, 2>(SmemLayoutVt{}),
make_shape(shape<2>(TileShape_MNK{}), shape<1>(TileShape_MNK{})),
_1{}));
using TMA_DS = decltype(make_tma_copy(
GmemTiledCopy{},
make_tensor(make_gmem_ptr(static_cast<float const*>(nullptr)), LayoutDS{}),
take<0, 2>(SmemLayoutDS{}),
make_shape(shape<0>(TileShape_MNK{}), shape<1>(TileShape_MNK{})),
_1{}));
using BlkScaledConfig = typename Ktraits::BlkScaledConfig;
using GmemTiledCopySF = typename Ktraits::GmemTiledCopySF;
using SmemLayoutSFQ = typename Ktraits::SmemLayoutSFQ;
using SmemLayoutSFK = typename Ktraits::SmemLayoutSFK;
using SmemLayoutSFV = typename Ktraits::SmemLayoutSFV;
using SmemLayoutSFVt = typename Ktraits::SmemLayoutSFVt;
using TMA_SFQ = decltype(make_tma_copy<uint16_t>(
GmemTiledCopySF{},
make_tensor(static_cast<ElementSF const*>(nullptr), LayoutSF{}),
SmemLayoutSFQ{},
make_shape(shape<0>(TileShape_MNK{}), shape<2>(TileShape_MNK{})),
_1{})); // No programmatic multicast
using TMA_SFKV = decltype(make_tma_copy<uint16_t>(
GmemTiledCopySF{},
make_tensor(static_cast<ElementSF const*>(nullptr), LayoutSF{}),
SmemLayoutSFK{}(_,_,cute::Int<0>{}),
make_shape(shape<1>(TileShape_MNK{}), shape<2>(TileShape_MNK{})),
_1{}));
using TMA_SFVt = decltype(make_tma_copy<uint16_t>(
GmemTiledCopySF{},
make_tensor(static_cast<ElementSF const*>(nullptr), LayoutSF{}),
SmemLayoutSFVt{}(_,_,cute::Int<0>{}),
make_shape(shape<2>(TileShape_MNK{}), shape<1>(TileShape_MNK{})),
_1{}));
using SmemCopyAtomQ = typename Ktraits::SmemCopyAtomQ;
using SmemCopyAtomKV = typename Ktraits::SmemCopyAtomKV;
using SmemCopyAtomSF = typename Ktraits::SmemCopyAtomSF;
using TiledMmaQK = typename Ktraits::TiledMmaQK;
using TiledMmaPV = typename Ktraits::TiledMmaPV;
static constexpr int NumMmaThreads = size(TiledMmaQK{});
using MainloopPipeline = typename Ktraits::MainloopPipeline;
using PipelineParams = typename MainloopPipeline::Params;
using PipelineState = typename MainloopPipeline::PipelineState;
using MainloopPipelineQ = typename Ktraits::MainloopPipelineQ;
using PipelineParamsQ = typename Ktraits::PipelineParamsQ;
using PipelineStateQ = typename Ktraits::PipelineStateQ;
using EpilogueBarrier = typename Ktraits::EpilogueBarrier;
// Set the bytes transferred in this TMA transaction (may involve multiple issues)
static constexpr uint32_t TmaTransactionBytesQ = static_cast<uint32_t>(
cutlass::bits_to_bytes(cosize((SmemLayoutSFQ{})) * cute::sizeof_bits_v<ElementSF>) +
cutlass::bits_to_bytes(size((SmemLayoutQ{})) * sizeof_bits<Element>::value));
static constexpr uint32_t TmaTransactionBytesK = static_cast<uint32_t>(
cutlass::bits_to_bytes(cosize(take<0,2>(SmemLayoutSFK{})) * cute::sizeof_bits_v<ElementSF>) +
cutlass::bits_to_bytes(cosize(take<0,2>(SmemLayoutDS{})) * cute::sizeof_bits_v<float>) +
cutlass::bits_to_bytes(size(take<0,2>(SmemLayoutK{})) * sizeof_bits<Element>::value));
static constexpr uint32_t TmaTransactionBytesV = static_cast<uint32_t>(
cutlass::bits_to_bytes(cosize(take<0,2>(SmemLayoutSFVt{})) * cute::sizeof_bits_v<ElementSF>) +
cutlass::bits_to_bytes(size(take<0,2>(SmemLayoutVt{})) * sizeof_bits<Element>::value));
// Host side kernel arguments
struct Arguments {
Element const* ptr_Q;
ShapeQKV const shape_Q;
StrideQKV const stride_Q;
Element const* ptr_K;
ShapeQKV const shape_K;
StrideQKV const stride_K;
ShapeQKV const unpadded_shape_K;
Element const* ptr_Vt;
ShapeQKV const shape_Vt;
StrideQKV const stride_Vt;
ElementSF const* ptr_SFQ{nullptr};
ShapeSF const shape_SFQ{};
ElementSF const* ptr_SFK{nullptr};
ShapeSF const shape_SFK{};
ElementSF const* ptr_SFVt{nullptr};
ShapeSF const shape_SFVt{};
float const* ptr_ds;
ShapeQKV const shape_ds;
StrideQKV const stride_ds;
float const softmax_scale_log2;
};
// Device side kernel params
struct Params {
ShapeQKV const shape_Q;
LayoutSF const layout_SFQ;
ShapeQKV const shape_K;
ShapeQKV const unpadded_shape_K;
LayoutSF const layout_SFK;
ShapeQKV const shape_Vt;
LayoutSF const layout_SFVt;
LayoutDS const layout_DS;
TMA_Q tma_load_Q;
TMA_SFQ tma_load_SFQ;
TMA_KV tma_load_K;
TMA_SFKV tma_load_SFK;
TMA_Vt tma_load_Vt;
TMA_SFVt tma_load_SFVt;
TMA_DS tma_load_DS;
float const softmax_scale_log2;
};
static Params
to_underlying_arguments(Arguments const& args) {
Tensor mQ = make_tensor(make_gmem_ptr(args.ptr_Q), args.shape_Q, args.stride_Q);
TMA_Q tma_load_Q = make_tma_copy(
GmemTiledCopy{},
mQ,
SmemLayoutQ{},
select<0, 2>(TileShape_MNK{}),
_1{}); // no mcast for Q
Tensor mK = make_tensor(make_gmem_ptr(args.ptr_K), args.shape_K, args.stride_K);
TMA_KV tma_load_K = make_tma_copy(
GmemTiledCopy{},
mK,
SmemLayoutK{}(_, _, _0{}),
select<1, 2>(TileShape_MNK{}),
_1{}); // mcast along M mode for this N load, if any
Tensor mVt = make_tensor(make_gmem_ptr(args.ptr_Vt), args.shape_Vt, args.stride_Vt);
TMA_Vt tma_load_Vt = make_tma_copy(
GmemTiledCopy{},
mVt,
SmemLayoutVt{}(_, _, _0{}),
make_shape(shape<2>(TileShape_MNK{}), shape<1>(TileShape_MNK{})),
_1{}); // mcast along M mode for this N load, if any
auto [Seqlen_Q, Seqlen_K, HeadNum, Batch] = args.shape_ds;
LayoutDS layout_ds = tile_to_shape(SmemLayoutAtomDS{}, make_shape(Seqlen_Q, Seqlen_K, HeadNum, Batch), Step<_2,_1,_3,_4>{});
Tensor mDS = make_tensor(make_gmem_ptr(args.ptr_ds), layout_ds);
TMA_DS tma_load_ds = make_tma_copy (
GmemTiledCopy{},
mDS,
SmemLayoutDS{}(_, _, _0{}),
make_shape(shape<0>(TileShape_MNK{}), shape<1>(TileShape_MNK{})),
_1{});
LayoutSF layout_sfq = BlkScaledConfig::tile_atom_to_shape_SFQKV(args.shape_SFQ);
Tensor mSFQ = make_tensor(make_gmem_ptr(args.ptr_SFQ), layout_sfq);
TMA_SFQ tma_load_sfq = make_tma_copy<uint16_t>(
GmemTiledCopySF{},
mSFQ,
SmemLayoutSFQ{},
make_shape(shape<0>(TileShape_MNK{}), shape<2>(TileShape_MNK{})),
_1{});
LayoutSF layout_sfk = BlkScaledConfig::tile_atom_to_shape_SFQKV(args.shape_SFK);
Tensor mSFK = make_tensor(make_gmem_ptr(args.ptr_SFK), layout_sfk);
TMA_SFKV tma_load_sfk = make_tma_copy<uint16_t>(
GmemTiledCopySF{},
mSFK,
SmemLayoutSFK{}(_, _, _0{}),
make_shape(shape<1>(TileShape_MNK{}), shape<2>(TileShape_MNK{})),
_1{});
LayoutSF layout_sfvt = BlkScaledConfig::tile_atom_to_shape_SFVt(args.shape_SFVt);
Tensor mSFVt = make_tensor(make_gmem_ptr(args.ptr_SFVt), layout_sfvt);
TMA_SFVt tma_load_sfvt = make_tma_copy<uint16_t>(
GmemTiledCopySF{},
mSFVt,
SmemLayoutSFVt{}(_, _, _0{}),
make_shape(shape<2>(TileShape_MNK{}), shape<1>(TileShape_MNK{})),
_1{});
return {args.shape_Q, layout_sfq,
args.shape_K, args.unpadded_shape_K, layout_sfk,
args.shape_Vt, layout_sfvt,
layout_ds,
tma_load_Q, tma_load_sfq,
tma_load_K, tma_load_sfk,
tma_load_Vt, tma_load_sfvt,
tma_load_ds,
args.softmax_scale_log2};
}
/// Issue Tma Descriptor Prefetch -- ideally from a single thread for best performance
CUTLASS_DEVICE
static void prefetch_tma_descriptors(Params const& mainloop_params) {
cute::prefetch_tma_descriptor(mainloop_params.tma_load_Q.get_tma_descriptor());
cute::prefetch_tma_descriptor(mainloop_params.tma_load_K.get_tma_descriptor());
cute::prefetch_tma_descriptor(mainloop_params.tma_load_Vt.get_tma_descriptor());
cute::prefetch_tma_descriptor(mainloop_params.tma_load_SFQ.get_tma_descriptor());
cute::prefetch_tma_descriptor(mainloop_params.tma_load_SFK.get_tma_descriptor());
cute::prefetch_tma_descriptor(mainloop_params.tma_load_SFVt.get_tma_descriptor());
cute::prefetch_tma_descriptor(mainloop_params.tma_load_DS.get_tma_descriptor());
}
CUTLASS_DEVICE
int get_n_block_max(Params const& mainloop_params, int m_block) {
static constexpr int kBlockM = get<0>(TileShape_MNK{});
static constexpr int kBlockN = get<1>(TileShape_MNK{});
int const seqlen_q = get<0>(mainloop_params.shape_Q);
int const seqlen_k = get<0>(mainloop_params.shape_K);
int n_block_max = cute::ceil_div(seqlen_k, kBlockN);
if constexpr (Is_causal) {
n_block_max = std::min(n_block_max,
cute::ceil_div((m_block + 1) * kBlockM + seqlen_k - seqlen_q, kBlockN));
}
return n_block_max;
}
template <class SFATensor, class Atom, class TiledThr, class TiledPerm>
CUTE_HOST_DEVICE constexpr
auto
thrfrg_SFA(SFATensor&& sfatensor, TiledMMA<Atom, TiledThr, TiledPerm>& mma)
{
CUTE_STATIC_ASSERT_V(rank(sfatensor) >= Int<2>{});
using AtomShape_MNK = typename Atom::Shape_MNK;
using AtomLayoutSFA_TV = typename Atom::Traits::SFALayout;
auto permutation_mnk = TiledPerm{};
auto thr_layout_vmnk = mma.get_thr_layout_vmnk();
// Reorder the tensor for the TiledAtom
auto t_tile = make_tile(get<0>(permutation_mnk),
get<2>(permutation_mnk));
auto t_tensor = logical_divide(sfatensor, t_tile); // (PermM,PermK)
// Tile the tensor for the Atom
auto a_tile = make_tile(make_layout(size<0>(AtomShape_MNK{})),
make_layout(size<2>(AtomShape_MNK{})));
auto a_tensor = zipped_divide(t_tensor, a_tile); // ((AtomM,AtomK),(RestM,RestK))
// Transform the Atom mode from (M,K) to (Thr,Val)
auto tv_tensor = a_tensor.compose(AtomLayoutSFA_TV{},_); // ((ThrV,FrgV),(RestM,RestK))
// Tile the tensor for the Thread
auto thr_tile = make_tile(_,
make_tile(make_layout(size<1>(thr_layout_vmnk)),
make_layout(size<3>(thr_layout_vmnk))));
auto thr_tensor = zipped_divide(tv_tensor, thr_tile); // ((ThrV,(ThrM,ThrK)),(FrgV,(RestM,RestK)))
return thr_tensor;
}
template <class SFBTensor, class Atom, class TiledThr, class TiledPerm>
CUTE_HOST_DEVICE constexpr
auto
thrfrg_SFB(SFBTensor&& sfbtensor, TiledMMA<Atom, TiledThr, TiledPerm>& mma)
{
CUTE_STATIC_ASSERT_V(rank(sfbtensor) >= Int<2>{});
using AtomShape_MNK = typename Atom::Shape_MNK;
using AtomLayoutSFB_TV = typename Atom::Traits::SFBLayout;
auto permutation_mnk = TiledPerm{};
auto thr_layout_vmnk = mma.get_thr_layout_vmnk();
// Reorder the tensor for the TiledAtom
auto t_tile = make_tile(get<1>(permutation_mnk),
get<2>(permutation_mnk));
auto t_tensor = logical_divide(sfbtensor, t_tile); // (PermN,PermK)
// Tile the tensor for the Atom
auto a_tile = make_tile(make_layout(size<1>(AtomShape_MNK{})),
make_layout(size<2>(AtomShape_MNK{})));
auto a_tensor = zipped_divide(t_tensor, a_tile); // ((AtomN,AtomK),(RestN,RestK))
// Transform the Atom mode from (M,K) to (Thr,Val)
auto tv_tensor = a_tensor.compose(AtomLayoutSFB_TV{},_); // ((ThrV,FrgV),(RestN,RestK))
// Tile the tensor for the Thread
auto thr_tile = make_tile(_,
make_tile(make_layout(size<2>(thr_layout_vmnk)),
make_layout(size<3>(thr_layout_vmnk))));
auto thr_tensor = zipped_divide(tv_tensor, thr_tile); // ((ThrV,(ThrN,ThrK)),(FrgV,(RestN,RestK)))
return thr_tensor;
}
template <class SFATensor, class ThrMma>
CUTE_HOST_DEVICE constexpr
auto
partition_fragment_SFA(SFATensor&& sfatensor, ThrMma& thread_mma)
{
using ValTypeSF = typename ThrMma::Atom::Traits::ValTypeSF;
auto thr_tensor = make_tensor(static_cast<SFATensor&&>(sfatensor).data(), thrfrg_SFA(sfatensor.layout(),thread_mma));
auto thr_vmnk = thread_mma.thr_vmnk_;
auto thr_vmk = make_coord(get<0>(thr_vmnk), make_coord(get<1>(thr_vmnk), get<3>(thr_vmnk)));
auto partition_SFA = thr_tensor(thr_vmk, make_coord(_, repeat<rank<1,1>(thr_tensor)>(_)));
return make_fragment_like<ValTypeSF>(partition_SFA);
}
template <class SFBTensor, class ThrMma>
CUTE_HOST_DEVICE constexpr
auto
partition_fragment_SFB(SFBTensor&& sfbtensor, ThrMma& thread_mma)
{
using ValTypeSF = typename ThrMma::Atom::Traits::ValTypeSF;
auto thr_tensor = make_tensor(static_cast<SFBTensor&&>(sfbtensor).data(), thrfrg_SFB(sfbtensor.layout(),thread_mma));
auto thr_vmnk = thread_mma.thr_vmnk_;
auto thr_vnk = make_coord(get<0>(thr_vmnk), make_coord(get<2>(thr_vmnk), get<3>(thr_vmnk)));
auto partition_SFB = thr_tensor(thr_vnk, make_coord(_, repeat<rank<1,1>(thr_tensor)>(_)));
return make_fragment_like<ValTypeSF>(partition_SFB);
}
template<class TiledMma>
CUTE_HOST_DEVICE constexpr
auto
get_layoutSFA_TV(TiledMma& mma)
{
// (M,K) -> (M,K)
auto tile_shape_mnk = tile_shape(mma);
auto ref_A = make_layout(make_shape(size<0>(tile_shape_mnk), size<2>(tile_shape_mnk)));
auto thr_layout_vmnk = mma.get_thr_layout_vmnk();
// (ThrV,(ThrM,ThrK)) -> (ThrV,(ThrM,ThrN,ThrK))
auto atile = make_tile(_,
make_tile(make_layout(make_shape (size<1>(thr_layout_vmnk), size<2>(thr_layout_vmnk)),
make_stride( Int<1>{} , Int<0>{} )),
_));
// thr_idx -> (ThrV,ThrM,ThrN,ThrK)
auto thridx_2_thrid = right_inverse(thr_layout_vmnk);
// (thr_idx,val) -> (M,K)
return thrfrg_SFA(ref_A, mma).compose(atile, _).compose(thridx_2_thrid, _);
}
template<class TiledMma>
CUTE_HOST_DEVICE constexpr
auto
get_layoutSFB_TV(TiledMma& mma)
{
// (N,K) -> (N,K)
auto tile_shape_mnk = tile_shape(mma);
auto ref_B = make_layout(make_shape(size<1>(tile_shape_mnk), size<2>(tile_shape_mnk)));
auto thr_layout_vmnk = mma.get_thr_layout_vmnk();
// (ThrV,(ThrM,ThrK)) -> (ThrV,(ThrM,ThrN,ThrK))
auto btile = make_tile(_,
make_tile(make_layout(make_shape (size<1>(thr_layout_vmnk), size<2>(thr_layout_vmnk)),
make_stride( Int<0>{} , Int<1>{} )),
_));
// thr_idx -> (ThrV,ThrM,ThrN,ThrK)
auto thridx_2_thrid = right_inverse(thr_layout_vmnk);
// (thr_idx,val) -> (M,K)
return thrfrg_SFB(ref_B, mma).compose(btile, _).compose(thridx_2_thrid, _);
}
template <typename SchedulerParams, typename SharedStorage, typename WorkTileInfo>
CUTLASS_DEVICE void
load(Params const& mainloop_params,
SchedulerParams const& scheduler_params,
MainloopPipelineQ pipeline_q,
MainloopPipeline pipeline_k,
MainloopPipeline pipeline_v,
PipelineStateQ& smem_pipe_write_q,
PipelineState& smem_pipe_write_k,
PipelineState& smem_pipe_write_v,
SharedStorage &shared_storage,
WorkTileInfo work_tile_info,
int& work_idx,
int& tile_count_semaphore
) {
static constexpr int kBlockM = get<0>(TileShape_MNK{});
static constexpr int kBlockN = get<1>(TileShape_MNK{});
auto [m_block, bidh, bidb] = work_tile_info.get_block_coord(scheduler_params);
int n_block_max = get_n_block_max(mainloop_params, m_block);
Tensor sQ = make_tensor(make_smem_ptr(shared_storage.smem_q.begin()), SmemLayoutQ{});
Tensor sK = make_tensor(make_smem_ptr(shared_storage.smem_k.begin()), SmemLayoutK{});
Tensor sVt = make_tensor(make_smem_ptr(shared_storage.smem_v.begin()), SmemLayoutVt{});
Tensor sSFQ = make_tensor(make_smem_ptr(shared_storage.smem_SFQ.begin()), SmemLayoutSFQ{});
Tensor sSFK = make_tensor(make_smem_ptr(shared_storage.smem_SFK.begin()), SmemLayoutSFK{});
Tensor sSFVt = make_tensor(make_smem_ptr(shared_storage.smem_SFV.begin()), SmemLayoutSFVt{});
Tensor sDS = make_tensor(make_smem_ptr(shared_storage.smem_ds.begin()), SmemLayoutDS{});
Tensor mQ = mainloop_params.tma_load_Q.get_tma_tensor(mainloop_params.shape_Q);
Tensor mK = mainloop_params.tma_load_K.get_tma_tensor(mainloop_params.shape_K);
Tensor mVt = mainloop_params.tma_load_Vt.get_tma_tensor(mainloop_params.shape_Vt);
Tensor mDS = mainloop_params.tma_load_DS.get_tma_tensor(shape(mainloop_params.layout_DS));
Tensor mSFQ = mainloop_params.tma_load_SFQ.get_tma_tensor(shape(mainloop_params.layout_SFQ));
Tensor mSFK = mainloop_params.tma_load_SFK.get_tma_tensor(shape(mainloop_params.layout_SFK));
Tensor mSFVt = mainloop_params.tma_load_SFVt.get_tma_tensor(shape(mainloop_params.layout_SFVt));
uint32_t block_rank_in_cluster = cute::block_rank_in_cluster();
constexpr uint32_t cluster_shape_x = get<0>(ClusterShape());
uint2 cluster_local_block_id = {block_rank_in_cluster % cluster_shape_x, block_rank_in_cluster / cluster_shape_x};
Tensor gQ = local_tile(mQ(_, _, bidh, bidb), select<0, 2>(TileShape_MNK{}), make_coord(m_block, _0{})); // (M, K)
Tensor gK = local_tile(mK(_, _, bidh, bidb), select<1, 2>(TileShape_MNK{}), make_coord(_, _0{})); // (N, K, _)
Tensor gVt = local_tile(mVt(_, _, bidh, bidb), make_shape(shape<2>(TileShape_MNK{}), shape<1>(TileShape_MNK{})), make_coord(_0{}, _)); // (N, K, _)
Tensor gDS = [&] {
if constexpr (BlockMean) {
return local_tile(mDS(_, _, bidh, bidb), select<0, 1>(TileShape_MNK{}), make_coord(m_block, _));
} else {
return local_tile(mDS(_, _, bidh, bidb), select<0, 1>(TileShape_MNK{}), make_coord(_0{}, _));
}
}();
Tensor gSFQ = local_tile(mSFQ(_, _, bidh, bidb), select<0, 2>(TileShape_MNK{}), make_coord(m_block, _0{}));
Tensor gSFK = local_tile(mSFK(_, _, bidh, bidb), select<1, 2>(TileShape_MNK{}), make_coord(_, _0{}));
Tensor gSFVt = local_tile(mSFVt(_, _, bidh, bidb), make_shape(shape<2>(TileShape_MNK{}), shape<1>(TileShape_MNK{})), make_coord(_0{}, _));
auto block_tma_q = mainloop_params.tma_load_Q.get_slice(_0{});
Tensor tQgQ = block_tma_q.partition_S(gQ);
Tensor tQsQ = block_tma_q.partition_D(sQ);
auto block_tma_sfq = mainloop_params.tma_load_SFQ.get_slice(_0{});
Tensor tQgSFQ = block_tma_sfq.partition_S(gSFQ);
Tensor tQsSFQ = block_tma_sfq.partition_D(sSFQ);
auto block_tma_k = mainloop_params.tma_load_K.get_slice(cluster_local_block_id.x);
Tensor tKgK = group_modes<0, 3>(block_tma_k.partition_S(gK));
Tensor tKsK = group_modes<0, 3>(block_tma_k.partition_D(sK));
auto block_tma_sfk = mainloop_params.tma_load_SFK.get_slice(cluster_local_block_id.x);
Tensor tKgSFK = group_modes<0, 3>(block_tma_sfk.partition_S(gSFK));
Tensor tKsSFK = group_modes<0, 3>(block_tma_sfk.partition_D(sSFK));
auto block_tma_vt = mainloop_params.tma_load_Vt.get_slice(cluster_local_block_id.x);
Tensor tVgVt = group_modes<0, 3>(block_tma_vt.partition_S(gVt));
Tensor tVsVt = group_modes<0, 3>(block_tma_vt.partition_D(sVt));
auto block_tma_sfvt = mainloop_params.tma_load_SFVt.get_slice(cluster_local_block_id.x);
Tensor tVgSFVt = group_modes<0, 3>(block_tma_sfvt.partition_S(gSFVt));
Tensor tVsSFVt = group_modes<0, 3>(block_tma_sfvt.partition_D(sSFVt));
auto block_tma_ds = mainloop_params.tma_load_DS.get_slice(cluster_local_block_id.x);
Tensor tDSgDS = group_modes<0, 3>(block_tma_ds.partition_S(gDS));
Tensor tDSsDS = group_modes<0, 3>(block_tma_ds.partition_D(sDS));
uint16_t mcast_mask_kv = 0;
int n_block = n_block_max - 1;
int lane_predicate = cute::elect_one_sync();
if (lane_predicate) {
pipeline_q.producer_acquire(smem_pipe_write_q);
copy(mainloop_params.tma_load_Q.with(*pipeline_q.producer_get_barrier(smem_pipe_write_q), 0), tQgQ, tQsQ);
copy(mainloop_params.tma_load_SFQ.with(*pipeline_q.producer_get_barrier(smem_pipe_write_q), 0), tQgSFQ, tQsSFQ);
++smem_pipe_write_q;
pipeline_k.producer_acquire(smem_pipe_write_k);
copy(mainloop_params.tma_load_K.with(*pipeline_k.producer_get_barrier(smem_pipe_write_k), mcast_mask_kv),
tKgK(_, n_block), tKsK(_, smem_pipe_write_k.index()));
copy(mainloop_params.tma_load_SFK.with(*pipeline_k.producer_get_barrier(smem_pipe_write_k), mcast_mask_kv),
tKgSFK(_, n_block), tKsSFK(_, smem_pipe_write_k.index()));
copy(mainloop_params.tma_load_DS.with(*pipeline_k.producer_get_barrier(smem_pipe_write_k), mcast_mask_kv),
tDSgDS(_, n_block), tDSsDS(_, smem_pipe_write_k.index()));
++smem_pipe_write_k;
pipeline_v.producer_acquire(smem_pipe_write_v);
copy(mainloop_params.tma_load_Vt.with(*pipeline_v.producer_get_barrier(smem_pipe_write_v), mcast_mask_kv),
tVgVt(_, n_block), tVsVt(_, smem_pipe_write_v.index()));
copy(mainloop_params.tma_load_SFVt.with(*pipeline_v.producer_get_barrier(smem_pipe_write_v), mcast_mask_kv),
tVgSFVt(_, n_block), tVsSFVt(_, smem_pipe_write_v.index()));
++smem_pipe_write_v;
}
n_block--;
if (lane_predicate) {
// CUTLASS_PRAGMA_NO_UNROLL
#pragma unroll 2
for (; n_block >= 0; --n_block) {
pipeline_k.producer_acquire(smem_pipe_write_k);
copy(mainloop_params.tma_load_K.with(*pipeline_k.producer_get_barrier(smem_pipe_write_k), mcast_mask_kv),
tKgK(_, n_block), tKsK(_, smem_pipe_write_k.index()));
copy(mainloop_params.tma_load_SFK.with(*pipeline_k.producer_get_barrier(smem_pipe_write_k), mcast_mask_kv),
tKgSFK(_, n_block), tKsSFK(_, smem_pipe_write_k.index()));
copy(mainloop_params.tma_load_DS.with(*pipeline_k.producer_get_barrier(smem_pipe_write_k), mcast_mask_kv),
tDSgDS(_, n_block), tDSsDS(_, smem_pipe_write_k.index()));
++smem_pipe_write_k;
pipeline_v.producer_acquire(smem_pipe_write_v);
copy(mainloop_params.tma_load_Vt.with(*pipeline_v.producer_get_barrier(smem_pipe_write_v), mcast_mask_kv),
tVgVt(_, n_block), tVsVt(_, smem_pipe_write_v.index()));
copy(mainloop_params.tma_load_SFVt.with(*pipeline_v.producer_get_barrier(smem_pipe_write_v), mcast_mask_kv),
tVgSFVt(_, n_block), tVsSFVt(_, smem_pipe_write_v.index()));
++smem_pipe_write_v;
}
}
++work_idx;
}
/// Perform a Producer Epilogue to prevent early exit of blocks in a Cluster
CUTLASS_DEVICE void
load_tail(MainloopPipelineQ pipeline_q,
MainloopPipeline pipeline_k,
MainloopPipeline pipeline_v,
PipelineStateQ& smem_pipe_write_q,
PipelineState& smem_pipe_write_k,
PipelineState& smem_pipe_write_v) {
int lane_predicate = cute::elect_one_sync();
// Issue the epilogue waits
if (lane_predicate) {
pipeline_q.producer_tail(smem_pipe_write_q);
pipeline_k.producer_tail(smem_pipe_write_k);
pipeline_v.producer_tail(smem_pipe_write_v);
}
}
template <typename SharedStorage, typename FrgTensorO, typename SoftmaxFused>
CUTLASS_DEVICE void
mma(Params const& mainloop_params,
MainloopPipelineQ pipeline_q,
MainloopPipeline pipeline_k,
MainloopPipeline pipeline_v,
PipelineStateQ& smem_pipe_read_q,
PipelineState& smem_pipe_read_k,
PipelineState& smem_pipe_read_v,
FrgTensorO& tOrO_store,
SoftmaxFused& softmax_fused,
int n_block_count,
int thread_idx,
int work_idx,
int m_block,
SharedStorage& shared_storage
) {
static_assert(is_rmem<FrgTensorO>::value, "O tensor must be rmem resident.");
static constexpr int kBlockM = get<0>(TileShape_MNK{});
static constexpr int kBlockN = get<1>(TileShape_MNK{});
static constexpr int kBlockK = get<2>(TileShape_MNK{});
Tensor sQ = make_tensor(make_smem_ptr(shared_storage.smem_q.begin()), SmemLayoutQ{});
Tensor sK = make_tensor(make_smem_ptr(shared_storage.smem_k.begin()), SmemLayoutK{});
Tensor sVt = make_tensor(make_smem_ptr(shared_storage.smem_v.begin()), SmemLayoutVt{});
Tensor sDS = make_tensor(make_smem_ptr(shared_storage.smem_ds.begin()), SmemLayoutDS{});
Tensor sSFQ = make_tensor(make_smem_ptr(shared_storage.smem_SFQ.begin()), SmemLayoutSFQ{});
Tensor sSFK = make_tensor(make_smem_ptr(shared_storage.smem_SFK.begin()), SmemLayoutSFK{});
Tensor sSFVt = make_tensor(make_smem_ptr(shared_storage.smem_SFV.begin()), SmemLayoutSFVt{});
Tensor cQ = make_identity_tensor(make_shape(size<0>(sQ), size<1>(sQ)));
Tensor cKV = make_identity_tensor(make_shape(size<0>(sK), size<1>(sK)));
TiledMmaQK tiled_mma_qk;
TiledMmaPV tiled_mma_pv;
auto thread_mma_qk = tiled_mma_qk.get_thread_slice(thread_idx);
auto thread_mma_pv = tiled_mma_pv.get_thread_slice(thread_idx);
Tensor tSrQ = thread_mma_qk.partition_fragment_A(sQ);
Tensor tSrK = thread_mma_qk.partition_fragment_B(sK(_,_,Int<0>{}));
Tensor tOrVt = thread_mma_pv.partition_fragment_B(sVt(_,_,Int<0>{}));
Tensor tOrP = make_tensor_like<Element>(LayoutP{});
Tensor tSrSFQ = partition_fragment_SFA(sSFQ, thread_mma_qk);
Tensor tSrSFK = partition_fragment_SFB(sSFK(_,_,Int<0>{}), thread_mma_qk);
Tensor tOrSFVt = partition_fragment_SFB(sSFVt(_,_,Int<0>{}), thread_mma_pv);
Tensor tOrSFP = make_tensor<ElementSF>(LayoutSFP{});
Tensor tOrSFP_flt = filter_zeros(tOrSFP);
Tensor tSrDS = make_tensor<float>(make_shape(_8{}, _4{}), make_stride(_1{}, _8{}));
// copy qk and sf from smem to rmem
auto smem_tiled_copy_Q = make_tiled_copy_A(SmemCopyAtomQ{}, tiled_mma_qk);
auto smem_thr_copy_Q = smem_tiled_copy_Q.get_thread_slice(thread_idx);
Tensor tSsQ = smem_thr_copy_Q.partition_S(as_position_independent_swizzle_tensor(sQ));
Tensor tSrQ_copy_view = smem_thr_copy_Q.retile_D(tSrQ);
auto smem_tiled_copy_K = make_tiled_copy_B(SmemCopyAtomKV{}, tiled_mma_qk);
auto smem_thr_copy_K = smem_tiled_copy_K.get_thread_slice(thread_idx);
Tensor tSsK = smem_thr_copy_K.partition_S(as_position_independent_swizzle_tensor(sK));
Tensor tSrK_copy_view = smem_thr_copy_K.retile_D(tSrK);
auto smem_tiled_copy_V = make_tiled_copy_B(SmemCopyAtomKV{}, tiled_mma_pv);
auto smem_thr_copy_V = smem_tiled_copy_V.get_thread_slice(thread_idx);
Tensor tOsVt = smem_thr_copy_V.partition_S(as_position_independent_swizzle_tensor(sVt));
Tensor tOrVt_copy_view = smem_thr_copy_V.retile_D(tOrVt);
auto tile_shape_mnk = tile_shape(tiled_mma_qk);
auto smem_tiled_copy_SFQ = make_tiled_copy_impl(SmemCopyAtomSF{},
get_layoutSFA_TV(tiled_mma_qk),
make_shape(size<0>(tile_shape_mnk), size<2>(tile_shape_mnk))
);
auto smem_thr_copy_SFQ = smem_tiled_copy_SFQ.get_thread_slice(thread_idx);
Tensor tSsSFQ = smem_thr_copy_SFQ.partition_S(as_position_independent_swizzle_tensor(sSFQ));
Tensor tSrSFQ_copy_view = smem_thr_copy_SFQ.retile_D(tSrSFQ);
auto smem_tiled_copy_SFK = make_tiled_copy_impl(SmemCopyAtomSF{},
get_layoutSFB_TV(tiled_mma_qk),
make_shape(size<1>(tile_shape_mnk), size<2>(tile_shape_mnk))
);
auto smem_thr_copy_SFK = smem_tiled_copy_SFK.get_thread_slice(thread_idx);
Tensor tSsSFK = smem_thr_copy_SFK.partition_S(as_position_independent_swizzle_tensor(sSFK));
Tensor tSrSFK_copy_view = smem_thr_copy_SFK.retile_D(tSrSFK);
auto smem_tiled_copy_SFV = make_tiled_copy_impl(SmemCopyAtomSF{},
get_layoutSFB_TV(tiled_mma_pv),
make_shape(size<1>(tile_shape_mnk), size<2>(tile_shape_mnk))
);
auto smem_thr_copy_SFV = smem_tiled_copy_SFV.get_thread_slice(thread_idx);
Tensor tOsSFVt = smem_thr_copy_SFV.partition_S(as_position_independent_swizzle_tensor(sSFVt));
Tensor tOrSFVt_copy_view = smem_thr_copy_SFV.retile_D(tOrSFVt);
auto consumer_wait = [](auto& pipeline, auto& smem_pipe_read) {
auto barrier_token = pipeline.consumer_try_wait(smem_pipe_read);
pipeline.consumer_wait(smem_pipe_read, barrier_token);
};
int const seqlen_q = get<0>(mainloop_params.shape_Q);
int const seqlen_k = get<0>(mainloop_params.shape_K);
int const unpadded_seqlen_k = get<0>(mainloop_params.unpadded_shape_K);
int n_block = n_block_count - 1;
auto copy_k_block = [&](auto block_id) {
auto tSsK_stage = tSsK(_, _, _, smem_pipe_read_k.index());
auto tSsSFK_stage = tSsSFK(_, _, _, smem_pipe_read_k.index());
copy(smem_tiled_copy_K, tSsK_stage(_, _, block_id), tSrK_copy_view(_, _, block_id));
copy(smem_tiled_copy_SFK, tSsSFK_stage(_, _, block_id), tSrSFK_copy_view(_, _, block_id));
};
auto copy_v_block = [&](auto block_id) {
auto tOsVt_stage = tOsVt(_, _, _, smem_pipe_read_v.index());
auto tOsSFVt_stage = tOsSFVt(_, _, _, smem_pipe_read_v.index());
copy(smem_tiled_copy_V, tOsVt_stage(_, _, block_id), tOrVt_copy_view(_, _, block_id));
copy(smem_tiled_copy_SFV, tOsSFVt_stage(_, _, block_id), tOrSFVt_copy_view(_, _, block_id));
};
// auto gemm_qk = [&](auto block_id) {
// cute::gemm(tiled_mma_qk, make_zip_tensor(tSrQ(_, _, block_id), tSrSFQ(_, _, block_id)), make_zip_tensor(tSrK(_, _, block_id), tSrSFK(_, _, block_id)), tSrS);
// };
// auto gemm_pv = [&](auto block_id) {
// cute::gemm(tiled_mma_pv, make_zip_tensor(tOrP(_, _, block_id), tOrSFP(_, _, block_id)), make_zip_tensor(tOrVt(_, _, block_id), tOrSFVt(_, _, block_id)), tOrO);
// };
auto add_delta_s = [&](auto& acc) {
auto tSsDS_stage = recast<float4>(sDS(_, _, smem_pipe_read_k.index()));
auto acc_float4 = recast<float4>(acc);
int quad_id = (threadIdx.x % 4) * 2;
for (int i = 0; i < 4; i++) {
auto num = quad_id + i * 8;
float4 delta_s_0 = tSsDS_stage(make_coord(_0{}, _0{}), make_coord(num, _0{}));
float4 delta_s_1 = tSsDS_stage(make_coord(_0{}, _0{}), make_coord(num + 1, _0{}));
acc_float4(make_coord(make_coord(_0{}, _0{}), _0{}), _0{}, i) = delta_s_0;
acc_float4(make_coord(make_coord(_0{}, _0{}), _1{}), _0{}, i) = delta_s_0;
acc_float4(make_coord(make_coord(_0{}, _1{}), _0{}), _0{}, i) = delta_s_1;
acc_float4(make_coord(make_coord(_0{}, _1{}), _1{}), _0{}, i) = delta_s_1;
}
};
consumer_wait(pipeline_q, smem_pipe_read_q);
copy(smem_tiled_copy_Q, tSsQ, tSrQ_copy_view);
copy(smem_tiled_copy_SFQ, tSsSFQ, tSrSFQ_copy_view);
pipeline_q.consumer_release(smem_pipe_read_q);
++smem_pipe_read_q;
Tensor tSrS = partition_fragment_C(tiled_mma_qk, select<0, 1>(TileShape_MNK{}));
Tensor tSrS_converion_view = make_tensor(tSrS.data(), flash::convert_to_conversion_layout(tSrS.layout()));
Tensor AbsMaxP = make_tensor_like<float>(
make_layout(shape(group<1, 4>(flatten(tSrS_converion_view.layout()(make_coord(_0{}, _), _, _)))))
);
consumer_wait(pipeline_k, smem_pipe_read_k);
copy_k_block(_0{});
add_delta_s(tSrS);
CUTLASS_PRAGMA_UNROLL
for (int k_block = 0; k_block < size<2>(tSrQ); ++k_block) {
cute::gemm(tiled_mma_qk, make_zip_tensor(tSrQ(_, _, k_block), tSrSFQ(_, _, k_block)),
make_zip_tensor(tSrK(_, _, k_block), tSrSFK(_, _, k_block)), tSrS);
if (k_block < size<2>(tSrQ) - 1) {
copy_k_block(k_block + 1);
} else {
pipeline_k.consumer_release(smem_pipe_read_k);
++smem_pipe_read_k;
}
}
auto col_limit_causal = [&](int row, int n_block) {
return row + 1 + seqlen_k - n_block * kBlockN - seqlen_q + m_block * kBlockM;
};
{
Tensor cS = cute::make_identity_tensor(select<0, 1>(TileShape_MNK{}));
Tensor tScS = thread_mma_qk.partition_C(cS);
CUTLASS_PRAGMA_UNROLL
for (int i = 0; i < size(tSrS); ++i) {
if constexpr (!Is_causal) { // Just masking based on col
if (int(get<1>(tScS(i))) >= int(unpadded_seqlen_k - n_block * kBlockN)) { tSrS(i) = -INFINITY; }
} else {
if (int(get<1>(tScS(i))) >= std::min(seqlen_k - n_block * kBlockN,
col_limit_causal(int(get<0>(tScS(i))), n_block))) {
tSrS(i) = -INFINITY;
}
}
}
}
auto quantize = [&](auto mma_k, auto acc_conversion_view) {
Tensor AbsMaxP_stagek = AbsMaxP(_, make_coord(_, _, mma_k));
Tensor acc_conversion_stagek = acc_conversion_view(_, _, mma_k);
Tensor SFP = make_tensor_like<cutlass::float_ue4m3_t>(AbsMaxP_stagek.layout());
Tensor SFP_uint32_view = recast<uint32_t>(SFP);
CUTLASS_PRAGMA_UNROLL
for (int i = 0; i < size(AbsMaxP_stagek); i += 4) {
uint32_t& tmp = SFP_uint32_view(i / 4);
flash::packed_float_to_ue4m3(
AbsMaxP_stagek(i),
AbsMaxP_stagek(i + 1),
AbsMaxP_stagek(i + 2),
AbsMaxP_stagek(i + 3),
tmp
);
}
int const quad_id = threadIdx.x & 3;
uint32_t MASK = (0xFF00FF) << ((quad_id & 1) * 8);
Tensor tOrSFP_uint32_view = recast<uint32_t>(tOrSFP(_, _, mma_k));
Tensor tOrP_uint32_view = recast<uint32_t>(tOrP(_, _, mma_k));
CUTLASS_PRAGMA_UNROLL
for (int mma_m = 0; mma_m < size<1>(tOrP); ++mma_m) {
CUTLASS_PRAGMA_UNROLL
for (int i = 0; i < 4; ++i) {
flash::packed_float_to_e2m1(
acc_conversion_stagek(make_coord(_0{}, i), mma_m),
acc_conversion_stagek(make_coord(_1{}, i), mma_m),
acc_conversion_stagek(make_coord(_2{}, i), mma_m),
acc_conversion_stagek(make_coord(_3{}, i), mma_m),
acc_conversion_stagek(make_coord(_4{}, i), mma_m),
acc_conversion_stagek(make_coord(_5{}, i), mma_m),
acc_conversion_stagek(make_coord(_6{}, i), mma_m),
acc_conversion_stagek(make_coord(_7{}, i), mma_m),
tOrP_uint32_view(i, mma_m)
);
}
uint32_t local_sfp = SFP_uint32_view(_0{}, _0{}, mma_m);
uint32_t peer_sfp = __shfl_xor_sync(int32_t(-1), local_sfp, 2);
if ((quad_id & 1) == 0) {
uint32_t sfp = (local_sfp & MASK) | ((peer_sfp & MASK) << 8);
tOrSFP_uint32_view(_0{}, mma_m) = sfp;
} else {
uint32_t sfp = (peer_sfp & MASK) | ((local_sfp & MASK) >> 8);
tOrSFP_uint32_view(_0{}, mma_m) = sfp;
}
}
};
softmax_fused.template online_softmax_with_quant</*Is_first=*/true>(tSrS, AbsMaxP, mainloop_params.softmax_scale_log2);
consumer_wait(pipeline_v, smem_pipe_read_v);
copy_v_block(_0{});
quantize(_0{}, tSrS_converion_view);
CUTLASS_PRAGMA_UNROLL
for (int v_block = 0; v_block < size<2>(tOrP); ++v_block) {
cute::gemm(tiled_mma_pv, make_zip_tensor(tOrP(_, _, v_block), tOrSFP(_, _, v_block)),
make_zip_tensor(tOrVt(_, _, v_block), tOrSFVt(_, _, v_block)), tOrO_store);
if (v_block < size<2>(tOrP) - 1) {
copy_v_block(v_block + 1);
quantize(v_block + 1, tSrS_converion_view);
} else {
pipeline_v.consumer_release(smem_pipe_read_v);
++smem_pipe_read_v;
}
}
n_block--;
constexpr int n_masking_steps = !Is_causal ? 1 : cute::ceil_div(kBlockM, kBlockN) + 1;
// // Only go through these if Is_causal, since n_masking_steps = 1 when !Is_causal
CUTLASS_PRAGMA_UNROLL
for (int masking_step = 0; masking_step < n_masking_steps - 1 && n_block >= 0; ++masking_step, --n_block) {
Tensor tSrS = partition_fragment_C(tiled_mma_qk, select<0, 1>(TileShape_MNK{}));
Tensor tSrS_converion_view = make_tensor(tSrS.data(), flash::convert_to_conversion_layout(tSrS.layout()));
consumer_wait(pipeline_k, smem_pipe_read_k);
copy_k_block(_0{});
add_delta_s(tSrS);
CUTLASS_PRAGMA_UNROLL
for (int k_block = 0; k_block < size<2>(tSrQ); ++k_block) {
cute::gemm(tiled_mma_qk, make_zip_tensor(tSrQ(_, _, k_block), tSrSFQ(_, _, k_block)),
make_zip_tensor(tSrK(_, _, k_block), tSrSFK(_, _, k_block)), tSrS);
if (k_block < size<2>(tSrQ) - 1) {
copy_k_block(k_block + 1);
}
}
pipeline_k.consumer_release(smem_pipe_read_k); // release K
++smem_pipe_read_k;
Tensor cS = cute::make_identity_tensor(select<0, 1>(TileShape_MNK{}));
Tensor tScS = thread_mma_qk.partition_C(cS);
#pragma unroll
for (int i = 0; i < size(tSrS); ++i) {
if (int(get<1>(tScS(i))) >= col_limit_causal(int(get<0>(tScS(i))), n_block - 1)) {
tSrS(i) = -INFINITY;
}
}
softmax_fused.template online_softmax_with_quant</*Is_first=*/false>(tSrS, AbsMaxP, mainloop_params.softmax_scale_log2);
Tensor tOrO = make_fragment_like(tOrO_store);
consumer_wait(pipeline_v, smem_pipe_read_v);
copy_v_block(_0{});
quantize(_0{}, tSrS_converion_view);
CUTLASS_PRAGMA_UNROLL
for (int v_block = 0; v_block < size<2>(tOrP); ++v_block) {
cute::gemm(tiled_mma_pv, make_zip_tensor(tOrP(_, _, v_block), tOrSFP(_, _, v_block)),
make_zip_tensor(tOrVt(_, _, v_block), tOrSFVt(_, _, v_block)), tOrO);
if (v_block < size<2>(tOrP) - 1) {
copy_v_block(v_block + 1);
quantize(v_block + 1, tSrS_converion_view);
}
}
pipeline_v.consumer_release(smem_pipe_read_v);
++smem_pipe_read_v;
if (masking_step > 0) { softmax_fused.rescale_o(tOrO_store, tOrO); }
}
#pragma unroll 1
for (; n_block >= 0; --n_block) {
Tensor tSrS = partition_fragment_C(tiled_mma_qk, select<0, 1>(TileShape_MNK{}));
Tensor tSrS_converion_view = make_tensor(tSrS.data(), flash::convert_to_conversion_layout(tSrS.layout()));
consumer_wait(pipeline_k, smem_pipe_read_k);
copy_k_block(_0{});
add_delta_s(tSrS);
CUTLASS_PRAGMA_UNROLL
for (int k_block = 0; k_block < size<2>(tSrQ); ++k_block) {
cute::gemm(tiled_mma_qk, make_zip_tensor(tSrQ(_, _, k_block), tSrSFQ(_, _, k_block)),
make_zip_tensor(tSrK(_, _, k_block), tSrSFK(_, _, k_block)), tSrS);
if (k_block < size<2>(tSrQ) - 1) {
copy_k_block(k_block + 1);
} else {
pipeline_k.consumer_release(smem_pipe_read_k);
++smem_pipe_read_k;
}
}
softmax_fused.template online_softmax_with_quant</*Is_first=*/false>(tSrS, AbsMaxP, mainloop_params.softmax_scale_log2);
Tensor tOrO = make_fragment_like(tOrO_store);
consumer_wait(pipeline_v, smem_pipe_read_v);
copy_v_block(_0{});
quantize(_0{}, tSrS_converion_view);
CUTLASS_PRAGMA_UNROLL
for (int v_block = 0; v_block < size<2>(tOrP); ++v_block) {
cute::gemm(tiled_mma_pv, make_zip_tensor(tOrP(_, _, v_block), tOrSFP(_, _, v_block)),
make_zip_tensor(tOrVt(_, _, v_block), tOrSFVt(_, _, v_block)), tOrO);
if (v_block < size<2>(tOrP) - 1) {
copy_v_block(v_block + 1);
quantize(v_block + 1, tSrS_converion_view);
} else {
pipeline_v.consumer_release(smem_pipe_read_v);
++smem_pipe_read_v;
}
}
softmax_fused.rescale_o(tOrO_store, tOrO);
}
softmax_fused.finalize(tOrO_store);
return;
}
};
} // namespace flash
@@ -0,0 +1,119 @@
/*
* Copyright (c) 2025 by SageAttention team.
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
#pragma once
#include "cutlass/arch/barrier.h"
#include "cutlass/pipeline/sm90_pipeline.hpp"
namespace flash {
enum class FP4NamedBarriers {
QueryEmpty = 1,
WarpSpecializedConsumer = 2,
WarpSpecializedPingPongConsumer1 = 3,
WarpSpecializedPingPongConsumer2 = 4,
ProducerEnd = 5,
ConsumerEnd = 6,
EpilogueBarrier = 7
};
template<int SequenceDepth, int SequenceLength>
struct OrderedSequenceBarrierVarGroupSizeSharedStorage {
using Barrier = cutlass::arch::ClusterBarrier;
Barrier barrier_[SequenceDepth][SequenceLength];
};
template<int SequenceDepth_, int SequenceLength_>
class OrderedSequenceBarrierVarGroupSize {
public:
static constexpr int SequenceDepth = SequenceDepth_;
static constexpr int SequenceLength = SequenceLength_;
using Barrier = cutlass::arch::ClusterBarrier;
using SharedStorage = flash::OrderedSequenceBarrierVarGroupSizeSharedStorage<SequenceDepth, SequenceLength>;
struct Params {
uint32_t group_id;
uint32_t* group_size_list;
};
private :
// In future this Params object can be replaced easily with a CG object
Params params_;
Barrier *barrier_ptr_;
cutlass::PipelineState<SequenceDepth> stage_;
static constexpr int Depth = SequenceDepth;
static constexpr int Length = SequenceLength;
public:
OrderedSequenceBarrierVarGroupSize() = delete;
OrderedSequenceBarrierVarGroupSize(const OrderedSequenceBarrierVarGroupSize&) = delete;
OrderedSequenceBarrierVarGroupSize(OrderedSequenceBarrierVarGroupSize&&) = delete;
OrderedSequenceBarrierVarGroupSize& operator=(const OrderedSequenceBarrierVarGroupSize&) = delete;
OrderedSequenceBarrierVarGroupSize& operator=(OrderedSequenceBarrierVarGroupSize&&) = delete;
~OrderedSequenceBarrierVarGroupSize() = default;
CUTLASS_DEVICE
OrderedSequenceBarrierVarGroupSize(SharedStorage& storage, Params const& params) :
params_(params),
barrier_ptr_(&storage.barrier_[0][0]),
// Group 0 - starts with an opposite phase
stage_({0, params.group_id == 0, 0}) {
int warp_idx = cutlass::canonical_warp_idx_sync();
int lane_predicate = cute::elect_one_sync();
// Barrier FULL, EMPTY init
// Init is done only by the one elected thread of the block
if (warp_idx == 0 && lane_predicate) {
for (int d = 0; d < Depth; ++d) {
for (int l = 0; l < Length; ++l) {
barrier_ptr_[d * Length + l].init(*(params.group_size_list + l));
}
}
}
cutlass::arch::fence_barrier_init();
}
// Wait on a stage to be unlocked
CUTLASS_DEVICE
void wait() {
get_barrier_for_current_stage(params_.group_id).wait(stage_.phase());
}
// Signal completion of Stage and move to the next stage
// (group_id) signals to (group_id+1)
CUTLASS_DEVICE
void arrive() {
int signalling_id = (params_.group_id + 1) % Length;
get_barrier_for_current_stage(signalling_id).arrive();
++stage_;
}
CUTLASS_DEVICE
void advance() {
++stage_;
}
private:
CUTLASS_DEVICE
Barrier& get_barrier_for_current_stage(int group_id) {
return barrier_ptr_[stage_.index() * Length + group_id];
}
};
} // flash
@@ -0,0 +1,180 @@
// Modified from the original SageAttention3 code
/*
* Copyright (c) 2025 by SageAttention team.
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
#pragma once
#include <cuda.h>
#include <vector>
#ifdef OLD_GENERATOR_PATH
#include <ATen/CUDAGeneratorImpl.h>
#else
#include <ATen/cuda/CUDAGeneratorImpl.h>
#endif
#include <ATen/cuda/CUDAGraphsUtils.cuh> // For at::cuda::philox::unpack
#include "cutlass/fast_math.h" // For cutlass::FastDivmod
////////////////////////////////////////////////////////////////////////////////////////////////////
struct Qkv_params {
using index_t = int64_t;
// The QKV matrices.
void *__restrict__ q_ptr;
void *__restrict__ k_ptr;
void *__restrict__ v_ptr;
void *__restrict__ delta_s_ptr;
// The QKV scale factor matrices.
void *__restrict__ sfq_ptr;
void *__restrict__ sfk_ptr;
void *__restrict__ sfv_ptr;
// The stride between rows of the Q, K and V matrices.
index_t q_batch_stride;
index_t k_batch_stride;
index_t v_batch_stride;
index_t q_row_stride;
index_t k_row_stride;
index_t v_row_stride;
index_t q_head_stride;
index_t k_head_stride;
index_t v_head_stride;
index_t ds_batch_stride;
index_t ds_row_stride;
index_t ds_head_stride;
// The stride of the Q, K and V scale factor matrices.
index_t sfq_batch_stride;
index_t sfk_batch_stride;
index_t sfv_batch_stride;
index_t sfq_row_stride;
index_t sfk_row_stride;
index_t sfv_row_stride;
index_t sfq_head_stride;
index_t sfk_head_stride;
index_t sfv_head_stride;
// The number of heads.
int h, h_k;
// In the case of multi-query and grouped-query attention (MQA/GQA), nheads_k could be
// different from nheads (query).
int h_h_k_ratio; // precompute h / h_k,
};
////////////////////////////////////////////////////////////////////////////////////////////////////
struct Flash_fwd_params : public Qkv_params {
// The O matrix (output).
void * __restrict__ o_ptr;
void * __restrict__ oaccum_ptr;
void * __restrict__ s_ptr;
// The stride between rows of O.
index_t o_batch_stride;
index_t o_row_stride;
index_t o_head_stride;
// The pointer to the P matrix.
void * __restrict__ p_ptr;
// The pointer to the softmax sum.
void * __restrict__ softmax_lse_ptr;
void * __restrict__ softmax_lseaccum_ptr;
// The dimensions.
int b, seqlen_q, seqlen_k, seqlen_knew, d, seqlen_q_rounded, seqlen_k_rounded, d_rounded, rotary_dim, unpadded_seqlen_k;
cutlass::FastDivmod head_divmod, m_block_divmod;
int total_blocks;
int seqlen_s;
// The scaling factors for the kernel.
float scale_softmax;
float scale_softmax_log2;
uint32_t scale_softmax_log2_half2;
// array of length b+1 holding starting offset of each sequence.
int * __restrict__ cu_seqlens_q;
int * __restrict__ cu_seqlens_k;
// If provided, the actual length of each k sequence.
int * __restrict__ seqused_k;
int *__restrict__ blockmask;
// The K_new and V_new matrices.
void * __restrict__ knew_ptr;
void * __restrict__ vnew_ptr;
// The stride between rows of the Q, K and V matrices.
index_t knew_batch_stride;
index_t vnew_batch_stride;
index_t knew_row_stride;
index_t vnew_row_stride;
index_t knew_head_stride;
index_t vnew_head_stride;
// The cos and sin matrices for rotary embedding.
void * __restrict__ rotary_cos_ptr;
void * __restrict__ rotary_sin_ptr;
// The indices to index into the KV cache.
int * __restrict__ cache_batch_idx;
// Paged KV cache
int * __restrict__ block_table;
index_t block_table_batch_stride;
int page_block_size;
// The dropout probability (probability of keeping an activation).
float p_dropout;
// uint32_t p_dropout_in_uint;
// uint16_t p_dropout_in_uint16_t;
uint8_t p_dropout_in_uint8_t;
// Scale factor of 1 / (1 - p_dropout).
float rp_dropout;
float scale_softmax_rp_dropout;
// Local window size
int window_size_left, window_size_right;
// Random state.
at::PhiloxCudaState philox_args;
// Pointer to the RNG seed (idx 0) and offset (idx 1).
uint64_t * rng_state;
bool is_bf16;
bool is_e4m3;
bool is_causal;
bool per_block_mean;
bool single_level_p_quant; // If true, use single-level 1x16 block scale quantization for P (like V), instead of two-level quantization
// If is_seqlens_k_cumulative, then seqlen_k is cu_seqlens_k[bidb + 1] - cu_seqlens_k[bidb].
// Otherwise it's cu_seqlens_k[bidb], i.e., we use cu_seqlens_k to store the sequence lengths of K.
bool is_seqlens_k_cumulative;
bool is_rotary_interleaved;
int num_splits; // For split-KV version
void * __restrict__ alibi_slopes_ptr;
index_t alibi_slopes_batch_stride;
int * __restrict__ tile_count_semaphore;
};
////////////////////////////////////////////////////////////////////////////////////////////////////
@@ -0,0 +1,190 @@
// Modified from the original SageAttention3 code
/*
* Copyright (c) 2025 by SageAttention team.
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
#pragma once
#include <cmath>
#include "cute/tensor.hpp"
#include "cutlass/numeric_types.h"
#include "utils.h"
namespace flash {
using namespace cute;
template <int Rows>
struct SoftmaxFused{
using TensorT = decltype(make_fragment_like<float>(Shape<Int<Rows>>{}));
TensorT row_sum, row_max, scores_scale;
static constexpr float fp8_scalexfp4_scale = 1.f / (448 * 6);
static constexpr float fp8_scalexfp4_scale_log2 = -11.392317422778762f; //log2f(fp8_scalexfp4_scale)
static constexpr float fp4_scale_log2 = -2.584962500721156f; // log2f(fp4_scale)
static constexpr int RowReductionThr = 4;
// If true, use single-level quantization: s_P2, P̂_2 = φ(P̃) directly (standard per-block FP4 quantization like V)
// If false (default), use two-level quantization: s_P1 = rowmax(P̃)/(448×6), then s_P2, P̂_2 = φ(P̃/s_P1)
bool single_level_p_quant;
CUTLASS_DEVICE SoftmaxFused(bool single_level = false) : single_level_p_quant(single_level) {};
template<bool FirstTile, bool InfCheck = false, typename TensorAcc, typename TensorMax>
CUTLASS_DEVICE auto online_softmax_with_quant(
TensorAcc& acc,
TensorMax& AbsMaxP,
const float softmax_scale_log2
) {
Tensor acc_reduction_view = make_tensor(acc.data(), flash::convert_to_reduction_layout(acc.layout()));
Tensor acc_conversion_view = make_tensor(acc.data(), flash::convert_to_conversion_layout(acc.layout()));
Tensor acc_conversion_flatten = group_modes<1, 5>(group_modes<0, 2>(flatten(acc_conversion_view)));
if constexpr (FirstTile) {
fill(row_max, -INFINITY);
clear(row_sum);
fill(scores_scale, 1.f);
CUTLASS_PRAGMA_UNROLL
for (int mi = 0; mi < size<0>(acc_reduction_view); mi++) {
CUTLASS_PRAGMA_UNROLL
for (int ni = 0; ni < size<1, 1>(acc_reduction_view); ni++) {
CUTLASS_PRAGMA_UNROLL
for (int ei = 0; ei < size<1, 0>(acc_reduction_view); ei++) {
AbsMaxP(mi, ni) = fmaxf(AbsMaxP(mi, ni), acc_reduction_view(mi, make_coord(ei, ni)));
}
float max_recv = __shfl_xor_sync(int32_t(-1), AbsMaxP(mi, ni), 1); // exchange max with neighbour thread of 8 elements
AbsMaxP(mi, ni) = fmaxf(AbsMaxP(mi, ni), max_recv);
row_max(mi) = fmaxf(row_max(mi), AbsMaxP(mi, ni));
}
float max_recv = __shfl_xor_sync(int32_t(-1), row_max(mi), 2); // exchange max in a quad in a row
row_max(mi) = fmaxf(row_max(mi), max_recv);
// Two-level P quantization (default): s_P1 = rowmax(P̃)/(448×6), then s_P2,P̂_2 = φ(P̃/s_P1)
// - Pre-scales P to [0, 448×6] range before φ, output scaled by s_P1
// Single-level P quantization: s_P2, P̂_2 = φ(P̃) directly (like V quantization)
// - No s_P1, just standard per-block FP4 quantization φ
const float s_P1_offset = single_level_p_quant ? 0.f : fp8_scalexfp4_scale_log2;
const float max_scaled = InfCheck
? (row_max(mi) == -INFINITY ? 0.f : (row_max(mi) * softmax_scale_log2 + s_P1_offset))
: (row_max(mi) * softmax_scale_log2 + s_P1_offset);
CUTLASS_PRAGMA_UNROLL
for (int ni = 0; ni < size<1>(acc_reduction_view); ni++) {
acc_reduction_view(mi, ni) = flash::ptx_exp2(acc_reduction_view(mi, ni) * softmax_scale_log2 - max_scaled);
}
// s_P2 = max(P_block)/6 — per-block scale factor from φ function (same formula for both modes)
// The difference is in max_scaled: two-level includes 448×6 pre-scaling, single-level doesn't
CUTLASS_PRAGMA_UNROLL
for (int sfi = 0; sfi < size<1>(AbsMaxP); sfi++) {
AbsMaxP(mi, sfi) = flash::ptx_exp2(AbsMaxP(mi, sfi) * softmax_scale_log2 - max_scaled + fp4_scale_log2);
}
}
CUTLASS_PRAGMA_UNROLL
for (int mi = 0; mi < size<0>(acc_reduction_view); mi++) {
CUTLASS_PRAGMA_UNROLL
for (int ni = 0; ni < size<1>(acc_reduction_view); ni++) {
row_sum(mi) += acc_reduction_view(mi, ni);
}
}
}
else {
Tensor scores_max_prev = make_fragment_like(row_max);
cute::copy(row_max, scores_max_prev);
CUTLASS_PRAGMA_UNROLL
for (int mi = 0; mi < size<0>(acc_reduction_view); mi++) {
CUTLASS_PRAGMA_UNROLL
for (int ni = 0; ni < size<1, 1>(acc_reduction_view); ni++) {
float local_max = -INFINITY;
CUTLASS_PRAGMA_UNROLL
for (int ei = 0; ei < size<1, 0>(acc_reduction_view); ei++) {
local_max = fmaxf(local_max, acc_reduction_view(mi, make_coord(ei, ni)));
}
float max_recv = __shfl_xor_sync(int32_t(-1), local_max, 1); // exchange max with neighbour thread of 8 elements
AbsMaxP(mi, ni) = fmaxf(local_max, max_recv);
row_max(mi) = fmaxf(row_max(mi), AbsMaxP(mi, ni));
}
float max_recv = __shfl_xor_sync(int32_t(-1), row_max(mi), 2); // exchange max in a quad in a row
row_max(mi) = fmaxf(row_max(mi), max_recv);
float scores_max_cur = !InfCheck
? row_max(mi)
: (row_max(mi) == -INFINITY ? 0.0f : row_max(mi));
scores_scale(mi) = flash::ptx_exp2((scores_max_prev(mi) - scores_max_cur) * softmax_scale_log2);
// Two-level P quantization (default): s_P1 = rowmax(P̃)/(448×6), then s_P2,P̂_2 = φ(P̃/s_P1)
// Single-level P quantization: s_P2, P̂_2 = φ(P̃) directly (like V quantization)
const float s_P1_offset = single_level_p_quant ? 0.f : fp8_scalexfp4_scale_log2;
const float max_scaled = InfCheck
? (row_max(mi) == -INFINITY ? 0.f : (row_max(mi) * softmax_scale_log2 + s_P1_offset))
: (row_max(mi) * softmax_scale_log2 + s_P1_offset);
row_sum(mi) = row_sum(mi) * scores_scale(mi);
CUTLASS_PRAGMA_UNROLL
for (int ni = 0; ni < size<1>(acc_reduction_view); ni++) {
acc_reduction_view(mi, ni) = flash::ptx_exp2(acc_reduction_view(mi, ni) * softmax_scale_log2 - max_scaled);
row_sum(mi) += acc_reduction_view(mi, ni);
}
// s_P2 = max(P_block)/6 — per-block scale factor from φ function
CUTLASS_PRAGMA_UNROLL
for (int sfi = 0; sfi < size<1>(AbsMaxP); sfi++) {
AbsMaxP(mi, sfi) = flash::ptx_exp2(AbsMaxP(mi, sfi) * softmax_scale_log2 - max_scaled + fp4_scale_log2);
}
// scores_scale(mi) = max_scaled;
}
}
CUTLASS_PRAGMA_UNROLL
for (int i = 0; i < size(AbsMaxP); ++i) {
CUTLASS_PRAGMA_UNROLL
for (int j = 0; j < size<0>(acc_conversion_flatten); ++j)
acc_conversion_flatten(j, i) /= AbsMaxP(i);
}
}
template<typename TensorAcc>
CUTLASS_DEVICE void finalize(TensorAcc& o_store) {
Tensor o_store_reduction_view = make_tensor(o_store.data(), flash::convert_to_reduction_layout(o_store.layout()));
CUTLASS_PRAGMA_UNROLL
for (int mi = 0; mi < size(row_max); ++mi) {
CUTLASS_PRAGMA_UNROLL
for (int i = 1; i < RowReductionThr; i <<= 1) {
float sum_recv = __shfl_xor_sync(int32_t(-1), row_sum(mi), i);
row_sum(mi) += sum_recv;
}
float sum = row_sum(mi);
float inv_sum = (sum == 0.f || sum != sum) ? 0.f : 1 / sum;
CUTLASS_PRAGMA_UNROLL
for (int ni = 0; ni < size<1>(o_store_reduction_view); ++ni) {
o_store_reduction_view(mi, ni) *= inv_sum;
}
}
}
template<typename TensorAcc>
CUTLASS_DEVICE void rescale_o(TensorAcc& o_store, TensorAcc const& o_tmp) {
Tensor o_store_reduction_view = make_tensor(o_store.data(), flash::convert_to_reduction_layout(o_store.layout()));
Tensor o_tmp_reduction_view = make_tensor(o_tmp.data(), flash::convert_to_reduction_layout(o_tmp.layout()));
CUTLASS_PRAGMA_UNROLL
for (int mi = 0; mi < size(row_max); ++mi) {
CUTLASS_PRAGMA_UNROLL
for (int ni = 0; ni < size<1>(o_store_reduction_view); ++ni) {
o_store_reduction_view(mi, ni) = o_store_reduction_view(mi, ni) * scores_scale(mi) + o_tmp_reduction_view(mi, ni);
}
}
}
};
} // namespace flash
@@ -0,0 +1,83 @@
// Inspired by
// https://github.com/NVIDIA/DALI/blob/main/include/dali/core/static_switch.h
// and https://github.com/pytorch/pytorch/blob/master/aten/src/ATen/Dispatch.h
#pragma once
/// @param COND - a boolean expression to switch by
/// @param CONST_NAME - a name given for the constexpr bool variable.
/// @param ... - code to execute for true and false
///
/// Usage:
/// ```
/// BOOL_SWITCH(flag, BoolConst, [&] {
/// some_function<BoolConst>(...);
/// });
/// ```
//
#define BOOL_SWITCH(COND, CONST_NAME, ...) \
[&] { \
if (COND) { \
constexpr static bool CONST_NAME = true; \
return __VA_ARGS__(); \
} else { \
constexpr static bool CONST_NAME = false; \
return __VA_ARGS__(); \
} \
}()
#define PREC_SWITCH(PRECTYPE, ...) \
[&] { \
if (PRECTYPE == 1) { \
using kPrecType = cutlass::half_t; \
constexpr static bool kSoftFp16 = false; \
constexpr static bool kHybrid = false; \
return __VA_ARGS__(); \
} else if (PRECTYPE == 2) { \
using kPrecType = cutlass::float_e4m3_t; \
constexpr static bool kSoftFp16 = false; \
constexpr static bool kHybrid = false; \
return __VA_ARGS__(); \
} else if (PRECTYPE == 3) { \
using kPrecType = cutlass::float_e4m3_t; \
constexpr static bool kSoftFp16 = false; \
constexpr static bool kHybrid = true; \
return __VA_ARGS__(); \
} else if (PRECTYPE == 4) { \
using kPrecType = cutlass::float_e4m3_t; \
constexpr static bool kSoftFp16 = true; \
constexpr static bool kHybrid = false; \
return __VA_ARGS__(); \
} \
}()
#define HEADDIM_SWITCH(HEADDIM, ...) \
[&] { \
if (HEADDIM == 64) { \
constexpr static int kHeadSize = 64; \
return __VA_ARGS__(); \
} else if (HEADDIM == 128) { \
constexpr static int kHeadSize = 128; \
return __VA_ARGS__(); \
} else if (HEADDIM == 256) { \
constexpr static int kHeadSize = 256; \
return __VA_ARGS__(); \
} \
}()
#define SEQLEN_SWITCH(USE_VAR_SEQ_LEN, SEQ_LEN_OUT_OF_BOUND_CHECK, ...) \
[&] { \
if (!USE_VAR_SEQ_LEN) { \
if (SEQ_LEN_OUT_OF_BOUND_CHECK) { \
using kSeqLenTraitsType = FixedSeqLenTraits<true>; \
return __VA_ARGS__(); \
} else { \
using kSeqLenTraitsType = FixedSeqLenTraits<false>; \
return __VA_ARGS__(); \
} \
} else { \
using kSeqLenTraitsType = VarSeqLenTraits; \
return __VA_ARGS__(); \
} \
}()
@@ -0,0 +1,304 @@
/*
* Copyright (c) 2025 by SageAttention team.
*
* This code is based on code from FlashAttention3, https://github.com/Dao-AILab/flash-attention
* Copyright (c) 2024, Jay Shah, Ganesh Bikshandi, Ying Zhang, Vijay Thakkar, Pradeep Ramani, Tri Dao.
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
#pragma once
#include "cutlass/fast_math.h"
namespace flash {
///////////////////////////////////////////////////////////////////////////////
class StaticPersistentTileSchedulerOld {
//
// Data members
//
private:
int current_work_linear_idx_;
cutlass::FastDivmod const &m_block_divmod, &head_divmod;
int const total_blocks;
public:
struct WorkTileInfo {
int M_idx = 0;
int H_idx = 0;
int B_idx = 0;
bool is_valid_tile = false;
CUTLASS_HOST_DEVICE
bool
is_valid() const {
return is_valid_tile;
}
CUTLASS_HOST_DEVICE
static WorkTileInfo
invalid_work_tile() {
return {-1, -1, -1, false};
}
};
public:
CUTLASS_DEVICE explicit StaticPersistentTileSchedulerOld(cutlass::FastDivmod const &m_block_divmod_,
cutlass::FastDivmod const &head_divmod_,
int const total_blocks_) :
m_block_divmod(m_block_divmod_), head_divmod(head_divmod_), total_blocks(total_blocks_) {
// MSVC requires protecting use of CUDA-specific nonstandard syntax,
// like blockIdx and gridDim, with __CUDA_ARCH__.
#if defined(__CUDA_ARCH__)
// current_work_linear_idx_ = blockIdx.x + blockIdx.y * gridDim.x + blockIdx.z * gridDim.x * gridDim.y;
current_work_linear_idx_ = blockIdx.x;
#else
CUTLASS_ASSERT(false && "This line should never be reached");
#endif
}
CUTLASS_DEVICE
WorkTileInfo
get_current_work() const {
return get_current_work_for_linear_idx(current_work_linear_idx_);
}
CUTLASS_DEVICE
WorkTileInfo
get_current_work_for_linear_idx(int linear_idx) const {
if (linear_idx >= total_blocks) {
return WorkTileInfo::invalid_work_tile();
}
// Map worker's linear index into the CTA tiled problem shape to the corresponding MHB indices
int M_idx, H_idx, B_idx;
int quotient = m_block_divmod.divmod(M_idx, linear_idx);
B_idx = head_divmod.divmod(H_idx, quotient);
return {M_idx, H_idx, B_idx, true};
}
CUTLASS_DEVICE
void
// advance_to_next_work(int advance_count = 1) {
advance_to_next_work() {
// current_work_linear_idx_ += int(gridDim.x * gridDim.y * gridDim.z);
current_work_linear_idx_ += int(gridDim.x);
}
CUTLASS_DEVICE
WorkTileInfo
fetch_next_work() {
WorkTileInfo new_work_tile_info;
advance_to_next_work();
new_work_tile_info = get_current_work();
return new_work_tile_info;
}
};
///////////////////////////////////////////////////////////////////////////////
class SingleTileScheduler {
public:
// Host side kernel arguments
struct Arguments {
int const num_blocks_m, num_head, num_batch;
int const* tile_count_semaphore = nullptr;
};
// Device side kernel params
struct Params {};
static Params
to_underlying_arguments(Arguments const& args) {
return {};
}
static dim3
get_grid_dim(Arguments const& args, int num_sm) {
return {uint32_t(args.num_blocks_m), uint32_t(args.num_head), uint32_t(args.num_batch)};
}
struct WorkTileInfo {
int M_idx = 0;
int H_idx = 0;
int B_idx = 0;
bool is_valid_tile = false;
CUTLASS_DEVICE
bool
is_valid(Params const& params) const {
return is_valid_tile;
}
CUTLASS_DEVICE
cute::tuple<int32_t, int32_t, int32_t>
get_block_coord(Params const& params) const {
return {M_idx, H_idx, B_idx};
}
CUTLASS_DEVICE
WorkTileInfo
get_next_work(Params const& params) const {
return {-1, -1, -1, false};
}
};
CUTLASS_DEVICE
WorkTileInfo
get_initial_work() const {
return {int(blockIdx.x), int(blockIdx.y), int(blockIdx.z), true};
}
CUTLASS_DEVICE
WorkTileInfo
get_next_work(Params const& params, WorkTileInfo const& current_work) const {
return {-1, -1, -1, false};
}
};
///////////////////////////////////////////////////////////////////////////////
class StaticPersistentTileScheduler {
public:
// Host side kernel arguments
struct Arguments {
int const num_blocks_m, num_head, num_batch;
int const* tile_count_semaphore = nullptr;
};
// Device side kernel params
struct Params {
int total_blocks;
cutlass::FastDivmod m_block_divmod, head_divmod;
};
static Params
to_underlying_arguments(Arguments const& args) {
return {args.num_blocks_m * args.num_head * args.num_batch,
cutlass::FastDivmod(args.num_blocks_m), cutlass::FastDivmod(args.num_head)};
}
static dim3
get_grid_dim(Arguments const& args, int num_sm) {
return {uint32_t(num_sm)};
}
struct WorkTileInfo {
int tile_idx;
CUTLASS_DEVICE
bool
is_valid(Params const& params) const {
return tile_idx < params.total_blocks;
}
CUTLASS_DEVICE
cute::tuple<int32_t, int32_t, int32_t>
get_block_coord(Params const& params) const {
int m_block, bidh, bidb;
bidb = params.head_divmod.divmod(bidh, params.m_block_divmod.divmod(m_block, tile_idx));
return {m_block, bidh, bidb};
}
};
CUTLASS_DEVICE
WorkTileInfo
get_initial_work() const {
return {int(blockIdx.x)};
}
CUTLASS_DEVICE
WorkTileInfo
get_next_work(Params const& params, WorkTileInfo const& current_work) const {
return {current_work.tile_idx + int(gridDim.x)};
}
};
class DynamicPersistentTileScheduler {
public:
// Host side kernel arguments
struct Arguments {
int const num_blocks_m, num_head, num_batch;
int const* tile_count_semaphore;
};
// Device side kernel params
struct Params {
int const total_blocks;
cutlass::FastDivmod const m_block_divmod, head_divmod;
int const* tile_count_semaphore;
};
static Params
to_underlying_arguments(Arguments const& args) {
return {args.num_blocks_m * args.num_head * args.num_batch,
cutlass::FastDivmod(args.num_blocks_m), cutlass::FastDivmod(args.num_head),
args.tile_count_semaphore};
}
static dim3
get_grid_dim(Arguments const& args, int num_sm) {
return {uint32_t(num_sm)};
}
using WorkTileInfo = StaticPersistentTileScheduler::WorkTileInfo;
// struct WorkTileInfo {
// int tile_idx;
// CUTLASS_DEVICE
// bool
// is_valid(Params const& params) const {
// return tile_idx < params.total_blocks;
// }
// CUTLASS_DEVICE
// cute::tuple<int32_t, int32_t, int32_t>
// get_block_coord(Params const& params) const {
// int m_block, bidh, bidb;
// bidb = params.head_divmod.divmod(bidh, params.m_block_divmod.divmod(m_block, tile_idx));
// return {m_block, bidh, bidb};
// }
// };
CUTLASS_DEVICE
WorkTileInfo
get_initial_work() const {
return {int(blockIdx.x)};
}
CUTLASS_DEVICE
WorkTileInfo
get_next_work(Params const& params, WorkTileInfo const& current_work) const {
return {current_work.tile_idx + int(gridDim.x)};
}
};
} // flash
@@ -0,0 +1,408 @@
/*
* Copyright (c) 2025 by SageAttention team.
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
#pragma once
#include <assert.h>
#include <stdint.h>
#include <stdlib.h>
#include <cuda_fp16.h>
#if defined(__CUDA_ARCH__) && __CUDA_ARCH__ >= 800
#include <cuda_bf16.h>
#endif
#include <cute/tensor.hpp>
#include <cutlass/array.h>
#include <cutlass/cutlass.h>
#include <cutlass/numeric_conversion.h>
#include <cutlass/numeric_types.h>
namespace flash {
using namespace cute;
////////////////////////////////////////////////////////////////////////////////////////////////////
template<typename T>
struct MaxOp {
__device__ __forceinline__ T operator()(T const & x, T const & y) { return x > y ? x : y; }
};
template <>
struct MaxOp<float> {
// This is slightly faster
__device__ __forceinline__ float operator()(float const &x, float const &y) { return max(x, y); }
};
////////////////////////////////////////////////////////////////////////////////////////////////////
template<typename T>
struct SumOp {
__device__ __forceinline__ T operator()(T const & x, T const & y) { return x + y; }
};
////////////////////////////////////////////////////////////////////////////////////////////////////
template<int THREADS>
struct Allreduce {
static_assert(THREADS == 32 || THREADS == 16 || THREADS == 8 || THREADS == 4);
template<typename T, typename Operator>
static __device__ __forceinline__ T run(T x, Operator &op) {
constexpr int OFFSET = THREADS / 2;
x = op(x, __shfl_xor_sync(uint32_t(-1), x, OFFSET));
return Allreduce<OFFSET>::run(x, op);
}
};
////////////////////////////////////////////////////////////////////////////////////////////////////
template<>
struct Allreduce<2> {
template<typename T, typename Operator>
static __device__ __forceinline__ T run(T x, Operator &op) {
x = op(x, __shfl_xor_sync(uint32_t(-1), x, 1));
return x;
}
};
////////////////////////////////////////////////////////////////////////////////////////////////////
template<bool zero_init=true, typename Engine0, typename Layout0, typename Engine1, typename Layout1, typename Operator>
__device__ __forceinline__ void thread_reduce_(Tensor<Engine0, Layout0> const &tensor, Tensor<Engine1, Layout1> &summary, Operator &op) {
static_assert(Layout0::rank == 2, "Only support 2D Tensor");
static_assert(Layout1::rank == 1, "Only support 1D Tensor");
CUTE_STATIC_ASSERT_V(size<0>(summary) == size<0>(tensor));
#pragma unroll
for (int mi = 0; mi < size<0>(tensor); mi++) {
summary(mi) = zero_init ? tensor(mi, 0) : op(summary(mi), tensor(mi, 0));
#pragma unroll
for (int ni = 1; ni < size<1>(tensor); ni++) {
summary(mi) = op(summary(mi), tensor(mi, ni));
}
}
}
template<typename Engine0, typename Layout0, typename Engine1, typename Layout1, typename Operator>
__device__ __forceinline__ void quad_allreduce_(Tensor<Engine0, Layout0> &dst, Tensor<Engine1, Layout1> &src, Operator &op) {
CUTE_STATIC_ASSERT_V(size(dst) == size(src));
#pragma unroll
for (int i = 0; i < size(dst); i++){
dst(i) = Allreduce<4>::run(src(i), op);
}
}
template<bool zero_init=true, typename Engine0, typename Layout0, typename Engine1, typename Layout1, typename Operator>
__device__ __forceinline__ void reduce_(Tensor<Engine0, Layout0> const& tensor, Tensor<Engine1, Layout1> &summary, Operator &op) {
thread_reduce_<zero_init>(tensor, summary, op);
quad_allreduce_(summary, summary, op);
}
template<bool zero_init=true, typename Engine0, typename Layout0, typename Engine1, typename Layout1>
__device__ __forceinline__ void reduce_max(Tensor<Engine0, Layout0> const& tensor, Tensor<Engine1, Layout1> &max){
MaxOp<float> max_op;
reduce_<zero_init>(tensor, max, max_op);
}
template<bool zero_init=true, bool warp_reduce=true, typename Engine0, typename Layout0, typename Engine1, typename Layout1>
__device__ __forceinline__ void reduce_sum(Tensor<Engine0, Layout0> const& tensor, Tensor<Engine1, Layout1> &sum){
SumOp<float> sum_op;
thread_reduce_<zero_init>(tensor, sum, sum_op);
if constexpr (warp_reduce) { quad_allreduce_(sum, sum, sum_op); }
}
__forceinline__ __device__ __half2 half_exp(__half2 x) {
uint32_t tmp_out, tmp_in;
tmp_in = reinterpret_cast<uint32_t&>(x);
asm ("ex2.approx.f16x2 %0, %1;\n"
: "=r"(tmp_out)
: "r"(tmp_in));
__half2 out = reinterpret_cast<__half2&>(tmp_out);
return out;
}
// Apply the exp to all the elements.
template <bool zero_init=false, typename Engine0, typename Layout0, typename Engine1, typename Layout1>
__forceinline__ __device__ void max_scale_exp2_sum(Tensor<Engine0, Layout0> &tensor, Tensor<Engine1, Layout1> &max, Tensor<Engine1, Layout1> &sum, const float scale) {
static_assert(Layout0::rank == 2, "Only support 2D Tensor"); static_assert(Layout1::rank == 1, "Only support 1D Tensor"); CUTE_STATIC_ASSERT_V(size<0>(max) == size<0>(tensor));
#pragma unroll
for (int mi = 0; mi < size<0>(tensor); ++mi) {
MaxOp<float> max_op;
max(mi) = zero_init ? tensor(mi, 0) : max_op(max(mi), tensor(mi, 0));
#pragma unroll
for (int ni = 1; ni < size<1>(tensor); ni++) {
max(mi) = max_op(max(mi), tensor(mi, ni));
}
max(mi) = Allreduce<4>::run(max(mi), max_op);
// If max is -inf, then all elements must have been -inf (possibly due to masking).
// We don't want (-inf - (-inf)) since that would give NaN.
const float max_scaled = max(mi) == -INFINITY ? 0.f : max(mi) * scale;
sum(mi) = 0;
#pragma unroll
for (int ni = 0; ni < size<1>(tensor); ++ni) {
// Instead of computing exp(x - max), we compute exp2(x * log_2(e) -
// max * log_2(e)) This allows the compiler to use the ffma
// instruction instead of fadd and fmul separately.
tensor(mi, ni) = exp2f(tensor(mi, ni) * scale - max_scaled);
sum(mi) += tensor(mi, ni);
}
}
}
// Apply the exp to all the elements.
template <bool Scale_max=true, bool Check_inf=true, typename Engine0, typename Layout0, typename Engine1, typename Layout1>
__forceinline__ __device__ void scale_apply_exp2(Tensor<Engine0, Layout0> &tensor, Tensor<Engine1, Layout1> const &max, const float scale) {
static_assert(Layout0::rank == 2, "Only support 2D Tensor");
static_assert(Layout1::rank == 1, "Only support 1D Tensor");
CUTE_STATIC_ASSERT_V(size<0>(max) == size<0>(tensor));
#pragma unroll
for (int mi = 0; mi < size<0>(tensor); ++mi) {
// If max is -inf, then all elements must have been -inf (possibly due to masking).
// We don't want (-inf - (-inf)) since that would give NaN.
// If we don't have float around M_LOG2E the multiplication is done in fp64.
const float max_scaled = Check_inf
? (max(mi) == -INFINITY ? 0.f : (max(mi) * (Scale_max ? scale : float(M_LOG2E))))
: (max(mi) * (Scale_max ? scale : float(M_LOG2E)));
#pragma unroll
for (int ni = 0; ni < size<1>(tensor); ++ni) {
// Instead of computing exp(x - max), we compute exp2(x * log_2(e) -
// max * log_2(e)) This allows the compiler to use the ffma
// instruction instead of fadd and fmul separately.
tensor(mi, ni) = exp2f(tensor(mi, ni) * scale - max_scaled);
}
}
}
////////////////////////////////////////////////////////////////////////////////////////////////////
__forceinline__ __device__ float ptx_exp2(float x) {
float y;
asm volatile("ex2.approx.ftz.f32 %0, %1;" : "=f"(y) : "f"(x));
return y;
}
CUTLASS_DEVICE void
packed_float_to_ue4m3(
float const &f0, float const &f1, float const &f2, float const &f3,
uint32_t &out
) {
asm volatile( \
"{\n" \
".reg .b16 lo;\n" \
".reg .b16 hi;\n" \
"cvt.rn.satfinite.e4m3x2.f32 lo, %2, %1;\n" \
"cvt.rn.satfinite.e4m3x2.f32 hi, %4, %3;\n" \
"mov.b32 %0, {lo, hi};\n" \
"}" \
: "=r"(out) : "f"(f0), "f"(f1), "f"(f2), "f"(f3));
}
CUTLASS_DEVICE void
packed_float_to_e2m1(
float const &f0, float const &f1, float const &f2, float const& f3,
float const &f4, float const &f5, float const &f6, float const& f7,
uint32_t &out
) {
asm volatile( \
"{\n" \
".reg .b8 byte0;\n" \
".reg .b8 byte1;\n" \
".reg .b8 byte2;\n" \
".reg .b8 byte3;\n" \
"cvt.rn.satfinite.e2m1x2.f32 byte0, %2, %1;\n" \
"cvt.rn.satfinite.e2m1x2.f32 byte1, %4, %3;\n" \
"cvt.rn.satfinite.e2m1x2.f32 byte2, %6, %5;\n" \
"cvt.rn.satfinite.e2m1x2.f32 byte3, %8, %7;\n" \
"mov.b32 %0, {byte0, byte1, byte2, byte3};\n" \
"}" \
: "=r"(out) : "f"(f0), "f"(f1), "f"(f2), "f"(f3),
"f"(f4), "f"(f5), "f"(f6), "f"(f7));
}
CUTLASS_DEVICE void
add(float2 & c,
float2 const& a,
float2 const& b)
{
asm volatile("add.f32x2 %0, %1, %2;\n"
: "=l"(reinterpret_cast<uint64_t &>(c))
: "l"(reinterpret_cast<uint64_t const&>(a)),
"l"(reinterpret_cast<uint64_t const&>(b)));
}
CUTLASS_DEVICE void
add_inplace(float2 &a,
float2 const& b)
{
asm volatile("add.f32x2 %0, %0, %1;\n"
: "+l"(reinterpret_cast<uint64_t &>(a)) // a: input/output
: "l"(reinterpret_cast<uint64_t const&>(b)) // b: input
);
}
CUTLASS_DEVICE void
sub(float2 & c,
float2 const& a,
float2 const& b)
{
asm volatile("sub.f32x2 %0, %1, %2;\n"
: "=l"(reinterpret_cast<uint64_t &>(c))
: "l"(reinterpret_cast<uint64_t const&>(a)),
"l"(reinterpret_cast<uint64_t const&>(b)));
}
CUTLASS_DEVICE void
sub_inplace(float2 &a,
float2 const& b)
{
asm volatile("sub.f32x2 %0, %0, %1;\n"
: "+l"(reinterpret_cast<uint64_t &>(a)) // a: input/output
: "l"(reinterpret_cast<uint64_t const&>(b)) // b: input
);
}
CUTLASS_DEVICE void
mul(float2 & c,
float2 const& a,
float2 const& b)
{
asm volatile("mul.f32x2 %0, %1, %2;\n"
: "=l"(reinterpret_cast<uint64_t &>(c))
: "l"(reinterpret_cast<uint64_t const&>(a)),
"l"(reinterpret_cast<uint64_t const&>(b)));
}
CUTLASS_DEVICE void
fma(float2 & d,
float2 const& a,
float2 const& b,
float2 const& c)
{
asm volatile("fma.rn.f32x2 %0, %1, %2, %3;\n"
: "=l"(reinterpret_cast<uint64_t &>(d))
: "l"(reinterpret_cast<uint64_t const&>(a)),
"l"(reinterpret_cast<uint64_t const&>(b)),
"l"(reinterpret_cast<uint64_t const&>(c)));
}
CUTLASS_DEVICE void
fma_inplace(float2 &a,
float2 const& b,
float2 const& c)
{
asm volatile("fma.rn.f32x2 %0, %0, %1, %2;\n"
: "+l"(reinterpret_cast<uint64_t &>(a))
: "l"(reinterpret_cast<uint64_t const&>(b)),
"l"(reinterpret_cast<uint64_t const&>(c)));
}
////////////////////////////////////////////////////////////////////////////////////////////////////
template <
class Layout
>
CUTLASS_DEVICE constexpr
auto convert_to_reduction_layout(Layout mma_layout) {
static_assert(rank(mma_layout) == 3, "Mma Layout should be (MmaAtom, MmaM, MmaN)");
static_assert(rank(get<0>(shape(mma_layout))) == 2, "MmaAtom should be (AtomN, AtomM)");
return make_layout(
make_layout(get<0,1>(mma_layout), get<1>(mma_layout)),
make_layout(get<0,0>(mma_layout), get<2>(mma_layout))
);
}
template <
class Tensor
>
CUTLASS_DEVICE constexpr
auto convert_to_reduction_tensor(Tensor mma_tensor) {
return make_tensor(mma_tensor.data(), convert_to_reduction_layout(mma_tensor.layout()));
}
template <
class Layout
>
CUTLASS_DEVICE constexpr
auto convert_to_conversion_layout(Layout mma_layout) {
static_assert(rank(mma_layout) == 3, "Mma Layout should be (MmaAtom, MmaM, MmaN)");
static_assert(rank(get<0>(shape(mma_layout))) == 2, "MmaAtom should be (AtomN, AtomM)");
constexpr int MmaAtomN = size<0, 0>(mma_layout);
constexpr int MmaAtomM = size<0, 1>(mma_layout);
constexpr int MmaM = size<1>(mma_layout);
constexpr int MmaN = size<2>(mma_layout);
static_assert(MmaAtomN == 8, "MmaAtomN should be 8.");
static_assert(MmaAtomM == 2, "MmaAtomM should be 2.");
static_assert(MmaN % 2 == 0, "MmaN should be multiple of 2.");
auto mma_n_division = zipped_divide(
layout<2>(mma_layout), make_tile(_2{})
);
return make_layout(
make_layout(layout<0,0>(mma_layout), make_layout(layout<0,1>(mma_layout), layout<0>(mma_n_division))),
layout<1>(mma_layout), layout<1>(mma_n_division)
);
}
template <
class Tensor
>
CUTLASS_DEVICE constexpr
auto convert_to_conversion_tensor(Tensor mma_tensor) {
return make_tensor(mma_tensor.data(), convert_to_conversion_layout(mma_tensor.layout()));
}
////////////////////////////////////////////////////////////////////////////////////////////////////
template <bool Is_even_MN=true, bool Is_even_K=true, bool Clear_OOB_MN=false, bool Clear_OOB_K=true,
typename TiledCopy, typename Engine0, typename Layout0, typename Engine1, typename Layout1,
typename Engine2, typename Layout2, typename Engine3, typename Layout3>
CUTLASS_DEVICE void copy(TiledCopy tiled_copy, Tensor<Engine0, Layout0> const &S,
Tensor<Engine1, Layout1> &D, Tensor<Engine2, Layout2> const &identity_MN,
Tensor<Engine3, Layout3> const &predicate_K, const int max_MN=0) {
CUTE_STATIC_ASSERT_V(rank(S) == Int<3>{});
CUTE_STATIC_ASSERT_V(rank(D) == Int<3>{});
CUTE_STATIC_ASSERT_V(size<0>(S) == size<0>(D)); // MMA
CUTE_STATIC_ASSERT_V(size<1>(S) == size<1>(D)); // MMA_M
CUTE_STATIC_ASSERT_V(size<2>(S) == size<2>(D)); // MMA_K
// There's no case where !Clear_OOB_K && Clear_OOB_MN
static_assert(!(Clear_OOB_MN && !Clear_OOB_K));
#pragma unroll
for (int m = 0; m < size<1>(S); ++m) {
if (Is_even_MN || get<0>(identity_MN(0, m, 0)) < max_MN) {
#pragma unroll
for (int k = 0; k < size<2>(S); ++k) {
if (Is_even_K || predicate_K(k)) {
cute::copy(tiled_copy, S(_, m, k), D(_, m, k));
} else if (Clear_OOB_K) {
cute::clear(D(_, m, k));
}
}
} else if (Clear_OOB_MN) {
cute::clear(D(_, m, _));
}
}
}
} // namespace flash
@@ -0,0 +1 @@
__version__ = "3.0.0.b1"
@@ -0,0 +1,90 @@
"""
Copyright (c) 2025 by SageAttention team.
Licensed under the Apache License, Version 2.0 (the "License");
you may not use this file except in compliance with the License.
You may obtain a copy of the License at
http://www.apache.org/licenses/LICENSE-2.0
Unless required by applicable law or agreed to in writing, software
distributed under the License is distributed on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
See the License for the specific language governing permissions and
limitations under the License.
"""
import torch
import fp4quant
from triton.tools.mxfp import MXFP4Tensor
from bench_utils import bench_kineto
b = 1
h = 32
n = 16384
d = 128
def test():
q = torch.randn((b, h, n, d), device="cuda", dtype=torch.float16)
o = torch.empty((b, h, n, d // 2), device="cuda", dtype=torch.uint8)
o_s = torch.empty((b, h, n, d // 16), device="cuda", dtype=torch.float8_e4m3fn)
fp4quant.scaled_fp4_quant_permute(q, o, o_s, 1)
test()
t = bench_kineto(test, "scaled_fp4_quant_kernel", suppress_kineto_output=True)
IO = b * h * n * d * 2 + b * h * n * d * 0.5 + b * h * n * d // 16 * 1
throughput = IO / t * 1e-9
print(f"Throughput: {throughput:.2f} GB/s")
def scale_and_fp4_tensor(x: torch.Tensor, packed_dim: int = 3, all_ones: bool = False, permuted: bool = False):
assert x.is_contiguous() and x.ndim == 4 and x.shape[-1] % 16 == 0
B, H, M, N = x.shape
x = x.view(B, H, M, N // 16, 16)
scales = (x.abs().amax(dim=-1, keepdim=True) / 6).to(torch.float32)
if all_ones:
scales = torch.ones_like(scales)
x_scaled = x / scales
packed_fp4 = MXFP4Tensor(x_scaled.flatten(start_dim=-2)).to_packed_tensor(dim=packed_dim)
dequant_x = (MXFP4Tensor(x_scaled).to(torch.float32) * scales.to(torch.float8_e4m3fn).to(torch.float32)).flatten(start_dim=-2)
fp8_scale = scales.flatten(start_dim=-2).to(torch.float8_e4m3fn)
permuted_fp8_scale = None
if permuted:
scales = scales.view(B, H // 64, 4, 16, M, N // 16).permute(0, 1, 3, 2, 4, 5).reshape(B, H, M, N // 16)
permuted_fp8_scale = scales.view(B, H // 64, 64, M, N // 64, 4).permute(0, 1, 4, 3, 2, 5).reshape(B, H, M, N // 16).to(torch.float8_e4m3fn)
return fp8_scale, packed_fp4, dequant_x, permuted_fp8_scale
b = 2
h = 4
n = 251
n_padded = (n + 127) // 128 * 128
d = 128
q = torch.randn(b, h, n, d, dtype=torch.float16, device='cuda')
o = torch.empty((b, h, n, d // 2), dtype=torch.uint8, device='cuda')
o_s = torch.empty((b, h, n, d // 16), dtype=torch.float8_e4m3fn, device='cuda')
fp4quant.scaled_fp4_quant(q, o, o_s, 1)
k_permute = [0, 1, 8, 9, 16, 17, 24, 25, 2, 3, 10, 11, 18, 19, 26, 27, 4, 5, 12, 13, 20, 21, 28, 29, 6, 7, 14, 15, 22, 23, 30, 31]
o_permuted = torch.empty((b, h, n_padded, d // 2), dtype=torch.uint8, device='cuda')
o_s_permuted = torch.empty((b, h, n_padded, d // 16), dtype=torch.float8_e4m3fn, device='cuda')
fp4quant.scaled_fp4_quant_permute(q, o_permuted, o_s_permuted, 1)
# padding
if n % 128 != 0:
o_permuted_gt = torch.cat([o, torch.zeros((b, h, n_padded - n, d // 2), dtype=torch.uint8, device='cuda')], dim=2)
o_s_permuted_gt = torch.cat([o_s, torch.zeros((b, h, n_padded - n, d // 16), dtype=torch.float8_e4m3fn, device='cuda')], dim=2)
else:
o_permuted_gt = o
o_s_permuted_gt = o_s
# use scale_and_fp4_tensor + torch permutation to get the ground truth
o_permuted_gt = o_permuted_gt.reshape(b, h, n_padded // 32, 32, d // 2)[:, :, :, k_permute, :].reshape(b, h, n_padded, d // 2)
o_s_permuted_gt = o_s_permuted_gt.reshape(b, h, n_padded // 32, 32, d // 16)[:, :, :, k_permute, :].reshape(b, h, n_padded, d // 16)
assert((o_permuted - o_permuted_gt).abs().max() == 0)
assert((o_s_permuted.float() - o_s_permuted_gt.float()).abs().max() == 0)
print("All tests passed!")
@@ -0,0 +1,86 @@
"""
Copyright (c) 2025 by SageAttention team.
Licensed under the Apache License, Version 2.0 (the "License");
you may not use this file except in compliance with the License.
You may obtain a copy of the License at
http://www.apache.org/licenses/LICENSE-2.0
Unless required by applicable law or agreed to in writing, software
distributed under the License is distributed on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
See the License for the specific language governing permissions and
limitations under the License.
"""
import torch
import fp4quant
from triton.tools.mxfp import MXFP4Tensor
from bench_utils import bench_kineto
b = 1
h = 32
n = 16384
d = 128
def test():
q = torch.randn((b, h, n, d), device="cuda", dtype=torch.float16)
o = torch.empty((b, h, n, d // 2), device="cuda", dtype=torch.uint8)
o_s = torch.empty((b, h, n, d // 16), device="cuda", dtype=torch.float8_e4m3fn)
fp4quant.scaled_fp4_quant(q, o, o_s, 1)
test()
t = bench_kineto(test, "scaled_fp4_quant_kernel", suppress_kineto_output=True)
IO = b * h * n * d * 2 + b * h * n * d * 0.5 + b * h * n * d // 16 * 1
throughput = IO / t * 1e-9
print(f"Throughput: {throughput:.2f} GB/s")
def scale_and_fp4_tensor(x: torch.Tensor, packed_dim: int = 3, all_ones: bool = False, permuted: bool = False):
assert x.is_contiguous() and x.ndim == 4 and x.shape[-1] % 16 == 0
B, H, M, N = x.shape
x = x.view(B, H, M, N // 16, 16)
scales = (x.abs().amax(dim=-1, keepdim=True) / 6).to(torch.float32)
if all_ones:
scales = torch.ones_like(scales)
x_scaled = x / scales
packed_fp4 = MXFP4Tensor(x_scaled.flatten(start_dim=-2)).to_packed_tensor(dim=packed_dim)
dequant_x = (MXFP4Tensor(x_scaled).to(torch.float32) * scales.to(torch.float8_e4m3fn).to(torch.float32)).flatten(start_dim=-2)
fp8_scale = scales.flatten(start_dim=-2).to(torch.float8_e4m3fn)
permuted_fp8_scale = None
if permuted:
scales = scales.view(B, H // 64, 4, 16, M, N // 16).permute(0, 1, 3, 2, 4, 5).reshape(B, H, M, N // 16)
permuted_fp8_scale = scales.view(B, H // 64, 64, M, N // 64, 4).permute(0, 1, 4, 3, 2, 5).reshape(B, H, M, N // 16).to(torch.float8_e4m3fn)
return fp8_scale, packed_fp4, dequant_x, permuted_fp8_scale
b = 2
h = 4
n = 251
d = 128
q = torch.randn(b, h, n, d, dtype=torch.float16, device='cuda')
o = torch.empty((b, h, n, d // 2), dtype=torch.uint8, device='cuda')
o_s = torch.empty((b, h, n, d // 16), dtype=torch.float8_e4m3fn, device='cuda')
fp4quant.scaled_fp4_quant(q, o, o_s, 1)
fp8_scale, packed_fp4, dequant_x, permuted_fp8_scale = scale_and_fp4_tensor(q, packed_dim=3)
assert((fp8_scale.float() - o_s.float()).abs().max() == 0)
o_binary = [
(int(bin_str[:4], 2), int(bin_str[4:], 2))
for bin_str in [format(x.item(), '08b') for x in o.view(-1)]
]
o_binary_gt = [
(int(bin_str[:4], 2), int(bin_str[4:], 2))
for bin_str in [format(x.item(), '08b') for x in packed_fp4.view(-1)]
]
for i in range(len(o_binary)):
# check contiguous 4 bits. Difference should be at most one
assert(abs(o_binary[i][0] - o_binary_gt[i][0]) <= 1)
assert(abs(o_binary[i][1] - o_binary_gt[i][1]) <= 1)
print("All tests passed!")
@@ -0,0 +1,86 @@
"""
Copyright (c) 2025 by SageAttention team.
Licensed under the Apache License, Version 2.0 (the "License");
you may not use this file except in compliance with the License.
You may obtain a copy of the License at
http://www.apache.org/licenses/LICENSE-2.0
Unless required by applicable law or agreed to in writing, software
distributed under the License is distributed on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
See the License for the specific language governing permissions and
limitations under the License.
"""
import torch
import fp4quant
from triton.tools.mxfp import MXFP4Tensor
from bench_utils import bench_kineto
b = 1
h = 32
n = 16384
d = 128
def test():
q = torch.randn((b, h, n, d), device="cuda", dtype=torch.float16)
o = torch.empty((b, h, d, n // 2), device="cuda", dtype=torch.uint8)
o_s = torch.empty((b, h, d, n // 16), device="cuda", dtype=torch.float8_e4m3fn)
fp4quant.scaled_fp4_quant_trans(q, o, o_s, 1)
test()
t = bench_kineto(test, "scaled_fp4_quant_trans_kernel", suppress_kineto_output=True)
IO = b * h * n * d * 2 + b * h * n * d * 0.5 + b * h * n * d // 16 * 1
throughput = IO / t * 1e-9
print(f"Throughput: {throughput:.2f} GB/s")
def scale_and_fp4_tensor(x: torch.Tensor, packed_dim: int = 3, all_ones: bool = False, permuted: bool = False):
assert x.is_contiguous() and x.ndim == 4 and x.shape[-1] % 16 == 0
B, H, M, N = x.shape
x = x.view(B, H, M, N // 16, 16)
scales = (x.abs().amax(dim=-1, keepdim=True) / 6).to(torch.float32)
if all_ones:
scales = torch.ones_like(scales)
x_scaled = x / scales
packed_fp4 = MXFP4Tensor(x_scaled.flatten(start_dim=-2)).to_packed_tensor(dim=packed_dim)
dequant_x = (MXFP4Tensor(x_scaled).to(torch.float32) * scales.to(torch.float8_e4m3fn).to(torch.float32)).flatten(start_dim=-2)
fp8_scale = scales.flatten(start_dim=-2).to(torch.float8_e4m3fn)
permuted_fp8_scale = None
if permuted:
scales = scales.view(B, H // 64, 4, 16, M, N // 16).permute(0, 1, 3, 2, 4, 5).reshape(B, H, M, N // 16)
permuted_fp8_scale = scales.view(B, H // 64, 64, M, N // 64, 4).permute(0, 1, 4, 3, 2, 5).reshape(B, H, M, N // 16).to(torch.float8_e4m3fn)
return fp8_scale, packed_fp4, dequant_x, permuted_fp8_scale
b = 2
h = 4
n = 491
n_padded = (n + 127) // 128 * 128
d = 128
q = torch.randn(b, h, n, d, dtype=torch.float16, device='cuda')
o = torch.empty((b, h, d, n_padded // 2), dtype=torch.uint8, device='cuda')
o_s = torch.empty((b, h, d, n_padded // 16), dtype=torch.float8_e4m3fn, device='cuda')
fp4quant.scaled_fp4_quant_trans(q, o, o_s, 1)
if n % 128 != 0:
q_padded = torch.cat([q, torch.zeros((b, h, n_padded - n, d), dtype=torch.float16, device='cuda')], dim=2)
else:
q_padded = q
# use torch transpose + scaled_fp4_quant to get the ground truth
q_padded = q_padded.transpose(2, 3).reshape(b, h, n_padded, d).contiguous()
o_gt = torch.empty((b, h, n_padded, d // 2), dtype=torch.uint8, device='cuda')
o_s_gt = torch.empty((b, h, n_padded, d // 16), dtype=torch.float8_e4m3fn, device='cuda')
fp4quant.scaled_fp4_quant(q_padded, o_gt, o_s_gt, 1)
o_gt = o_gt.reshape(b, h, d, n_padded // 2).contiguous()
o_s_gt = o_s_gt.reshape(b, h, d, n_padded // 16).contiguous()
assert((o_s_gt.float() - o_s.float()).abs().max() == 0)
assert((o_gt - o).abs().max() == 0)
print("All tests passed!")
@@ -0,0 +1,169 @@
"""
Copyright (c) 2025 by SageAttention team.
Licensed under the Apache License, Version 2.0 (the "License");
you may not use this file except in compliance with the License.
You may obtain a copy of the License at
http://www.apache.org/licenses/LICENSE-2.0
Unless required by applicable law or agreed to in writing, software
distributed under the License is distributed on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
See the License for the specific language governing permissions and
limitations under the License.
"""
import os
import sys
import torch
import torch.distributed as dist
def bench(fn, num_warmups: int = 5, num_tests: int = 10,
high_precision: bool = False):
# Flush L2 cache with 256 MB data
torch.cuda.synchronize()
cache = torch.empty(int(256e6 // 4), dtype=torch.int, device='cuda')
cache.zero_()
# Warmup
for _ in range(num_warmups):
fn()
# Add a large kernel to eliminate the CPU launch overhead
if high_precision:
x = torch.randn((8192, 8192), dtype=torch.float, device='cuda')
y = torch.randn((8192, 8192), dtype=torch.float, device='cuda')
x @ y
# Testing
start_event = torch.cuda.Event(enable_timing=True)
end_event = torch.cuda.Event(enable_timing=True)
start_event.record()
for i in range(num_tests):
fn()
end_event.record()
torch.cuda.synchronize()
return start_event.elapsed_time(end_event) / num_tests
class empty_suppress:
def __enter__(self):
return self
def __exit__(self, *_):
pass
class suppress_stdout_stderr:
def __enter__(self):
self.outnull_file = open(os.devnull, 'w')
self.errnull_file = open(os.devnull, 'w')
self.old_stdout_fileno_undup = sys.stdout.fileno()
self.old_stderr_fileno_undup = sys.stderr.fileno()
self.old_stdout_fileno = os.dup(sys.stdout.fileno())
self.old_stderr_fileno = os.dup(sys.stderr.fileno())
self.old_stdout = sys.stdout
self.old_stderr = sys.stderr
os.dup2(self.outnull_file.fileno(), self.old_stdout_fileno_undup)
os.dup2(self.errnull_file.fileno(), self.old_stderr_fileno_undup)
sys.stdout = self.outnull_file
sys.stderr = self.errnull_file
return self
def __exit__(self, *_):
sys.stdout = self.old_stdout
sys.stderr = self.old_stderr
os.dup2(self.old_stdout_fileno, self.old_stdout_fileno_undup)
os.dup2(self.old_stderr_fileno, self.old_stderr_fileno_undup)
os.close(self.old_stdout_fileno)
os.close(self.old_stderr_fileno)
self.outnull_file.close()
self.errnull_file.close()
def bench_kineto(fn, kernel_names, num_tests: int = 30, suppress_kineto_output: bool = False,
trace_path: str = None, barrier_comm_profiling: bool = False, flush_l2: bool = False):
# Conflict with Nsight Systems
using_nsys = os.environ.get('DG_NSYS_PROFILING', False)
# For some auto-tuning kernels with prints
fn()
# Profile
suppress = suppress_stdout_stderr if suppress_kineto_output and not using_nsys else empty_suppress
with suppress():
schedule = torch.profiler.schedule(wait=0, warmup=1, active=1, repeat=1) if not using_nsys else None
profiler = torch.profiler.profile(activities=[torch.profiler.ProfilerActivity.CUDA], schedule=schedule) if not using_nsys else empty_suppress()
with profiler:
for i in range(2):
# NOTES: use a large kernel and a barrier to eliminate the unbalanced CPU launch overhead
if barrier_comm_profiling:
lhs = torch.randn((8192, 8192), dtype=torch.float, device='cuda')
rhs = torch.randn((8192, 8192), dtype=torch.float, device='cuda')
lhs @ rhs
dist.all_reduce(torch.ones(1, dtype=torch.float, device='cuda'))
for _ in range(num_tests):
if flush_l2:
torch.empty(int(256e6 // 4), dtype=torch.int, device='cuda').zero_()
fn()
if not using_nsys:
profiler.step()
# Return 1 if using Nsight Systems
if using_nsys:
return 1
# Parse the profiling table
assert isinstance(kernel_names, str) or isinstance(kernel_names, tuple)
is_tupled = isinstance(kernel_names, tuple)
prof_lines = profiler.key_averages().table(sort_by='cuda_time_total', max_name_column_width=100).split('\n')
kernel_names = (kernel_names, ) if isinstance(kernel_names, str) else kernel_names
assert all([isinstance(name, str) for name in kernel_names])
for name in kernel_names:
assert sum([name in line for line in prof_lines]) == 1, f'Errors of the kernel {name} in the profiling table'
# Save chrome traces
if trace_path is not None:
profiler.export_chrome_trace(trace_path)
# Return average kernel times
units = {'ms': 1e3, 'us': 1e6}
kernel_times = []
for name in kernel_names:
for line in prof_lines:
if name in line:
time_str = line.split()[-2]
for unit, scale in units.items():
if unit in time_str:
kernel_times.append(float(time_str.replace(unit, '')) / scale)
break
break
return tuple(kernel_times) if is_tupled else kernel_times[0]
def calc_diff(x, y):
x, y = x.double(), y.double()
denominator = (x * x + y * y).sum()
sim = 2 * (x * y).sum() / denominator
return 1 - sim
def count_bytes(tensors):
total = 0
for t in tensors:
if isinstance(t, tuple):
total += count_bytes(t)
else:
total += t.numel() * t.element_size()
return total
@@ -0,0 +1,52 @@
#pragma once
#include <stdio.h>
#if defined(__HIPCC__)
#define HOST_DEVICE_INLINE __host__ __device__
#define DEVICE_INLINE __device__
#define HOST_INLINE __host__
#elif defined(__CUDACC__) || defined(_NVHPC_CUDA)
#define HOST_DEVICE_INLINE __host__ __device__ __forceinline__
#define DEVICE_INLINE __device__ __forceinline__
#define HOST_INLINE __host__ __forceinline__
#else
#define HOST_DEVICE_INLINE inline
#define DEVICE_INLINE inline
#define HOST_INLINE inline
#endif
#define CUDA_CHECK(cmd) \
do { \
cudaError_t e = cmd; \
if (e != cudaSuccess) { \
printf("Failed: Cuda error %s:%d '%s'\n", __FILE__, __LINE__, \
cudaGetErrorString(e)); \
exit(EXIT_FAILURE); \
} \
} while (0)
int64_t get_device_attribute(int64_t attribute, int64_t device_id) {
static int value = [=]() {
int device = static_cast<int>(device_id);
if (device < 0) {
CUDA_CHECK(cudaGetDevice(&device));
}
int value;
CUDA_CHECK(cudaDeviceGetAttribute(
&value, static_cast<cudaDeviceAttr>(attribute), device));
return static_cast<int>(value);
}();
return value;
}
namespace cuda_utils {
template <typename T>
HOST_DEVICE_INLINE constexpr std::enable_if_t<std::is_integral_v<T>, T>
ceil_div(T a, T b) {
return (a + b - 1) / b;
}
}; // namespace cuda_utils
@@ -0,0 +1,629 @@
/*
* Copyright (c) 2025 by SageAttention team.
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
#include <torch/all.h>
#include <torch/python.h>
#include <torch/nn/functional.h>
#include <ATen/cuda/CUDAContext.h>
#include <c10/cuda/CUDAGuard.h>
#include <cuda_runtime_api.h>
#include <cuda_runtime.h>
#include <ATen/cuda/CUDAContext.h>
#include <c10/cuda/CUDAGuard.h>
#include <cuda_fp8.h>
#include "cuda_utils.h"
#include "../blackwell/block_config.h"
#define DISPATCH_PYTORCH_DTYPE_TO_CTYPE_FP16(pytorch_dtype, c_type, ...) \
if (pytorch_dtype == at::ScalarType::Half) { \
using c_type = half; \
__VA_ARGS__ \
} else if (pytorch_dtype == at::ScalarType::BFloat16) { \
using c_type = nv_bfloat16; \
__VA_ARGS__ \
} else { \
std::ostringstream oss; \
oss << __PRETTY_FUNCTION__ << " failed to dispatch data type " << pytorch_dtype; \
TORCH_CHECK(false, oss.str()); \
}
#define DISPATCH_HEAD_DIM(head_dim, HEAD_DIM, ...) \
if (head_dim == 64) { \
constexpr int HEAD_DIM = 64; \
__VA_ARGS__ \
} else if (head_dim == 128) { \
constexpr int HEAD_DIM = 128; \
__VA_ARGS__ \
} else { \
std::ostringstream err_msg; \
err_msg << "Unsupported head dim: " << int(head_dim); \
throw std::invalid_argument(err_msg.str()); \
}
#define CHECK_CUDA(x) \
TORCH_CHECK(x.is_cuda(), "Tensor " #x " must be on CUDA")
#define CHECK_DTYPE(x, true_dtype) \
TORCH_CHECK(x.dtype() == true_dtype, \
"Tensor " #x " must have dtype (" #true_dtype ")")
#define CHECK_DIMS(x, true_dim) \
TORCH_CHECK(x.dim() == true_dim, \
"Tensor " #x " must have dimension number (" #true_dim ")")
#define CHECK_SHAPE(x, ...) \
TORCH_CHECK(x.sizes() == torch::IntArrayRef({__VA_ARGS__}), \
"Tensor " #x " must have shape (" #__VA_ARGS__ ")")
#define CHECK_CONTIGUOUS(x) \
TORCH_CHECK(x.is_contiguous(), "Tensor " #x " must be contiguous")
#define CHECK_LASTDIM_CONTIGUOUS(x) \
TORCH_CHECK(x.stride(-1) == 1, \
"Tensor " #x " must be contiguous at the last dimension")
constexpr int CVT_FP4_ELTS_PER_THREAD = 16;
// Convert 4 float2 values into 8 e2m1 values (represented as one uint32_t).
inline __device__ uint32_t fp32_vec_to_e2m1(float2 *array) {
#if defined(__CUDA_ARCH__) && (__CUDA_ARCH__ >= 1000)
uint32_t val;
asm volatile(
"{\n"
".reg .b8 byte0;\n"
".reg .b8 byte1;\n"
".reg .b8 byte2;\n"
".reg .b8 byte3;\n"
"cvt.rn.satfinite.e2m1x2.f32 byte0, %2, %1;\n"
"cvt.rn.satfinite.e2m1x2.f32 byte1, %4, %3;\n"
"cvt.rn.satfinite.e2m1x2.f32 byte2, %6, %5;\n"
"cvt.rn.satfinite.e2m1x2.f32 byte3, %8, %7;\n"
"mov.b32 %0, {byte0, byte1, byte2, byte3};\n"
"}"
: "=r"(val)
: "f"(array[0].x), "f"(array[0].y), "f"(array[1].x), "f"(array[1].y),
"f"(array[2].x), "f"(array[2].y), "f"(array[3].x), "f"(array[3].y));
return val;
#else
return 0;
#endif
}
// Get type2 from type or vice versa (applied to half and bfloat16)
template <typename T>
struct TypeConverter {
using Type = half2;
}; // keep for generality
template <>
struct TypeConverter<half2> {
using Type = half;
};
template <>
struct TypeConverter<half> {
using Type = half2;
};
template <>
struct TypeConverter<__nv_bfloat162> {
using Type = __nv_bfloat16;
};
template <>
struct TypeConverter<__nv_bfloat16> {
using Type = __nv_bfloat162;
};
// Define a 32 bytes packed data type.
template <class Type>
struct PackedVec {
typename TypeConverter<Type>::Type elts[8];
};
template <uint32_t head_dim, uint32_t BLOCK_SIZE, bool permute, typename T>
__global__ void scaled_fp4_quant_kernel(
const T* input, uint8_t* output, uint8_t* output_sf,
int batch_size, int num_heads, int num_tokens,
int stride_bz_input, int stride_h_input, int stride_seq_input,
int stride_bz_output, int stride_h_output, int stride_seq_output,
int stride_bz_output_sf, int stride_h_output_sf, int stride_seq_output_sf) {
static_assert(std::is_same<T, half>::value || std::is_same<T, nv_bfloat16>::value, "Only half and bfloat16 input are supported");
using PackedVec = PackedVec<T>;
const int batch_id = blockIdx.y;
const int head_id = blockIdx.z;
const int token_block_id = blockIdx.x;
static_assert(CVT_FP4_ELTS_PER_THREAD == 8 || CVT_FP4_ELTS_PER_THREAD == 16,
"CVT_FP4_ELTS_PER_THREAD must be 8 or 16");
static_assert(sizeof(PackedVec) == sizeof(T) * CVT_FP4_ELTS_PER_THREAD,
"Vec size is not matched.");
constexpr uint32_t NUM_THREADS_PER_TOKEN = head_dim / CVT_FP4_ELTS_PER_THREAD;
// load input
const int token_id = token_block_id * BLOCK_SIZE + threadIdx.x / NUM_THREADS_PER_TOKEN;
int load_token_id;
if constexpr (!permute) {
load_token_id = token_id;
} else {
int local_token_id = threadIdx.x / NUM_THREADS_PER_TOKEN;
int local_token_id_residue = local_token_id % 32;
// [0, 1, 8, 9, 16, 17, 24, 25, 2, 3, 10, 11, 18, 19, 26, 27, 4, 5, 12, 13, 20, 21, 28, 29, 6, 7, 14, 15, 22, 23, 30, 31]
load_token_id = token_block_id * BLOCK_SIZE + (local_token_id / 32) * 32 +
(local_token_id_residue / 8) * 2 +
((local_token_id_residue % 8) / 2) * 8 +
(local_token_id_residue % 8) % 2;
}
PackedVec in_vec;
#pragma unroll
for (int i = 0; i < CVT_FP4_ELTS_PER_THREAD / 2; i++) {
reinterpret_cast<uint32_t&>(in_vec.elts[i]) = 0;
}
if (load_token_id < num_tokens) {
in_vec = reinterpret_cast<PackedVec const*>(input +
batch_id * stride_bz_input + // batch dim
head_id * stride_h_input + // head dim
load_token_id * stride_seq_input + // seq dim
(threadIdx.x % NUM_THREADS_PER_TOKEN) * CVT_FP4_ELTS_PER_THREAD)[0]; // feature dim
}
// calculate max of every consecutive 16 elements
auto localMax = __habs2(in_vec.elts[0]);
#pragma unroll
for (int i = 1; i < CVT_FP4_ELTS_PER_THREAD / 2; i++) { // local max
localMax = __hmax2(localMax, __habs2(in_vec.elts[i]));
}
if constexpr (CVT_FP4_ELTS_PER_THREAD == 8) { // shuffle across two threads
localMax = __hmax2(__shfl_xor_sync(0xffffffff, localMax, 1, 32), localMax);
}
float vecMax = float(__hmax(localMax.x, localMax.y));
// scaling factor
float SFValue = vecMax / 6.0f;
uint8_t SFValueFP8;
reinterpret_cast<__nv_fp8_e4m3&>(SFValueFP8) = __nv_fp8_e4m3(SFValue);
SFValue = float(reinterpret_cast<__nv_fp8_e4m3&>(SFValueFP8));
float SFValueInv = (SFValue == 0.0f) ? 0.0f : 1.0f / SFValue;
// convert input to float2 and apply scale
float2 fp2Vals[CVT_FP4_ELTS_PER_THREAD / 2];
#pragma unroll
for (int i = 0; i < CVT_FP4_ELTS_PER_THREAD / 2; i++) {
if constexpr (std::is_same<T, half>::value) {
fp2Vals[i] = __half22float2(in_vec.elts[i]);
} else {
fp2Vals[i] = __bfloat1622float2(in_vec.elts[i]);
}
fp2Vals[i].x = fp2Vals[i].x * SFValueInv;
fp2Vals[i].y = fp2Vals[i].y * SFValueInv;
}
// convert to e2m1
uint32_t e2m1Vals[CVT_FP4_ELTS_PER_THREAD / 8];
#pragma unroll
for (int i = 0; i < CVT_FP4_ELTS_PER_THREAD / 8; i++) {
e2m1Vals[i] = fp32_vec_to_e2m1(fp2Vals + i * 4);
}
// save, do not check range
if constexpr (CVT_FP4_ELTS_PER_THREAD == 8) {
reinterpret_cast<uint32_t*>(output +
batch_id * stride_bz_output +
head_id * stride_h_output +
token_id * stride_seq_output +
(threadIdx.x % NUM_THREADS_PER_TOKEN) * CVT_FP4_ELTS_PER_THREAD / 2)[0] = e2m1Vals[0];
} else {
reinterpret_cast<uint64_t*>(output +
batch_id * stride_bz_output +
head_id * stride_h_output +
token_id * stride_seq_output +
(threadIdx.x % NUM_THREADS_PER_TOKEN) * CVT_FP4_ELTS_PER_THREAD / 2)[0] = reinterpret_cast<uint64_t*>(e2m1Vals)[0];
}
uint8_t* output_sf_save_base = output_sf + batch_id * stride_bz_output_sf + head_id * stride_h_output_sf + (token_id / 64) * 64 * stride_seq_output_sf;
uint32_t token_id_local = token_id % 64;
if constexpr (CVT_FP4_ELTS_PER_THREAD == 16) {
uint32_t col_id_local = threadIdx.x % NUM_THREADS_PER_TOKEN;
uint32_t offset_local = (col_id_local / 4) * 256 + (col_id_local % 4) +
(token_id_local / 16) * 4 + (token_id_local % 16) * 16;
reinterpret_cast<uint8_t*>(output_sf_save_base + offset_local)[0] = SFValueFP8;
} else {
if (threadIdx.x % 2 == 0) {
uint32_t col_id_local = (threadIdx.x % NUM_THREADS_PER_TOKEN) / 2;
uint32_t offset_local = (col_id_local / 4) * 256 + (col_id_local % 4) +
(token_id_local / 16) * 4 + (token_id_local % 16) * 16;
reinterpret_cast<uint8_t*>(output_sf_save_base + offset_local)[0] = SFValueFP8;
}
}
}
template <uint32_t head_dim, uint32_t BLOCK_SIZE, typename T>
__global__ void scaled_fp4_quant_trans_kernel(
const T* input, uint8_t* output, uint8_t* output_sf,
int batch_size, int num_heads, int num_tokens,
int stride_bz_input, int stride_h_input, int stride_seq_input,
int stride_bz_output, int stride_h_output, int stride_d_output,
int stride_bz_output_sf, int stride_h_output_sf, int stride_d_output_sf) {
static_assert(std::is_same<T, half>::value || std::is_same<T, nv_bfloat16>::value, "Only half and bfloat16 input are supported");
using PackedVec = PackedVec<T>;
const int batch_id = blockIdx.y;
const int head_id = blockIdx.z;
const int token_block_id = blockIdx.x;
static_assert(CVT_FP4_ELTS_PER_THREAD == 8 || CVT_FP4_ELTS_PER_THREAD == 16,
"CVT_FP4_ELTS_PER_THREAD must be 8 or 16");
static_assert(sizeof(PackedVec) == sizeof(T) * CVT_FP4_ELTS_PER_THREAD,
"Vec size is not matched.");
constexpr uint32_t NUM_THREADS_PER_TOKEN = head_dim / CVT_FP4_ELTS_PER_THREAD;
constexpr uint32_t NUM_THREADS_PER_SEQ = BLOCK_SIZE / CVT_FP4_ELTS_PER_THREAD;
// load input
const int token_id = token_block_id * BLOCK_SIZE + threadIdx.x / NUM_THREADS_PER_TOKEN;
PackedVec in_vec;
#pragma unroll
for (int i = 0; i < CVT_FP4_ELTS_PER_THREAD / 2; i++) {
reinterpret_cast<uint32_t&>(in_vec.elts[i]) = 0;
}
if (token_id < num_tokens) {
in_vec = reinterpret_cast<PackedVec const*>(input +
batch_id * stride_bz_input + // batch dim
head_id * stride_h_input + // head dim
token_id * stride_seq_input + // seq dim
(threadIdx.x % NUM_THREADS_PER_TOKEN) * CVT_FP4_ELTS_PER_THREAD)[0]; // feature dim
}
// transpose
__shared__ T shared_input[BLOCK_SIZE * head_dim];
reinterpret_cast<PackedVec*>(shared_input)[threadIdx.x] = in_vec;
__syncthreads();
#pragma unroll
for (int i = 0; i < CVT_FP4_ELTS_PER_THREAD / 2; i++) {
in_vec.elts[i].x = shared_input[(threadIdx.x / NUM_THREADS_PER_SEQ) + ((threadIdx.x % NUM_THREADS_PER_SEQ) * CVT_FP4_ELTS_PER_THREAD + 2 * i) * head_dim];
in_vec.elts[i].y = shared_input[(threadIdx.x / NUM_THREADS_PER_SEQ) + ((threadIdx.x % NUM_THREADS_PER_SEQ) * CVT_FP4_ELTS_PER_THREAD + 2 * i + 1) * head_dim];
}
// calculate max of every consecutive 16 elements
auto localMax = __habs2(in_vec.elts[0]);
#pragma unroll
for (int i = 1; i < CVT_FP4_ELTS_PER_THREAD / 2; i++) { // local max
localMax = __hmax2(localMax, __habs2(in_vec.elts[i]));
}
if constexpr (CVT_FP4_ELTS_PER_THREAD == 8) { // shuffle across two threads
localMax = __hmax2(__shfl_xor_sync(0xffffffff, localMax, 1, 32), localMax);
}
float vecMax = float(__hmax(localMax.x, localMax.y));
// scaling factor
float SFValue = vecMax / 6.0f;
uint8_t SFValueFP8;
reinterpret_cast<__nv_fp8_e4m3&>(SFValueFP8) = __nv_fp8_e4m3(SFValue);
SFValue = float(reinterpret_cast<__nv_fp8_e4m3&>(SFValueFP8));
float SFValueInv = (SFValue == 0.0f) ? 0.0f : 1.0f / SFValue;
// convert input to float2 and apply scale
float2 fp2Vals[CVT_FP4_ELTS_PER_THREAD / 2];
#pragma unroll
for (int i = 0; i < CVT_FP4_ELTS_PER_THREAD / 2; i++) {
if constexpr (std::is_same<T, half>::value) {
fp2Vals[i] = __half22float2(in_vec.elts[i]);
} else {
fp2Vals[i] = __bfloat1622float2(in_vec.elts[i]);
}
fp2Vals[i].x = fp2Vals[i].x * SFValueInv;
fp2Vals[i].y = fp2Vals[i].y * SFValueInv;
}
// convert to e2m1
uint32_t e2m1Vals[CVT_FP4_ELTS_PER_THREAD / 8];
#pragma unroll
for (int i = 0; i < CVT_FP4_ELTS_PER_THREAD / 8; i++) {
e2m1Vals[i] = fp32_vec_to_e2m1(fp2Vals + i * 4);
}
// save
if constexpr (CVT_FP4_ELTS_PER_THREAD == 8) {
reinterpret_cast<uint32_t*>(output +
batch_id * stride_bz_output +
head_id * stride_h_output +
(threadIdx.x / NUM_THREADS_PER_SEQ) * stride_d_output +
(token_block_id * BLOCK_SIZE + (threadIdx.x % NUM_THREADS_PER_SEQ) * CVT_FP4_ELTS_PER_THREAD) / 2)[0] = e2m1Vals[0];
} else {
reinterpret_cast<uint64_t*>(output +
batch_id * stride_bz_output +
head_id * stride_h_output +
(threadIdx.x / NUM_THREADS_PER_SEQ) * stride_d_output +
(token_block_id * BLOCK_SIZE + (threadIdx.x % NUM_THREADS_PER_SEQ) * CVT_FP4_ELTS_PER_THREAD) / 2)[0] = reinterpret_cast<uint64_t*>(e2m1Vals)[0];
}
uint8_t *output_sf_save_base = output_sf +
batch_id * stride_bz_output_sf +
head_id * stride_h_output_sf +
(threadIdx.x / NUM_THREADS_PER_SEQ / 64) * 64 * stride_d_output_sf;
uint32_t row_id_local = (threadIdx.x / NUM_THREADS_PER_SEQ) % 64;
if constexpr (CVT_FP4_ELTS_PER_THREAD == 16) {
uint32_t col_id_local = token_block_id * BLOCK_SIZE / CVT_FP4_ELTS_PER_THREAD + threadIdx.x % NUM_THREADS_PER_SEQ;
uint32_t offset_local = (col_id_local / 4) * 256 + (col_id_local % 4) +
(row_id_local / 16) * 4 + (row_id_local % 16) * 16;
reinterpret_cast<uint8_t*>(output_sf_save_base + offset_local)[0] = SFValueFP8;
} else {
if (threadIdx.x % 2 == 0) {
uint32_t col_id_local = token_block_id * BLOCK_SIZE / CVT_FP4_ELTS_PER_THREAD + (threadIdx.x % NUM_THREADS_PER_SEQ) / 2;
uint32_t offset_local = (col_id_local / 4) * 256 + (col_id_local % 4) +
(row_id_local / 16) * 4 + (row_id_local % 16) * 16;
reinterpret_cast<uint8_t*>(output_sf_save_base + offset_local)[0] = SFValueFP8;
}
}
}
void scaled_fp4_quant(torch::Tensor const& input,
torch::Tensor const& output,
torch::Tensor const& output_sf,
int tensor_layout) {
constexpr int BLOCK_SIZE = flash::BLOCK_M;
CHECK_CUDA(input);
CHECK_CUDA(output);
CHECK_CUDA(output_sf);
CHECK_LASTDIM_CONTIGUOUS(input);
CHECK_LASTDIM_CONTIGUOUS(output);
CHECK_LASTDIM_CONTIGUOUS(output_sf);
CHECK_DTYPE(output, at::ScalarType::Byte);
CHECK_DTYPE(output_sf, at::ScalarType::Float8_e4m3fn);
CHECK_DIMS(input, 4);
CHECK_DIMS(output, 4);
CHECK_DIMS(output_sf, 4);
const int batch_size = input.size(0);
const int head_dim = input.size(3);
const int stride_bz_input = input.stride(0);
const int stride_bz_output = output.stride(0);
const int stride_bz_output_sf = output_sf.stride(0);
int num_tokens, num_heads;
int stride_seq_input, stride_seq_output, stride_seq_output_sf;
int stride_h_input, stride_h_output, stride_h_output_sf;
if (tensor_layout == 0) {
num_tokens = input.size(1);
num_heads = input.size(2);
stride_seq_input = input.stride(1);
stride_seq_output = output.stride(1);
stride_seq_output_sf = output_sf.stride(1);
stride_h_input = input.stride(2);
stride_h_output = output.stride(2);
stride_h_output_sf = output_sf.stride(2);
CHECK_SHAPE(output, batch_size, num_tokens, num_heads, head_dim / 2);
CHECK_SHAPE(output_sf, batch_size, num_tokens, num_heads, head_dim / 16);
} else {
num_tokens = input.size(2);
num_heads = input.size(1);
stride_seq_input = input.stride(2);
stride_seq_output = output.stride(2);
stride_seq_output_sf = output_sf.stride(2);
stride_h_input = input.stride(1);
stride_h_output = output.stride(1);
stride_h_output_sf = output_sf.stride(1);
CHECK_SHAPE(output, batch_size, num_heads, num_tokens, head_dim / 2);
CHECK_SHAPE(output_sf, batch_size, num_heads, num_tokens, head_dim / 16);
}
auto input_dtype = input.scalar_type();
auto stream = at::cuda::getCurrentCUDAStream(input.get_device());
DISPATCH_PYTORCH_DTYPE_TO_CTYPE_FP16(input_dtype, c_type, {
DISPATCH_HEAD_DIM(head_dim, HEAD_DIM, {
dim3 block(BLOCK_SIZE * HEAD_DIM / CVT_FP4_ELTS_PER_THREAD, 1, 1);
dim3 grid((num_tokens + BLOCK_SIZE - 1) / BLOCK_SIZE, batch_size, num_heads);
scaled_fp4_quant_kernel<HEAD_DIM, BLOCK_SIZE, false, c_type>
<<<grid, block, 0, stream>>>(
reinterpret_cast<c_type*>(input.data_ptr()),
reinterpret_cast<uint8_t*>(output.data_ptr()),
reinterpret_cast<uint8_t*>(output_sf.data_ptr()),
batch_size, num_heads, num_tokens,
stride_bz_input, stride_h_input, stride_seq_input,
stride_bz_output, stride_h_output, stride_seq_output,
stride_bz_output_sf, stride_h_output_sf, stride_seq_output_sf);
});
});
}
void scaled_fp4_quant_permute(torch::Tensor const& input,
torch::Tensor const& output,
torch::Tensor const& output_sf,
int tensor_layout) {
constexpr int BLOCK_SIZE = flash::BLOCK_M;
CHECK_CUDA(input);
CHECK_CUDA(output);
CHECK_CUDA(output_sf);
CHECK_LASTDIM_CONTIGUOUS(input);
CHECK_LASTDIM_CONTIGUOUS(output);
CHECK_LASTDIM_CONTIGUOUS(output_sf);
CHECK_DTYPE(output, at::ScalarType::Byte);
CHECK_DTYPE(output_sf, at::ScalarType::Float8_e4m3fn);
CHECK_DIMS(input, 4);
CHECK_DIMS(output, 4);
CHECK_DIMS(output_sf, 4);
const int batch_size = input.size(0);
const int head_dim = input.size(3);
const int stride_bz_input = input.stride(0);
const int stride_bz_output = output.stride(0);
const int stride_bz_output_sf = output_sf.stride(0);
int num_tokens, num_heads;
int stride_seq_input, stride_seq_output, stride_seq_output_sf;
int stride_h_input, stride_h_output, stride_h_output_sf;
if (tensor_layout == 0) {
num_tokens = input.size(1);
num_heads = input.size(2);
stride_seq_input = input.stride(1);
stride_seq_output = output.stride(1);
stride_seq_output_sf = output_sf.stride(1);
stride_h_input = input.stride(2);
stride_h_output = output.stride(2);
stride_h_output_sf = output_sf.stride(2);
CHECK_SHAPE(output, batch_size, ((num_tokens + BLOCK_SIZE - 1) / BLOCK_SIZE) * BLOCK_SIZE, num_heads, head_dim / 2);
CHECK_SHAPE(output_sf, batch_size, ((num_tokens + BLOCK_SIZE - 1) / BLOCK_SIZE) * BLOCK_SIZE, num_heads, head_dim / 16);
} else {
num_tokens = input.size(2);
num_heads = input.size(1);
stride_seq_input = input.stride(2);
stride_seq_output = output.stride(2);
stride_seq_output_sf = output_sf.stride(2);
stride_h_input = input.stride(1);
stride_h_output = output.stride(1);
stride_h_output_sf = output_sf.stride(1);
CHECK_SHAPE(output, batch_size, num_heads, ((num_tokens + BLOCK_SIZE - 1) / BLOCK_SIZE) * BLOCK_SIZE, head_dim / 2);
CHECK_SHAPE(output_sf, batch_size, num_heads, ((num_tokens + BLOCK_SIZE - 1) / BLOCK_SIZE) * BLOCK_SIZE, head_dim / 16);
}
auto input_dtype = input.scalar_type();
auto stream = at::cuda::getCurrentCUDAStream(input.get_device());
DISPATCH_PYTORCH_DTYPE_TO_CTYPE_FP16(input_dtype, c_type, {
DISPATCH_HEAD_DIM(head_dim, HEAD_DIM, {
constexpr int BLOCK_SIZE = flash::BLOCK_M;
dim3 block(BLOCK_SIZE * HEAD_DIM / CVT_FP4_ELTS_PER_THREAD, 1, 1);
dim3 grid((num_tokens + BLOCK_SIZE - 1) / BLOCK_SIZE, batch_size, num_heads);
scaled_fp4_quant_kernel<HEAD_DIM, BLOCK_SIZE, true, c_type>
<<<grid, block, 0, stream>>>(
reinterpret_cast<c_type*>(input.data_ptr()),
reinterpret_cast<uint8_t*>(output.data_ptr()),
reinterpret_cast<uint8_t*>(output_sf.data_ptr()),
batch_size, num_heads, num_tokens,
stride_bz_input, stride_h_input, stride_seq_input,
stride_bz_output, stride_h_output, stride_seq_output,
stride_bz_output_sf, stride_h_output_sf, stride_seq_output_sf);
});
});
}
void scaled_fp4_quant_trans(torch::Tensor const& input,
torch::Tensor const& output,
torch::Tensor const& output_sf,
int tensor_layout) {
constexpr int BLOCK_SIZE = flash::BLOCK_M;
CHECK_CUDA(input);
CHECK_CUDA(output);
CHECK_CUDA(output_sf);
CHECK_LASTDIM_CONTIGUOUS(input);
CHECK_LASTDIM_CONTIGUOUS(output);
CHECK_LASTDIM_CONTIGUOUS(output_sf);
CHECK_DTYPE(output, at::ScalarType::Byte);
CHECK_DTYPE(output_sf, at::ScalarType::Float8_e4m3fn);
CHECK_DIMS(input, 4);
CHECK_DIMS(output, 4);
CHECK_DIMS(output_sf, 4);
const int batch_size = input.size(0);
const int head_dim = input.size(3);
const int stride_bz_input = input.stride(0);
const int stride_bz_output = output.stride(0);
const int stride_bz_output_sf = output_sf.stride(0);
int num_tokens, num_heads;
int stride_seq_input;
int stride_d_output, stride_d_output_sf;
int stride_h_input, stride_h_output, stride_h_output_sf;
if (tensor_layout == 0) {
num_tokens = input.size(1);
num_heads = input.size(2);
stride_seq_input = input.stride(1);
stride_d_output = output.stride(1);
stride_d_output_sf = output_sf.stride(1);
stride_h_input = input.stride(2);
stride_h_output = output.stride(2);
stride_h_output_sf = output_sf.stride(2);
CHECK_SHAPE(output, batch_size, head_dim, num_heads, ((num_tokens + BLOCK_SIZE - 1) / BLOCK_SIZE) * BLOCK_SIZE / 2);
CHECK_SHAPE(output_sf, batch_size, head_dim, num_heads, ((num_tokens + BLOCK_SIZE - 1) / BLOCK_SIZE) * BLOCK_SIZE / 16);
} else {
num_tokens = input.size(2);
num_heads = input.size(1);
stride_seq_input = input.stride(2);
stride_d_output = output.stride(2);
stride_d_output_sf = output_sf.stride(2);
stride_h_input = input.stride(1);
stride_h_output = output.stride(1);
stride_h_output_sf = output_sf.stride(1);
CHECK_SHAPE(output, batch_size, num_heads, head_dim, ((num_tokens + BLOCK_SIZE - 1) / BLOCK_SIZE) * BLOCK_SIZE / 2);
CHECK_SHAPE(output_sf, batch_size, num_heads, head_dim, ((num_tokens + BLOCK_SIZE - 1) / BLOCK_SIZE) * BLOCK_SIZE / 16);
}
auto input_dtype = input.scalar_type();
auto stream = at::cuda::getCurrentCUDAStream(input.get_device());
DISPATCH_PYTORCH_DTYPE_TO_CTYPE_FP16(input_dtype, c_type, {
DISPATCH_HEAD_DIM(head_dim, HEAD_DIM, {
dim3 block(BLOCK_SIZE * HEAD_DIM / CVT_FP4_ELTS_PER_THREAD, 1, 1);
dim3 grid((num_tokens + BLOCK_SIZE - 1) / BLOCK_SIZE, batch_size, num_heads);
scaled_fp4_quant_trans_kernel<HEAD_DIM, BLOCK_SIZE, c_type>
<<<grid, block, 0, stream>>>(
reinterpret_cast<c_type*>(input.data_ptr()),
reinterpret_cast<uint8_t*>(output.data_ptr()),
reinterpret_cast<uint8_t*>(output_sf.data_ptr()),
batch_size, num_heads, num_tokens,
stride_bz_input, stride_h_input, stride_seq_input,
stride_bz_output, stride_h_output, stride_d_output,
stride_bz_output_sf, stride_h_output_sf, stride_d_output_sf);
});
});
}
PYBIND11_MODULE(TORCH_EXTENSION_NAME, m) {
m.def("scaled_fp4_quant", &scaled_fp4_quant);
m.def("scaled_fp4_quant_permute", &scaled_fp4_quant_permute);
m.def("scaled_fp4_quant_trans", &scaled_fp4_quant_trans);
}
+14
View File
@@ -0,0 +1,14 @@
"""Make local benchmark scripts runnable from common repo entrypoints."""
from pathlib import Path
import sys
BENCHMARKS_DIR = Path(__file__).resolve().parent
KERNEL_ROOT = BENCHMARKS_DIR.parent
REPO_ROOT = KERNEL_ROOT.parent
for path in (KERNEL_ROOT, REPO_ROOT):
path_str = str(path)
if path_str not in sys.path:
sys.path.insert(0, path_str)
@@ -0,0 +1,287 @@
import sys
import traceback
import _bootstrap # noqa: F401
import torch
from attn_qat_infer.api import (
blockscaled_fp4_attn,
preprocess_qkv,
scale_and_quant_fp4,
scale_and_quant_fp4_permute,
scale_and_quant_fp4_transpose,
)
from attn_qat_infer.quantization.bench.bench_utils import bench
def calculate_attention_flops(batch_size, num_heads, seq_len_q, seq_len_k, head_dim, is_causal=False):
"""Calculate FLOPs for attention (FlashAttention standard - matmuls only)."""
f = 4 * batch_size * num_heads * seq_len_q * seq_len_k * head_dim
if is_causal:
f = f // 2
return f
def benchmark_blockscaled_fp4_attn(batch_size, num_heads, seq_len, head_dim,
is_causal=False, dtype=torch.bfloat16,
per_block_mean=True, single_level_p_quant=False,
num_warmups=100, num_tests=1000):
"""
Benchmark blockscaled_fp4_attn function (excluding quantization overhead).
This benchmarks ONLY the core FP4 attention kernel, with pre-quantized inputs.
The quantization step is performed once before benchmarking.
Args:
batch_size: Batch size
num_heads: Number of attention heads
seq_len: Sequence length (same for Q, K, V)
head_dim: Head dimension
is_causal: Whether to use causal masking
dtype: Data type (torch.bfloat16 or torch.float16)
per_block_mean: Whether to use per-block mean for Q smoothing
single_level_p_quant: If True, use single-level quantization for P matrix
num_warmups: Number of warmup iterations
num_tests: Number of test iterations
Returns:
dict with performance metrics
"""
device = 'cuda'
if not torch.cuda.is_available():
raise RuntimeError("CUDA is not available. This benchmark requires a CUDA device.")
# Create input tensors
q = torch.randn(batch_size, num_heads, seq_len, head_dim,
device=device, dtype=dtype)
k = torch.randn(batch_size, num_heads, seq_len, head_dim,
device=device, dtype=dtype)
v = torch.randn(batch_size, num_heads, seq_len, head_dim,
device=device, dtype=dtype)
# Pre-process and quantize inputs (done once, not included in benchmark)
is_bf16 = dtype == torch.bfloat16
KL = k.size(2)
q_processed, k_processed, v_processed, delta_s = preprocess_qkv(q, k, v, per_block_mean)
qlist = scale_and_quant_fp4(q_processed)
klist = scale_and_quant_fp4_permute(k_processed)
vlist = scale_and_quant_fp4_transpose(v_processed)
# Synchronize to ensure quantization is complete
torch.cuda.synchronize()
# Create closure for benchmarking (only the attention kernel)
def run_attention():
return blockscaled_fp4_attn(
qlist, klist, vlist,
delta_s,
KL,
is_causal=is_causal,
per_block_mean=per_block_mean,
is_bf16=is_bf16,
single_level_p_quant=single_level_p_quant
)
# Benchmark using the bench utility (handles warmup and timing)
avg_time_ms = bench(run_attention, num_warmups=num_warmups, num_tests=num_tests)
avg_time_s = avg_time_ms / 1000.0
# Calculate FLOPs (use processed sequence length after padding)
processed_seq_len = q_processed.size(2)
total_flops = calculate_attention_flops(
batch_size, num_heads, processed_seq_len, processed_seq_len, head_dim, is_causal
)
# Calculate TFLOPs
tflops = total_flops / (avg_time_s * 1e12)
# Calculate throughput (tokens/sec) - use original seq_len for meaningful metric
tokens_per_second = (batch_size * seq_len) / avg_time_s
return {
'batch_size': batch_size,
'num_heads': num_heads,
'seq_len': seq_len,
'processed_seq_len': processed_seq_len,
'head_dim': head_dim,
'is_causal': is_causal,
'dtype': str(dtype),
'per_block_mean': per_block_mean,
'single_level_p_quant': single_level_p_quant,
'avg_time_ms': avg_time_ms,
'avg_time_s': avg_time_s,
'total_flops': total_flops,
'tflops': tflops,
'tokens_per_second': tokens_per_second,
}
def print_results(results):
"""Print benchmark results in a formatted table."""
print("\n" + "="*100)
print("blockscaled_fp4_attn Benchmark Results (Kernel Only, No Quantization)")
print("="*100)
print(f"Configuration:")
print(f" Batch Size: {results['batch_size']}")
print(f" Num Heads: {results['num_heads']}")
print(f" Sequence Length: {results['seq_len']}")
print(f" Processed Seq Length: {results['processed_seq_len']} (after padding)")
print(f" Head Dimension: {results['head_dim']}")
print(f" Causal: {results['is_causal']}")
print(f" Data Type: {results['dtype']}")
print(f" Per Block Mean: {results['per_block_mean']}")
print(f" Single Level P Quant: {results['single_level_p_quant']}")
print(f"\nPerformance:")
print(f" Average Time: {results['avg_time_ms']:.3f} ms")
print(f" Total FLOPs: {results['total_flops']/1e12:.4f} TFLOPs (theoretical)")
print(f" Throughput: {results['tflops']:.4f} TFLOPs/s")
print(f" Tokens/sec: {results['tokens_per_second']:,.0f}")
print("="*100 + "\n")
sys.stdout.flush()
def run_benchmark_suite():
"""Run a comprehensive benchmark suite with various configurations."""
print("Starting blockscaled_fp4_attn Benchmark Suite (Kernel Only)...")
print(f"CUDA Device: {torch.cuda.get_device_name(0)}")
print(f"CUDA Version: {torch.version.cuda}")
print(f"PyTorch Version: {torch.__version__}\n")
print("Note: This benchmark measures only the FP4 attention kernel,")
print(" excluding quantization overhead.\n")
sys.stdout.flush()
# Default configurations to test
# (batch_size, num_heads, seq_len, head_dim, is_causal, dtype)
configs = [
(1, 16, 512, 64, False, torch.bfloat16),
(1, 16, 1024, 64, False, torch.bfloat16),
(1, 16, 2048, 64, False, torch.bfloat16),
(1, 16, 4096, 64, False, torch.bfloat16),
(1, 16, 8192, 64, False, torch.bfloat16),
(1, 16, 16384, 64, False, torch.bfloat16),
(1, 16, 512, 128, False, torch.bfloat16),
(1, 16, 1024, 128, False, torch.bfloat16),
(1, 16, 2048, 128, False, torch.bfloat16),
(1, 16, 4096, 128, False, torch.bfloat16),
(1, 16, 8192, 128, False, torch.bfloat16),
(1, 16, 16384, 128, False, torch.bfloat16),
(1, 32, 512, 64, False, torch.bfloat16),
(1, 32, 1024, 64, False, torch.bfloat16),
(1, 32, 2048, 64, False, torch.bfloat16),
(1, 32, 4096, 64, False, torch.bfloat16),
(1, 32, 8192, 64, False, torch.bfloat16),
(1, 32, 16384, 64, False, torch.bfloat16),
(1, 32, 512, 128, False, torch.bfloat16),
(1, 32, 1024, 128, False, torch.bfloat16),
(1, 32, 2048, 128, False, torch.bfloat16),
(1, 32, 4096, 128, False, torch.bfloat16),
(1, 32, 8192, 128, False, torch.bfloat16),
(1, 32, 16384, 128, False, torch.bfloat16),
]
all_results = []
for config in configs:
batch_size, num_heads, seq_len, head_dim, is_causal, dtype = config
print(f"\nBenchmarking: B={batch_size}, H={num_heads}, L={seq_len}, D={head_dim}, "
f"Causal={is_causal}, dtype={dtype}...")
sys.stdout.flush()
try:
results = benchmark_blockscaled_fp4_attn(
batch_size=batch_size,
num_heads=num_heads,
seq_len=seq_len,
head_dim=head_dim,
is_causal=is_causal,
dtype=dtype,
num_warmups=10,
num_tests=50
)
print_results(results)
all_results.append(results)
except Exception as e:
print(f"Error benchmarking configuration {config}:")
print(f" Exception: {e}")
traceback.print_exc()
sys.stdout.flush()
continue
# Print summary table
print("\n" + "="*120)
print("Summary Table (blockscaled_fp4_attn Kernel Only)")
print("="*120)
print(f"{'B':<4} {'H':<4} {'L':<6} {'D':<4} {'Causal':<7} {'Time (ms)':<12} {'TFLOPs/s':<12} {'Tokens/s':<15}")
print("-"*120)
for r in all_results:
print(f"{r['batch_size']:<4} {r['num_heads']:<4} {r['seq_len']:<6} {r['head_dim']:<4} "
f"{str(r['is_causal']):<7} {r['avg_time_ms']:<12.3f} {r['tflops']:<12.4f} "
f"{r['tokens_per_second']:<15,.0f}")
print("="*120 + "\n")
sys.stdout.flush()
if __name__ == "__main__":
import argparse
parser = argparse.ArgumentParser(description='Benchmark blockscaled_fp4_attn kernel (excluding quantization)')
parser.add_argument('--batch-size', type=int, default=None, help='Batch size')
parser.add_argument('--num-heads', type=int, default=None, help='Number of attention heads')
parser.add_argument('--seq-len', type=int, default=None, help='Sequence length')
parser.add_argument('--head-dim', type=int, default=None, help='Head dimension')
parser.add_argument('--causal', action='store_true', help='Use causal attention')
parser.add_argument('--dtype', type=str, default='bfloat16', choices=['bfloat16', 'float16'],
help='Data type')
parser.add_argument('--per-block-mean', action='store_true', default=True,
help='Use per-block mean for Q smoothing (default: True)')
parser.add_argument('--no-per-block-mean', action='store_false', dest='per_block_mean',
help='Disable per-block mean for Q smoothing')
parser.add_argument('--single-level-p-quant', action='store_true', default=False,
help='Use single-level P quantization (default: False)')
parser.add_argument('--two-level-p-quant', action='store_false', dest='single_level_p_quant',
help='Use two-level P quantization')
parser.add_argument('--num-warmups', type=int, default=10, help='Number of warmup iterations')
parser.add_argument('--num-tests', type=int, default=50, help='Number of test iterations')
parser.add_argument('--suite', action='store_true', help='Run full benchmark suite')
args = parser.parse_args()
dtype_map = {
'bfloat16': torch.bfloat16,
'float16': torch.float16
}
if args.suite:
run_benchmark_suite()
elif args.batch_size and args.num_heads and args.seq_len and args.head_dim:
results = benchmark_blockscaled_fp4_attn(
batch_size=args.batch_size,
num_heads=args.num_heads,
seq_len=args.seq_len,
head_dim=args.head_dim,
is_causal=args.causal,
dtype=dtype_map[args.dtype],
per_block_mean=args.per_block_mean,
single_level_p_quant=args.single_level_p_quant,
num_warmups=args.num_warmups,
num_tests=args.num_tests
)
print_results(results)
else:
print("Running default benchmark suite. Use --suite for full suite or provide all parameters.")
print(
"Example: python benchmarks/benchmark_blockscaled_fp4_attn.py "
"--batch-size 1 --num-heads 16 --seq-len 4096 --head-dim 128"
)
print("\nNote: This benchmark measures only the FP4 attention kernel,")
print(" excluding quantization overhead.")
sys.stdout.flush()
run_benchmark_suite()
@@ -0,0 +1,380 @@
import argparse
import sys
from typing import Dict, List, Optional
import _bootstrap # noqa: F401
import matplotlib
matplotlib.use('Agg') # Use non-interactive backend for server environments
import matplotlib.pyplot as plt
import numpy as np
import torch
from flash_attn import flash_attn_func
from attn_qat_infer.quantization.bench.bench_utils import bench
# Import SageAttn components for direct control
from attn_qat_infer.api import (
preprocess_qkv,
scale_and_quant_fp4,
scale_and_quant_fp4_permute,
scale_and_quant_fp4_transpose,
blockscaled_fp4_attn
)
def calculate_attention_flops(batch_size, num_heads, seq_len_q, seq_len_k, head_dim, is_causal=False):
"""Calculate FLOPs for attention (FlashAttention standard - matmuls only)."""
f = 4 * batch_size * num_heads * seq_len_q * seq_len_k * head_dim
if is_causal:
f = f // 2
return f
def sageattn_blackwell_configurable(q, k, v, is_causal=False, per_block_mean=True,
single_level_p_quant=True,
enable_smoothing_q=False, enable_smoothing_k=False):
"""
Configurable SageAttention3 Blackwell kernel with explicit smoothing control.
Args:
q: Query tensor [B, H, L, D]
k: Key tensor [B, H, L, D]
v: Value tensor [B, H, L, D]
is_causal: Whether to use causal masking
per_block_mean: Whether to use per-block mean for Q smoothing
single_level_p_quant: If True, use single-level quantization for P matrix
enable_smoothing_q: Enable Q smoothing
enable_smoothing_k: Enable K smoothing
Returns:
Output tensor [B, H, L, D]
"""
QL = q.size(2)
KL = k.size(2)
is_bf16 = q.dtype == torch.bfloat16
# Preprocess with explicit smoothing control
q, k, v, delta_s = preprocess_qkv(q, k, v, per_block_mean, enable_smoothing_q, enable_smoothing_k)
qlist_from_cuda = scale_and_quant_fp4(q)
klist_from_cuda = scale_and_quant_fp4_permute(k)
vlist_from_cuda = scale_and_quant_fp4_transpose(v)
o_fp4 = blockscaled_fp4_attn(
qlist_from_cuda,
klist_from_cuda,
vlist_from_cuda,
delta_s,
KL,
is_causal,
per_block_mean,
is_bf16,
single_level_p_quant
)[0][:, :, :QL, :].contiguous()
return o_fp4
def benchmark_flashattn2(batch_size, num_heads, seq_len, head_dim,
is_causal=False, dtype=torch.bfloat16,
num_warmups=10, num_tests=50):
"""Benchmark FlashAttention2."""
device = 'cuda'
# FlashAttention2 expects (batch, seq_len, num_heads, head_dim)
q = torch.randn(batch_size, seq_len, num_heads, head_dim, device=device, dtype=dtype)
k = torch.randn(batch_size, seq_len, num_heads, head_dim, device=device, dtype=dtype)
v = torch.randn(batch_size, seq_len, num_heads, head_dim, device=device, dtype=dtype)
def run_attention():
return flash_attn_func(q, k, v, causal=is_causal)
avg_time_ms = bench(run_attention, num_warmups=num_warmups, num_tests=num_tests)
return avg_time_ms
def benchmark_sageattn3(batch_size, num_heads, seq_len, head_dim,
is_causal=False, dtype=torch.bfloat16,
per_block_mean=True, single_level_p_quant=False,
enable_smoothing_q=True, enable_smoothing_k=True,
num_warmups=10, num_tests=50):
"""Benchmark SageAttention3 with configurable smoothing."""
device = 'cuda'
# SageAttn expects (batch, num_heads, seq_len, head_dim)
q = torch.randn(batch_size, num_heads, seq_len, head_dim, device=device, dtype=dtype)
k = torch.randn(batch_size, num_heads, seq_len, head_dim, device=device, dtype=dtype)
v = torch.randn(batch_size, num_heads, seq_len, head_dim, device=device, dtype=dtype)
def run_attention():
return sageattn_blackwell_configurable(
q, k, v,
is_causal=is_causal,
per_block_mean=per_block_mean,
single_level_p_quant=single_level_p_quant,
enable_smoothing_q=enable_smoothing_q,
enable_smoothing_k=enable_smoothing_k
)
avg_time_ms = bench(run_attention, num_warmups=num_warmups, num_tests=num_tests)
return avg_time_ms
def time_to_tflops(time_ms, batch_size, num_heads, seq_len, head_dim, is_causal=False):
"""Convert time to TFLOPs (Tera FLOPs per Second)."""
total_flops = calculate_attention_flops(batch_size, num_heads, seq_len, seq_len, head_dim, is_causal)
time_s = time_ms / 1000.0
tflops = total_flops / (time_s * 1e12)
return tflops
def run_benchmark_suite(head_dim=64, is_causal=False, num_heads=12, batch_size=1,
num_warmups=10, num_tests=50,
seq_lens=None, output_file="benchmark_attention.png"):
"""
Run comprehensive benchmark suite and generate plot.
Args:
head_dim: Head dimension (64 or 128)
is_causal: Whether to use causal attention
num_heads: Number of attention heads
batch_size: Batch size
num_warmups: Number of warmup iterations
num_tests: Number of test iterations
seq_lens: List of sequence lengths to test
output_file: Output plot filename
"""
if seq_lens is None:
seq_lens = [1024, 2048, 4096, 8192, 16384, 32768]
device_name = torch.cuda.get_device_name(0)
# Extract short name (e.g., "RTX5090" from full name)
short_name = device_name.split()[-1] if 'RTX' in device_name or 'A100' in device_name else device_name[:20]
print(f"Starting Combined Attention Benchmark Suite...")
print(f"CUDA Device: {device_name}")
print(f"CUDA Version: {torch.version.cuda}")
print(f"PyTorch Version: {torch.__version__}")
print(f"Head Dim: {head_dim}, Causal: {is_causal}, Num Heads: {num_heads}, Batch Size: {batch_size}")
print("="*80)
sys.stdout.flush()
# Results storage: {method_name: {seq_len: tflops}}
results: Dict[str, Dict[int, Optional[float]]] = {
'FlashAttn': {},
'SageAttn3': {},
'FP4': {},
}
dtype = torch.bfloat16
for seq_len in seq_lens:
print(f"\n--- Sequence Length: {seq_len} ---")
sys.stdout.flush()
# FlashAttention2
print(f" Benchmarking FlashAttn2...", end=" ")
sys.stdout.flush()
try:
time_ms = benchmark_flashattn2(
batch_size, num_heads, seq_len, head_dim,
is_causal=is_causal, dtype=dtype,
num_warmups=num_warmups, num_tests=num_tests
)
tflops = time_to_tflops(time_ms, batch_size, num_heads, seq_len, head_dim, is_causal)
results['FlashAttn'][seq_len] = tflops
print(f"{tflops:.0f} TFLOPs ({time_ms:.3f} ms)")
except Exception as e:
print(f"OOM or Error: {e}")
results['FlashAttn'][seq_len] = None
sys.stdout.flush()
# SageAttn3 (with smoothing: single_level_p_quant=False, enable_smoothing_q=True, enable_smoothing_k=True)
print(f" Benchmarking SageAttn3 (smoothing ON)...", end=" ")
sys.stdout.flush()
try:
time_ms = benchmark_sageattn3(
batch_size, num_heads, seq_len, head_dim,
is_causal=is_causal, dtype=dtype,
per_block_mean=True,
single_level_p_quant=False,
enable_smoothing_q=True,
enable_smoothing_k=True,
num_warmups=num_warmups, num_tests=num_tests
)
tflops = time_to_tflops(time_ms, batch_size, num_heads, seq_len, head_dim, is_causal)
results['SageAttn3'][seq_len] = tflops
print(f"{tflops:.0f} TFLOPs ({time_ms:.3f} ms)")
except Exception as e:
print(f"OOM or Error: {e}")
results['SageAttn3'][seq_len] = None
sys.stdout.flush()
# FP4 (no smoothing: single_level_p_quant=True, enable_smoothing_q=False, enable_smoothing_k=False)
print(f" Benchmarking FP4 (smoothing OFF)...", end=" ")
sys.stdout.flush()
try:
time_ms = benchmark_sageattn3(
batch_size, num_heads, seq_len, head_dim,
is_causal=is_causal, dtype=dtype,
per_block_mean=True,
single_level_p_quant=True,
enable_smoothing_q=False,
enable_smoothing_k=False,
num_warmups=num_warmups, num_tests=num_tests
)
tflops = time_to_tflops(time_ms, batch_size, num_heads, seq_len, head_dim, is_causal)
results['FP4'][seq_len] = tflops
print(f"{tflops:.0f} TFLOPs ({time_ms:.3f} ms)")
except Exception as e:
print(f"OOM or Error: {e}")
results['FP4'][seq_len] = None
sys.stdout.flush()
# Print summary table
print("\n" + "="*100)
print("Summary Table (TFLOPs)")
print("="*100)
header = f"{'SeqLen':<10}"
for method in results.keys():
header += f"{method:<15}"
print(header)
print("-"*100)
for seq_len in seq_lens:
row = f"{seq_len:<10}"
for method in results.keys():
val = results[method].get(seq_len)
if val is not None:
row += f"{val:<15.0f}"
else:
row += f"{'OOM':<15}"
print(row)
print("="*100)
sys.stdout.flush()
# Generate plot
generate_plot(results, seq_lens, head_dim, is_causal, short_name, output_file)
return results
def generate_plot(results: Dict[str, Dict[int, Optional[float]]],
seq_lens: List[int],
head_dim: int,
is_causal: bool,
device_name: str,
output_file: str):
"""Generate bar plot comparing attention implementations."""
# Prepare data
methods = list(results.keys())
x_labels = [f"{sl//1024}K" for sl in seq_lens]
# Colors for each method (red, blue, green scheme)
colors = {
'FlashAttn': '#1E90FF', # Blue (Dodger Blue)
'SageAttn3': '#228B22', # Green (Forest Green)
'FP4': '#DC143C', # Red (Crimson)
}
# Number of methods and positions
n_methods = len(methods)
n_positions = len(seq_lens)
# Bar width and positions
bar_width = 0.25
x = np.arange(n_positions)
# Create figure
fig, ax = plt.subplots(figsize=(12, 6))
# Plot bars for each method
for i, method in enumerate(methods):
values = []
for seq_len in seq_lens:
val = results[method].get(seq_len)
values.append(val if val is not None else 0)
offset = (i - n_methods/2 + 0.5) * bar_width
bars = ax.bar(x + offset, values, bar_width,
label=method, color=colors.get(method, f'C{i}'),
edgecolor='black', linewidth=0.5)
# Add value labels on top of bars
for bar, val, seq_len in zip(bars, values, seq_lens):
if results[method].get(seq_len) is None:
label = 'OOM'
else:
label = f'{int(val)}'
height = bar.get_height()
ax.annotate(label,
xy=(bar.get_x() + bar.get_width() / 2, height),
xytext=(0, 3), # 3 points vertical offset
textcoords="offset points",
ha='center', va='bottom',
fontsize=8, rotation=0)
# Customize plot
ax.set_xlabel('Sequence Length', fontsize=12, fontweight='bold')
ax.set_ylabel('Speed (TFLOPs)', fontsize=12, fontweight='bold')
ax.set_title(f'{device_name}, (Head dim = {head_dim}, causal = {is_causal})', fontsize=14, fontweight='bold')
ax.set_xticks(x)
ax.set_xticklabels(x_labels)
legend = ax.legend(loc='upper left', ncol=len(methods), fontsize=10)
# Make legend text bold
for text in legend.get_texts():
text.set_fontweight('bold')
# Set y-axis to start from 0
ax.set_ylim(bottom=0)
# Add grid for readability
ax.yaxis.grid(True, linestyle='--', alpha=0.7)
ax.set_axisbelow(True)
# Tight layout
plt.tight_layout()
# Save plot
plt.savefig(output_file, dpi=150, bbox_inches='tight')
print(f"\nPlot saved to: {output_file}")
# Also save as PDF for high quality
pdf_file = output_file.rsplit('.', 1)[0] + '.pdf'
plt.savefig(pdf_file, bbox_inches='tight')
print(f"PDF saved to: {pdf_file}")
plt.close()
if __name__ == "__main__":
parser = argparse.ArgumentParser(description='Combined Attention Benchmark (FlashAttn2 vs SageAttn3 vs FP4)')
parser.add_argument('--batch-size', type=int, default=1, help='Batch size')
parser.add_argument('--num-heads', type=int, default=16, help='Number of attention heads')
parser.add_argument('--head-dim', type=int, default=64, choices=[64, 128], help='Head dimension')
parser.add_argument('--causal', action='store_true', help='Use causal attention')
parser.add_argument('--num-warmups', type=int, default=10, help='Number of warmup iterations')
parser.add_argument('--num-tests', type=int, default=50, help='Number of test iterations')
parser.add_argument('--seq-lens', type=int, nargs='+',
default=[1024, 2048, 4096, 8192, 16384, 32768],
help='Sequence lengths to benchmark')
parser.add_argument('--output', type=str, default='benchmark_attention.png',
help='Output plot filename')
args = parser.parse_args()
if not torch.cuda.is_available():
raise RuntimeError("CUDA is not available. This benchmark requires a CUDA device.")
run_benchmark_suite(
head_dim=args.head_dim,
is_causal=args.causal,
num_heads=args.num_heads,
batch_size=args.batch_size,
num_warmups=args.num_warmups,
num_tests=args.num_tests,
seq_lens=args.seq_lens,
output_file=args.output
)
@@ -0,0 +1,234 @@
import sys
import traceback
import _bootstrap # noqa: F401
import torch
from flash_attn import flash_attn_func
from attn_qat_infer.quantization.bench.bench_utils import bench
def calculate_attention_flops(batch_size, num_heads, seq_len_q, seq_len_k, head_dim, is_causal=False):
"""Calculate FLOPs for attention (FlashAttention standard - matmuls only)."""
f = 4 * batch_size * num_heads * seq_len_q * seq_len_k * head_dim
if is_causal:
f = f // 2
return f
def benchmark_flashattn2(batch_size, num_heads, seq_len, head_dim,
is_causal=False, dtype=torch.bfloat16,
num_warmups=100, num_tests=1000):
"""
Benchmark FlashAttention2 and return performance metrics.
Args:
batch_size: Batch size
num_heads: Number of attention heads
seq_len: Sequence length (same for Q, K, V)
head_dim: Head dimension
is_causal: Whether to use causal masking
dtype: Data type (torch.bfloat16 or torch.float16)
num_warmups: Number of warmup iterations
num_tests: Number of test iterations
Returns:
dict with performance metrics
"""
device = 'cuda'
if not torch.cuda.is_available():
raise RuntimeError("CUDA is not available. This benchmark requires a CUDA device.")
# Create input tensors - FlashAttention2 expects (batch, seq_len, num_heads, head_dim)
q = torch.randn(batch_size, seq_len, num_heads, head_dim,
device=device, dtype=dtype)
k = torch.randn(batch_size, seq_len, num_heads, head_dim,
device=device, dtype=dtype)
v = torch.randn(batch_size, seq_len, num_heads, head_dim,
device=device, dtype=dtype)
# Create closure for benchmarking
def run_attention():
return flash_attn_func(
q, k, v,
causal=is_causal,
)
# Benchmark using the bench utility (handles warmup and timing)
avg_time_ms = bench(run_attention, num_warmups=num_warmups, num_tests=num_tests)
avg_time_s = avg_time_ms / 1000.0
# Calculate FLOPs
total_flops = calculate_attention_flops(
batch_size, num_heads, seq_len, seq_len, head_dim, is_causal
)
# Calculate TFLOPs
tflops = total_flops / (avg_time_s * 1e12)
# Calculate throughput (tokens/sec)
tokens_per_second = (batch_size * seq_len) / avg_time_s
return {
'batch_size': batch_size,
'num_heads': num_heads,
'seq_len': seq_len,
'head_dim': head_dim,
'is_causal': is_causal,
'dtype': str(dtype),
'avg_time_ms': avg_time_ms,
'avg_time_s': avg_time_s,
'total_flops': total_flops,
'tflops': tflops,
'tokens_per_second': tokens_per_second,
}
def print_results(results):
"""Print benchmark results in a formatted table."""
print("\n" + "="*100)
print("FlashAttention2 Benchmark Results")
print("="*100)
print(f"Configuration:")
print(f" Batch Size: {results['batch_size']}")
print(f" Num Heads: {results['num_heads']}")
print(f" Sequence Length: {results['seq_len']}")
print(f" Head Dimension: {results['head_dim']}")
print(f" Causal: {results['is_causal']}")
print(f" Data Type: {results['dtype']}")
print(f"\nPerformance:")
print(f" Average Time: {results['avg_time_ms']:.3f} ms")
print(f" Total FLOPs: {results['total_flops']/1e12:.4f} TFLOPs (theoretical)")
print(f" Throughput: {results['tflops']:.4f} TFLOPs/s")
print(f" Tokens/sec: {results['tokens_per_second']:,.0f}")
print("="*100 + "\n")
sys.stdout.flush()
def run_benchmark_suite():
"""Run a comprehensive benchmark suite with various configurations."""
print("Starting FlashAttention2 Benchmark Suite...")
print(f"CUDA Device: {torch.cuda.get_device_name(0)}")
print(f"CUDA Version: {torch.version.cuda}")
print(f"PyTorch Version: {torch.__version__}\n")
sys.stdout.flush()
# Default configurations to test
# (batch_size, num_heads, seq_len, head_dim, is_causal, dtype)
configs = [
(1, 16, 1024, 64, False, torch.bfloat16),
(1, 16, 2048, 64, False, torch.bfloat16),
(1, 16, 4096, 64, False, torch.bfloat16),
(1, 16, 8192, 64, False, torch.bfloat16),
(1, 16, 16384, 64, False, torch.bfloat16),
(1, 16, 1024, 128, False, torch.bfloat16),
(1, 16, 2048, 128, False, torch.bfloat16),
(1, 16, 4096, 128, False, torch.bfloat16),
(1, 16, 8192, 128, False, torch.bfloat16),
(1, 16, 16384, 128, False, torch.bfloat16),
(1, 32, 1024, 64, False, torch.bfloat16),
(1, 32, 2048, 64, False, torch.bfloat16),
(1, 32, 4096, 64, False, torch.bfloat16),
(1, 32, 8192, 64, False, torch.bfloat16),
(1, 32, 16384, 64, False, torch.bfloat16),
(1, 32, 1024, 128, False, torch.bfloat16),
(1, 32, 2048, 128, False, torch.bfloat16),
(1, 32, 4096, 128, False, torch.bfloat16),
(1, 32, 8192, 128, False, torch.bfloat16),
(1, 32, 16384, 128, False, torch.bfloat16),
]
all_results = []
for config in configs:
batch_size, num_heads, seq_len, head_dim, is_causal, dtype = config
print(f"\nBenchmarking: B={batch_size}, H={num_heads}, L={seq_len}, D={head_dim}, "
f"Causal={is_causal}, dtype={dtype}...")
sys.stdout.flush()
try:
results = benchmark_flashattn2(
batch_size=batch_size,
num_heads=num_heads,
seq_len=seq_len,
head_dim=head_dim,
is_causal=is_causal,
dtype=dtype,
num_warmups=10,
num_tests=50
)
print_results(results)
all_results.append(results)
except Exception as e:
print(f"Error benchmarking configuration {config}:")
print(f" Exception: {e}")
traceback.print_exc()
sys.stdout.flush()
continue
# Print summary table
print("\n" + "="*120)
print("Summary Table")
print("="*120)
print(f"{'B':<4} {'H':<4} {'L':<6} {'D':<4} {'Causal':<7} {'Time (ms)':<12} {'TFLOPs/s':<12} {'Tokens/s':<15}")
print("-"*120)
for r in all_results:
print(f"{r['batch_size']:<4} {r['num_heads']:<4} {r['seq_len']:<6} {r['head_dim']:<4} "
f"{str(r['is_causal']):<7} {r['avg_time_ms']:<12.3f} {r['tflops']:<12.4f} "
f"{r['tokens_per_second']:<15,.0f}")
print("="*120 + "\n")
sys.stdout.flush()
if __name__ == "__main__":
import argparse
parser = argparse.ArgumentParser(description='Benchmark FlashAttention2 in TFLOPs')
parser.add_argument('--batch-size', type=int, default=None, help='Batch size')
parser.add_argument('--num-heads', type=int, default=None, help='Number of attention heads')
parser.add_argument('--seq-len', type=int, default=None, help='Sequence length')
parser.add_argument('--head-dim', type=int, default=None, help='Head dimension')
parser.add_argument('--causal', action='store_true', help='Use causal attention')
parser.add_argument('--dtype', type=str, default='bfloat16', choices=['bfloat16', 'float16'],
help='Data type')
parser.add_argument('--num-warmups', type=int, default=10, help='Number of warmup iterations')
parser.add_argument('--num-tests', type=int, default=50, help='Number of test iterations')
parser.add_argument('--suite', action='store_true', help='Run full benchmark suite')
args = parser.parse_args()
dtype_map = {
'bfloat16': torch.bfloat16,
'float16': torch.float16
}
if args.suite:
run_benchmark_suite()
elif args.batch_size and args.num_heads and args.seq_len and args.head_dim:
results = benchmark_flashattn2(
batch_size=args.batch_size,
num_heads=args.num_heads,
seq_len=args.seq_len,
head_dim=args.head_dim,
is_causal=args.causal,
dtype=dtype_map[args.dtype],
num_warmups=args.num_warmups,
num_tests=args.num_tests
)
print_results(results)
else:
print("Running default benchmark suite. Use --suite for full suite or provide all parameters.")
print(
"Example: python benchmarks/benchmark_flashattn2.py "
"--batch-size 1 --num-heads 16 --seq-len 4096 --head-dim 128"
)
sys.stdout.flush()
run_benchmark_suite()
@@ -0,0 +1,253 @@
import sys
import traceback
import _bootstrap # noqa: F401
import torch
from attn_qat_infer.api import sageattn_blackwell
from attn_qat_infer.quantization.bench.bench_utils import bench
def calculate_attention_flops(batch_size, num_heads, seq_len_q, seq_len_k, head_dim, is_causal=False):
"""Calculate FLOPs for attention (FlashAttention standard - matmuls only)."""
f = 4 * batch_size * num_heads * seq_len_q * seq_len_k * head_dim
if is_causal:
f = f // 2
return f
def benchmark_sageattn3(batch_size, num_heads, seq_len, head_dim,
is_causal=False, dtype=torch.bfloat16,
per_block_mean=True, single_level_p_quant=False,
num_warmups=100, num_tests=1000):
"""
Benchmark SageAttention3 and return performance metrics.
Args:
batch_size: Batch size
num_heads: Number of attention heads
seq_len: Sequence length (same for Q, K, V)
head_dim: Head dimension
is_causal: Whether to use causal masking
dtype: Data type (torch.bfloat16 or torch.float16)
per_block_mean: Whether to use per-block mean for Q smoothing
single_level_p_quant: If True, use single-level quantization for P matrix
num_warmups: Number of warmup iterations
num_tests: Number of test iterations
Returns:
dict with performance metrics
"""
device = 'cuda'
if not torch.cuda.is_available():
raise RuntimeError("CUDA is not available. This benchmark requires a CUDA device.")
# Create input tensors
q = torch.randn(batch_size, num_heads, seq_len, head_dim,
device=device, dtype=dtype)
k = torch.randn(batch_size, num_heads, seq_len, head_dim,
device=device, dtype=dtype)
v = torch.randn(batch_size, num_heads, seq_len, head_dim,
device=device, dtype=dtype)
# Create closure for benchmarking (no extra stream needed - bench handles synchronization)
def run_attention():
return sageattn_blackwell(
q, k, v,
is_causal=is_causal,
per_block_mean=per_block_mean,
single_level_p_quant=single_level_p_quant
)
# Benchmark using the bench utility (handles warmup and timing)
avg_time_ms = bench(run_attention, num_warmups=num_warmups, num_tests=num_tests)
avg_time_s = avg_time_ms / 1000.0
# Calculate FLOPs
total_flops = calculate_attention_flops(
batch_size, num_heads, seq_len, seq_len, head_dim, is_causal
)
# Calculate TFLOPs
tflops = total_flops / (avg_time_s * 1e12)
# Calculate throughput (tokens/sec)
tokens_per_second = (batch_size * seq_len) / avg_time_s
return {
'batch_size': batch_size,
'num_heads': num_heads,
'seq_len': seq_len,
'head_dim': head_dim,
'is_causal': is_causal,
'dtype': str(dtype),
'per_block_mean': per_block_mean,
'single_level_p_quant': single_level_p_quant,
'avg_time_ms': avg_time_ms,
'avg_time_s': avg_time_s,
'total_flops': total_flops,
'tflops': tflops,
'tokens_per_second': tokens_per_second,
}
def print_results(results):
"""Print benchmark results in a formatted table."""
print("\n" + "="*100)
print("SageAttention3 Benchmark Results")
print("="*100)
print(f"Configuration:")
print(f" Batch Size: {results['batch_size']}")
print(f" Num Heads: {results['num_heads']}")
print(f" Sequence Length: {results['seq_len']}")
print(f" Head Dimension: {results['head_dim']}")
print(f" Causal: {results['is_causal']}")
print(f" Data Type: {results['dtype']}")
print(f" Per Block Mean: {results['per_block_mean']}")
print(f" Single Level P Quant: {results['single_level_p_quant']}")
print(f"\nPerformance:")
print(f" Average Time: {results['avg_time_ms']:.3f} ms")
print(f" Total FLOPs: {results['total_flops']/1e12:.4f} TFLOPs (theoretical)")
print(f" Throughput: {results['tflops']:.4f} TFLOPs/s")
print(f" Tokens/sec: {results['tokens_per_second']:,.0f}")
print("="*100 + "\n")
sys.stdout.flush()
def run_benchmark_suite():
"""Run a comprehensive benchmark suite with various configurations."""
print("Starting SageAttention3 Benchmark Suite...")
print(f"CUDA Device: {torch.cuda.get_device_name(0)}")
print(f"CUDA Version: {torch.version.cuda}")
print(f"PyTorch Version: {torch.__version__}\n")
sys.stdout.flush()
# Default configurations to test
# (batch_size, num_heads, seq_len, head_dim, is_causal, dtype)
configs = [
(1, 16, 1024, 64, False, torch.bfloat16),
(1, 16, 2048, 64, False, torch.bfloat16),
(1, 16, 4096, 64, False, torch.bfloat16),
(1, 16, 8192, 64, False, torch.bfloat16),
(1, 16, 16384, 64, False, torch.bfloat16),
(1, 16, 1024, 128, False, torch.bfloat16),
(1, 16, 2048, 128, False, torch.bfloat16),
(1, 16, 4096, 128, False, torch.bfloat16),
(1, 16, 8192, 128, False, torch.bfloat16),
(1, 16, 16384, 128, False, torch.bfloat16),
(1, 32, 1024, 64, False, torch.bfloat16),
(1, 32, 2048, 64, False, torch.bfloat16),
(1, 32, 4096, 64, False, torch.bfloat16),
(1, 32, 8192, 64, False, torch.bfloat16),
(1, 32, 16384, 64, False, torch.bfloat16),
(1, 32, 1024, 128, False, torch.bfloat16),
(1, 32, 2048, 128, False, torch.bfloat16),
(1, 32, 4096, 128, False, torch.bfloat16),
(1, 32, 8192, 128, False, torch.bfloat16),
(1, 32, 16384, 128, False, torch.bfloat16),
]
all_results = []
for config in configs:
batch_size, num_heads, seq_len, head_dim, is_causal, dtype = config
print(f"\nBenchmarking: B={batch_size}, H={num_heads}, L={seq_len}, D={head_dim}, "
f"Causal={is_causal}, dtype={dtype}...")
sys.stdout.flush()
try:
results = benchmark_sageattn3(
batch_size=batch_size,
num_heads=num_heads,
seq_len=seq_len,
head_dim=head_dim,
is_causal=is_causal,
dtype=dtype,
num_warmups=10,
num_tests=50
)
print_results(results)
all_results.append(results)
except Exception as e:
print(f"Error benchmarking configuration {config}:")
print(f" Exception: {e}")
traceback.print_exc()
sys.stdout.flush()
continue
# Print summary table
print("\n" + "="*120)
print("Summary Table")
print("="*120)
print(f"{'B':<4} {'H':<4} {'L':<6} {'D':<4} {'Causal':<7} {'Time (ms)':<12} {'TFLOPs/s':<12} {'Tokens/s':<15}")
print("-"*120)
for r in all_results:
print(f"{r['batch_size']:<4} {r['num_heads']:<4} {r['seq_len']:<6} {r['head_dim']:<4} "
f"{str(r['is_causal']):<7} {r['avg_time_ms']:<12.3f} {r['tflops']:<12.4f} "
f"{r['tokens_per_second']:<15,.0f}")
print("="*120 + "\n")
sys.stdout.flush()
if __name__ == "__main__":
import argparse
parser = argparse.ArgumentParser(description='Benchmark SageAttention3 in TFLOPs')
parser.add_argument('--batch-size', type=int, default=None, help='Batch size')
parser.add_argument('--num-heads', type=int, default=None, help='Number of attention heads')
parser.add_argument('--seq-len', type=int, default=None, help='Sequence length')
parser.add_argument('--head-dim', type=int, default=None, help='Head dimension')
parser.add_argument('--causal', action='store_true', help='Use causal attention')
parser.add_argument('--dtype', type=str, default='bfloat16', choices=['bfloat16', 'float16'],
help='Data type')
parser.add_argument('--per-block-mean', action='store_true', default=True,
help='Use per-block mean for Q smoothing (default: True)')
parser.add_argument('--no-per-block-mean', action='store_false', dest='per_block_mean',
help='Disable per-block mean for Q smoothing')
parser.add_argument('--single-level-p-quant', action='store_true', default=False,
help='Use single-level P quantization (default: True)')
parser.add_argument('--two-level-p-quant', action='store_false', dest='single_level_p_quant',
help='Use two-level P quantization')
parser.add_argument('--num-warmups', type=int, default=10, help='Number of warmup iterations')
parser.add_argument('--num-tests', type=int, default=50, help='Number of test iterations')
parser.add_argument('--suite', action='store_true', help='Run full benchmark suite')
args = parser.parse_args()
dtype_map = {
'bfloat16': torch.bfloat16,
'float16': torch.float16
}
if args.suite:
run_benchmark_suite()
elif args.batch_size and args.num_heads and args.seq_len and args.head_dim:
results = benchmark_sageattn3(
batch_size=args.batch_size,
num_heads=args.num_heads,
seq_len=args.seq_len,
head_dim=args.head_dim,
is_causal=args.causal,
dtype=dtype_map[args.dtype],
per_block_mean=args.per_block_mean,
single_level_p_quant=args.single_level_p_quant,
num_warmups=args.num_warmups,
num_tests=args.num_tests
)
print_results(results)
else:
print("Running default benchmark suite. Use --suite for full suite or provide all parameters.")
print(
"Example: python benchmarks/benchmark_sageattn3.py "
"--batch-size 1 --num-heads 16 --seq-len 4096 --head-dim 128"
)
sys.stdout.flush()
run_benchmark_suite()
+1 -1
View File
@@ -32,4 +32,4 @@ dependencies = [
[tool.scikit-build]
cmake.build-type = "Release"
minimum-version = "build-system.requires"
wheel.packages = ["python/fastvideo_kernel"]
wheel.packages = ["python/fastvideo_kernel", "attn_qat_infer"]
@@ -0,0 +1,5 @@
"""Triton kernel entrypoints exposed by ``fastvideo_kernel``."""
from .fused_attention import attention as fused_attention
__all__ = ["fused_attention"]
File diff suppressed because it is too large Load Diff
@@ -0,0 +1,55 @@
"""Compatibility shim for the legacy non-QAT Triton attention import path.
Historically callers imported
``fastvideo_kernel.triton_kernels.fused_attention`` directly. The shared
implementation now lives in ``attn_qat_train.py`` and is parameterized by the
``IS_QAT`` flag. This module preserves the original public API for tests and
downstream users while always dispatching to the non-QAT configuration.
"""
from __future__ import annotations
import torch
from .attn_qat_train import attention as _attention
def attention(
q: torch.Tensor,
k: torch.Tensor,
v: torch.Tensor,
causal: bool,
sm_scale: float,
warp_specialize: bool = True,
) -> torch.Tensor:
"""Run the shared Triton attention kernel in non-QAT mode."""
use_qat_qkv_backward = True
smooth_k = False
is_qat = False
two_level_quant_p = False
fake_quant_p = False
use_high_prec_o = False
smooth_q = False
use_global_sf_p = False
use_global_sf_qkv = False
return _attention(
q,
k,
v,
causal,
sm_scale,
use_qat_qkv_backward,
smooth_k,
warp_specialize,
is_qat,
two_level_quant_p,
fake_quant_p,
use_high_prec_o,
smooth_q,
use_global_sf_p,
use_global_sf_qkv,
)
__all__ = ["attention"]
@@ -0,0 +1,237 @@
# SPDX-License-Identifier: Apache-2.0
# Adapted from https://github.com/triton-lang/triton/blob/main/python/triton_kernels/triton_kernels/numerics_details/mxfp_details/_upcast_from_mxfp.py
# and https://github.com/triton-lang/triton/blob/main/python/triton_kernels/triton_kernels/numerics_details/mxfp_details/_downcast_to_mxfp.py
import triton
import triton.language as tl
from triton.language.target_info import cuda_capability_geq
MXFP_BLOCK_SIZE = tl.constexpr(16)
@triton.jit
def _compute_quant_and_scale(
src_tensor,
valid_src_mask,
mx_tensor_dtype: tl.constexpr = tl.uint8,
use_global_sf=True,
two_level_quant_P=False,
):
BLOCK_SIZE_OUT_DIM: tl.constexpr = src_tensor.shape[0]
BLOCK_SIZE_QUANT_DIM: tl.constexpr = src_tensor.shape[1]
BLOCK_SIZE_QUANT_MX_SCALE: tl.constexpr = src_tensor.shape[1] // MXFP_BLOCK_SIZE
is_fp4: tl.constexpr = mx_tensor_dtype == tl.uint8
tl.static_assert(
is_fp4
or mx_tensor_dtype == tl.float8e4nv
or mx_tensor_dtype == tl.float8e5,
"mx_tensor_dtype must be uint8, float8e4nv, or float8e5",
)
# Explicit cast to fp32 since most ops are not supported on bfloat16. We avoid needless conversions to and from bf16
f32_tensor = src_tensor.to(tl.float32)
abs_tensor = tl.abs(f32_tensor)
abs_tensor = tl.where(valid_src_mask, abs_tensor, -1.0) # Don't consider padding tensors in scale computation
if two_level_quant_P:
# row max from SageAttn3 paper
global_max_val = tl.max(f32_tensor, axis=1, keep_dims=True) # (BLOCK_SIZE_OUT_DIM, 1)
global_max_val = tl.maximum(global_max_val, 1e-8)
s_enc = ((6 * 448) / global_max_val).reshape([BLOCK_SIZE_OUT_DIM, 1, 1])
s_dec = (1 / s_enc)
abs_tensor = tl.reshape(abs_tensor, [BLOCK_SIZE_OUT_DIM, BLOCK_SIZE_QUANT_MX_SCALE, MXFP_BLOCK_SIZE])
if use_global_sf and not two_level_quant_P:
global_max_val = tl.max(abs_tensor)
# Avoid division by zero: if all values are padding (max is 0), use a default scale
global_max_val = tl.maximum(global_max_val, 1e-8)
s_enc = (6 * 448) / global_max_val
s_dec = (1 / s_enc)
elif not two_level_quant_P and not use_global_sf:
s_dec = 1.0
s_enc = 1.0
max_val = tl.max(abs_tensor, axis=2, keep_dims=True) # (BLOCK_SIZE_OUT_DIM, BLOCK_SIZE_QUANT_MX_SCALE, 1) # per block maxima
s_dec_b = max_val / 6 # (BLOCK_SIZE_OUT_DIM, BLOCK_SIZE_QUANT_MX_SCALE, 1)
s_dec_b_e4m3 = (s_dec_b * s_enc).to(tl.float8e4nv) # (BLOCK_SIZE_OUT_DIM, BLOCK_SIZE_QUANT_MX_SCALE, 1)
s_enc_b = 1 / (s_dec_b_e4m3.to(tl.float32) * s_dec) # (BLOCK_SIZE_OUT_DIM, BLOCK_SIZE_QUANT_MX_SCALE, 1)
f32_tensor = tl.reshape(f32_tensor, [BLOCK_SIZE_OUT_DIM, BLOCK_SIZE_QUANT_MX_SCALE, MXFP_BLOCK_SIZE])
quant_tensor = f32_tensor * s_enc_b
# Reshape the tensors after scaling
quant_tensor = quant_tensor.reshape([BLOCK_SIZE_OUT_DIM, BLOCK_SIZE_QUANT_DIM])
# Set the invalid portions of the tensor to 0. This will ensure that any padding tensors are 0 in the mx format.
quant_tensor = tl.where(valid_src_mask, quant_tensor, 0.0)
dequant_scale = s_dec_b_e4m3.reshape([BLOCK_SIZE_OUT_DIM, BLOCK_SIZE_QUANT_MX_SCALE])
if is_fp4 and cuda_capability_geq(10, 0):
# Convert scaled values to two f32 lanes and use PTX cvt to e2m1x2 with two f32 operands.
pairs = tl.reshape(quant_tensor, [BLOCK_SIZE_OUT_DIM, BLOCK_SIZE_QUANT_DIM // 2, 2])
lo_f, hi_f = tl.split(pairs)
lo_f32 = lo_f.to(tl.float32)
hi_f32 = hi_f.to(tl.float32)
# Inline PTX: cvt.rn.satfinite.e2m1x2.f32 takes two f32 sources and produces one .b8 packed e2m1x2.
out_tensor = tl.inline_asm_elementwise(
"""
{
.reg .b8 r;
cvt.rn.satfinite.e2m1x2.f32 r, $1, $2;
mov.b32 $0, {r, r, r, r};
}
""",
constraints="=r,f,f",
args=[hi_f32, lo_f32],
dtype=tl.uint8,
is_pure=True,
pack=1,
)
elif is_fp4:
quant_tensor = quant_tensor.to(tl.uint32, bitcast=True)
signs = quant_tensor & 0x80000000
exponents = (quant_tensor >> 23) & 0xFF
mantissas_orig = (quant_tensor & 0x7FFFFF)
# For RTNE: 0.25 < x < 0.75 maps to 0.5 (denormal); exactly 0.25 maps to 0.0
E8_BIAS = 127
E2_BIAS = 1
# Move implicit bit 1 at the beginning to mantissa for denormals
is_subnormal = exponents < E8_BIAS
adjusted_exponents = tl.core.sub(E8_BIAS, exponents + 1, sanitize_overflow=False)
mantissas_pre = (0x400000 | (mantissas_orig >> 1))
mantissas = tl.where(is_subnormal, mantissas_pre >> adjusted_exponents, mantissas_orig)
# For normal numbers, we change the bias from 127 to 1, and for subnormals, we keep exponent as 0.
exponents = tl.maximum(exponents, E8_BIAS - E2_BIAS) - (E8_BIAS - E2_BIAS)
# Combine sign, exponent, and mantissa, while saturating
# Round to nearest, ties to even (RTNE): use guard/sticky and LSB to decide increment
m2bits = mantissas >> 21
lsb_keep = (m2bits >> 1) & 0x1
guard = m2bits & 0x1
IS_SRC_FP32: tl.constexpr = src_tensor.dtype == tl.float32
if IS_SRC_FP32:
bit0_dropped = (mantissas_orig & 0x1) != 0
mask = (1 << tl.minimum(adjusted_exponents, 31)) - 1
dropped_post = (mantissas_pre & mask) != 0
sticky = is_subnormal & (bit0_dropped | dropped_post)
sticky |= ((mantissas & 0x1FFFFF) != 0).to(tl.uint32)
else:
sticky = ((mantissas & 0x1FFFFF) != 0).to(tl.uint32)
round_inc = guard & (sticky | lsb_keep)
e2m1_tmp = tl.minimum((((exponents << 2) | m2bits) + round_inc) >> 1, 0x7)
e2m1_value = ((signs >> 28) | e2m1_tmp).to(tl.uint8)
e2m1_value = tl.reshape(e2m1_value, [BLOCK_SIZE_OUT_DIM, BLOCK_SIZE_QUANT_DIM // 2, 2])
evens, odds = tl.split(e2m1_value)
out_tensor = evens | (odds << 4)
else:
out_tensor = quant_tensor.to(mx_tensor_dtype)
return out_tensor, dequant_scale, s_dec
@triton.jit
def _compute_dequant(
mx_tensor,
scale,
s_dec,
BLOCK_SIZE_OUT_DIM: tl.constexpr,
BLOCK_SIZE_QUANT_DIM: tl.constexpr,
dst_dtype: tl.constexpr,
):
tl.static_assert(BLOCK_SIZE_QUANT_DIM % MXFP_BLOCK_SIZE == 0, f"Block size along quantization block must be a multiple of {MXFP_BLOCK_SIZE=}")
# uint8 signifies two fp4 e2m1 values packed into a single byte
mx_tensor_dtype: tl.constexpr = mx_tensor.dtype
tl.static_assert(dst_dtype == tl.float16 or dst_dtype == tl.bfloat16 or dst_dtype == tl.float32)
tl.static_assert(
mx_tensor_dtype == tl.uint8
or ((mx_tensor_dtype == tl.float8e4nv or mx_tensor_dtype == tl.float8e5) or mx_tensor_dtype == dst_dtype),
"mx_tensor_ptr must be uint8 or float8 or dst_dtype")
tl.static_assert(scale.dtype == tl.float8e4nv, "scale must be float8e4nv")
# Determine if we are dealing with fp8 types.
is_fp4: tl.constexpr = mx_tensor_dtype == tl.uint8
BLOCK_SIZE_QUANT_MX_SCALE: tl.constexpr = BLOCK_SIZE_QUANT_DIM // MXFP_BLOCK_SIZE
# Upcast the scale to the destination type.
if dst_dtype == tl.bfloat16:
dst_scale = scale.to(tl.bfloat16)
else:
dst_scale = scale.to(tl.float32)
if dst_dtype == tl.float16:
dst_scale = dst_scale.to(tl.float16)
# Now upcast the tensor.
intermediate_dtype: tl.constexpr = tl.bfloat16 if dst_dtype == tl.float32 else dst_dtype
if cuda_capability_geq(10, 0):
assert is_fp4
packed_u32 = tl.inline_asm_elementwise(
asm="""
{
.reg .b8 in_8;
.reg .f16x2 out;
cvt.u8.u32 in_8, $1;
cvt.rn.f16x2.e2m1x2 out, in_8;
mov.b32 $0, out;
}
""",
constraints="=r,r",
args=[mx_tensor], # tl.uint8 passed in as a 32-bit reg with value in low 8 bits
dtype=tl.uint32,
is_pure=True,
pack=1,
)
lo_u16 = (packed_u32 & 0xFFFF).to(tl.uint16)
hi_u16 = (packed_u32 >> 16).to(tl.uint16)
lo_f16 = lo_u16.to(tl.float16, bitcast=True)
hi_f16 = hi_u16.to(tl.float16, bitcast=True)
if intermediate_dtype == tl.float16:
x0, x1 = lo_f16, hi_f16
else:
x0 = lo_f16.to(intermediate_dtype)
x1 = hi_f16.to(intermediate_dtype)
dst_tensor = tl.interleave(x0, x1)
else:
assert is_fp4
dst_bias: tl.constexpr = 127 if intermediate_dtype == tl.bfloat16 else 15 # exponent bias
dst_0p5: tl.constexpr = 16128 if intermediate_dtype == tl.bfloat16 else 0x3800
dst_m_bits: tl.constexpr = 7 if intermediate_dtype == tl.bfloat16 else 10 # mantissa bits
# e2m1
em0 = mx_tensor & 0x07
em1 = mx_tensor & 0x70
x0 = (em0.to(tl.uint16) << (dst_m_bits - 1)) | ((mx_tensor & 0x08).to(tl.uint16) << 12)
x1 = (em1.to(tl.uint16) << (dst_m_bits - 5)) | ((mx_tensor & 0x80).to(tl.uint16) << 8)
# Three cases:
# 1) x is normal and non-zero: Correct bias
x0 = tl.where((em0 & 0x06) != 0, x0 + ((dst_bias - 1) << dst_m_bits), x0)
x1 = tl.where((em1 & 0x60) != 0, x1 + ((dst_bias - 1) << dst_m_bits), x1)
# 2) x is subnormal (x == 0bs001 where s is the sign): Map to +-0.5 in the dst type
x0 = tl.where(em0 == 0x01, dst_0p5 | (x0 & 0x8000), x0)
x1 = tl.where(em1 == 0x10, dst_0p5 | (x1 & 0x8000), x1)
# 3) x is zero, do nothing
dst_tensor = tl.interleave(x0, x1).to(intermediate_dtype, bitcast=True)
dst_tensor = dst_tensor.to(dst_dtype)
# Reshape for proper broadcasting: the scale was stored with a 16‐sized “inner” grouping.
dst_tensor = dst_tensor.reshape([BLOCK_SIZE_OUT_DIM, BLOCK_SIZE_QUANT_MX_SCALE, MXFP_BLOCK_SIZE])
dst_scale = dst_scale.reshape([BLOCK_SIZE_OUT_DIM, BLOCK_SIZE_QUANT_MX_SCALE, 1])
scale = scale.reshape(dst_scale.shape)
out_tensor = dst_tensor * dst_scale * s_dec # NVFP4 has the additional global scale factor
if dst_dtype == tl.float32:
max_fin = 3.4028234663852886e+38
elif dst_dtype == tl.bfloat16:
max_fin = 3.3895313892515355e+38
else:
tl.static_assert(dst_dtype == tl.float16)
max_fin = 65504
out_tensor = tl.clamp(out_tensor, min=-max_fin, max=max_fin)
out_tensor = out_tensor.reshape([BLOCK_SIZE_OUT_DIM, BLOCK_SIZE_QUANT_DIM])
out_tensor = out_tensor.to(dst_dtype)
return out_tensor
@@ -0,0 +1,80 @@
import triton
import triton.language as tl
from .nvfp4_utils import _compute_quant_and_scale, _compute_dequant
@triton.jit
def fake_quantize(src_tensor, valid_src_mask, BLOCK_SIZE_OUT_DIM: tl.constexpr,
BLOCK_SIZE_QUANT_DIM: tl.constexpr,
dst_dtype: tl.constexpr,
mx_tensor_dtype: tl.constexpr = tl.uint8,
use_global_sf: tl.constexpr = True,
two_level_quant_P: tl.constexpr = False):
high_prec_src_tensor = src_tensor
src_tensor, src_scale, src_s_dec = _compute_quant_and_scale(src_tensor=src_tensor,
valid_src_mask=valid_src_mask,
mx_tensor_dtype=mx_tensor_dtype,
use_global_sf=use_global_sf,
two_level_quant_P=two_level_quant_P)
src_tensor = _compute_dequant(mx_tensor=src_tensor,
scale=src_scale,
s_dec=src_s_dec,
BLOCK_SIZE_OUT_DIM=BLOCK_SIZE_OUT_DIM,
BLOCK_SIZE_QUANT_DIM=BLOCK_SIZE_QUANT_DIM,
dst_dtype=dst_dtype)
return src_tensor, high_prec_src_tensor.to(src_tensor.dtype)
@triton.jit
def fake_quantize_q(Q, fake_Q, stride_z_q, stride_h_q,
stride_tok_q, stride_d_q,
fake_stride_z_q, fake_stride_h_q,
fake_stride_tok_q, fake_stride_d_q,
H, N_CTX_Q,
BLOCK_M: tl.constexpr,
HEAD_DIM: tl.constexpr,
use_global_sf: tl.constexpr = True):
bhid = tl.program_id(1)
adj_q = (stride_h_q * (bhid % H) + stride_z_q * (bhid // H))
fake_adj_q = (fake_stride_h_q * (bhid % H) + fake_stride_z_q * (bhid // H))
Q += adj_q
fake_Q += fake_adj_q
pid = tl.program_id(0)
start_m = pid * BLOCK_M
offs_m = start_m + tl.arange(0, BLOCK_M)
offs_k = tl.arange(0, HEAD_DIM)
q_valid = offs_m < N_CTX_Q
q = tl.load(Q + offs_m[:, None] * stride_tok_q + offs_k[None, :] * stride_d_q, mask=q_valid[:, None], other=0.0)
q, _ = fake_quantize(src_tensor=q, valid_src_mask=q_valid[:, None], BLOCK_SIZE_OUT_DIM=BLOCK_M, BLOCK_SIZE_QUANT_DIM=HEAD_DIM, dst_dtype=q.dtype, use_global_sf=use_global_sf)
tl.store(fake_Q + offs_m[:, None] * fake_stride_tok_q + offs_k[None, :] * fake_stride_d_q, q, mask=q_valid[:, None])
@triton.jit
def fake_quantize_kv(K, V, fake_K, fake_V, stride_z_kv, stride_h_kv,
stride_tok_kv, stride_d_kv,
fake_stride_z_kv, fake_stride_h_kv,
fake_stride_tok_kv, fake_stride_d_kv,
H, N_CTX_KV,
BLOCK_N: tl.constexpr,
HEAD_DIM: tl.constexpr,
use_global_sf: tl.constexpr = True):
bhid = tl.program_id(1)
adj_kv = (stride_h_kv * (bhid % H) + stride_z_kv * (bhid // H))
fake_adj_kv = (fake_stride_h_kv * (bhid % H) + fake_stride_z_kv * (bhid // H))
K += adj_kv
V += adj_kv
fake_K += fake_adj_kv
fake_V += fake_adj_kv
pid = tl.program_id(0)
start_n = pid * BLOCK_N
offs_n = start_n + tl.arange(0, BLOCK_N)
offs_k = tl.arange(0, HEAD_DIM)
kv_valid = offs_n < N_CTX_KV
k_block = tl.load(K + offs_n[:, None] * stride_tok_kv + offs_k[None, :] * stride_d_kv, mask=kv_valid[:, None], other=0.0)
v_block = tl.load(V + offs_n[:, None] * stride_tok_kv + offs_k[None, :] * stride_d_kv, mask=kv_valid[:, None], other=0.0)
k, _ = fake_quantize(src_tensor=k_block, valid_src_mask=kv_valid[:, None], BLOCK_SIZE_OUT_DIM=BLOCK_N, BLOCK_SIZE_QUANT_DIM=HEAD_DIM, dst_dtype=k_block.dtype, use_global_sf=use_global_sf)
v, _ = fake_quantize(src_tensor=v_block, valid_src_mask=kv_valid[:, None], BLOCK_SIZE_OUT_DIM=BLOCK_N, BLOCK_SIZE_QUANT_DIM=HEAD_DIM, dst_dtype=v_block.dtype, use_global_sf=use_global_sf)
tl.store(fake_K + offs_n[:, None] * fake_stride_tok_kv + offs_k[None, :] * fake_stride_d_kv, k, mask=kv_valid[:, None])
tl.store(fake_V + offs_n[:, None] * fake_stride_tok_kv + offs_k[None, :] * fake_stride_d_kv, v, mask=kv_valid[:, None])
+4
View File
@@ -0,0 +1,4 @@
from ._bootstrap import ensure_local_kernel_sources_first
ensure_local_kernel_sources_first()
+56
View File
@@ -0,0 +1,56 @@
"""Test import helpers for preferring the in-tree fastvideo-kernel sources."""
from __future__ import annotations
import importlib
import sys
from pathlib import Path
def _prepend_import_path(path: Path) -> None:
path_str = str(path)
if path_str not in sys.path:
sys.path.insert(0, path_str)
def _purge_package(name: str) -> None:
prefix = f"{name}."
for module_name in tuple(sys.modules):
if module_name == name or module_name.startswith(prefix):
sys.modules.pop(module_name, None)
def _module_is_from_checkout(module_file: str | None, checkout_root: Path) -> bool:
if module_file is None:
return False
try:
module_path = Path(module_file).resolve()
except OSError:
return False
return module_path.is_relative_to(checkout_root.resolve())
def ensure_local_kernel_sources_first() -> None:
tests_root = Path(__file__).resolve().parent
kernel_root = tests_root.parent
repo_root = kernel_root.parent
kernel_python_root = kernel_root / "python"
# Keep the in-tree kernel sources ahead of any preinstalled wheel so tests
# exercise the checkout under review.
for path in (repo_root, kernel_root, kernel_python_root):
_prepend_import_path(path)
importlib.invalidate_caches()
loaded_kernel = sys.modules.get("fastvideo_kernel")
if loaded_kernel is None:
return
if not _module_is_from_checkout(
getattr(loaded_kernel, "__file__", None),
kernel_python_root,
):
_purge_package("fastvideo_kernel")
+4
View File
@@ -0,0 +1,4 @@
from ._bootstrap import ensure_local_kernel_sources_first
ensure_local_kernel_sources_first()
File diff suppressed because it is too large Load Diff
+30
View File
@@ -0,0 +1,30 @@
from __future__ import annotations
import sys
import types
from pathlib import Path
from tests._bootstrap import ensure_local_kernel_sources_first
def test_bootstrap_prefers_local_kernel_checkout(monkeypatch) -> None:
tests_root = Path(__file__).resolve().parent
kernel_root = tests_root.parent
repo_root = kernel_root.parent
kernel_python_root = kernel_root / "python"
stale_module = types.ModuleType("fastvideo_kernel")
stale_module.__file__ = (
"/tmp/site-packages/fastvideo_kernel/__init__.py"
)
monkeypatch.setitem(sys.modules, "fastvideo_kernel", stale_module)
monkeypatch.setattr(sys, "path", ["/tmp/site-packages"])
ensure_local_kernel_sources_first()
assert "fastvideo_kernel" not in sys.modules
assert sys.path[:3] == [
str(kernel_python_root),
str(kernel_root),
str(repo_root),
]
+503
View File
@@ -0,0 +1,503 @@
#!/usr/bin/env python3
"""
Precision test to compare numerical differences for fake_quantize triton function:
1. fake_quantize (triton implementation) vs reference implementations
2. Tests various shapes, dtypes, and value ranges
3. Evaluates cosine similarity, max diff, and mean diff between input and output
"""
import torch
import triton
import triton.language as tl
from flashinfer import SfLayout, nvfp4_quantize, e2m1_and_ufp8sf_scale_to_float
from fastvideo_kernel.triton_kernels.nvfp4_utils import (
_compute_quant_and_scale,
_compute_dequant,
)
from typing import Optional
# MXFP_BLOCK_SIZE is 16 - use Python int for runtime checks
MXFP_BLOCK_SIZE = 16
DEVICE = torch.device("cuda")
def cosine_similarity(tensor1, tensor2):
"""
Compute cosine similarity between two tensors.
Args:
tensor1: First tensor
tensor2: Second tensor (same shape as tensor1)
Returns:
Cosine similarity value (scalar)
- Returns 1.0 if both tensors are zero (identical zero vectors)
- Returns 0.0 if only one tensor is zero (orthogonal to non-zero vector)
"""
# Flatten tensors for computation
t1_flat = tensor1.flatten().float()
t2_flat = tensor2.flatten().float()
# Compute cosine similarity: (A · B) / (||A|| * ||B||)
dot_product = torch.dot(t1_flat, t2_flat)
norm1 = torch.norm(t1_flat)
norm2 = torch.norm(t2_flat)
# Handle zero vectors
if norm1 == 0 and norm2 == 0:
# Both are zero vectors - they are identical, so similarity is 1.0
return 1.0
elif norm1 == 0 or norm2 == 0:
# One is zero, one is not - they are orthogonal, so similarity is 0.0
return 0.0
cos_sim = dot_product / (norm1 * norm2)
return cos_sim.item()
@triton.jit
def fake_quantize(src_tensor, valid_src_mask, BLOCK_SIZE_OUT_DIM: tl.constexpr,
BLOCK_SIZE_QUANT_DIM: tl.constexpr,
dst_dtype: tl.constexpr,
mx_tensor_dtype: tl.constexpr = tl.uint8):
"""
Fake quantize function - matches API from attn_qat_train.py.
"""
high_prec_src_tensor = src_tensor
src_tensor, src_scale, src_s_dec = _compute_quant_and_scale(
src_tensor=src_tensor,
valid_src_mask=valid_src_mask,
mx_tensor_dtype=mx_tensor_dtype
)
src_tensor = _compute_dequant(
mx_tensor=src_tensor,
scale=src_scale,
s_dec=src_s_dec,
BLOCK_SIZE_OUT_DIM=BLOCK_SIZE_OUT_DIM,
BLOCK_SIZE_QUANT_DIM=BLOCK_SIZE_QUANT_DIM,
dst_dtype=dst_dtype
)
return src_tensor, high_prec_src_tensor
def get_fake_quant_reference(x: torch.Tensor):
"""
Reference implementation using FlashInfer for comparison.
"""
orig_shape = x.shape
orig_dtype = x.dtype
device = x.device
x = x.view(-1, x.shape[-1])
x_global_sf = (448 * 6) / x.float().abs().nan_to_num().max()
x_fp4, x_scale = nvfp4_quantize(x, x_global_sf, sfLayout=SfLayout.layout_128x4, do_shuffle=False)
x_dequant = e2m1_and_ufp8sf_scale_to_float(x_fp4, x_scale, 1 / x_global_sf)
return x_dequant.view(orig_shape).to(orig_dtype).to(device)
@triton.jit
def fake_quantize_kernel(
src_ptr,
dst_ptr,
BLOCK_SIZE_OUT_DIM: tl.constexpr,
BLOCK_SIZE_QUANT_DIM: tl.constexpr,
dst_dtype: tl.constexpr,
mx_tensor_dtype: tl.constexpr,
stride_src_outer,
stride_src_quant,
stride_dst_outer,
stride_dst_quant,
outer_dim,
quant_dim,
):
"""
Kernel wrapper to call fake_quantize on a block of data.
"""
outer_idx = tl.program_id(0)
quant_idx = tl.program_id(1)
# Compute offsets
start_outer = outer_idx * BLOCK_SIZE_OUT_DIM
start_quant = quant_idx * BLOCK_SIZE_QUANT_DIM
# Create offset arrays
offs_outer = tl.arange(0, BLOCK_SIZE_OUT_DIM)[:, None]
offs_quant = tl.arange(0, BLOCK_SIZE_QUANT_DIM)[None, :]
# Create masks for valid elements
mask_outer = (start_outer + offs_outer) < outer_dim
mask_quant = (start_quant + offs_quant) < quant_dim
full_mask = mask_outer & mask_quant
# Load source tensor
src_offsets = (start_outer + offs_outer) * stride_src_outer + (start_quant + offs_quant) * stride_src_quant
src_tensor = tl.load(src_ptr + src_offsets, mask=full_mask, other=0.0)
# Call fake_quantize with valid_src_mask parameter
quantized_tensor, high_prec_tensor = fake_quantize(
src_tensor=src_tensor,
valid_src_mask=full_mask,
BLOCK_SIZE_OUT_DIM=BLOCK_SIZE_OUT_DIM,
BLOCK_SIZE_QUANT_DIM=BLOCK_SIZE_QUANT_DIM,
dst_dtype=dst_dtype,
mx_tensor_dtype=mx_tensor_dtype
)
# Store result
dst_offsets = (start_outer + offs_outer) * stride_dst_outer + (start_quant + offs_quant) * stride_dst_quant
tl.store(dst_ptr + dst_offsets, quantized_tensor, mask=full_mask)
def triton_fake_quantize(
x: torch.Tensor,
BLOCK_SIZE_OUT_DIM: int = 128,
BLOCK_SIZE_QUANT_DIM: int = 128,
use_fp4: bool = True, # True for fp4 (uint8), False for fp8 (float8e4nv)
dst_dtype: Optional[torch.dtype] = None
) -> torch.Tensor:
"""
Call fake_quantize triton function on a tensor.
Args:
x: Input tensor (2D or can be reshaped to 2D)
BLOCK_SIZE_OUT_DIM: Block size for outer dimension
BLOCK_SIZE_QUANT_DIM: Block size for quantization dimension (must be multiple of 16)
use_fp4: If True, use fp4 (uint8), else use fp8 (float8e4nv)
dst_dtype: Output dtype (defaults to input dtype)
Returns:
Fake quantized tensor
"""
assert x.is_cuda, "Input must be on CUDA"
assert BLOCK_SIZE_QUANT_DIM % 16 == 0, f"BLOCK_SIZE_QUANT_DIM must be multiple of 16"
orig_shape = x.shape
orig_dtype = x.dtype
# Reshape to 2D
x_2d = x.view(-1, x.shape[-1])
outer_dim, quant_dim = x_2d.shape
if dst_dtype is None:
dst_dtype = orig_dtype
# Map torch dtype to triton dtype
dtype_map = {
torch.float32: tl.float32,
torch.float16: tl.float16,
torch.bfloat16: tl.bfloat16,
}
triton_dst_dtype = dtype_map.get(dst_dtype, tl.float16)
# Allocate output
output = torch.empty_like(x_2d, dtype=dst_dtype)
# Launch kernel with appropriate quantization dtype
grid = (
triton.cdiv(outer_dim, BLOCK_SIZE_OUT_DIM),
triton.cdiv(quant_dim, BLOCK_SIZE_QUANT_DIM),
)
if use_fp4:
# Use fp4 (uint8)
fake_quantize_kernel[grid](
src_ptr=x_2d,
dst_ptr=output,
BLOCK_SIZE_OUT_DIM=BLOCK_SIZE_OUT_DIM,
BLOCK_SIZE_QUANT_DIM=BLOCK_SIZE_QUANT_DIM,
dst_dtype=triton_dst_dtype,
mx_tensor_dtype=tl.uint8, # fp4 uses uint8
stride_src_outer=x_2d.stride(0),
stride_src_quant=x_2d.stride(1),
stride_dst_outer=output.stride(0),
stride_dst_quant=output.stride(1),
outer_dim=outer_dim,
quant_dim=quant_dim,
)
else:
# Use fp8 (float8e4nv)
fake_quantize_kernel[grid](
src_ptr=x_2d,
dst_ptr=output,
BLOCK_SIZE_OUT_DIM=BLOCK_SIZE_OUT_DIM,
BLOCK_SIZE_QUANT_DIM=BLOCK_SIZE_QUANT_DIM,
dst_dtype=triton_dst_dtype,
mx_tensor_dtype=tl.float8e4nv, # fp8 uses float8e4nv
stride_src_outer=x_2d.stride(0),
stride_src_quant=x_2d.stride(1),
stride_dst_outer=output.stride(0),
stride_dst_quant=output.stride(1),
outer_dim=outer_dim,
quant_dim=quant_dim,
)
return output.view(orig_shape).to(dst_dtype)
def test_fake_quantize_basic():
"""Test basic functionality of fake_quantize."""
torch.manual_seed(42)
# Test parameters
shape = (128, 128)
dtype = torch.bfloat16
# Create input tensor
x = torch.randn(shape, dtype=dtype, device=DEVICE)
# Test triton fake_quantize
x_fq = triton_fake_quantize(x, BLOCK_SIZE_OUT_DIM=128, BLOCK_SIZE_QUANT_DIM=128, use_fp4=True, dst_dtype=dtype)
# Check that outputs have correct shape
assert x_fq.shape == x.shape
assert x_fq.dtype == x.dtype
# Check that outputs are finite
assert torch.isfinite(x_fq).all()
# Compare input and output
max_diff = (x_fq - x).abs().max()
mean_diff = (x_fq - x).abs().mean()
cos_sim = cosine_similarity(x_fq, x)
print(f" Input vs Output - Max diff: {max_diff.item():.6f}, Mean diff: {mean_diff.item():.6f}, Cosine sim: {cos_sim:.6f}")
print("✓ Basic test passed.")
def test_fake_quantize_different_shapes():
"""Test fake_quantize with different input shapes."""
torch.manual_seed(42)
test_configs = [
(128, 64), # Small
(256, 128), # Medium
(512, 256), # Large
(128, 256), # Rectangular
(256, 128), # Rectangular (reversed)
]
dtype = torch.bfloat16
for outer_dim, quant_dim in test_configs:
# Create input tensor
x = torch.randn((outer_dim, quant_dim), dtype=dtype, device=DEVICE)
# Test triton fake_quantize
x_fq = triton_fake_quantize(x, BLOCK_SIZE_OUT_DIM=128, BLOCK_SIZE_QUANT_DIM=128, use_fp4=True, dst_dtype=dtype)
assert x_fq.shape == x.shape
assert torch.isfinite(x_fq).all()
# Compare input and output
max_diff = (x_fq - x).abs().max()
mean_diff = (x_fq - x).abs().mean()
cos_sim = cosine_similarity(x_fq, x)
print(f" Shape ({outer_dim}, {quant_dim}) - Max diff: {max_diff.item():.6f}, Mean diff: {mean_diff.item():.6f}, Cosine sim: {cos_sim:.6f}")
print(f"✓ Shape test passed for ({outer_dim}, {quant_dim})")
def test_fake_quantize_different_dtypes():
"""Test fake_quantize with different input dtypes."""
torch.manual_seed(42)
shape = (128, 128)
dtypes = [torch.float32, torch.float16, torch.bfloat16]
for dtype in dtypes:
# Create input tensor
x = torch.randn(shape, dtype=dtype, device=DEVICE)
# Test triton fake_quantize
x_fq = triton_fake_quantize(x, BLOCK_SIZE_OUT_DIM=128, BLOCK_SIZE_QUANT_DIM=128, use_fp4=True, dst_dtype=dtype)
assert x_fq.shape == x.shape
assert x_fq.dtype == x.dtype
assert torch.isfinite(x_fq).all()
# Compare input and output
max_diff = (x_fq - x).abs().max()
mean_diff = (x_fq - x).abs().mean()
cos_sim = cosine_similarity(x_fq, x)
print(f" Dtype {dtype} - Max diff: {max_diff.item():.6f}, Mean diff: {mean_diff.item():.6f}, Cosine sim: {cos_sim:.6f}")
print(f"✓ Dtype test passed for {dtype}")
def test_fake_quantize_3d_4d_tensors():
"""Test fake_quantize with 3D and 4D tensors (reshaped to 2D internally)."""
torch.manual_seed(42)
dtype = torch.bfloat16
# Test 3D tensor (B, H, D)
print("\nTesting 3D tensor (B, H, D)")
x_3d = torch.randn((2, 8, 128), dtype=dtype, device=DEVICE)
x_fq_3d = triton_fake_quantize(x_3d, BLOCK_SIZE_OUT_DIM=128, BLOCK_SIZE_QUANT_DIM=128, use_fp4=True, dst_dtype=dtype)
assert x_fq_3d.shape == x_3d.shape
assert torch.isfinite(x_fq_3d).all()
max_diff = (x_fq_3d - x_3d).abs().max()
mean_diff = (x_fq_3d - x_3d).abs().mean()
cos_sim = cosine_similarity(x_fq_3d, x_3d)
print(f" 3D (2, 8, 128) - Max diff: {max_diff.item():.6f}, Mean diff: {mean_diff.item():.6f}, Cosine sim: {cos_sim:.6f}")
# Test 4D tensor (B, H, L, D)
print("\nTesting 4D tensor (B, H, L, D)")
x_4d = torch.randn((1, 8, 256, 128), dtype=dtype, device=DEVICE)
x_fq_4d = triton_fake_quantize(x_4d, BLOCK_SIZE_OUT_DIM=128, BLOCK_SIZE_QUANT_DIM=128, use_fp4=True, dst_dtype=dtype)
assert x_fq_4d.shape == x_4d.shape
assert torch.isfinite(x_fq_4d).all()
max_diff = (x_fq_4d - x_4d).abs().max()
mean_diff = (x_fq_4d - x_4d).abs().mean()
cos_sim = cosine_similarity(x_fq_4d, x_4d)
print(f" 4D (1, 8, 256, 128) - Max diff: {max_diff.item():.6f}, Mean diff: {mean_diff.item():.6f}, Cosine sim: {cos_sim:.6f}")
print("✓ 3D/4D tensor test passed.")
def test_fake_quantize_edge_cases():
"""Test edge cases like zeros, ones, extreme values."""
torch.manual_seed(42)
dtype = torch.bfloat16
shape = (128, 128)
test_cases = [
("zeros", torch.zeros(shape, dtype=dtype, device=DEVICE)),
("ones", torch.ones(shape, dtype=dtype, device=DEVICE)),
("negative_ones", -torch.ones(shape, dtype=dtype, device=DEVICE)),
("very_small", torch.randn(shape, dtype=dtype, device=DEVICE) * 1e-6),
("very_large", torch.randn(shape, dtype=dtype, device=DEVICE) * 1e6),
]
for name, x in test_cases:
x_fq = triton_fake_quantize(x, BLOCK_SIZE_OUT_DIM=128, BLOCK_SIZE_QUANT_DIM=128, use_fp4=True, dst_dtype=dtype)
assert x_fq.shape == x.shape
assert torch.isfinite(x_fq).all()
max_diff = (x_fq - x).abs().max()
mean_diff = (x_fq - x).abs().mean()
cos_sim = cosine_similarity(x_fq, x)
print(f" {name} - Max diff: {max_diff.item():.6f}, Mean diff: {mean_diff.item():.6f}, Cosine sim: {cos_sim:.6f}")
print("✓ Edge cases test passed.")
def test_fake_quantize_vs_reference():
"""Test triton fake_quantize vs reference implementation."""
torch.manual_seed(42)
dtype = torch.bfloat16
shape = (128, 128)
# Create input tensor
x = torch.randn(shape, dtype=dtype, device=DEVICE)
# Test triton fake_quantize
x_fq_triton = triton_fake_quantize(x, BLOCK_SIZE_OUT_DIM=128, BLOCK_SIZE_QUANT_DIM=128, use_fp4=True, dst_dtype=dtype)
# Test reference implementation
x_fq_ref = get_fake_quant_reference(x)
# Compare triton vs reference
max_diff = (x_fq_triton - x_fq_ref).abs().max()
mean_diff = (x_fq_triton - x_fq_ref).abs().mean()
cos_sim = cosine_similarity(x_fq_triton, x_fq_ref)
print(f" Triton vs Reference - Max diff: {max_diff.item():.6f}, Mean diff: {mean_diff.item():.6f}, Cosine sim: {cos_sim:.6f}")
# Also compare each vs input
max_diff_triton = (x_fq_triton - x).abs().max()
mean_diff_triton = (x_fq_triton - x).abs().mean()
cos_sim_triton = cosine_similarity(x_fq_triton, x)
print(f" Triton vs Input - Max diff: {max_diff_triton.item():.6f}, Mean diff: {mean_diff_triton.item():.6f}, Cosine sim: {cos_sim_triton:.6f}")
max_diff_ref = (x_fq_ref - x).abs().max()
mean_diff_ref = (x_fq_ref - x).abs().mean()
cos_sim_ref = cosine_similarity(x_fq_ref, x)
print(f" Reference vs Input - Max diff: {max_diff_ref.item():.6f}, Mean diff: {mean_diff_ref.item():.6f}, Cosine sim: {cos_sim_ref:.6f}")
print("✓ Reference comparison test passed.")
def test_fake_quantize_attention_shapes():
"""Test fake_quantize with attention-like shapes (similar to test_attn_qat_train.py)."""
torch.manual_seed(42)
test_configs = [
(2, 4, 128, 64), # Z, H, N_CTX, HEAD_DIM - Medium
(1, 8, 256, 128), # Large head dim
(1, 40, 9360, 128), # WAN shape
]
dtype = torch.bfloat16
for Z, H, N_CTX, HEAD_DIM in test_configs:
# Create input tensor in BLHD format (B, L, H, D) = (Z, N_CTX, H, HEAD_DIM)
x = torch.randn((Z, N_CTX, H, HEAD_DIM), dtype=dtype, device=DEVICE)
# Test triton fake_quantize
x_fq = triton_fake_quantize(x, BLOCK_SIZE_OUT_DIM=128, BLOCK_SIZE_QUANT_DIM=128, use_fp4=True, dst_dtype=dtype)
assert x_fq.shape == x.shape
assert torch.isfinite(x_fq).all()
# Compare input and output
max_diff = (x_fq - x).abs().max()
mean_diff = (x_fq - x).abs().mean()
cos_sim = cosine_similarity(x_fq, x)
print(f" (Z={Z}, H={H}, N_CTX={N_CTX}, HEAD_DIM={HEAD_DIM}) - Max diff: {max_diff.item():.6f}, Mean diff: {mean_diff.item():.6f}, Cosine sim: {cos_sim:.6f}")
print(f"✓ Attention shape test passed for (Z={Z}, H={H}, N_CTX={N_CTX}, HEAD_DIM={HEAD_DIM})")
def test_fake_quantize_non_divisible_blocks():
"""Test fake_quantize with shapes that are not divisible by block sizes."""
torch.manual_seed(42)
dtype = torch.bfloat16
# Test with non-divisible dimensions
test_shapes = [
(100, 100), # Not divisible by 128
(150, 200), # Not divisible by 128
(256, 100), # One dimension divisible, one not
]
for shape in test_shapes:
x = torch.randn(shape, dtype=dtype, device=DEVICE)
# Test triton fake_quantize
x_fq = triton_fake_quantize(x, BLOCK_SIZE_OUT_DIM=128, BLOCK_SIZE_QUANT_DIM=128, use_fp4=True, dst_dtype=dtype)
assert x_fq.shape == x.shape
assert torch.isfinite(x_fq).all()
max_diff = (x_fq - x).abs().max()
mean_diff = (x_fq - x).abs().mean()
cos_sim = cosine_similarity(x_fq, x)
print(f" Shape {shape} - Max diff: {max_diff.item():.6f}, Mean diff: {mean_diff.item():.6f}, Cosine sim: {cos_sim:.6f}")
print(f"✓ Non-divisible block test passed for {shape}")
if __name__ == "__main__":
print("Running fake_quantize tests...")
print(f"Device: {DEVICE}")
print()
test_fake_quantize_basic()
test_fake_quantize_different_shapes()
test_fake_quantize_different_dtypes()
test_fake_quantize_3d_4d_tensors()
test_fake_quantize_edge_cases()
test_fake_quantize_vs_reference()
test_fake_quantize_attention_shapes()
test_fake_quantize_non_divisible_blocks()
print()
print("All tests passed! ✓")
+2
View File
@@ -107,6 +107,8 @@ def legacy_from_pretrained_to_config(
engine[key] = value
elif key == "override_text_encoder_quant":
quantization["text_encoder_quant"] = value
elif key == "transformer_quant":
quantization["transformer_quant"] = value
elif key == "workload_type":
pipeline["workload_type"] = value
elif key == "lora_path":
+6 -2
View File
@@ -2,7 +2,7 @@
# Adapted from vllm: https://github.com/vllm-project/vllm/blob/v0.7.3/vllm/attention/backends/abstract.py
from abc import ABC, abstractmethod
from dataclasses import dataclass, fields
from dataclasses import dataclass, field, fields
from typing import TYPE_CHECKING, Any, Generic, Protocol, TypeVar
if TYPE_CHECKING:
@@ -53,6 +53,10 @@ class AttentionMetadata:
"""Attention metadata for prefill and decode batched together."""
# Current step of diffusion process
current_timestep: int
VSA_sparsity: float = field(default=0.0, kw_only=True)
def __getattr__(self, name: str) -> Any:
raise AttributeError(f"'{type(self).__name__}' object has no attribute '{name}'")
def asdict_zerocopy(self, skip_fields: set[str] | None = None) -> dict[str, Any]:
"""Similar to dataclasses.asdict, but avoids deepcopying."""
@@ -82,7 +86,7 @@ class AttentionMetadataBuilder(ABC, Generic[T]):
@abstractmethod
def build(
self,
**kwargs: dict[str, Any],
**kwargs: Any,
) -> AttentionMetadata:
"""Build attention metadata with on-device tensors."""
raise NotImplementedError
@@ -0,0 +1,121 @@
# SPDX-License-Identifier: Apache-2.0
import importlib
import sys
from collections.abc import Callable
from pathlib import Path
import torch
from fastvideo.attention.backends.abstract import (
AttentionBackend,
AttentionImpl,
AttentionMetadata,
AttentionMetadataBuilder,
)
from fastvideo.logger import init_logger
logger = init_logger(__name__)
_project_root = Path(__file__).resolve().parent.parent.parent.parent
_kernel_root = _project_root / "fastvideo-kernel"
_kernel_python_root = _kernel_root / "python"
_attn_qat_infer: Callable[..., torch.Tensor] | None = None
_attn_qat_infer_import_attempted = False
def _ensure_kernel_paths() -> None:
for path in (_project_root, _kernel_root, _kernel_python_root):
path_str = str(path)
if path_str not in sys.path:
sys.path.insert(0, path_str)
def _get_attn_qat_infer() -> Callable[..., torch.Tensor] | None:
global _attn_qat_infer
global _attn_qat_infer_import_attempted
if _attn_qat_infer_import_attempted:
return _attn_qat_infer
_attn_qat_infer_import_attempted = True
_ensure_kernel_paths()
try:
# Prefer the in-repo kernel implementation during local development.
_attn_qat_infer = importlib.import_module("attn_qat_infer").sageattn_blackwell
except ImportError:
_attn_qat_infer = None
return _attn_qat_infer
def is_attn_qat_infer_available() -> bool:
return _get_attn_qat_infer() is not None
class AttnQatInferBackend(AttentionBackend):
accept_output_buffer: bool = True
@staticmethod
def get_supported_head_sizes() -> list[int]:
return [64, 128]
@staticmethod
def get_name() -> str:
return "ATTN_QAT_INFER"
@staticmethod
def get_impl_cls() -> type["AttnQatInferImpl"]:
return AttnQatInferImpl
@staticmethod
def get_metadata_cls() -> type["AttentionMetadata"]:
raise NotImplementedError
@staticmethod
def get_builder_cls() -> type["AttentionMetadataBuilder"]:
raise NotImplementedError
class AttnQatInferImpl(AttentionImpl):
def __init__(
self,
num_heads: int,
head_size: int,
causal: bool,
softmax_scale: float,
num_kv_heads: int | None = None,
prefix: str = "",
**extra_impl_args,
) -> None:
self.causal = causal
self.softmax_scale = softmax_scale
self.dropout = extra_impl_args.get("dropout_p", 0.0)
def forward(
self,
query: torch.Tensor,
key: torch.Tensor,
value: torch.Tensor,
attn_metadata: AttentionMetadata,
) -> torch.Tensor:
attn_qat_infer = _get_attn_qat_infer()
if attn_qat_infer is None:
raise ImportError("attn_qat_infer is not available. Please ensure the "
"attn_qat_infer kernel package is installed.")
query = query.transpose(1, 2).contiguous()
key = key.transpose(1, 2).contiguous()
value = value.transpose(1, 2).contiguous()
output = attn_qat_infer(
query,
key,
value,
attn_mask=None,
is_causal=self.causal,
)
return output.transpose(1, 2).contiguous()
@@ -0,0 +1,145 @@
# SPDX-License-Identifier: Apache-2.0
import importlib
import sys
from collections.abc import Callable
from pathlib import Path
import torch
from fastvideo.attention.backends.abstract import (
AttentionBackend,
AttentionImpl,
AttentionMetadata,
AttentionMetadataBuilder,
)
from fastvideo.logger import init_logger
logger = init_logger(__name__)
_project_root = Path(__file__).resolve().parent.parent.parent.parent
_kernel_root = _project_root / "fastvideo-kernel"
_kernel_python_root = _kernel_root / "python"
_attn_qat_train_attention: Callable[..., torch.Tensor] | None = None
_attn_qat_train_import_attempted = False
def _ensure_kernel_paths() -> None:
for path in (_project_root, _kernel_root, _kernel_python_root):
path_str = str(path)
if path_str not in sys.path:
sys.path.insert(0, path_str)
def _get_attn_qat_train_attention() -> Callable[..., torch.Tensor] | None:
global _attn_qat_train_attention
global _attn_qat_train_import_attempted
if _attn_qat_train_import_attempted:
return _attn_qat_train_attention
_attn_qat_train_import_attempted = True
_ensure_kernel_paths()
try:
_attn_qat_train_attention = importlib.import_module("fastvideo_kernel.triton_kernels.attn_qat_train").attention
except ImportError:
_attn_qat_train_attention = None
return _attn_qat_train_attention
def attn_qat_train(q_BLHD: torch.Tensor,
k_BLHD: torch.Tensor,
v_BLHD: torch.Tensor,
is_causal: bool = False) -> torch.Tensor:
attention = _get_attn_qat_train_attention()
if attention is None:
raise ImportError("fastvideo_kernel.triton_kernels.attn_qat_train is not available. "
"Please ensure the FastVideo kernel package is installed.")
q_BHLD = q_BLHD.permute(0, 2, 1, 3).contiguous()
k_BHLD = k_BLHD.permute(0, 2, 1, 3).contiguous()
v_BHLD = v_BLHD.permute(0, 2, 1, 3).contiguous()
use_qat_qkv_backward = True
smooth_k = False
warp_specialize = True
is_qat = True
two_level_quant_p_sage3 = False
fake_quant_p_bwd = True
use_high_prec_o = True
smooth_q = False
sm_scale = 1.0 / (q_BHLD.shape[-1]**0.5)
use_global_sf_qkv = False
use_global_sf_p = False
o_BHLD = attention(
q_BHLD,
k_BHLD,
v_BHLD,
is_causal,
sm_scale,
use_qat_qkv_backward,
smooth_k,
warp_specialize,
is_qat,
two_level_quant_p_sage3,
fake_quant_p_bwd,
use_high_prec_o,
smooth_q,
use_global_sf_p,
use_global_sf_qkv,
)
return o_BHLD.permute(0, 2, 1, 3).contiguous()
class AttnQatTrainBackend(AttentionBackend):
accept_output_buffer: bool = True
@staticmethod
def get_supported_head_sizes() -> list[int]:
return [64, 96, 128, 160, 192, 224, 256]
@staticmethod
def get_name() -> str:
return "ATTN_QAT_TRAIN"
@staticmethod
def get_impl_cls() -> type["AttnQatTrainImpl"]:
return AttnQatTrainImpl
@staticmethod
def get_metadata_cls() -> type["AttentionMetadata"]:
raise NotImplementedError
@staticmethod
def get_builder_cls() -> type["AttentionMetadataBuilder"]:
raise NotImplementedError
class AttnQatTrainImpl(AttentionImpl):
def __init__(
self,
num_heads: int,
head_size: int,
causal: bool,
softmax_scale: float,
num_kv_heads: int | None = None,
prefix: str = "",
**extra_impl_args,
) -> None:
self.causal = causal
self.softmax_scale = softmax_scale
self.dropout = extra_impl_args.get("dropout_p", 0.0)
def forward(
self,
query: torch.Tensor,
key: torch.Tensor,
value: torch.Tensor,
attn_metadata: AttentionMetadata,
) -> torch.Tensor:
return attn_qat_train(query, key, value, is_causal=self.causal)
+3 -5
View File
@@ -33,13 +33,11 @@ from fastvideo.logger import init_logger
try:
from fastvideo.attention.utils.flash_attn_no_pad import (
flash_attn_varlen_func_impl, )
FLASH_ATTN_AVAILABLE = True
except ImportError:
try:
from flash_attn import flash_attn_varlen_func as flash_attn_varlen_func_impl
FLASH_ATTN_AVAILABLE = True
except ImportError:
FLASH_ATTN_AVAILABLE = False
flash_attn_varlen_func_impl = None
FLASH_ATTN_AVAILABLE = False
logger = init_logger(__name__)
+1 -1
View File
@@ -16,7 +16,7 @@ class SageAttention3Backend(AttentionBackend):
@staticmethod
def get_supported_head_sizes() -> list[int]:
return [64, 128, 256]
return [64, 128]
@staticmethod
def get_name() -> str:
@@ -133,7 +133,6 @@ class VideoSparseAttentionBackend(AttentionBackend):
class VideoSparseAttentionMetadata(AttentionMetadata):
current_timestep: int
dit_seq_shape: list[int]
VSA_sparsity: float
num_tiles: list[int]
total_seq_length: int
tile_partition_indices: torch.LongTensor
@@ -144,10 +143,10 @@ class VideoSparseAttentionMetadata(AttentionMetadata):
class VideoSparseAttentionMetadataBuilder(AttentionMetadataBuilder):
def __init__(self):
def __init__(self) -> None:
pass
def prepare(self):
def prepare(self) -> None:
pass
def build( # type: ignore
+8 -4
View File
@@ -61,10 +61,10 @@ class VideoMobaAttentionMetadata(AttentionMetadata):
class VideoMobaAttentionMetadataBuilder(AttentionMetadataBuilder):
def __init__(self):
def __init__(self) -> None:
pass
def prepare(self):
def prepare(self) -> None:
pass
def build( # type: ignore
@@ -81,7 +81,7 @@ class VideoMobaAttentionMetadataBuilder(AttentionMetadataBuilder):
moba_select_mode: str = 'threshold',
moba_threshold: float = 0.25,
moba_threshold_type: str = 'query_head',
device: torch.device = None,
device: torch.device | None = None,
first_full_layer: int = 0,
first_full_step: int = 12,
temporal_layer: int = 1,
@@ -142,7 +142,7 @@ class VMOBAAttentionImpl(AttentionImpl):
query: torch.Tensor,
key: torch.Tensor,
value: torch.Tensor,
attn_metadata: AttentionMetadata,
attn_metadata: VideoMobaAttentionMetadata,
) -> torch.Tensor:
"""
query: [B, L, H, D]
@@ -154,7 +154,9 @@ class VMOBAAttentionImpl(AttentionImpl):
# select chunk type according to layer idx:
loop_layer_num = attn_metadata.temporal_layer + attn_metadata.spatial_layer + attn_metadata.st_layer
assert self.layer_idx is not None, "VMoBA attention requires layer_idx to be set"
moba_layer = self.layer_idx - attn_metadata.first_full_layer
moba_chunk_size: int | tuple[int, int] | tuple[int, int, int]
if moba_layer % loop_layer_num < attn_metadata.temporal_layer:
moba_chunk_size = attn_metadata.temporal_chunk_size
moba_topk = attn_metadata.temporal_topk
@@ -164,6 +166,8 @@ class VMOBAAttentionImpl(AttentionImpl):
elif moba_layer % loop_layer_num < attn_metadata.temporal_layer + attn_metadata.spatial_layer + attn_metadata.st_layer:
moba_chunk_size = attn_metadata.st_chunk_size
moba_topk = attn_metadata.st_topk
else:
raise ValueError(f"Invalid MoBA layer selection for layer {moba_layer}")
query, chunk_size = process_moba_input(query, attn_metadata.patch_resolution, moba_chunk_size)
key, chunk_size = process_moba_input(key, attn_metadata.patch_resolution, moba_chunk_size)
+2
View File
@@ -158,6 +158,7 @@ class DistributedAttention_VSA(DistributedAttention):
replicated_v: torch.Tensor | None = None,
gate_compress: torch.Tensor | None = None,
freqs_cis: tuple[torch.Tensor, torch.Tensor] | None = None,
attention_mask: torch.Tensor | None = None,
) -> tuple[torch.Tensor, torch.Tensor | None]:
"""Forward pass for distributed attention.
@@ -170,6 +171,7 @@ class DistributedAttention_VSA(DistributedAttention):
replicated_q (Optional[torch.Tensor]): Replicated query tensor, typically for text tokens
replicated_k (Optional[torch.Tensor]): Replicated key tensor
replicated_v (Optional[torch.Tensor]): Replicated value tensor
attention_mask (Optional[torch.Tensor]): Attention mask [batch_size, seq_len]
Returns:
Tuple[torch.Tensor, Optional[torch.Tensor]]: A tuple containing:
+23 -16
View File
@@ -85,7 +85,26 @@ def get_attn_backend(
supported_attention_backends: tuple[AttentionBackendEnum, ...]
| None = None,
) -> type[AttentionBackend]:
return _cached_get_attn_backend(head_size, dtype, supported_attention_backends)
selected_backend, is_forced = _resolve_backend_override()
return _cached_get_attn_backend(
head_size,
dtype,
supported_attention_backends,
selected_backend,
is_forced,
)
def _resolve_backend_override() -> tuple[AttentionBackendEnum | None, bool]:
backend_by_global_setting = get_global_forced_attn_backend()
if backend_by_global_setting is not None:
return backend_by_global_setting, True
backend_by_env_var: str | None = envs.FASTVIDEO_ATTENTION_BACKEND
if backend_by_env_var is not None:
return backend_name_to_enum(backend_by_env_var), False
return None, False
@cache
@@ -94,28 +113,16 @@ def _cached_get_attn_backend(
dtype: torch.dtype,
supported_attention_backends: tuple[AttentionBackendEnum, ...]
| None = None,
selected_backend: AttentionBackendEnum | None = None,
is_forced_backend: bool = False,
) -> type[AttentionBackend]:
# Check whether a particular choice of backend was
# previously forced.
#
# THIS SELECTION OVERRIDES THE FASTVIDEO_ATTENTION_BACKEND
# ENVIRONMENT VARIABLE.
if not supported_attention_backends:
raise ValueError("supported_attention_backends is empty")
selected_backend = None
backend_by_global_setting: AttentionBackendEnum | None = (get_global_forced_attn_backend())
if backend_by_global_setting is not None:
selected_backend = backend_by_global_setting
else:
# Check the environment variable and override if specified
backend_by_env_var: str | None = envs.FASTVIDEO_ATTENTION_BACKEND
if backend_by_env_var is not None:
selected_backend = backend_name_to_enum(backend_by_env_var)
# get device-specific attn_backend
from fastvideo.platforms import current_platform
if selected_backend not in supported_attention_backends:
if not is_forced_backend and selected_backend not in supported_attention_backends:
selected_backend = None
attention_cls = current_platform.get_attn_backend_cls(selected_backend, head_size, dtype)
if not attention_cls:
+49 -35
View File
@@ -14,30 +14,44 @@
# of rights and permissions under this agreement.
# See the License for the specific language governing permissions and limitations under the License.
from einops import rearrange
from flash_attn.bert_padding import pad_input, unpad_input
from flash_attn import flash_attn_varlen_qkvpacked_func
from typing import Any
try:
from fastvideo.attention.utils.flash_attn_cute import (
flash_attn_varlen_func as flash_attn_varlen_func_impl, )
except ImportError:
import torch
from einops import rearrange
from flash_attn import flash_attn_varlen_qkvpacked_func
from flash_attn.bert_padding import pad_input, unpad_input
def _resolve_flash_attn_varlen_func() -> Any:
try:
from flash_attn_interface import (
flash_attn_varlen_func as flash_attn_varlen_func_impl, )
from fastvideo.attention.utils.flash_attn_cute import (
flash_attn_varlen_func as flash_attn_varlen_func_cute, )
return flash_attn_varlen_func_cute
except ImportError:
from flash_attn import (
flash_attn_varlen_func as flash_attn_varlen_func_impl, )
try:
from flash_attn_interface import (
flash_attn_varlen_func as flash_attn_varlen_func_interface, )
return flash_attn_varlen_func_interface
except ImportError:
from flash_attn import (
flash_attn_varlen_func as flash_attn_varlen_func_flash, )
return flash_attn_varlen_func_flash
flash_attn_varlen_func_impl = _resolve_flash_attn_varlen_func()
def flash_attn_no_pad(
qkv,
key_padding_mask,
causal=False,
dropout_p=0.0,
softmax_scale=None,
deterministic=False,
):
qkv: torch.Tensor,
key_padding_mask: torch.Tensor,
causal: bool = False,
dropout_p: float = 0.0,
softmax_scale: float | None = None,
deterministic: bool = False,
) -> torch.Tensor:
batch_size = qkv.shape[0]
seqlen = qkv.shape[1]
nheads = qkv.shape[-2]
@@ -68,13 +82,13 @@ def flash_attn_no_pad(
def flash_attn_no_pad_v3(
qkv,
key_padding_mask,
causal=False,
dropout_p=0.0,
softmax_scale=None,
deterministic=False,
):
qkv: torch.Tensor,
key_padding_mask: torch.Tensor,
causal: bool = False,
dropout_p: float = 0.0,
softmax_scale: float | None = None,
deterministic: bool = False,
) -> torch.Tensor:
from flash_attn_interface import (
flash_attn_varlen_func as flash_attn_varlen_func_v3, )
@@ -120,16 +134,16 @@ def flash_attn_no_pad_v3(
def flash_attn_varlen_qk_no_pad(
query,
key,
value,
query_padding_mask,
key_padding_mask,
causal=False,
dropout_p=0.0,
softmax_scale=None,
deterministic=False,
):
query: torch.Tensor,
key: torch.Tensor,
value: torch.Tensor,
query_padding_mask: torch.Tensor,
key_padding_mask: torch.Tensor,
causal: bool = False,
dropout_p: float = 0.0,
softmax_scale: float | None = None,
deterministic: bool = False,
) -> torch.Tensor:
batch_size, q_seqlen, nheads, _ = query.shape
query_unpad, q_indices, cu_seqlens_q, max_seqlen_q, _ = unpad_input(rearrange(query, "b s h d -> b s (h d)"),
+14 -4
View File
@@ -12,8 +12,15 @@ logger = init_logger(__name__)
# 3. Any field in ArchConfig is fixed upon initialization, and should be hidden away from users
@dataclass
class ArchConfig:
stacked_params_mapping: list[tuple[str, str, str]] = field(
stacked_params_mapping: list[tuple[str, str, str | int]] = field(
default_factory=list) # mapping from huggingface weight names to custom names
output_hidden_states: bool = False
def __post_init__(self) -> None:
pass
def __getattr__(self, name: str) -> Any:
raise AttributeError(f"'{type(self).__name__}' object has no attribute '{name}'")
@dataclass
@@ -24,19 +31,22 @@ class ModelConfig:
# FastVideo-specific parameters here
def __getattr__(self, name):
def __post_init__(self) -> None:
pass
def __getattr__(self, name: str) -> Any:
# Only called if 'name' is not found in ModelConfig directly
if hasattr(self.arch_config, name):
return getattr(self.arch_config, name)
raise AttributeError(f"'{type(self).__name__}' object has no attribute '{name}'")
def __getstate__(self):
def __getstate__(self) -> dict[str, Any]:
# Return a dictionary of attributes to pickle
# Convert to dict and exclude any problematic attributes
state = self.__dict__.copy()
return state
def __setstate__(self, state):
def __setstate__(self, state: dict[str, Any]) -> None:
# Restore instance attributes from the unpickled state
self.__dict__.update(state)
+16 -3
View File
@@ -19,13 +19,19 @@ class DiTArchConfig(ArchConfig):
AttentionBackendEnum.TORCH_SDPA,
AttentionBackendEnum.VIDEO_SPARSE_ATTN,
AttentionBackendEnum.VMOBA_ATTN, AttentionBackendEnum.SAGE_ATTN_THREE,
AttentionBackendEnum.SLA_ATTN, AttentionBackendEnum.SAGE_SLA_ATTN)
AttentionBackendEnum.ATTN_QAT_INFER,
AttentionBackendEnum.ATTN_QAT_TRAIN, AttentionBackendEnum.SLA_ATTN,
AttentionBackendEnum.SAGE_SLA_ATTN)
hidden_size: int = 0
num_attention_heads: int = 0
num_channels_latents: int = 0
in_channels: int = 0
out_channels: int = 0
in_channels: int | None = 0
out_channels: int | None = 0
patch_size: int | tuple[int, int, int] | None = None
expand_timesteps: bool = False
num_layers: int = 0
ffn_dim: int = 0
exclude_lora_layers: list[str] = field(default_factory=list)
boundary_ratio: float | None = None
@@ -41,6 +47,13 @@ class DiTConfig(ModelConfig):
# FastVideoDiT-specific parameters
prefix: str = ""
quant_config: QuantizationConfig | None = None
expand_timesteps: bool = False
boundary_ratio: float | None = None
def __post_init__(self) -> None:
super().__post_init__()
self.arch_config.expand_timesteps = self.expand_timesteps
self.arch_config.boundary_ratio = self.boundary_ratio
@staticmethod
def add_cli_args(parser: Any, prefix: str = "dit-config") -> Any:
+2 -2
View File
@@ -85,7 +85,7 @@ class HYWorldArchConfig(DiTArchConfig):
reverse_param_names_mapping: dict = field(default_factory=lambda: {})
# Parameters from HY-WorldPlay config.json (loaded from checkpoint)
patch_size: list | tuple | int = field(default_factory=lambda: [1, 1, 1])
patch_size: tuple[int, int, int] = (1, 1, 1)
# Base latent channels - will be expanded in __post_init__ if concat_condition=True
in_channels: int = 32
concat_condition: bool = True
@@ -120,7 +120,7 @@ class HYWorldArchConfig(DiTArchConfig):
task_type: str = "i2v"
exclude_lora_layers: list[str] = field(default_factory=lambda: ["img_in", "txt_in", "time_in", "vector_in"])
def __post_init__(self):
def __post_init__(self) -> None:
super().__post_init__()
# Convert HY-WorldPlay naming to FastVideo naming conventions
self.num_attention_heads: int = self.heads_num
+6 -5
View File
@@ -24,15 +24,16 @@ class TextEncoderArchConfig(EncoderArchConfig):
hidden_size: int = 0
num_hidden_layers: int = 0
num_attention_heads: int = 0
pad_token_id: int = 0
eos_token_id: int = 0
pad_token_id: int | None = 0
eos_token_id: int | None = 0
text_len: int = 0
hidden_state_skip_layer: int = 0
decoder_start_token_id: int = 0
output_past: bool = True
scalable_attention: bool = True
tie_word_embeddings: bool = False
stacked_params_mapping: list[tuple[str, str, str]] = field(
padding_side: str = "right"
stacked_params_mapping: list[tuple[str, str, str | int]] = field(
default_factory=list) # mapping from huggingface weight names to custom names
tokenizer_kwargs: dict[str, Any] = field(default_factory=dict)
_fsdp_shard_conditions: list = field(default_factory=lambda: [])
@@ -70,10 +71,10 @@ class EncoderConfig(ModelConfig):
@dataclass
class TextEncoderConfig(EncoderConfig):
arch_config: ArchConfig = field(default_factory=TextEncoderArchConfig)
arch_config: TextEncoderArchConfig = field(default_factory=TextEncoderArchConfig)
is_chat_model: bool = False
@dataclass
class ImageEncoderConfig(EncoderConfig):
arch_config: ArchConfig = field(default_factory=ImageEncoderArchConfig)
arch_config: ImageEncoderArchConfig = field(default_factory=ImageEncoderArchConfig)
+2 -2
View File
@@ -32,7 +32,7 @@ class CLIPTextArchConfig(TextEncoderArchConfig):
bos_token_id: int = 49406
eos_token_id: int = 49407
text_len: int = 77
stacked_params_mapping: list[tuple[str, str, str]] = field(default_factory=lambda: [
stacked_params_mapping: list[tuple[str, str, str | int]] = field(default_factory=lambda: [
# (param_name, shard_name, shard_id)
("qkv_proj", "q_proj", "q"),
("qkv_proj", "k_proj", "k"),
@@ -57,7 +57,7 @@ class CLIPVisionArchConfig(ImageEncoderArchConfig):
attention_dropout: float = 0.0
initializer_range: float = 0.02
initializer_factor: float = 1.0
stacked_params_mapping: list[tuple[str, str, str]] = field(default_factory=lambda: [
stacked_params_mapping: list[tuple[str, str, str | int]] = field(default_factory=lambda: [
# (param_name, shard_name, shard_id)
("qkv_proj", "q_proj", "q"),
("qkv_proj", "k_proj", "k"),
+2 -2
View File
@@ -41,7 +41,7 @@ class T5ArchConfig(TextEncoderArchConfig):
text_len: int = 512
dtype: str | None = None
gradient_checkpointing: bool = False
stacked_params_mapping: list[tuple[str, str, str]] = field(default_factory=lambda: [
stacked_params_mapping: list[tuple[str, str, str | int]] = field(default_factory=lambda: [
# (param_name, shard_name, shard_id)
(".qkv_proj", ".q", "q"),
(".qkv_proj", ".k", "k"),
@@ -51,7 +51,7 @@ class T5ArchConfig(TextEncoderArchConfig):
default_factory=lambda: [_is_transformer_layer, _is_embeddings, _is_final_layernorm])
# Referenced from https://github.com/huggingface/transformers/blob/main/src/transformers/models/t5/configuration_t5.py
def __post_init__(self):
def __post_init__(self) -> None:
super().__post_init__()
act_info = self.feed_forward_proj.split("-")
self.dense_act_fn: str = act_info[-1]
+7 -1
View File
@@ -13,6 +13,8 @@ from fastvideo.utils import StoreBoolean
@dataclass
class VAEArchConfig(ArchConfig):
scaling_factor: float | torch.Tensor = 0
scale_factor_temporal: int = 1
scale_factor_spatial: int = 1
temporal_compression_ratio: int = 4
spatial_compression_ratio: int = 8
@@ -37,8 +39,12 @@ class VAEConfig(ModelConfig):
use_tiling: bool = True
use_temporal_tiling: bool = True
use_parallel_tiling: bool = True
ltx2_spatial_tile_size_in_pixels: int | None = None
ltx2_spatial_tile_overlap_in_pixels: int | None = None
ltx2_temporal_tile_size_in_frames: int | None = None
ltx2_temporal_tile_overlap_in_frames: int | None = None
def __post_init__(self):
def __post_init__(self) -> None:
self.blend_num_frames = self.tile_sample_min_num_frames - self.tile_sample_stride_num_frames
@staticmethod
+9 -4
View File
@@ -31,7 +31,7 @@ class PipelineConfig:
pipeline_config_path: str | None = None
# Video generation parameters
embedded_cfg_scale: float = 6.0
embedded_cfg_scale: float | None = 6.0
flow_shift: float | None = None
flow_shift_sr: float | None = None
disable_autocast: bool = False
@@ -40,7 +40,7 @@ class PipelineConfig:
# Model configuration
dit_config: DiTConfig = field(default_factory=DiTConfig)
dit_precision: str = "bf16"
upsampler_config: UpsamplerConfig = field(default_factory=UpsamplerConfig)
upsampler_config: UpsamplerConfig | tuple[UpsamplerConfig, ...] = field(default_factory=UpsamplerConfig)
upsampler_precision: str = "fp32"
# VAE configuration
@@ -58,8 +58,7 @@ class PipelineConfig:
text_encoder_configs: tuple[EncoderConfig, ...] = field(default_factory=lambda: (EncoderConfig(), ))
text_encoder_precisions: tuple[str, ...] = field(default_factory=lambda: ("fp32", ))
preprocess_text_funcs: tuple[Callable[[str], str], ...] = field(default_factory=lambda: (preprocess_text, ))
postprocess_text_funcs: tuple[Callable[[BaseEncoderOutput], torch.tensor],
...] = field(default_factory=lambda: (postprocess_text, ))
postprocess_text_funcs: tuple[Callable[..., Any], ...] = field(default_factory=lambda: (postprocess_text, ))
# DMD parameters
dmd_denoising_steps: list[int] | None = field(default=None)
@@ -71,6 +70,12 @@ class PipelineConfig:
# Compilation
# enable_torch_compile: bool = False
def __post_init__(self) -> None:
pass
def __getattr__(self, name: str) -> Any:
raise AttributeError(f"'{type(self).__name__}' object has no attribute '{name}'")
@staticmethod
def add_cli_args(parser: FlexibleArgumentParser, prefix: str = "") -> FlexibleArgumentParser:
prefix_with_dot = f"{prefix}." if (prefix.strip() != "") else ""
+4 -4
View File
@@ -41,9 +41,9 @@ class Cosmos25Config(PipelineConfig):
in_channels=16,
out_channels=16,
num_layers=28,
patch_size=[1, 2, 2],
max_size=[128, 240, 240],
rope_scale=[1.0, 3.0, 3.0],
patch_size=(1, 2, 2),
max_size=(128, 240, 240),
rope_scale=(1.0, 3.0, 3.0),
text_embed_dim=1024,
mlp_ratio=4.0,
adaln_lora_dim=256,
@@ -75,7 +75,7 @@ class Cosmos25Config(PipelineConfig):
vae_tiling: bool = False
vae_sp: bool = False
def __post_init__(self):
def __post_init__(self) -> None:
self.vae_config.load_encoder = True
self.vae_config.load_decoder = True
self._vae_latent_dim = 16
+1 -1
View File
@@ -22,7 +22,7 @@ class _Gen3CT5LargeArchConfig(T5LargeArchConfig):
refactor [PR#1142](https://github.com/hao-ai-lab/FastVideo/pull/1142).
"""
def __post_init__(self):
def __post_init__(self) -> None:
super().__post_init__()
self.tokenizer_kwargs["padding"] = "max_length"
+1 -1
View File
@@ -112,7 +112,7 @@ class Hunyuan15T2V480PConfig(PipelineConfig):
vae_tiling: bool = True
def __post_init__(self):
def __post_init__(self) -> None:
self.vae_config.load_encoder = False
self.vae_config.load_decoder = True
if self.text_encoder_configs:
@@ -44,7 +44,7 @@ class HunyuanGameCraftPipelineConfig(PipelineConfig):
# Denoising parameters
# Official GameCraft does NOT use embedded guidance (passes guidance=None)
# It uses standard CFG with guidance_scale=6.0 instead
embedded_cfg_scale = None
embedded_cfg_scale: float | None = None
flow_shift: int = 5 # Official GameCraft uses flow_shift=5.0
# Text encoding stage - same as HunyuanVideo
@@ -60,7 +60,7 @@ class HunyuanGameCraftPipelineConfig(PipelineConfig):
vae_precision: str = "fp16"
text_encoder_precisions: tuple[str, ...] = field(default_factory=lambda: ("fp16", "fp16"))
def __post_init__(self):
def __post_init__(self) -> None:
# VAE only needs decoder for inference
self.vae_config.load_encoder = False
self.vae_config.load_decoder = True
+1 -1
View File
@@ -22,7 +22,7 @@ class HYWorldConfig(Hunyuan15T2V480PConfig):
# Text encoding
text_encoder_precisions: tuple[str, ...] = field(default_factory=lambda: ("fp16", "fp32"))
def __post_init__(self):
def __post_init__(self) -> None:
super().__post_init__()
self.vae_config.load_encoder = True
self.vae_config.load_decoder = True
+2 -2
View File
@@ -32,7 +32,7 @@ class LongCatDiTArchConfig(DiTArchConfig):
num_heads: int = 32
out_channels: int = 16
text_tokens_zero_pad: bool = True
patch_size: list[int] = field(default_factory=lambda: [1, 2, 2])
patch_size: tuple[int, int, int] = (1, 2, 2)
cp_split_hw: list[int] | None = None
bsa_params: dict | None = None
@@ -314,7 +314,7 @@ ASPECT_RATIO_960_F256 = {
}
def get_bucket_config(resolution, scale_factor_spatial):
def get_bucket_config(resolution: str, scale_factor_spatial: int) -> dict[str, tuple[list[int], int]]:
if resolution == '480p':
if scale_factor_spatial == 16 or scale_factor_spatial == 32:
return ASPECT_RATIO_627
+6 -2
View File
@@ -47,7 +47,7 @@ class SamplingParam:
# Text inputs
prompt: str | list[str] | None = None
negative_prompt: str = "Bright tones, overexposed, static, blurred details, subtitles, style, works, paintings, images, static, overall gray, worst quality, low quality, JPEG compression residue, ugly, incomplete, extra fingers, poorly drawn hands, poorly drawn faces, deformed, disfigured, misshapen limbs, fused fingers, still picture, messy background, three legs, many people in the background, walking backwards"
negative_prompt: str | None = "Bright tones, overexposed, static, blurred details, subtitles, style, works, paintings, images, static, overall gray, worst quality, low quality, JPEG compression residue, ugly, incomplete, extra fingers, poorly drawn hands, poorly drawn faces, deformed, disfigured, misshapen limbs, fused fingers, still picture, messy background, three legs, many people in the background, walking backwards"
prompt_path: str | None = None
output_path: str = "outputs/"
output_video_name: str | None = None
@@ -68,6 +68,7 @@ class SamplingParam:
num_inference_steps: int = 50
num_inference_steps_sr: int = 50
guidance_scale: float = 1.0
guidance_scale_2: float | None = None
guidance_rescale: float = 0.0
boundary_ratio: float | None = None
sigmas: list[float] | None = None
@@ -89,7 +90,10 @@ class SamplingParam:
def __post_init__(self) -> None:
self.data_type = "video" if self.num_frames > 1 else "image"
def check_sampling_param(self):
def __getattr__(self, name: str) -> Any:
raise AttributeError(f"'{type(self).__name__}' object has no attribute '{name}'")
def check_sampling_param(self) -> None:
if self.prompt_path and not self.prompt_path.endswith(".txt"):
raise ValueError("prompt_path must be a txt file")
@@ -134,3 +134,18 @@ class PyHcclCommunicator:
buffer = buffer_type(tensor.data_ptr())
self.hccl.hcclBroadcast(buffer, tensor.numel(), hcclDataTypeEnum.from_torch(tensor.dtype), src, self.comm,
aclrtStream_t(stream.npu_stream))
def send(self, tensor: torch.Tensor, dst: int) -> None:
if isinstance(self.group, StatelessProcessGroup):
self.group.send_obj(tensor.cpu(), dst)
return
dist.send(tensor, dst=dist.get_process_group_ranks(self.group)[dst], group=self.group)
def recv(self, tensor: torch.Tensor, src: int) -> None:
if isinstance(self.group, StatelessProcessGroup):
received = self.group.recv_obj(src)
if not isinstance(received, torch.Tensor):
raise TypeError(f"Expected a tensor from rank {src}, got {type(received)!r}")
tensor.copy_(received.to(device=tensor.device, dtype=tensor.dtype))
return
dist.recv(tensor, src=dist.get_process_group_ranks(self.group)[src], group=self.group)
+1 -1
View File
@@ -256,7 +256,7 @@ class StreamingVideoGenerator(VideoGenerator):
return frames
def shutdown(self):
def shutdown(self) -> None:
if self.writer:
self.writer.close()
self.writer = None
+6 -3
View File
@@ -65,6 +65,7 @@ _FROM_PRETRAINED_CONVENIENCE_KWARGS = frozenset({
"pin_cpu_memory",
"enable_torch_compile",
"torch_compile_kwargs",
"transformer_quant",
})
@@ -324,7 +325,7 @@ class VideoGenerator:
results.extend(wrapped)
continue
wrapped.prompt_index = index
if wrapped.prompt is None:
if wrapped.prompt is None and isinstance(prompt, str):
wrapped.prompt = prompt
results.append(wrapped)
return results
@@ -357,7 +358,7 @@ class VideoGenerator:
def _generate_video_impl(
self,
prompt: str | None = None,
prompt: str | list[str] | None = None,
sampling_param: SamplingParam | None = None,
mouse_cond: torch.Tensor | None = None,
keyboard_cond: torch.Tensor | None = None,
@@ -429,6 +430,8 @@ class VideoGenerator:
# Single prompt generation (original behavior)
if prompt is None:
raise ValueError("Either prompt or prompt_txt must be provided")
if not isinstance(prompt, str):
raise ValueError("Single-prompt generation expects a string prompt")
output_path = self._prepare_output_path(sampling_param.output_path, prompt)
kwargs["output_path"] = output_path
return self._generate_single_video(
@@ -786,7 +789,7 @@ class VideoGenerator:
def merge_lora_weights(self) -> None:
self.executor.merge_lora_weights()
def shutdown(self):
def shutdown(self) -> None:
"""
Shutdown the video generator.
"""
+2
View File
@@ -201,6 +201,8 @@ environment_variables: dict[str, Callable[[], Any]] = {
# - "VIDEO_SPARSE_ATTN": use Video Sparse Attention
# - "SAGE_ATTN": use Sage Attention
# - "SAGE_ATTN_THREE": use Sage Attention 3
# - "ATTN_QAT_INFER": use the in-repo attn_qat_infer inference backend
# - "ATTN_QAT_TRAIN": use the FastVideoKernel Triton attn_qat_train backend
"FASTVIDEO_ATTENTION_BACKEND":
lambda: os.getenv("FASTVIDEO_ATTENTION_BACKEND", None),
+17 -4
View File
@@ -172,6 +172,7 @@ class FastVideoArgs:
override_text_encoder_safetensors: str | None = None # path to safetensors file for text encoder override
override_text_encoder_quant: QuantizationMethods = None
transformer_quant: QuantizationMethods = None
override_transformer_cls_name: str | None = None
init_weights_from_safetensors: str = "" # path to safetensors file for initial weight loading
@@ -183,7 +184,7 @@ class FastVideoArgs:
# dmd_denoising_steps: List[int] | None = field(default=None)
# MoE parameters used by Wan2.2
boundary_ratio: float | None = 0.875
boundary_ratio: float = 0.875
@property
def training_mode(self) -> bool:
@@ -201,6 +202,9 @@ class FastVideoArgs:
self._apply_ltx2_vae_overrides()
self.check_fastvideo_args()
def __getattr__(self, name: str) -> Any:
raise AttributeError(f"'{type(self).__name__}' object has no attribute '{name}'")
def _apply_ltx2_vae_overrides(self) -> None:
if self.pipeline_config is None:
return
@@ -461,8 +465,7 @@ class FastVideoArgs:
"--use-fsdp-inference",
action=StoreBoolean,
help=
"Use FSDP for inference by sharding the model weights. FSDP helps reduce GPU memory usage but may introduce"
+ " weight transfer overhead depending on the specific setup. Enable if run out of memory.",
"Use FSDP for inference by sharding the model weights. Latency is very low due to prefetch--enable if run out of memory.",
)
parser.add_argument(
"--text-encoder-cpu-offload",
@@ -528,6 +531,13 @@ class FastVideoArgs:
default=FastVideoArgs.override_text_encoder_quant,
help="Quantization method for text encoder override",
)
parser.add_argument(
"--transformer-quant",
type=str,
choices=QUANTIZATION_METHODS,
default=FastVideoArgs.transformer_quant,
help="Quantization method for transformer loading",
)
parser.add_argument(
"--override-transformer-cls-name",
type=str,
@@ -782,7 +792,8 @@ class TrainingArgs(FastVideoArgs):
trackers: list[str] = dataclasses.field(default_factory=list)
tracker_project_name: str = ""
wandb_run_name: str = ""
seed: int | None = None
seed: int = 0
_loading_teacher_critic_model: bool = False
# output
output_dir: str = ""
@@ -855,6 +866,8 @@ class TrainingArgs(FastVideoArgs):
# simulate generator forward to match inference
simulate_generator_forward: bool = False
warp_denoising_step: bool = False
generator_4bit_attn: bool = False
generator_4bit_linear: bool = False
# Self-forcing specific arguments
num_frame_per_block: int = 3
+115
View File
@@ -0,0 +1,115 @@
from typing import Any
import torch
try:
import flashinfer
except ImportError:
flashinfer = None
def _require_flashinfer() -> Any:
if flashinfer is None:
raise ImportError("flashinfer is required for FP4 linear layers. "
"Please install flashinfer to use this path.")
return flashinfer
class _LinearFWD4BWD16Fn(torch.autograd.Function):
@staticmethod
def forward(ctx, x, weight, bias, backend="cutlass", block_size=16, use_128x4_sf_layout=True):
flashinfer_mod = _require_flashinfer()
# assert activation dtype
if x.dtype not in (torch.float16, torch.bfloat16):
x = x.to(dtype=torch.bfloat16)
# cast params (can be fp32) to activation dtype for quantization
weight_cast = weight.to(dtype=x.dtype)
bias_cast = bias.to(dtype=x.dtype) if bias is not None else None
# shapes
orig_shape = x.shape
k = weight_cast.shape[1]
n = weight_cast.shape[0]
x2d = x.reshape(-1, k).contiguous()
M = x2d.shape[0]
out2d = torch.empty((M, n), device=x.device, dtype=x.dtype)
@torch.compile
def _global_sf(t: torch.Tensor) -> torch.Tensor:
maxabs = t.float().abs().nan_to_num().max()
maxabs = torch.maximum(maxabs, torch.tensor(1e-12, device=t.device, dtype=maxabs.dtype))
return (448.0 * 6.0) / maxabs
a_sf_layout = (flashinfer_mod.SfLayout.layout_128x4
if use_128x4_sf_layout else flashinfer_mod.SfLayout.layout_8x4)
global_sf_a = _global_sf(x2d)
global_sf_b = _global_sf(weight_cast)
a_fp4, a_inv_s = flashinfer_mod.nvfp4_quantize(
x2d,
global_sf_a,
sfLayout=a_sf_layout,
do_shuffle=False,
)
b_fp4, b_inv_s = flashinfer_mod.nvfp4_quantize(
weight_cast,
global_sf_b,
sfLayout=flashinfer_mod.SfLayout.layout_128x4,
do_shuffle=False,
)
alpha = 1.0 / (global_sf_a * global_sf_b)
flashinfer_mod.mm_fp4(
a_fp4,
b_fp4.T,
a_inv_s,
b_inv_s.T,
alpha,
x.dtype,
out2d,
block_size=block_size,
use_8x4_sf_layout=(not use_128x4_sf_layout),
backend=backend,
)
if bias_cast is not None:
out2d.add_(bias_cast)
# save tensors for backward (keep original dtypes)
ctx.save_for_backward(x2d, weight, bias)
ctx.k = k
ctx.n = n
ctx.orig_shape = orig_shape
return out2d.reshape(*orig_shape[:-1], n)
@staticmethod
def backward(ctx, grad_out):
x2d, weight, bias = ctx.saved_tensors
M = x2d.shape[0]
n = ctx.n
grad_out_2d = grad_out.reshape(M, n).contiguous()
# cast to grad dtype for matmuls
weight_cast = weight.to(dtype=grad_out.dtype)
x_cast = x2d.to(dtype=grad_out.dtype)
grad_x = grad_out_2d.matmul(weight_cast).reshape(*ctx.orig_shape)
grad_w = grad_out_2d.t().matmul(x_cast)
grad_b = grad_out_2d.sum(dim=0) if bias is not None else None
# None for the three extra forward args
return grad_x, grad_w, grad_b, None, None, None
def fp4_linear_forward(self, x: torch.Tensor) -> tuple[torch.Tensor, torch.Tensor | None]:
# pass config **positionally**; autograd.Function.apply ignores kwargs
bias = self.bias if not self.skip_bias_add else None
output = _LinearFWD4BWD16Fn.apply(x, self.weight, bias, "cutlass", 16, True)
output_bias = self.bias if self.skip_bias_add else None
return output, output_bias
+80 -33
View File
@@ -211,36 +211,30 @@ class ReplicatedLinear(LinearBase):
(e.g. model.layers.0.qkv_proj)
"""
def __init__(
self,
input_size: int,
output_size: int,
bias: bool = True,
skip_bias_add: bool = False,
params_dtype: torch.dtype | None = None,
quant_config: QuantizationConfig | None = None,
prefix: str = "",
):
super().__init__(
input_size,
output_size,
skip_bias_add,
params_dtype,
quant_config,
prefix=prefix,
)
enable_shape_tracking = False
_unique_shapes: set[tuple[torch.Size, torch.Size]] = set()
_shape_to_layer_types: dict[tuple[torch.Size, torch.Size], list[str]] = {}
def __init__(self,
input_size: int,
output_size: int,
bias: bool = True,
skip_bias_add: bool = False,
params_dtype: torch.dtype | None = None,
quant_config: QuantizationConfig | None = None,
prefix: str = ""):
super().__init__(input_size, output_size, skip_bias_add, params_dtype, quant_config, prefix=prefix)
# All the linear layer supports quant method.
assert self.quant_method is not None
self.quant_method.create_weights(
self,
self.input_size,
[self.output_size],
self.input_size,
self.output_size,
self.params_dtype,
weight_loader=self.weight_loader,
)
if self.quant_method is None:
self.quant_method = UnquantizedLinearMethod()
self.quant_method.create_weights(self,
self.input_size, [self.output_size],
self.input_size,
self.output_size,
self.params_dtype,
weight_loader=self.weight_loader)
if bias:
self.bias = Parameter(torch.empty(
@@ -269,8 +263,13 @@ class ReplicatedLinear(LinearBase):
def forward(self, x: torch.Tensor) -> tuple[torch.Tensor, Parameter | None]:
bias = self.bias if not self.skip_bias_add else None
assert self.quant_method is not None
if self.quant_method is None:
self.quant_method = UnquantizedLinearMethod()
output = self.quant_method.apply(self, x, bias)
if self.enable_shape_tracking:
self._track_shape(x.shape, output.shape)
output_bias = self.bias if self.skip_bias_add else None
return output, output_bias
@@ -280,6 +279,48 @@ class ReplicatedLinear(LinearBase):
s += f", bias={self.bias is not None}"
return s
@classmethod
def get_shape_mapping(cls) -> dict:
"""Get the mapping from (input_shape, output_shape) to layer types."""
return cls._shape_to_layer_types.copy()
@classmethod
def reset_shape_tracking(cls) -> None:
"""Clear tracked shapes and layer type mappings."""
cls._unique_shapes.clear()
cls._shape_to_layer_types.clear()
def _track_shape(self, input_shape: torch.Size, output_shape: torch.Size) -> None:
shape_key = (input_shape, output_shape)
if shape_key not in self._unique_shapes:
self._unique_shapes.add(shape_key)
self._shape_to_layer_types[shape_key] = []
print(f"Layer: {self.prefix} | input shape: {input_shape} --> "
f"output shape: {output_shape}, Quant Method: "
f"{self.quant_method.__class__.__name__}")
layer_type = self.__class__.__name__
if layer_type not in self._shape_to_layer_types[shape_key]:
self._shape_to_layer_types[shape_key].append(layer_type)
@classmethod
def print_shape_summary(cls):
"""Print a summary of all unique shapes and their layer types."""
if not cls._shape_to_layer_types:
print("No shapes have been processed yet.")
return
print("\n=== Matrix Multiplication Shape Summary ===")
print(f"Total unique shapes: {len(cls._shape_to_layer_types)}")
print()
for i, (shape_key, layer_types) in enumerate(cls._shape_to_layer_types.items(), 1):
input_shape, output_shape = shape_key
print(f"{i}. Input: {input_shape} → Output: {output_shape}")
print(f" Layer types: {', '.join(layer_types)}")
print()
class ColumnParallelLinear(LinearBase):
"""Linear layer with column parallelism.
@@ -340,7 +381,9 @@ class ColumnParallelLinear(LinearBase):
if output_sizes is None:
output_sizes = [output_size]
assert self.quant_method is not None
if self.quant_method is None:
self.quant_method = UnquantizedLinearMethod()
self.quant_method.create_weights(
layer=self,
input_size_per_partition=self.input_size_per_partition,
@@ -399,7 +442,8 @@ class ColumnParallelLinear(LinearBase):
bias = self.bias if not self.skip_bias_add else None
# Matrix multiply.
assert self.quant_method is not None
if self.quant_method is None:
self.quant_method = UnquantizedLinearMethod()
output_parallel = self.quant_method.apply(self, input_, bias)
# All-gather across the partitions if needed.
output = tensor_model_parallel_all_gather(output_parallel) if self.gather_output else output_parallel
@@ -916,7 +960,9 @@ class RowParallelLinear(LinearBase):
self.input_is_parallel = input_is_parallel
self.reduce_results = reduce_results
assert self.quant_method is not None
if self.quant_method is None:
self.quant_method = UnquantizedLinearMethod()
self.quant_method.create_weights(
layer=self,
input_size_per_partition=self.input_size_per_partition,
@@ -983,7 +1029,8 @@ class RowParallelLinear(LinearBase):
input_parallel = splitted_input[tp_rank].contiguous()
# Matrix multiply.
assert self.quant_method is not None
if self.quant_method is None:
self.quant_method = UnquantizedLinearMethod()
# Only fuse bias add into GEMM for rank 0 (this ensures that
# bias will not get added more than once in TP>1 case)
bias_ = None if (self.tp_rank > 0 or self.skip_bias_add) else self.bias
+20 -2
View File
@@ -5,6 +5,7 @@ import torch.nn as nn
from fastvideo.layers.activation import get_act_fn
from fastvideo.layers.linear import ReplicatedLinear
from fastvideo.layers.quantization import QuantizationConfig
class MLP(nn.Module):
@@ -21,18 +22,35 @@ class MLP(nn.Module):
act_type: str = "gelu_pytorch_tanh",
dtype: torch.dtype | None = None,
prefix: str = "",
quant_config: QuantizationConfig | None = None,
):
super().__init__()
self.fc_in = ReplicatedLinear(
input_dim,
mlp_hidden_dim, # For activation func like SiLU that need 2x width
bias=bias,
params_dtype=dtype)
params_dtype=dtype,
quant_config=quant_config,
prefix=f"{prefix}.fc_in",
)
if quant_config is not None:
quant_method = self.fc_in.quant_config.get_quant_method(self.fc_in, f"{prefix}.fc_in")
if quant_method is not None:
quant_method.process_weights_after_loading(self.fc_in)
self.act = get_act_fn(act_type)
if output_dim is None:
output_dim = input_dim
self.fc_out = ReplicatedLinear(mlp_hidden_dim, output_dim, bias=bias, params_dtype=dtype)
self.fc_out = ReplicatedLinear(mlp_hidden_dim,
output_dim,
bias=bias,
params_dtype=dtype,
quant_config=quant_config,
prefix=f"{prefix}.fc_out")
if quant_config is not None:
quant_method = self.fc_out.quant_config.get_quant_method(self.fc_out, f"{prefix}.fc_out")
if quant_method is not None:
quant_method.process_weights_after_loading(self.fc_out)
def forward(self, x: torch.Tensor) -> torch.Tensor:
x, _ = self.fc_in(x)

Some files were not shown because too many files have changed in this diff Show More