Compare commits

..
Author SHA1 Message Date
rlsu9 34a64597f4 update 2024-12-04 01:42:35 +04:00
rlsu9 4998121d6d update 2024-12-04 01:32:29 +04:00
rlsu9 1c25e3ed55 update 2024-12-04 01:24:30 +04:00
rlsu9 e98886f32f update 2024-12-03 10:03:26 +04:00
rlsu9 8619d2942d add l2 2024-12-03 09:53:44 +04:00
rlsu9 1876244531 update 2024-12-03 09:22:01 +04:00
rlsu9 34f709725e update 2024-12-03 01:44:07 +04:00
rlsu9 c5d6341f98 update 2024-12-02 22:45:29 +04:00
rlsu9 5325aa2cd2 xMerge branch 'main' of https://github.com/jzhang38/FastVideo-OSP into peiyuan 2024-12-02 22:31:48 +04:00
rlsu9 74d6fc92a9 update 2024-12-02 22:30:34 +04:00
haoanyscale daaefc5f63 fix finetune code bug 2024-12-02 11:45:04 +00:00
rlsu9 3eee42e917 update 2024-12-02 11:06:16 +04:00
rlsu9 28c0a5dce9 update 2024-12-02 11:03:44 +04:00
rlsu9 275dce4700 update 2024-12-02 10:53:14 +04:00
rlsu9 83fc3449e8 update 2024-12-02 10:44:00 +04:00
rlsu9 e96f359236 update 2024-12-02 05:12:47 +04:00
rlsu9 e89273a0da update 2024-12-02 04:18:38 +04:00
rlsu9 407e918676 runlong 2024-12-02 02:55:26 +04:00
rlsu9 94d86c168b wandb offline and dir 2024-12-02 01:31:24 +04:00
rlsu9 5a4c227572 update 2024-12-02 00:50:03 +04:00
rlsu9 554f934d37 update 2024-12-02 00:46:34 +04:00
rlsu9 51b72e7434 update 2024-12-01 23:05:29 +04:00
rlsu9 ebfbf6046a update 2024-12-01 21:56:36 +04:00
rlsu9 892324ae59 typo 2024-12-01 21:44:29 +04:00
rlsu9 b66e1a0427 new script 2024-12-01 21:40:25 +04:00
Hao Zhang 02ff44fcd1 update experiment 10 2024-12-01 05:43:24 +00:00
Hao Zhang eb551efe7f update scripts 2024-12-01 05:38:39 +00:00
rlsu9 4af3a99f9e typo 2024-12-01 09:36:42 +04:00
rlsu9 b3d16e0284 typo 2024-12-01 07:58:24 +04:00
rlsu9 c466fe9d73 Merge branch 'main' of https://github.com/jzhang38/FastVideo-OSP into peiyuan 2024-12-01 07:38:25 +04:00
rlsu9 494129344f debug gradient accumulation loss 2024-12-01 07:38:13 +04:00
Hao Zhang 8b9ecfdbcb update gitignore 2024-12-01 02:04:18 +00:00
rlsu9 58440f43de Merge branch 'main' of https://github.com/jzhang38/FastVideo-OSP into peiyuan 2024-12-01 05:26:04 +04:00
rlsu9 c8238e1f7b typo 2024-12-01 05:25:23 +04:00
Zhang Peiyuan 5ff08541f6 [Debug] Typo (#65) 2024-11-30 17:14:14 -08:00
rlsu9 f771899001 update 2024-12-01 05:13:51 +04:00
rlsu9 c609c28e63 update 2024-12-01 05:04:33 +04:00
rlsu9 c19f72f722 typo 2024-12-01 04:44:42 +04:00
rlsu9 43b4778474 Merge branch 'main' of https://github.com/jzhang38/FastVideo-OSP into peiyuan 2024-12-01 04:44:29 +04:00
Zhang Peiyuan 08f5489406 [Feat & Debug] fix uncond; multi guidance validaiton; multiphase schedule; linear range (#64) 2024-11-30 16:43:24 -08:00
rlsu9 7e6e52e02a update 2024-12-01 04:39:44 +04:00
rlsu9 e4114d8fd4 merge main 2024-12-01 04:35:01 +04:00
rlsu9 a19ed188c8 add script 2024-12-01 04:32:22 +04:00
rlsu9 6d3a71b681 linear range; fix uncond; multi guidance validation 2024-12-01 04:31:53 +04:00
Zhang Peiyuan 81103cd48d [Feat] EMA Distill; Distributed validation (#63) 2024-11-29 21:46:40 -08:00
rlsu9 a3d4a1cef4 add gupload 2024-11-30 09:41:37 +04:00
rlsu9 954dc8e35c distributed validation 2024-11-30 09:39:00 +04:00
rlsu9 278e8bb8f3 Merge branch 'main' into peiyuan 2024-11-30 05:28:37 +04:00
rlsu9 b378896b82 readme 2024-11-30 05:28:02 +04:00
rlsu9 1d2a7f233c distributed validation 2024-11-30 05:24:05 +04:00
rlsu9 b4aed1721f add aws efo env 2024-11-30 03:44:43 +04:00
rlsu9 3447ded21b revert to sp=4, sp bs=2, full shard 2024-11-30 03:43:39 +04:00
rlsu9 0c01223d3d update script; no sp 2024-11-30 02:51:22 +04:00
rlsu9 f8089afbf1 ema 2024-11-30 02:44:36 +04:00
rlsu9 2b0c1e66cb update env 2024-11-29 22:52:19 +04:00
rlsu9 79e7dac0ac ok 2024-11-29 21:50:49 +04:00
rlsu9 2b3f3cddd4 remove hardcode 2024-11-29 19:35:30 +04:00
rlsu9 064195363c remove harcode 2024-11-29 19:34:02 +04:00
rlsu9 ed1e17f37f add ema transformer 2024-11-29 11:42:17 +04:00
rlsu9 e32d137aec add ema_transform 2024-11-29 11:11:46 +04:00
rlsu9 62a5530c21 add 2024-11-29 11:08:32 +04:00
rlsu9 b3be22e1e2 add upload command 2024-11-29 09:44:13 +04:00
rlsu9 7156b8d7d5 typo 2024-11-29 09:12:22 +04:00
Zhang Peiyuan 6dd7980ec2 [Feat] Refactor GAN; State saving & Resume; Experiments script (#59) 2024-11-28 21:09:31 -08:00
Zhang Peiyuan b3c54b9e5b [Feat] Training precision (#57) 2024-11-27 16:25:07 -08:00
Zhang Peiyuan 333488ac09 [Feat][Debug] linear quadratic distill; HF precision bug (#56) 2024-11-27 15:13:20 -08:00
Zhang Peiyuan c7526707cc [Fix] Squeeze bug (#55) 2024-11-26 13:29:03 -08:00
Zhang Peiyuan a0ffbe2927 [Feat] PCM Distill; Refactor FM logit to be compatible with all SD3/Flux scheduler. (#54) 2024-11-26 12:26:24 -08:00
rlsu9andRunlong faef037f2e [feat]: Add batchy data preprocess (#53)
Co-authored-by:Runlong <rlsu9@ucsd.edu>
2024-11-25 21:52:36 -08:00
rlsu9andrunlong 9c45a15f53 [feat]: Add Image-Video Mixture training to main repo (#50)
Co-authored-by: runlong <r3su@ucsd.edu>
2024-11-24 14:36:45 -08:00
Zhang Peiyuan e6ff7ae2a3 Resolve config bug and seed (#51)
Co-authored-by: Peiyuan Zhang <a1286225768@gmail.com>
2024-11-15 20:29:48 -08:00
Zhang Peiyuan 897afb8b94 Add LADD (#45)
Co-authored-by: Peiyuan Zhang <a1286225768@gmail.com>
2024-11-15 19:28:41 -08:00
f8178a5a14 add lr scheduler; precision bug fix; add naive dataloader resume (#49)
Co-authored-by: Peiyuan Zhang <a1286225768@gmail.com>
Co-authored-by: Yongqi Chen <yongqich@gl1712.arc-ts.umich.edu>
Co-authored-by: Yongqi Chen <yongqich@gl-login3.arc-ts.umich.edu>
Co-authored-by: Yongqi Chen <yongqich@gl-login1.arc-ts.umich.edu>
2024-11-14 19:57:38 -08:00
4c76b3cc7a [Feat] Lora resume (#48)
Co-authored-by: Peiyuan Zhang <a1286225768@gmail.com>
Co-authored-by: Yongqi Chen <yongqich@gl1712.arc-ts.umich.edu>
Co-authored-by: Yongqi Chen <yongqich@gl-login3.arc-ts.umich.edu>
Co-authored-by: Yongqi Chen <yongqich@gl-login1.arc-ts.umich.edu>
2024-11-12 22:30:58 -08:00
3e0e8cf534 Add lora (#47)
Co-authored-by: Yongqi Chen <144848849+BrianChen1129@users.noreply.github.com>
Co-authored-by: Yongqi Chen <yongqich@gl1712.arc-ts.umich.edu>
Co-authored-by: Yongqi Chen <yongqich@gl-login3.arc-ts.umich.edu>
Co-authored-by: Yongqi Chen <yongqich@gl-login1.arc-ts.umich.edu>
2024-11-11 13:12:14 -08:00
Zhang Peiyuan 635fb2350d [Refactor] Switch to FSDP (#42) 2024-11-09 15:46:25 -08:00
Zhang Peiyuan a52f29374c No checkout mochi 2024-11-07 20:49:14 -08:00
rlsu9andhaoanyscale 31827c9e0a [feat]: Add adaptive fps dataloader and remove redundant code (#41)
Co-authored-by: haoanyscale <c-hao.zhang@anyscale.com>
2024-11-06 19:52:48 -08:00
rlsu9andhaoanyscale 74688e56b5 [feat]: Add vae encoder embedded generator to main (#30)
Co-authored-by: haoanyscale <c-hao.zhang@anyscale.com>
2024-11-06 11:58:15 -08:00
Zhang Peiyuan e56b2ae0c4 Add validation logging with SP (#36) 2024-11-05 14:46:20 -08:00
Zhang Peiyuan 6c6d4b34e7 [Feat] Sequence Parallel (#31) 2024-11-05 08:13:01 -08:00
Zhang Peiyuan 93d6ee49ba Merge pull request #25 from jzhang38/peiyuan
SP inference and Overfit training done.
2024-11-01 17:27:30 -07:00
rlsu9 e678eeb2dc Merge branch 'main' of https://github.com/jzhang38/FastVideo-OSP into peiyuan 2024-11-02 04:26:41 +04:00
rlsu9 432403b12d SP inference done! 2024-11-02 04:21:45 +04:00
rlsu9 8e899953ff Merge branch 'hl/diffusers' of https://github.com/jzhang38/FastVideo-OSP into peiyuan 2024-11-01 20:41:49 +04:00
rlsu9 fd80dc653d switch to deepspeed dummyoptim 2024-11-01 20:41:05 +04:00
foreverpiano 9fb510b618 fix some bugs; still output green 2024-11-01 12:30:38 +00:00
rlsu9 97f48ef433 Merge branch 'hl/diffusers' of https://github.com/jzhang38/FastVideo-OSP into peiyuan 2024-11-01 04:44:23 +04:00
rlsu9 5975deed62 OK 2024-11-01 04:26:21 +04:00
rlsu9 60c420731d Add zero3 2024-10-31 23:42:49 +04:00
rlsu9 a9a66a22a0 Debug successful! 2024-10-31 23:20:46 +04:00
Zhang Peiyuan fd415488df Merge pull request #21 from jzhang38/data_preprocess
[feat]: update data processor with multi GPU data parallel
2024-10-30 17:38:37 -07:00
haoanyscale a496c1c656 update data preprocess 2024-10-31 00:13:05 +00:00
rlsu9 2788f9351a Merge pull request #20 from jzhang38/main
Merge pull request #18 from jzhang38/peiyuan
2024-10-30 17:10:17 -07:00
rlsu9 5209b6ebdd debugging .. 2024-10-30 21:25:48 +04:00
foreverpiano db2c68e879 update inference sp code / can run / still has bug / don't output normal mp4 2024-10-30 16:36:50 +00:00
rlsu9 682774d730 Merge pull request #18 from jzhang38/peiyuan
[feat]: Merge training code into main branch
2024-10-30 09:29:50 -07:00
foreverpiano 2f4d6caa90 rename 2024-10-30 12:43:53 +00:00
haoanyscale dcd8e95c7b fix typos in readme and latent dataset debug file 2024-10-30 05:47:51 +00:00
a1286225768@gmail.com 9e2c9eaa01 overfitting .... 2024-10-30 04:14:04 +00:00
Peiyuan Zhang 567f9f66f8 debug 2024-10-29 21:10:50 +00:00
foreverpiano 48a82aba37 sp enable & still has bug 2024-10-29 15:28:27 +00:00
foreverpiano 4143cd1f95 random seed args 2024-10-29 15:27:52 +00:00
foreverpiano 5b3bffcca9 small bug 2024-10-29 14:17:54 +00:00
foreverpiano 37b091f2ef load 2024-10-29 12:47:48 +00:00
foreverpiano 70f3407032 update optimizer 2024-10-29 12:42:04 +00:00
Peiyuan Zhang d22d40c74a 14 2024-10-29 03:47:08 +00:00
Peiyuan Zhang 6fe5521667 Update generate_synthetic.sh and deepspeed_zero2_config.yaml 2024-10-29 01:00:31 +00:00
Peiyuan Zhang d1422b646d ok 2024-10-29 00:39:26 +00:00
Peiyuan Zhang d4e071be20 training 2024-10-29 00:38:40 +00:00
Peiyuan Zhang 7455bc9e7f commit first 2024-10-28 23:39:15 +00:00
Peiyuan Zhang 45c89c46c5 commit first 2024-10-28 21:46:01 +00:00
Peiyuan Zhang 7ec6f8f720 Deleted unnecessary files 2024-10-28 19:21:00 +00:00
Peiyuan Zhang cd6f8f14f1 remove files 2024-10-28 17:56:34 +00:00
Peiyuan Zhang 8824f74609 clean up name changing 2024-10-28 17:54:38 +00:00
Peiyuan Zhang 1ff9d45044 Remove OSP modeling 2024-10-28 17:53:13 +00:00
Peiyuan Zhang ffe11218dd Change name to fast video 2024-10-28 17:52:49 +00:00
Peiyuan Zhang 6c723e75f3 amend log validation 2024-10-28 17:50:34 +00:00
Peiyuan Zhang 4c3a83a0b3 Delete Open Sora Plan Modeling 2024-10-28 17:20:12 +00:00
Peiyuan Zhang d8489bf368 Generate synthetic dataset 2024-10-28 00:44:02 +00:00
Peiyuan Zhang f6a6366c58 Original pipeline 2024-10-27 23:30:04 +00:00
Peiyuan Zhang 5f8b952ede typo 2024-10-27 22:57:54 +00:00
Peiyuan Zhang a07aee80c3 Mochi Inferenfce & Diffusers 2024-10-27 22:57:14 +00:00
Peiyuan Zhang 235b6498b0 Refactor OpenSora sample_t2v.py and update download_hf 2024-10-26 21:59:22 +00:00
Peiyuan Zhang a89be390e7 remove vae loss 2024-10-26 19:20:04 +00:00
Peiyuan Zhang 3b8c76ea64 Remove files in causalvae 2024-10-26 19:06:46 +00:00
Peiyuan Zhang 63950da0d4 normalize 255; vae reconstruct 2024-10-26 18:57:07 +00:00
Peiyuan Zhang f245e4b4f8 Add mochi download & Change output dir 2024-10-26 18:08:50 +00:00
Peiyuan Zhang 1a8f9196ea Delete merge_data.txt 2024-10-26 18:00:16 +00:00
Peiyuan Zhang ea372da285 Remove unused adaptor files 2024-10-26 17:59:04 +00:00
Peiyuan Zhang fac510d043 Remove unused arguments in train_t2v_diffusers.py 2024-10-26 17:54:43 +00:00
Peiyuan Zhang b6b1a2aa92 Update T5Base 2024-10-26 17:53:51 +00:00
Peiyuan Zhang 464c8a421f Update PyTorch installation command 2024-10-26 17:36:23 +00:00
Peiyuan Zhang 602bf19d93 Update model path and cache directory 2024-10-25 10:21:31 +00:00
Peiyuan Zhang 84cf3b6795 Update t2v_debug_multi.sh with video_length_tolerance_range and dataloader_num_workers 2024-10-25 09:58:41 +00:00
Peiyuan Zhang 83a0280a14 update version 2024-10-25 09:22:16 +00:00
Peiyuan Zhang 182f4082f5 Update max height and width for video processing 2024-10-25 04:28:31 +00:00
Peiyuan Zhang 8d3f651217 Add pretrained model for OpenSoraT2V-ROPE-L 2024-10-25 04:26:57 +00:00
Peiyuan Zhang 330717768d include pretrained open-sora 2024-10-25 04:17:36 +00:00
Peiyuan Zhang ef653171dd Merge branch 'peiyuan' of https://github.com/jzhang38/FastVideo-OSP into peiyuan 2024-10-25 03:57:05 +00:00
Peiyuan Zhang 15fbad6dd5 Remove UDIT and inpaint 2024-10-25 03:55:15 +00:00
Peiyuan Zhang c00cc4b531 Update PyTorch index URLs and video length tolerance range 2024-10-25 03:18:16 +00:00
Peiyuan Zhang 1f423fca06 Fix warning with dataset handling and model loading 2024-10-25 02:31:07 +00:00
rlsu9 2cc0e17ade fix typo for dataset download 2024-10-25 01:40:26 +00:00
Peiyuan Zhang b1564eb141 Remove compress kv 2024-10-25 00:34:07 +00:00
Peiyuan Zhang 97f2b7fdaa Delete npu related stuff and remove inpaint module 2024-10-24 22:57:31 +00:00
Peiyuan Zhang 83825e2060 Update EMA model and t2v_debug.sh script 2024-10-24 22:19:48 +00:00
Peiyuan Zhang da9c1ab7c1 Update torchvision imports 2024-10-24 22:11:25 +00:00
Peiyuan Zhang 7c5b0a6ccd Remove all npu code 2024-10-24 21:35:27 +00:00
Peiyuan Zhang 5f21e3a979 Delete unused files and code 2024-10-24 21:09:24 +00:00
Peiyuan Zhang b47585c68d Update dependencies; Setup code for debugging 2024-10-24 21:01:32 +00:00
Peiyuan Zhang eb1669dab7 Add environment setup and training instructions to README.md 2024-10-24 17:47:34 +00:00
jzhang38 c1eb81f0f9 Remove NPU related code and update training process 2024-10-24 02:36:18 +00:00
jzhang38 8fb5c38381 Remove unused scripts and update TODO list 2024-10-24 02:27:03 +00:00
jzhang38 6e313fad97 Delete unnecessary files 2024-10-24 02:23:33 +00:00
Guangyi Liuandguangyi 294993ca78 [fix] fix the path typo for google/mt5-xxl in gradio_web_server.py (#405)
* [fix] fix the path typo for google/mt5-xxl in gradio_web_server.py

* [fix] fix the issue: the cache of pretrained mt5-xxl weights is inconsistent.

---------

Co-authored-by: guangyi <guangyi.liu@mbz-h100-029.core42.ai>
2024-08-23 11:14:30 +08:00
lb203 de7fc3150a Update train_inpaint.sh
do not set seed
2024-08-20 15:10:56 +08:00
lb203 eb1311b9a2 Update README.md 2024-08-20 11:23:51 +08:00
lb203 d3ea50fc05 Update Report-v1.2.0.md
add 29x480p link
2024-08-20 11:22:06 +08:00
yunyang GeandLinB203 842435c016 release Open-Sora Plan v1.2.0 i2v (#389)
* gitignore

* inpaint

* Create condition_image_path.txt

* Rename condition_image_path.txt to condition_images_path.txt

* Update sample_inpaint.py

* Update sample_inpaint.sh

* inpaint

* fix bug

* Update README.md

* inpainting

* release v1.2.0 i2v

* release v1.2.0 i2v

* Update Report-v1.2.0.md

* Update pipeline_inpaint_sp.py

* Update sample_inpaint_ddp.py

* Update sample_inpaint_sp.py

* Update sample_inpaint.py

* Grammar Error Correction

* Grammar Error Correction

---------

Co-authored-by: LinB203 <2267330597@qq.com>
2024-08-14 12:39:51 +08:00
lb203 394c6444aa Update train_t2v_diffusers.py 2024-08-05 15:23:05 +08:00
lb203 7ef00d7749 Update Report-v1.2.0.md 2024-08-01 13:06:37 +08:00
lb203 16e3afed62 Update README.md 2024-08-01 13:05:36 +08:00
lb203 3c08cb2fef Delete opensora/train/train_t2v_diffusers_lora.py
lora bug
2024-07-27 23:20:25 +08:00
LinB203 74744be314 fix sample 2024-07-27 07:59:13 +00:00
lb203 cc4ba38e1a Update README.md 2024-07-26 21:42:19 +08:00
lb203 39e00fd1c4 Update Report-v1.2.0.md 2024-07-26 21:40:26 +08:00
lb203 adb2a20a3d Merge pull request #352 from cxh0519/main
Update Report-v1.2.0.md
2024-07-25 14:12:33 +08:00
lb203 a6eb95f471 Merge branch 'main' into main 2024-07-25 14:11:05 +08:00
lb203 f0667d8db2 Update Report-v1.2.0.md 2024-07-25 14:10:17 +08:00
lb203 20b395624b Update README.md 2024-07-25 14:09:12 +08:00
Xinhua Cheng 40e0423a1e Update Report-v1.2.0.md 2024-07-25 14:05:45 +08:00
lb203 7e42eef228 Update Report-v1.2.0.md 2024-07-25 13:26:35 +08:00
lb203 08ae0a8379 Update t2v_datasets.py 2024-07-24 21:38:47 +08:00
lb203 b50b51ad92 Update Report-v1.2.0.md 2024-07-24 19:48:17 +08:00
lb203 a4300446d7 Update Report-v1.2.0.md 2024-07-24 19:46:43 +08:00
lb203 ebdfdc49e7 Update Report-v1.2.0.md 2024-07-24 19:46:14 +08:00
lb203 fbf13b7688 Update pyproject.toml 2024-07-24 19:45:28 +08:00
lb203 3a8c34efd0 Update README.md 2024-07-24 19:14:11 +08:00
lb203 ae5679185c Update Report-v1.2.0.md 2024-07-24 18:32:15 +08:00
lb203 067870f28f Update README.md 2024-07-24 18:32:02 +08:00
LinB203 535cb6330f release v1.2.0 2024-07-24 10:31:04 +00:00
lb203 b08681f697 Update Report-v1.1.0.md 2024-06-11 16:40:10 +08:00
lb203 d3ea240481 Update README.md 2024-06-07 18:35:08 +08:00
lb203 40bb7a55d7 Rename sample_video_513.sh to sample_video_221.sh 2024-06-01 21:25:35 +08:00
lb203 27c7046472 Update sample_video_513.sh 2024-06-01 21:25:27 +08:00
lb203 8d62c7273e Update gradio_utils.py 2024-06-01 18:28:18 +08:00
lb203 9dd3c8c494 Update README.md 2024-06-01 18:27:03 +08:00
lb203 46566a713d Update gradio_web_server.py 2024-06-01 18:24:50 +08:00
lb203 fb21d9938a Merge pull request #290 from Linzy19/0528lzy
fix the bug
2024-05-28 18:08:32 +08:00
lb203 be284a0274 Update README.md 2024-05-28 18:07:33 +08:00
ZongyingLin c58d27384a fix the bug 2024-05-28 15:16:49 +08:00
lb203 c421a6ed11 Merge pull request #286 from qqingzheng/causalvideovae-docs
Update CausalVideoVAE docs
2024-05-28 00:40:19 +08:00
New User 336403dd98 fix vis 2024-05-28 00:34:03 +08:00
New User 23aeb03f91 fix train bug 2024-05-27 23:50:36 +08:00
lb203 1a0872ac8e Update LICENSE 2024-05-27 23:40:13 +08:00
Zongjian 0360c1b6be [docs] update CausalVideoVAE docs 2024-05-27 23:36:18 +08:00
New User 559fbf3c36 fix wrong code 2024-05-27 23:33:49 +08:00
lb203 183b05173a Update t2v_datasets.py 2024-05-27 23:01:17 +08:00
lb203 ebefa85284 Update train_t2v.py 2024-05-27 23:00:56 +08:00
lb203 b2976643f8 Update train_t2v.py 2024-05-27 22:59:19 +08:00
lb203 b5b2c0968e Update modeling_latte.py 2024-05-27 22:57:34 +08:00
lb203 dc12340706 Update dataset_utils.py 2024-05-27 22:54:38 +08:00
lb203 b2a50079ac Update README.md 2024-05-27 21:01:02 +08:00
lb203 c9a7881dc5 Update pyproject.toml 2024-05-27 19:44:45 +08:00
lb203 71597994d1 Merge pull request #285 from cxh0519/patch-1
Update Report-v1.1.0.md
2024-05-27 19:44:26 +08:00
Xinhua Cheng 5b78e8e00f Update Report-v1.1.0.md
solve typos
2024-05-27 19:37:15 +08:00
YuanLi 858696d591 Update LICENSE 2024-05-27 19:20:26 +08:00
lb203 2a8b2328a5 Update README.md 2024-05-27 17:42:43 +08:00
lb203 f920d640d0 Update README.md 2024-05-27 17:12:39 +08:00
lb203 1e16f6be29 Update README.md 2024-05-27 16:39:16 +08:00
lb203 194ee0307a Update README.md 2024-05-27 16:33:23 +08:00
lb203 06eedf5ea3 Update README.md 2024-05-27 16:28:09 +08:00
lb203 a19488e33a Update README.md 2024-05-27 16:27:58 +08:00
lb203 60773f4860 Update README.md 2024-05-27 16:11:40 +08:00
lb203 da48eca111 Update README.md 2024-05-27 16:08:09 +08:00
lb203 e3e81fd4f1 Update README.md 2024-05-27 15:57:42 +08:00
lb203 91be347ad2 Update README.md 2024-05-27 15:51:22 +08:00
lb203 b3b0d60536 Update README.md 2024-05-27 15:47:46 +08:00
lb203 6e5515df28 Create Report-v1.1.0.md 2024-05-27 15:44:58 +08:00
lb203 732c672cb4 Update README.md 2024-05-27 15:44:27 +08:00
lb203 e2ae46b718 Update README.md 2024-05-27 15:44:00 +08:00
lb203 9cd5a906fd Merge pull request #284 from PKU-YuanGroup/dev
released v1.1.0
2024-05-27 15:43:39 +08:00
LinB203 27a98335b9 update prompt 2024-05-27 07:41:41 +00:00
LinB203 65ec02d036 5.27 2024-05-27 03:47:55 +00:00
LinB203 c21ea81533 fix demo 2024-05-26 07:39:11 +00:00
LinB203 8aebbc64c0 prepare scripts 2024-05-26 04:17:03 +00:00
root de3f7d8a9f prepare release 2024-05-25 02:54:40 +00:00
root 9b2951969f 5.15 2024-05-15 12:29:03 +00:00
root cfd61fdbb5 fix mask 2024-05-04 12:48:09 +00:00
root 78272b0941 update 2024-05-04 01:49:56 +00:00
root aa6d088e91 train vis 2024-04-30 15:23:43 +00:00
node106 594d98265b fix dataset 2024-04-29 13:15:41 +00:00
LinB203 8dc49b8a6b multi-data 2024-04-29 19:01:10 +08:00
LinB203 5e3a3c6f78 mask loss 2024-04-27 14:53:23 +08:00
LinB203 e767ef3a2e update dataset 2024-04-22 11:10:56 +08:00
LinB203 e92a28ba49 fix rope with compress 2024-04-21 23:30:46 +08:00
root 8358b3014c compress kv and rope pi 2024-04-21 11:44:26 +00:00
LinB203 3287e525c5 abs and rope 2024-04-17 15:45:54 +00:00
LinB203 c59023066e vae temporal tiling 2024-04-16 16:49:43 +00:00
LinB203 350138480f support newvae 2024-04-15 15:04:35 +00:00
LinB203 f1086e192c img training 2024-04-15 12:27:54 +00:00
LinB203 1323daa7b3 read image from folder 2024-04-14 12:03:04 +00:00
LinB203 8a6db0b399 refactor dynamic training 2024-04-14 11:37:16 +00:00
LinB203 8fa3e614a1 add 2drope and dynamic training 2024-04-14 02:04:14 +00:00
lb203 bec0e85238 Update pyproject.toml 2024-04-13 19:58:11 +08:00
lb203 098ecbbd5d Update Report-v1.0.0.md 2024-04-12 14:21:12 +08:00
YuanLi 8d3cd692e0 Update LICENSE 2024-04-12 09:49:31 +08:00
lb203 0183e89cb7 Merge pull request #218 from qqingzheng/vae_rec
Updated content related to VAE reconstruction
2024-04-11 19:06:38 +08:00
lb203 58b64d2662 Update README.md 2024-04-11 16:10:40 +08:00
qqingzheng f58e57d8f5 [refactor] fix hardcode 2024-04-11 03:56:55 +00:00
qqingzheng cdc626b14e [docs] fix typo 2024-04-11 02:36:10 +00:00
qqingzheng 8349763576 [feat] add time chunk inference 2024-04-11 02:31:20 +00:00
qqingzheng 02ad56275d [docs] update inference example. 2024-04-11 02:30:59 +00:00
lb203 461a4b2c97 Merge pull request #214 from JJJYmmm/fix_name_error_emavq
[fix]: Fix variable naming errors
2024-04-10 23:20:28 +08:00
lb203 65a3f0fe8a Merge pull request #207 from AlonzoLeeeooo/main
Fix typos
2024-04-10 23:17:53 +08:00
lb203 0cf62a3e0c Merge pull request #206 from digger-yu/patch1
fix typo
2024-04-10 23:17:15 +08:00
JJJYmmm e2492578c7 [fix]: Fix variable naming errors 2024-04-10 19:40:55 +08:00
USTC-liuchang d581ba621a Fix typos
line 24: `github -> GitHub`
line 54: `a -> an`
line 85: `re-organizes -> re-organize, modulizes -> modulize`
line 87: `opened -> open` (modified according to the changelogs of other dates)
line 94: `a -> an`
2024-04-10 11:51:18 +08:00
digger yu d65994469a fix typo 2024-04-10 08:59:00 +08:00
lb203 193b0398a8 Update README.md 2024-04-10 01:50:40 +08:00
lb203 e08fe61f2b Delete assets/we_want_you.jpg 2024-04-10 01:45:49 +08:00
lb203 c3cd4da606 Merge pull request #201 from qqingzheng/make_vae_better
[docs] update docs and train.sh

Former-commit-id: af001aec7d9bbf38d1e7e1f6a8ace5a77670b85e [formerly e6c3c58c77c2d8fdab1d5b464cd861ff78cc2e43]
Former-commit-id: 6ad032635deb1854b76d5a63a67f10442db0e655
2024-04-09 21:04:29 +08:00
qqingzheng a4fe6a634d [docs] update train.sh
Former-commit-id: eeb00ac046692bc97a064ec1b47c994436798ab1 [formerly 6d0fd85e1b13a7824b8181ba9ccebec2a3c76d40]
Former-commit-id: 1e8e669ad9967aeeeb7622bc04a0759f348cb26f
2024-04-09 13:03:34 +00:00
qqingzheng 933141c5ea [docs] add evaluation detail
Former-commit-id: c15a717cc61159e5f6adafb718237ab578ac81e7 [formerly 9e92772776fa4dd18772c7fec939a42ad6dfbc0a]
Former-commit-id: 98c2a38af5064f27888d1b102c9fded87d2420d5
2024-04-09 12:58:13 +00:00
qqingzheng 3f92ec2c53 [fix] fix bug in #185
Former-commit-id: d74efbd0195380ca383598c06d3d8185dd9cc9bc [formerly 0f2687266bc2271e11b2f06edfe514f466e57b0f]
Former-commit-id: cb6bb234befc49b3870c5690653469a0714890c5
2024-04-09 12:48:32 +00:00
qqingzheng 71541c766a [docs] update docs and train.sh
Former-commit-id: 1d9358e84d12863681b396c7621ae7244caa4dc2 [formerly b801491889b3abad1211225d7f07af3ae0389895]
Former-commit-id: 9e6a59ae2f246b50ace0a033cbd6548660241204
2024-04-09 12:43:19 +00:00
lb203 0623ee9a36 Update train_t2v_feature.py
Former-commit-id: 4a4f7ea2366f1dff5669d6d4834238ebd790e6e2 [formerly ca7d3158fb01f8e7228f833cea8e6741be127c2a]
Former-commit-id: 389f0e7dc66826026046b8e12d7a9f8692d32e42
2024-04-09 19:56:14 +08:00
lb203 4e54046c5f fix using pretrained bug
Former-commit-id: 2dfbb0683b2f3bd1a1d619e5e4fabb5bd4c92d79 [formerly 27f352e822f732cb351d64ac5726c5df3b62c1bb]
Former-commit-id: 722d60ae5ef28499c47292e0a4807a87b9658b01
2024-04-09 19:55:49 +08:00
lb203 326b6cbbe6 Update train_videoae_65x512x512.sh
Former-commit-id: 8e40db106efa2fd879cca42bd9ce2d9b79d777f8 [formerly 3e11781b045a838a5aa98ea75e274198f2b26a44]
Former-commit-id: f3b0804a7cf063e7ff04ee57dccee0320c1f32a5
2024-04-09 18:33:53 +08:00
lb203 71eb96c306 Update train_videoae_65x256x256.sh
Former-commit-id: b2f5cfe72452dfb369aba166e4a101dfcf5758c9 [formerly 557b78ab38a61d7520e91044137f818e50ec24cb]
Former-commit-id: 0e46c9f261e7f5fd4d08e3d7735c17c7975b0e72
2024-04-09 18:33:17 +08:00
lb203 a20766c262 Update README.md
Former-commit-id: be00a17729ac889493f63fd1d649054d7980ca42 [formerly 9b4d8dcfc8e6e2abe4ca727e01561d27d1ba3968]
Former-commit-id: f1b6f5013b99d0d288af0f05c46cb8300d79a8aa
2024-04-09 17:13:42 +08:00
YuanLi 72aef089ff Update README.md
Former-commit-id: 68eaf2e6b00266cf0858b33f9b4d09681efbd2bc [formerly d8dcad250ed8357964ab8ee570b9973b45e29fcb]
Former-commit-id: 75f51b3d84203be3672f6b07bb3c70ba0ae5556c
2024-04-09 16:56:19 +08:00
YuanLi 9b011137b3 Update README.md
Former-commit-id: c48eba0bb15d594007809d4cc605dc46ba029cec [formerly 7deb8a0decf7a9757af4475e3029d415d78d185f]
Former-commit-id: 7f43bb2f1d9308bd08154171d78e9b36e00a837f
2024-04-09 16:50:28 +08:00
YuanLi 14dacb1ca1 Update README.md
Former-commit-id: 54bad83bf054a2ff574199931121c686b4e28008 [formerly 06bb190a39a9b98c8de5e385fc304a86cc8c9fb2]
Former-commit-id: 23e8a235d6828d9ab7c02ef59d40fe13a46fa4b1
2024-04-09 16:47:05 +08:00
lb203 fd3198d7db Update README.md
Former-commit-id: 6fd60c73b6ae69d6ad9dc14c7371bd9bff086b44 [formerly 2ed5d5b7bdcd88044da26ca94237c50af7ab47dd]
Former-commit-id: 443c333353963bc1ec64a4946afbe5e17f9b5a87
2024-04-09 15:26:19 +08:00
lb203 1556d341ed Add files via upload
Former-commit-id: 943dd9a063baaa05d7f0d0bd52cbde35cf2b3d98 [formerly 5542359738243a8e948a8f08ed4e8dedf6b4b818]
Former-commit-id: fbee07f6fef2ac86bc15ff66f2e2a1a68b64ed01
2024-04-09 15:25:46 +08:00
lb203 3794a468fa Update Report-v1.0.0-cn.md
Former-commit-id: 83124efeaa5a8cd528a15760bdf88acc403142bd
2024-04-09 13:36:35 +08:00
lb203 b21d386205 Update Report-v1.0.0-cn.md
Former-commit-id: 59edce70edbaed957ba40f1d95b4a5050ece12bf
2024-04-09 11:34:43 +08:00
lb203 9b72cb3e6f Update Report-v1.0.0-cn.md
Former-commit-id: 0edac24fcaa447a534755bf3aac1f2027a491c06
2024-04-09 11:34:17 +08:00
lb203 5b5159e8fc Update Report-v1.0.0.md
Former-commit-id: 3c8eb75dc2045154e8654f860b1fabd7e597094b
2024-04-09 11:33:34 +08:00
lb203 113bb8c5df Update Data.md
Former-commit-id: 15defc1a3846335fd68e18fa9b8ffbe38e3e7100
2024-04-09 11:32:24 +08:00
lb203 4ad069ffdc Update Report-v1.0.0-cn.md
Former-commit-id: d5cf6aed7c1673562a6d2022c3fd7f7b16006cfd
2024-04-09 11:09:02 +08:00
lb203 421aa9b66e Update README.md
Former-commit-id: 0cc27a5b065b61f37d44dc2c90ba2782075eefb2
2024-04-09 11:06:18 +08:00
lb203 8c40057697 Update README.md
Former-commit-id: c437a18dbec7ac6e02287565a84d75f96664ee99
2024-04-09 11:06:05 +08:00
lb203 791a48a088 Update README.md
Former-commit-id: e386ec112337271bb9f55ea6ec872eaba94aff50
2024-04-09 11:02:06 +08:00
lb203 b85795c263 Create Report-v1.0.0-cn.md
Former-commit-id: d484754e080bd65f6b35619b840040581fa6b69c
2024-04-09 11:00:43 +08:00
lb203 bd895c4d90 Update README.md
Former-commit-id: 88e7f2062c494c199caff468f5f14f3165f763de
2024-04-09 09:46:28 +08:00
lb203 4731a994b0 Merge pull request #187 from qqingzheng/add_causalvae_docs
add causalvae doc

Former-commit-id: d4166b1cf6356e4415bbdba315c34e23ffdd0231
2024-04-09 09:39:48 +08:00
Chestnut bf4364b50a 更新 README.md
Former-commit-id: 42607786f3f3d0eba9336f774b0a2ae076cce793
2024-04-09 00:25:46 +08:00
Chestnut 1817a15c17 Update README.md
Former-commit-id: 567ade18047a9cc81648997c5d9287fa8900614d
2024-04-08 23:43:00 +08:00
Chestnut 8bc9abb45b Update Train_And_Eval_CausalVideoVAE.md
Former-commit-id: 156739c35b5dda594138b2e3b1fd6ec99556af23
2024-04-08 23:41:49 +08:00
Chestnut 2dc20a2cd1 Update Train_And_Eval_CausalVideoVAE.md
Former-commit-id: 589657b089930379bb75f3d49063ebb72671efc6
2024-04-08 23:40:07 +08:00
qqingzheng 7d9f03db56 add causalvae doc
Former-commit-id: 4e6d9e038c8edb81f8402834bd01eeeb823924f9
2024-04-08 12:25:43 +00:00
lb203 5e5c6a64dc Update Data.md
Former-commit-id: d9a5c8fbd36a34b7baf720bc27d69604ae5c9d57
2024-04-08 19:28:52 +08:00
lb203 9906697152 Update pyproject.toml
Former-commit-id: bb037420b75303b58486ba79da9180b2bf63cf88
2024-04-08 19:00:26 +08:00
LinB203 1f7edca9fd clean
Former-commit-id: a7f2c1c8b2243587cedd6e152e64298842287095
2024-04-08 18:57:06 +08:00
lb203 a7c3aebf27 Update README.md
Former-commit-id: 828be4267f0795969293e4a321e4e0bfa8e1f1e6
2024-04-08 16:08:12 +08:00
lb203 029d7eba30 Update README.md
Former-commit-id: dc923fe2413b3be31d8d88e7dd83a931fcbf9458
2024-04-08 14:44:58 +08:00
lb203 22b86b0e32 Update Report-v1.0.0.md
Former-commit-id: 37f181f34f64ca1141a1345b7c1f2af283be1cad
2024-04-08 14:27:11 +08:00
lb203 0e8162191f Update Report-v1.0.0.md
Former-commit-id: 400f708a080f0799e02697bb8e66d8430ffb1f56
2024-04-08 10:54:11 +08:00
lb203 8139823ff5 Update README.md
Former-commit-id: 2873d0eaedfad392c627a714448118949d7b7e13
2024-04-08 10:40:23 +08:00
lb203 b37a6ff3a7 Update Report-v1.0.0.md
Former-commit-id: 6ebe2c62c0904142026d3796027e95ce43afe16a
2024-04-08 10:37:04 +08:00
lb203 172e1bcc58 Update Report-v1.0.0.md
Former-commit-id: 38b779022418cc41cdf9fc508e461f2f6299fd60
2024-04-07 23:48:08 +08:00
lb203 22f61c0272 Update README.md
Former-commit-id: 9c0f82ae9953b56f2540e9c7ada021e4e1bc6165
2024-04-07 23:47:36 +08:00
lb203 cd99a6759a Merge pull request #183 from chaojie/patch-2
Update README.md fix colab link

Former-commit-id: 257c9275a0bdd8ae931592b7e5cf302f0d97d900
2024-04-07 23:07:21 +08:00
YuanLi 8cd4d9671e Update README.md
Former-commit-id: 5731e810e2efb7ca7eda1aa8954eb7fd260f4250
2024-04-07 22:55:49 +08:00
chaojie 459369d3b6 Update README.md
Former-commit-id: 1b54444249fa91b29334b8fc364fb1978c25fb2d
2024-04-07 22:52:01 +08:00
YuanLi 15113e2e7a Update README.md
Former-commit-id: bbf3cef43a129a25a80ce134a8895796e24353c6
2024-04-07 22:50:49 +08:00
YuanLi 9016110a90 Update README.md
Former-commit-id: c4a976e9b93536299b2e350454c780fec54fbe95
2024-04-07 22:45:48 +08:00
lb203 154070c2d0 Update README.md
Former-commit-id: c1492211ced3cdbec93ef28678a1467245e122e9
2024-04-07 22:26:58 +08:00
lb203 ed570f7b18 Merge pull request #178 from qqingzheng/causalvae_release_pr
Fix some compatibility issues

Former-commit-id: 11af500907302d699b761799408462a0ebfb9b31
2024-04-07 19:45:14 +08:00
qqingzheng 492dbab0d6 fix bugs in inference
Former-commit-id: e17c7a7b7704920cb86894419bf573d0fb0c06fa
2024-04-07 11:36:57 +00:00
qqingzheng 0044ec6623 Fix bugs in inference
Former-commit-id: c251a86ac7c91d6f2bb12ae645628e9cf4b83d3d
2024-04-07 11:11:16 +00:00
lb203 e1fc7291d5 Update README.md
Former-commit-id: ce2bbf82045041495ed448b46d5f4bdba84aa740
2024-04-07 18:46:43 +08:00
lb203 d3bbaa7a55 Update Report-v1.0.0.md
Former-commit-id: 7cec40420601f7da57449586508e50119bd02a61
2024-04-07 18:11:34 +08:00
lb203 d58cc4504a Update README.md
Former-commit-id: 9f926abb08f156807091dd044c133788649a6bac
2024-04-07 16:57:38 +08:00
lb203 d30e3653f2 Update README.md
Former-commit-id: 9bae31c73ea902581e1ee94cfd9012efdf78d349
2024-04-07 16:51:14 +08:00
lb203 b08bf8d907 Add files via upload
Former-commit-id: 9a207435d9efac0b19771df5d2288160af1bdd6e
2024-04-07 16:49:45 +08:00
lb203 d86ff4f933 Update README.md
Former-commit-id: f709d75e7a7d8a08ec60132aa906f6b880af8e58
2024-04-07 16:38:47 +08:00
lb203 79bf57e77c Update Report-v1.0.0.md
Former-commit-id: aea1f4a82a525d3e6e1d1a1bb7631326b294cdcc
2024-04-07 16:33:28 +08:00
lb203 3afaf0c5fe Update Report-v1.0.0.md
Former-commit-id: bd9aa7991212d98a6224b855a3bfb2c12feb5950
2024-04-07 16:31:48 +08:00
lb203 fcd94894da Update README.md
Former-commit-id: f8cb7f38751d7a6ef586c293fb769f5d60ad5c59
2024-04-07 16:30:47 +08:00
lb203 a3cf68c883 Update README.md
Former-commit-id: 30f61958e221c3d41c4aa5917f4511e8c0482cf9
2024-04-07 16:28:34 +08:00
lb203 5df5af0629 Update README.md
Former-commit-id: d5ab411b2e2fd5d6abeda500895e3339167572ab
2024-04-07 16:27:37 +08:00
lb203 deaea23969 Add files via upload
Former-commit-id: a232ec2d98b245a9b502a2c49fc4f468a86fbd64
2024-04-07 16:22:43 +08:00
lb203 3e6abd244d Update README.md
Former-commit-id: 3210474a46845ae3a6c0bde7fb6c7f9845866628
2024-04-07 16:13:48 +08:00
lb203 760029dbfe Update README.md
Former-commit-id: 7d4118566b77e09de4a3615612a5b368c76c293e
2024-04-07 16:12:52 +08:00
lb203 f08fa0b9d5 Update README.md
Former-commit-id: 037d7d3e084d7ff426347db49c4b210a2ee3927d
2024-04-07 15:58:09 +08:00
YuanLi c335b0b70c Update Report-v1.0.0.md
Former-commit-id: 92d906b340f5970a77377586d7295f46b5de99ec
2024-04-07 15:57:37 +08:00
lb203 cdd41c2aa3 Add files via upload
Former-commit-id: 292cbcfa4e4f384fe6958badc27d3067173119f9
2024-04-07 15:57:35 +08:00
lb203 d02d16e6e2 Add files via upload
Former-commit-id: 1ae8512b7846e9faafdb38ca532898d7f1bdcce2
2024-04-07 15:50:18 +08:00
lb203 7434308e92 Update README.md
Former-commit-id: a2ec90a19eacb099cb9f36c1b5dd6e24be307fae
2024-04-07 15:08:09 +08:00
lb203 511ea03617 Update README.md
Former-commit-id: a61200bea2eea4fd22bd8095f192fef4c22c49b4
2024-04-07 15:04:51 +08:00
lb203 8d5f0281ae Update README.md
Former-commit-id: e6fb5d864ac8af8214cdc89c71b42f296dda14e7
2024-04-07 14:59:24 +08:00
lb203 68ec69bc6c Update and rename CausalVideoVAE.md to Report-v1.0.0.md
Former-commit-id: 3b848f8de33cf4f0ed802428e9ad31d2fa26de63
2024-04-07 14:57:39 +08:00
LinB203 ef0157a168 update gradio demo
Former-commit-id: 80c2cc0bcc2a7532bc2440df16c1f811b0b30516
2024-04-07 09:44:23 +08:00
LinB203 20c3149b62 clean
Former-commit-id: 56ed87088e13cdd93929e2ceedbfae693e2f013a
2024-04-06 23:41:01 +08:00
qqingzheng 35f5e4cff8 Fix some compatibility issues
Former-commit-id: 942d8598b28516369780c3519607929869b427b7
2024-04-06 14:08:02 +00:00
LinB203 282f973d9b v1.0.0
Former-commit-id: 93f44f3b425c090d9bc92687a9c50a26b4f0d7c8
2024-04-06 22:04:52 +08:00
lb203 4b8b80835b Merge pull request #177 from qqingzheng/causalvae_release_pr
CausalVideoVAE release version

Former-commit-id: 50dd03b07b4a57e708ca01f703d806e9fc568d2a
2024-04-06 19:48:26 +08:00
qqingzheng a554dcc696 release
Former-commit-id: 52fe2ac6610661d6a13a08390c84a8237f37986a
2024-04-06 11:40:05 +00:00
LinB203 a4a1b93bbc refactor vae to hf
Former-commit-id: a938d3f7f962709f7ecff5e9d49c9bf1ce40d7fa
2024-04-06 19:26:08 +08:00
LinB203 cea0c041df update train script
Former-commit-id: ac16f77d19488cd46130587369cd7a5ea378052b
2024-04-06 13:28:29 +08:00
LinB203 8137d40a92 add tile training
Former-commit-id: 216079dc925beefe1ec722a4e87e46a4e1ca56be
2024-04-04 22:00:38 +08:00
LinB203 e202525517 sample pipeline
Former-commit-id: 8964b16c0dd65445a20050c9a0e82d909b55eaff
2024-04-04 10:50:29 +08:00
LinB203 8084949b14 released v1.0.0
Former-commit-id: 8df5897a4ffc341ced45faf617687fb8ebf1286c
2024-04-04 10:46:29 +08:00
lb203 b3f6d06941 tile only2d
Former-commit-id: 7f31ad03fdbaac97aea51a98bd29dc7327b700c9
2024-04-01 12:27:40 +08:00
lb203 b7f0d770b8 Merge pull request #172 from SamitHuang/fix_attn3d
[Bug fix] Fix reshape bugs in AttnBlock3D in CausalVideoVAE

Former-commit-id: 8a6b0947c126c87d13809adbd0adfdf65716c27f
2024-03-31 15:07:13 +08:00
Samit ee0b749421 fix reshape bugs in AttnBlock3D
Former-commit-id: 299622e168a6cbee14ef2056c47cae722c20e005
2024-03-31 14:52:28 +08:00
LinB203 0c86a4ff11 tile conv
Former-commit-id: 5133615bcba221c4b86ae4e8b09b82ee5d34c66d
2024-03-30 20:42:20 +08:00
LinB203 2aaf448a24 update sample
Former-commit-id: 116726b9d565c689496123d4b6e62b98400536a6
2024-03-30 15:30:29 +08:00
lb203 a50abb7b49 Update __init__.py
Former-commit-id: 2bdc29d86ebff3d6f29c38ded57c7b0dd1e6bbdb
2024-03-28 22:04:27 +08:00
lb203 b88dfd6d1e Update README.md
Former-commit-id: 2d9ae56309717a91e0690bd5c354fe3cbe5d5ed6
2024-03-28 21:16:34 +08:00
lb203 5dc588f47c Update CausalVideoVAE.md
Former-commit-id: fcde4d0ff783448c2006d6f4751a0e4f9cc23d0f
2024-03-28 20:16:49 +08:00
LinB203 f48b3da80a fix videovae trainer
Former-commit-id: 79e1feb412b6eeead3306f7f0d55a01f9b2a543e
2024-03-28 20:04:15 +08:00
LinB203 9462239286 update vae
Former-commit-id: 24298df5d374d5aaa5696f6200d4e23ae887715f
2024-03-28 19:57:15 +08:00
LinB203 c70648b740 update model
Former-commit-id: e4e24650376d9dce974b290752efb416d5061983
2024-03-28 19:10:24 +08:00
LinB203 45187a0d22 fix training bug
Former-commit-id: 09a6283ddd9ab11479f30324de788bd7415702ae
2024-03-28 19:07:46 +08:00
lb203 b88ae0d76d Merge pull request #165 from qqingzheng/add_eval
[feat] add eval code

Former-commit-id: 2b555c515ed71b8f699627423b492a5c43218b84
2024-03-28 19:02:32 +08:00
qqingzheng c564dc569d [feat] eval
Former-commit-id: d2c3cf5ef8d18feb7b0544d68d9796656e90ba8d
2024-03-28 11:00:20 +00:00
lb203 55a595ece8 Update README.md
Former-commit-id: 02b0422fd745b4b7ec091cc6ee81d1d237245fe6
2024-03-27 23:13:29 +08:00
lb203 c7a9f094c8 Update README.md
Former-commit-id: f294a0a28f288ae592e7de004320d3eb5e48e945
2024-03-27 23:13:01 +08:00
lb203 2dbd3bd700 Update README.md
Former-commit-id: 90d6d298eaee6255098b342b31444e7a347fc50e
2024-03-27 23:10:14 +08:00
lb203 c19b099618 Update CausalVideoVAE.md
Former-commit-id: 1cd665f699632e9dc9357379225c085d9c1278d5
2024-03-27 22:59:36 +08:00
lb203 7db5b81522 Update README.md
Former-commit-id: a41674f5ef71ced59b374226dddfa09c15a99f25
2024-03-27 22:57:59 +08:00
lb203 be1eb20b95 Update README.md
Former-commit-id: 8d70a890fc4191bfc98c3aca6dd3c8a70534fb3d
2024-03-27 22:57:40 +08:00
lb203 ce836de93d Rename causalvideovae.md to CausalVideoVAE.md
Former-commit-id: 0a78e4193fcaec4662b7c6465a9d80643b354cd6
2024-03-27 22:55:50 +08:00
lb203 44eb12b157 Create causalvideovae.md
Former-commit-id: 98414c2ad7dd39f740440b2d9059140f93492e83
2024-03-27 22:55:32 +08:00
lb203 5f9ff24a3d Merge pull request #162 from qqingzheng/add_causal_vae
[feat] add causalvae ✨

Former-commit-id: 4c1b27c277f6dee020d7aa9d6f719415f0eed898
2024-03-27 21:43:31 +08:00
qqingzheng ec81e2427c add causalvae
Former-commit-id: ac936276fd3213ee1815dfe096cd64cecb601428
2024-03-27 12:49:30 +00:00
LinB203 f78a599fa3 scripts
Former-commit-id: 047046f684c2831e68e0cf256cb46fc3790cda33
2024-03-25 16:33:14 +08:00
LinB203 6c12429fd2 train t2v feature
Former-commit-id: 7f16b162cae0e3533c8ba9aa03186a846d9e02bf
2024-03-25 16:25:01 +08:00
lb203 c6d7ed5c3f Update Data.md
Former-commit-id: d1c72e3209769d0204efb8e7d705bdf4f22b82ea
2024-03-22 10:30:04 +08:00
lb203 b3fb8cf3c5 Update README.md
Former-commit-id: 859c0f60dac1562b8a6bbb72b793a4709dbc3597
2024-03-21 16:02:23 +08:00
lb203 ebaa135355 Merge pull request #152 from Ytimed2020/main
Add CLIP support and example

Former-commit-id: 3c3f80caa5d24fac1f6bf5c01b4cb8df86f2edfd
2024-03-21 15:59:24 +08:00
lb203 211c7409bc Merge branch 'main' into main
Former-commit-id: e4c99f4b5d084171e3097bc8939d4806f32485a8
2024-03-21 15:59:03 +08:00
lb203 dfef484db5 t2v attention_mode
Former-commit-id: 4fd8bd87e0d0a98d5d0a3201d5db9a5ba055b894
2024-03-20 23:04:02 +08:00
lb203 4d476bc759 Update README.md
Former-commit-id: 95b28706d257d7769049abe30a7121b60e39a53f
2024-03-20 22:35:11 +08:00
lb203 a4b273f255 Update README.md
Former-commit-id: 9065d12ee867c76452ce370fb71cc83bdc01470b
2024-03-20 22:33:39 +08:00
LinB203 6b51d0ccfd support attention_mode
Former-commit-id: c72f354fbfd948e8cd4b0cc63c157bc4688c7665
2024-03-20 22:23:24 +08:00
LinB203 ff36d291d8 train with image
Former-commit-id: 209b9a8ad2f19a50e46f9b7ec68dd781898e491a
2024-03-19 23:21:57 +08:00
Ytimed2020 73ad660277 Update clip.py
Former-commit-id: 1ee53d152630b6acc69369be33287abfb143224d
2024-03-18 23:03:33 +08:00
Ytimed2020 0817d81290 Create clip.py
Former-commit-id: e3397c567efcfeef6f58dfef2eb4251bedbbf33a
2024-03-18 23:01:34 +08:00
Ytimed2020 f2ac960431 Update __init__.py
Former-commit-id: 7c8af2909c4549e96bc0893d09fcc53d71fb6cc7
2024-03-18 23:01:04 +08:00
lb203 81cc4190fd Merge pull request #145 from qqingzheng/add_casual_vqvae
[feat] add casual vqvae ✨

Former-commit-id: f034a4cfe8bc84fee5a50a1da833a39c9499f213
2024-03-18 13:10:12 +08:00
lb203 1f34030723 Update pyproject.toml
Former-commit-id: 5637656a79faba1b397b56918c42022cc209e731
2024-03-17 20:22:03 +08:00
lb203 757268eccd Merge pull request #146 from glgh/doc-fix
[docs]: fix VQVAE training script path

Former-commit-id: 6ea832b958e6c8e5494a7b199874ffce061325aa
2024-03-17 09:41:01 +08:00
gl 83cf1dbd8e [docs]: fix VQVAE script path
Former-commit-id: d5f5c681b8c50622a84256881d193d5803ff3c28
2024-03-16 12:20:59 -07:00
qqingzheng b3c17829e9 [feat] add casual vqvae
Former-commit-id: 89087970b267908c8667984aba42b1ade2505467
2024-03-17 02:46:33 +08:00
lb203 862f627607 Update README.md
Former-commit-id: 6729039f059a23bf71faa944e2498657ef2c7442
2024-03-17 00:00:45 +08:00
lb203 73161c3248 Update train_t2v.py
Former-commit-id: d25b39ffac4474abd40b551fe0e51c421cb7bf5e
2024-03-16 22:23:36 +08:00
lb203 1d1a6a5b63 Update train.py
Former-commit-id: 783faf5cd86c148a713ea986ceed8527b108f2b8
2024-03-16 22:23:22 +08:00
LinB203 085062a53f refactor and fix resume bug
Former-commit-id: 1b1136671b9a7e0b15b70168e2c44dbff04a22e2
2024-03-16 22:21:15 +08:00
LinB203 516082e06c resume and compress kv
Former-commit-id: 0af7d58f88c1b65b4089da6f3e5c70456614da7c
2024-03-15 22:48:09 +08:00
lb203 054bf4bfb0 Update README.md
Former-commit-id: d97edd0e714a237a54f191dcebb6fd39f73185d6
2024-03-15 22:22:42 +08:00
lb203 510434d9a1 Update README.md
Former-commit-id: 34e529fd02712d4466ab2bc1a63f9bd69976b6c3
2024-03-15 20:14:47 +08:00
lb203 befdbd5943 Update pyproject.toml
Former-commit-id: 7fb921440b608e47781d2f541b2a8d893a48193e
2024-03-15 20:13:18 +08:00
LinB203 daee16c393 update scripts
Former-commit-id: 8b2a794095f2deb9b1e8090486cb8242bf81031d
2024-03-15 19:09:18 +08:00
lb203 04034c2e50 Merge pull request #143 from anapple-hub/my-script-branch
[fix]: fix sample.sh

Former-commit-id: 7ceacc5f9544d0e27d8cea4981ed12ee784efd6f
2024-03-15 19:00:35 +08:00
LinB203 d2a8e64a83 del t2v
Former-commit-id: 5daa2952997a29e74e2b5d235001d6d9dabf9ff6
2024-03-15 18:43:44 +08:00
anapple-hub 99bbe2ecbe [fix]: fix sample.sh
Former-commit-id: 09916dfcfb0b80a4217016919e0394a2cb665cfd
2024-03-15 18:30:49 +08:00
lb203 210690e431 Update README.md
Former-commit-id: 7b96fd89eaf3743a4a9fbc995e6a2a791c734f2e
2024-03-15 13:47:29 +08:00
lb203 17c430042e Update sample.sh
Former-commit-id: b149a300f1aff53d9ab6e66a91cc2bcd37900347
2024-03-14 13:21:23 +08:00
lb203 4b8c154784 Update sample.py
Former-commit-id: eb906691abec030d7aa2078b63fc9c7d678662ed
2024-03-14 13:21:04 +08:00
LinB203 99a789bd5d del t2v
Former-commit-id: 910398ca297be8a188d59c59c121857666774af1
2024-03-14 08:51:10 +08:00
lb203 d82c3b9e60 Merge pull request #139 from qqingzheng/fix_vqvae
[bug] add quantization loss

Former-commit-id: 58bb160e834dcd74650741e2003720d6ae5cad5f
2024-03-13 18:47:15 +08:00
qqingzheng 08d8ef6e72 [bug] fix quantization loss
Former-commit-id: 111fcf55f969ffe4c9be284a317ae5712570a527
2024-03-13 18:16:28 +08:00
lb203 51581271b4 Update README.md
Former-commit-id: 2d5b6815f95e94a98847b8a0376473f01bb15d14
2024-03-13 14:02:33 +08:00
lb203 014e5b4933 Merge pull request #130 from sennnnn/dit_deepspeed
⭐ [Feature] Support deepspeed training for DiT

Former-commit-id: 9b25f038f32a03dd56552a132118757d0835c428
2024-03-13 13:40:03 +08:00
lb203 18a652afff Update README.md
Former-commit-id: cf1c56c0058288427c1dfd651c8c0e9a981c5653
2024-03-13 13:33:48 +08:00
lb203 8aa4e60cf4 Update README.md
Former-commit-id: 7e74356a3279b02d062bab9c9f361f61eb756179
2024-03-13 13:22:16 +08:00
sennnnn 9750ead38b Fix typos.
Former-commit-id: 6add5ee7ef39a973e91b7a7c1b3a4b0980130b88
2024-03-12 21:23:44 +08:00
sennnnn 66f085252e merge main branch.
Former-commit-id: 5f78da5eae2b18305bb4a0a6dc7ab771e9b96f60
2024-03-12 21:11:56 +08:00
lb203 f8b0701a4b Merge pull request #123 from yunyangge/main
[docs]: frame interpolation update

Former-commit-id: ced46982cd41903a212a841500618a614e7c9b8a
2024-03-12 14:44:21 +08:00
yunyang Ge 1c72bea528 Merge branch 'PKU-YuanGroup:main' into main
Former-commit-id: 314be102744ec83be3bc37750ea4d6674e383de4
2024-03-12 14:21:45 +08:00
lb203 c8811aa2d5 Update train_256.sh
Former-commit-id: a2dfc9a8c70845446e731c88acd222e9b2db9122
2024-03-12 14:21:14 +08:00
yunyang Ge 07d818282d Update readme.md
Former-commit-id: e287f94838007f2eda434a46bcd250099b627097
2024-03-12 14:19:52 +08:00
yunyang Ge e4ea15e66a Update and rename Frame Interpolation.md to readme.md
Former-commit-id: 87bdf1ebecb002c086497b3c47acb3f3cb2aaf34
2024-03-12 14:19:16 +08:00
yunyang Ge 59e2050bc3 Add files via upload
Former-commit-id: b707b1968fa59d85891040604e7d54bef61b4549
2024-03-12 14:18:00 +08:00
lb203 a0f4a001a3 Update README.md
Former-commit-id: 8ec7524f4f171e650d7a25cc88ed29f7b5017454
2024-03-12 14:16:18 +08:00
yunyang Ge 78b9cb8775 Update interpolation.py
Former-commit-id: e10e3416433e5f396aa13e7d8beaf451ff47a383
2024-03-12 14:13:54 +08:00
lb203 ff617764b7 Merge pull request #104 from sennnnn/videogpt_deepspeed
⭐ [Feature] Support deepspeed for videogpt training.

Former-commit-id: dc639030e3544ca54e9c579624375bdea486a0ef
2024-03-12 14:12:51 +08:00
lb203 f8ca07a1d8 Update sample.py
Former-commit-id: 6e09fc266a626f465109abc199d9fb452657929e
2024-03-12 14:10:10 +08:00
lb203 61306c7510 Update README.md
Former-commit-id: 0d7dd6233811ec3c8741affb436d8004d8c676b8
2024-03-12 14:06:10 +08:00
lb203 429391840f Merge pull request #121 from sysuyy/add_multi_node_script
[feat]: use accelerate on multi-node

Former-commit-id: 67b01dc1d934d3d86d7ecca00c5dd1a11fd4d22d
2024-03-12 14:05:08 +08:00
lb203 aa3ff1e9c3 Update README.md
Former-commit-id: c5eea8d3492872255a6a34d422230509d1c3531c
2024-03-12 13:35:03 +08:00
lb203 e0dac4fb40 Create train_256.sh
Former-commit-id: bf62151617feb49663f8dfe0c78aa01f41e62092
2024-03-12 12:55:41 +08:00
lb203 486218b252 Update README.md
Former-commit-id: 5717aab628fceba311e284d1bdddeacc4c43c7c5
2024-03-12 11:52:28 +08:00
lb203 21383acca7 Update README.md
Former-commit-id: dd00e04220ff5232c5c4503a095e409471476a51
2024-03-12 11:49:20 +08:00
lb203 e4949048e0 Update README.md
Former-commit-id: 266f9f97ea8f327fdefb2ca959c23d08b7084137
2024-03-12 11:47:12 +08:00
lb203 c39a13d6b0 Merge pull request #120 from touale/main
[fix]: correct attention_mode to attention-mode

Former-commit-id: 6deb9393f771edbf8db8a95e7b9f2107b2682c76
2024-03-12 11:00:50 +08:00
sennnnn ddf7e82b61 Update training arguments of videogpt training.
Former-commit-id: c9660af891c0dae1f51a1075ef5b2df012f65e06
2024-03-12 10:30:41 +08:00
sennnnn 064a503c31 use imageio for write video which has better compatibility.
Former-commit-id: 005119dc9f78ea2a611d22a53772fd59d97d7139
2024-03-12 10:30:01 +08:00
sennnnn 91409c89cb Can't fix zero loss bug.
Former-commit-id: 1f32f9d889743b7961b34b823b276bfd03974191
2024-03-12 10:17:24 +08:00
sysuyy 330e46e9e8 [feat]: use accelerate on multi-node
Former-commit-id: 0b12fc31f1d2a63f9da38821174ab63d9b3a2d72
2024-03-11 14:45:23 +00:00
touale 76e172e08c [fix]: correct attention_mode to attention-mode
Former-commit-id: 9957e2eebb2166b02600c6052fb54908f5d12bad
2024-03-11 22:14:34 +08:00
sennnnn 896d5e3657 Fix deepspeed zero loss bug.
Former-commit-id: e24af415bc5d6fd38d72a24d6312f4e65d3627dd
2024-03-11 21:37:23 +08:00
sennnnn a360da1ec7 update videogpt.
Former-commit-id: 2e211563c684a8b6ef62dfc0c4f1897224dfb58d
2024-03-11 20:35:01 +08:00
sennnnn 46126baadf use fp16 temporarily.
Former-commit-id: 93381968e10aae7bbec23ec1ef40e7cb311fddf0
2024-03-11 17:03:23 +08:00
lb203 2eba66af88 Update README.md
Former-commit-id: c2c64665d4f281e264d87c20cc1cfd38e7e5f168
2024-03-11 16:58:01 +08:00
sennnnn a5d891a8bc Add deepspeed script for videogpt training.
Former-commit-id: 3bda2a05cb789fe785b2355a613385b3eb7abbbc
2024-03-11 16:29:11 +08:00
lb203 d1c7015eaa Update README.md
Former-commit-id: abb8ed57a29b1f66adec66d854429f36f9199540
2024-03-11 16:10:31 +08:00
sennnnn 56ed84acbd Add hidden_size argument for videogpt config.
Former-commit-id: 1be9f22767fe9c80ad90cb17184751b97aff9012
2024-03-11 15:46:54 +08:00
sennnnn 9fbd96d66c Merge branch 'main' into videogpt_deepspeed
Former-commit-id: c815b5eae7d60f89df0366f762a9a4afc8c7655e
2024-03-11 15:45:15 +08:00
LinB203 0e2a86b2ac clean dit
Former-commit-id: 117ec4196a6f2cd83da223b993b14e28f7a693f8
2024-03-11 15:44:53 +08:00
sennnnn 59030ac60c Merge branch 'resolution_typo' into videogpt_deepspeed
Former-commit-id: acaac6af0e35fcc611a9ecd15a39d6bb41aed4aa
2024-03-11 14:36:07 +08:00
sennnnn 2b17f4fac2 Merge main branch.
Former-commit-id: 800ca74fad0d9021f1e96ddf32cd93cdb37cbbe0
2024-03-11 14:35:17 +08:00
lb203 523670a478 Merge pull request #117 from sennnnn/resolution_typo
Fix resolution typos

Former-commit-id: b9acfa97f5514513f6c983f3e815300b435d3076
2024-03-11 14:32:59 +08:00
sennnnn 1365640065 Fix resolution typos.
Former-commit-id: 3ad55d28eaf13f9c8fd23912d2691965737527e9
2024-03-11 14:26:05 +08:00
lb203 59e2407366 Merge pull request #97 from HowardLi1984/refiner
[feat]: Caption Refiner

Former-commit-id: 6af3b3689679a99acb741218416dc3a9cde29f45
2024-03-11 14:05:35 +08:00
sennnnn 1321e1f977 refactor dit for supporting deepspeed.
Former-commit-id: e1909745cb05ea2f28f3a391c194f7656593a236
2024-03-11 11:51:46 +08:00
lb203 14e217a698 Update README.md
Former-commit-id: 1f94f2721f35670aeae8dbfb7c0a5abd06cd2aee
2024-03-10 21:37:01 +08:00
lb203 d4aa360c50 Update README.md
Former-commit-id: 2808edf11005e04aa730971a786d1a4e3bca88b5
2024-03-10 21:16:15 +08:00
lb203 93b624e62d Update README.md
Former-commit-id: 64d147863368aeef479ad25f2e5463fdcb76de46
2024-03-10 20:27:46 +08:00
lb203 f423fc7538 Update README.md
Former-commit-id: 6d41cfe906f97dbed1cea8c9ed06a1f6dc84308e
2024-03-10 20:25:48 +08:00
LinB203 a7165b506e clean script
Former-commit-id: 02c61f63a8539a31516b92f5bb7c2d2e428ca7d7
2024-03-10 18:34:37 +08:00
lb203 740363460f Update README.md
Former-commit-id: 7a62dec060c91b02ca556451266e6f3ef28e03e7
2024-03-10 18:27:50 +08:00
lb203 aeb3081341 Update pyproject.toml
Former-commit-id: 5c379b6e4515ad75b89cdebbc5cd490c9a821d78
2024-03-10 18:21:50 +08:00
lb203 fb62b8ff7e Update README.md
Former-commit-id: 059d9e863958cfd679dbb8d365de6cd933e284fc
2024-03-10 18:20:28 +08:00
lb203 74494966bd Update README.md
Former-commit-id: 87636c1dc2275c4da77c6368a0f4425bea15dc12
2024-03-10 18:15:21 +08:00
LinB203 b57709108a train 1080p video
Former-commit-id: 74625142ef892f9e7fa459b2f98f2f7ae5b36d6a
2024-03-10 18:08:23 +08:00
lb203 1f4d4cbdfd Merge pull request #112 from sennnnn/diffusion_deepspeed_fix
🐛 [BUG] Add base config for Latte

Former-commit-id: 825aa78ded6a5436d53d488b2c2250ac41b2bff9
2024-03-10 15:29:50 +08:00
sennnnn 995454b518 Add base config.
Former-commit-id: d37c551a0421d6b73dee26060b517c46d8217e4f
2024-03-10 15:24:40 +08:00
lb203 3d3ea0c818 Update README.md
Former-commit-id: 582070d4f97c4a4398dcb9a6699f4ab6e980c812
2024-03-10 15:17:18 +08:00
LinB203 6dae0b1c0c deepspeed
Former-commit-id: 6179a4b0e7d06caed4201a27679abc64615febe6
2024-03-10 15:08:12 +08:00
lb203 557b062028 Merge pull request #110 from sennnnn/diffusion_deepspeed
⭐[Feature] Support deepspeed for latte training.

Former-commit-id: 59e731cb98915a2635547cad904ced7d9fcf73f4
2024-03-10 14:48:03 +08:00
sennnnn 28b7fc4bae Fix typos.
Former-commit-id: c10330d373466f3c1ab455bfb391569b9175bc3a
2024-03-10 13:17:44 +08:00
sennnnn 61f9c5bced Add some comments.
Former-commit-id: d5b5bd7743edde6a63dc74457bded0f8c56be1c2
2024-03-10 12:23:42 +08:00
sennnnn f3274235d8 Support deepspeed zero2 and zero2_offload for latte training.
Former-commit-id: c87e95dd357e6fa6d6fb2dcfee781bc49c039e13
2024-03-10 11:27:33 +08:00
sennnnn 81c455e878 Refactor latte.
Former-commit-id: 7408fd0ee16247afc3d4ba43b8c8e7dfb1091a97
2024-03-10 11:25:56 +08:00
sennnnn 443bab1364 Refactor latte.
Former-commit-id: 304534269f105578097bc16ab497636ad59fd9b2
2024-03-10 11:25:50 +08:00
sennnnn bfe4f15694 Fix mixed_precision bugs of latte.
Former-commit-id: d70a551972e04d95444d46cdbe2bfb16ad099fcb
2024-03-10 11:25:07 +08:00
sennnnn ee3810a585 Decouple mixed_precision and gradient_accumulation to command args.
Former-commit-id: 903d473d1432eb492a87efe184d03439f643b92d
2024-03-10 11:24:39 +08:00
lb203 d6d54da131 typo
Former-commit-id: 9e87366e906ed19726c8905afb27b5df4d4e1512
2024-03-10 11:15:14 +08:00
LinB203 f034826c6d fixed attn_mask in flash-attn
Former-commit-id: 5e2504f429a342b75bc7a9004cee7b3f75582ee3
2024-03-10 11:12:12 +08:00
lb203 74ee558f49 Merge pull request #107 from jpthu17/fix_2d_RoPE_init_bug
[fix] 2d RoPE init

Former-commit-id: e873ffa66f01268cce600f8da9ad297dc3e8aaaf
2024-03-10 10:21:43 +08:00
jpthu17 75938cfaeb [fix] 2d RoPE init
Former-commit-id: ff5931bae6efa94675b9ef6bc25f1b76217d128e
2024-03-09 22:04:15 +08:00
LinB203 9a0d0bc843 safe dit
Former-commit-id: f407cab7dc0e53257972d430e873f4fd77bb039d
2024-03-09 21:53:43 +08:00
LinB203 5ebb382d85 xformers and flashattn
Former-commit-id: 002cad95253f7da36344237ce8c8410ffb517711
2024-03-09 21:52:03 +08:00
lb203 c08f985c8c Update README.md
Former-commit-id: 6630c22baa2340a8b55badd61472daab7177bc63
2024-03-09 21:19:47 +08:00
lb203 c891f85022 Merge pull request #106 from jpthu17/add_2d_RoPE
[feat] Add 2D RoPE

Former-commit-id: 85c46fa0e728c9cfc5af4cf6523a7e6a64f96a93
2024-03-09 21:18:17 +08:00
lb203 ad54bd8f94 Merge pull request #105 from sennnnn/einops_bug
Fix einops bug: module 'keras.backend' has no attribute 'is_tensor'

Former-commit-id: c4f3a3a1172b68a4a34ddfac8ea69be34d11038b
2024-03-09 21:13:29 +08:00
jpthu17 aef95b043c [feat] Add 2D RoPE
Former-commit-id: ccae80d339e34fb23706827565b5095ae2c2f320
2024-03-09 19:53:48 +08:00
sennnnn 5834bc0a8b Merge branch 'main' into videogpt_deepspeed
Former-commit-id: 04dbc4c5f04d1e82815e02c6215d8ff6e098049f
2024-03-09 19:24:12 +08:00
sennnnn 29e167fecf Fix einops bug: module 'keras.backend' has no attribute 'is_tensor'
Former-commit-id: 849f8ed815295bac78c3c10d319e6731aa26ca02
2024-03-09 18:18:26 +08:00
sennnnn efad4348d5 Fix config typos.
Former-commit-id: 455973b57bda19379d41e59aa418421fd269281e
2024-03-09 17:39:03 +08:00
sennnnn 3b4e881b11 Support deepspeed for videogpt training.
Former-commit-id: 1f3dfe038845d91b81b0ecc9b6f3cabb06eed5d5
2024-03-09 17:33:30 +08:00
lb203 e415e25897 Update README.md
Former-commit-id: 513883df1ad853c9a5e7457c364227261df17a83
2024-03-09 17:13:18 +08:00
lb203 e2b563ab2d Merge pull request #91 from RuslanPeresy/typecheck
[refactor] add static type checking

Former-commit-id: 969dfa6dedaeb64cf776d977b466b14568cc4b1c
2024-03-09 16:54:19 +08:00
LinB203 6df13b07bb flash-attn train and sample
Former-commit-id: e336565bd5ff2aaf4d0b4ad5f42732e89cd216b3
2024-03-09 16:30:09 +08:00
lb203 39e398e815 Merge pull request #103 from rain305f/new_eval
[docs]: update the eval code

Former-commit-id: 5ee073e7493a608c27c257b2a341585192c5b68f
2024-03-09 15:47:54 +08:00
rain305f e780368bb6 [docs]:update the eval code
Former-commit-id: a8c9e6cb9e39a9e929dab34077280a8d80d85573
2024-03-09 06:22:04 +00:00
rain305f 8a987de2a8 [docs]:update the eval code
Former-commit-id: 937f2fc87d34ab9d4724cb95a6457b54a6bf4ebc
2024-03-09 05:21:43 +00:00
lb203 a60217616e Update README.md
Former-commit-id: 38e0e290e7a5909706c2cce2e4dd047647134fb8
2024-03-09 11:49:17 +08:00
lb203 fa401d1d3f Update README.md
Former-commit-id: 88e9f21cfc8ec775dbc9a65f740043e91e26bd91
2024-03-09 11:44:12 +08:00
lb203 81f5e207d0 Update README.md
Former-commit-id: 826cc5e9162888bed82413088c6be662c278176b
2024-03-09 11:41:37 +08:00
lb203 7774b0d1eb Update pyproject.toml
Former-commit-id: 36da5d0e950625e629aebe954f89f92827397d0d
2024-03-09 11:37:47 +08:00
lb203 82cc72a739 Merge pull request #96 from Jason-fan20/my-docs-branch
Update README.md

Former-commit-id: 426cbdfa78282e098739dc1740d802d8e9c7d6de
2024-03-09 10:58:50 +08:00
lb203 39848c3dff Merge pull request #102 from sennnnn/vqvae_data
🐛 [BUG] Fix the memory error bug when loading metadata pickle file of vqvae dataset

Former-commit-id: 3b458d91fdc646e63f004031ebab9984a34934d2
2024-03-09 10:58:18 +08:00
Li Hao c8ba21979e [BUG]: Shorten the refined caption
Shorten the refined caption using GPT-3.5 summary


Former-commit-id: bb159f0d5fa928258475b2058983397ceb91af9d
2024-03-09 10:22:09 +08:00
lb203 38c2c2ccaf Merge pull request #99 from jialin-zhao/jialin-zhao-xformers-inputs-revised
Update latte.py about the xformers input dims

Former-commit-id: 421b49c351124170ab476e17cba6e45bd7cf706e
2024-03-09 09:58:51 +08:00
lb203 c42a86310d Merge pull request #94 from rain305f/eval_code
[feat]:update the eval_code for calculating the FVD, clip_score etc. metric

Former-commit-id: 363b237f71387e51164d02ed622c5b997fd5dde3
2024-03-09 09:43:25 +08:00
LinB203 b0a9e72008 extract script
Former-commit-id: 4139f2f68ae09695302ce49afdb7318df83cabfd
2024-03-09 01:03:05 +08:00
lb203 7573495d70 Update README.md
Former-commit-id: 99b0066895af072a8887737f2b603ec7fc893ccd
2024-03-09 00:33:09 +08:00
rain305f 130fcd397a [docs]: update the evaluation
Former-commit-id: 02e3789b74daa036fbbff6d13dfc7e98372260ed
2024-03-08 15:38:54 +00:00
sennnnn 41b14c6a13 Fix the memory error bug when loading metadata pickle file of vqvae dataset.
Former-commit-id: 97ab243bd12915cea7aa2deecfac4b99417f78a3
2024-03-08 22:54:17 +08:00
lb203 f751508193 Merge pull request #55 from SimonLeeGit/main
[refactor]: add docker development support

Former-commit-id: 159477551119f68a07adeab349d37083b390e533
2024-03-08 20:51:55 +08:00
lb203 8eef9df1b4 Merge pull request #61 from Mon-ius/main
[feat] - add CI for docker autobuild

Former-commit-id: f9b004a99e67a32be46c1bb3267ad20378b1e89b
2024-03-08 20:50:48 +08:00
lb203 61643ba9e8 Merge pull request #100 from Linzy19/230308
[Updata] updata the video super resolution

Former-commit-id: ebfe0c526f0ee977a91ef0214658783972afc0ac
2024-03-08 20:49:49 +08:00
ZongyingLin 80d857ac86 [Updata] updata the video super resolution
updata the run.py and add README.md


Former-commit-id: 1e6d6ace7a0963149e228813801210b022aceacf
2024-03-08 20:45:47 +08:00
Li Hao 99fe427ac8 [feat]: Caption Refiner
Caption Refiner for Video Caption


Former-commit-id: 97c16bccb9979acf19fb91ae8d7b84ac4a3b4d97
2024-03-08 20:04:41 +08:00
chialin cd2ffd7229 Update latte.py
revise the xformers inputs

Former-commit-id: 60db66c912192c3fbafe55ebe16157514b884e5b
2024-03-08 20:02:31 +08:00
Jasonfan 825fe9eada Update README.md
Former-commit-id: 8f7b41118a3f491d1d5b5789c53790c46b023191
2024-03-08 19:59:51 +08:00
lb203 e3f002e280 Update README.md
Former-commit-id: 70f6ebf421f0830a51dcc2838c2c3783cfb67376
2024-03-08 19:24:44 +08:00
LinB203 a23d24fd11 fix extract feature
Former-commit-id: d4dce5ec1aaa058e452a41f15cd35fe1713f0b78
2024-03-08 19:22:36 +08:00
LinB203 2fd41ff003 update sample.py
Former-commit-id: 10d8621051d1510270c6a55abe686eac86297170
2024-03-08 17:28:16 +08:00
LinB203 bc4e4f6fa6 del use_fp16
Former-commit-id: 5424c0112e04b56d1df30ac76cbc104e848406a3
2024-03-08 17:27:22 +08:00
LinB203 eaee962fa9 update train.py
Former-commit-id: 844173711342f2dcc021294e87971ca39854bfe4
2024-03-08 17:25:48 +08:00
LinB203 b07538bf8a feature dataset
Former-commit-id: 9683caae8648735552270b871de22fc540ba2b55
2024-03-08 17:24:23 +08:00
LinB203 21fad0d34a update script
Former-commit-id: 43f539b68af150c45040d79918a84de1dce47c33
2024-03-08 17:21:06 +08:00
rain305f 77a79e67b6 [docs]:update the eval_code for calculate the FVD, clip_score, ssim, lpips, psnr
Former-commit-id: b504623d7ec2ac0b508530d21ffc88978f8a4f71
2024-03-08 09:19:42 +00:00
rain305f cf2e4f48d0 [docs]:update the eval_code for calculate the FVD, clip_score, ssim, lpips, psnr
Former-commit-id: 1ee226c338126fbdc5178020b12271b980a7ce0c
2024-03-08 09:11:05 +00:00
SimonLee 9c406ca0d7 Merge branch 'main' of https://github.com/PKU-YuanGroup/Open-Sora-Plan
Former-commit-id: aa4a88029885ff0b6e4e96174c07927116150434
2024-03-08 15:44:23 +08:00
RuslanPeresy 7e7dcbe406 [refactor] add static type checking
Former-commit-id: 12c4240dc28888c61f1f79243627e7cdb445c840
2024-03-08 15:37:47 +08:00
SimonLee 9711db6896 update docker scripts and configs
Former-commit-id: 4567e6368f1bf413d8f6319a97a852b8289ca770
2024-03-08 15:32:17 +08:00
lb203 e37dae81ac Delete scripts/train_vqvae.sh
Former-commit-id: 952a4ce3a3b0da159d599ff3a0a052265f48c2dd
2024-03-08 11:21:00 +08:00
lb203 20ca4c089c Update README.md
Former-commit-id: c6b1ea8f1ef01096aa8c3e5272bc9fb255b5532b
2024-03-08 11:18:43 +08:00
SimonLee 518ce56f09 [refactor]: add docker development support
Former-commit-id: fded3baff7915c0cfc9cec3765c271259359831c
2024-03-08 10:58:28 +08:00
lb203 997738438f Update README.md
Former-commit-id: 97fe914aaa2b9bd7c2ff5d4239381631c47e8356
2024-03-08 10:01:16 +08:00
lb203 b76c71e634 Update __init__.py
Former-commit-id: 03a45aaf60d294b7a68d8fc44132d3d671e4472b
2024-03-08 09:46:17 +08:00
Liu Hanchen 203105e504 Update README.md
Former-commit-id: b37da2202e730a4930a194a80a703b591a535adf
2024-03-08 01:27:35 +08:00
Liu Hanchen 76ce11567a Update README.md
Former-commit-id: bca7070fac79d5ae5c37830cd55ca1e81a289197
2024-03-08 01:24:29 +08:00
Liu Hanchen d6980acdd1 Update README.md
Former-commit-id: c4d7aafa3ceb6e4d48bf85b402ae2b2ce9394e53
2024-03-08 01:16:28 +08:00
Liu Hanchen 2448aa7c8c commit
Former-commit-id: 81359cf67248b6c594ff79f121a8a38558c1cc70
2024-03-08 01:04:16 +08:00
lb203 bb167c3a70 Update Data.md
Former-commit-id: 86be6b2bf5a9be1030bb3af661a2ec85e710a5c3
2024-03-07 23:00:54 +08:00
lb203 85232cc966 Update Data.md
Former-commit-id: 9531e33666f842125b331cb266eb26a3a03a4009
2024-03-07 22:35:54 +08:00
lb203 ad5632cc50 Delete opensora/models/super_resolution/placeholder
Former-commit-id: 8dd5a6a3c85dc28989ff95e6ad41be7fdb0f0811
2024-03-07 22:19:00 +08:00
LinB203 9c8b50a29a train script
Former-commit-id: 3fd1ccf099c2deed20af50bf9556e2fc3b7a4382
2024-03-07 22:17:39 +08:00
lb203 68cf1dd673 Merge pull request #88 from Linzy19/2307
[updata] updata video super resolution

Former-commit-id: 7f512b4f186a795358b7e2c922068e2f881aabae
2024-03-07 21:53:22 +08:00
ZongyingLin ae766cb6dd [updata] updata video super resolution
updata video super resolution


Former-commit-id: 82397545859791d112d6f0569e33fe23437563df
2024-03-07 21:27:28 +08:00
lb203 2e84d17e50 Update README.md
Former-commit-id: 7cd398721e556db6199f1d06c7a0d1de2a9720dd
2024-03-07 21:22:24 +08:00
lb203 9f4a42baf9 Update README.md
Former-commit-id: 076e6e70e717888f24265d624c6a7a8ffa2f5d1b
2024-03-07 20:48:32 +08:00
lb203 7a03fe4ccc Update README.md
Former-commit-id: deddba1f03bf697fb1252a0ce4fdb5df5bcae8de
2024-03-07 20:44:55 +08:00
lb203 af8ade6655 Update README.md
Former-commit-id: a1d1f07c83eb0c03d93b023b5c9162a6a3c26cd9
2024-03-07 20:44:12 +08:00
LinB203 af49b2a9e9 init pos embed
Former-commit-id: 28f8076136ec214c66039ead7547c9b25cfe7bfc
2024-03-07 20:01:10 +08:00
lb203 0ca3104739 Update README.md
Former-commit-id: e458366e47d4b13737a71bbf866e20ec10961ade
2024-03-07 19:52:52 +08:00
lb203 781c8ca81d Update README.md
Former-commit-id: 3b4eb14dcd8862b8251b72f30cba523a5dc1def8
2024-03-07 19:52:06 +08:00
lb203 2c7fa77589 Merge pull request #84 from Nyx-177/main
[docs] Modify README.md to fix spelling and grammar issues

Former-commit-id: 096a602e57b5dc3a99d98fc55d608f7569b17af6
2024-03-07 19:42:57 +08:00
Nyx 4cd6d293ca [docs] fix merge conflict
Former-commit-id: d8b1afe957765f3844bd9c1d88dd925efa57330e
2024-03-07 21:30:20 +10:00
Nyx 5149e05147 [docs] Modify README.md to fix other grammatical errors
Former-commit-id: f0de27fcc38bb4dab0eeafac225a2924080a3218
2024-03-07 21:15:56 +10:00
lb203 fbd9b89c01 Update README.md
Former-commit-id: 8a191abc1903ad5998fe83149c9c9ef5c8226b9e
2024-03-07 19:14:49 +08:00
lb203 a0a8e4fb76 Update README.md
Former-commit-id: a881ff5198e1125032418f4f02fd95d2179ffb3a
2024-03-07 19:14:12 +08:00
Nyx d57553d520 [docs] Modify README.md to fix translation error in name "CloseGPT"
Former-commit-id: dd36043df4b0408e592a82f9046f6d654a5375c3
2024-03-07 20:55:02 +10:00
lb203 8e0e8ba91e Update README.md
Former-commit-id: 2a68a1aa399a53f58246df4f9f64dfdcdd130bcf
2024-03-07 18:46:14 +08:00
lb203 28ea2cc1b2 Update README.md
Former-commit-id: 7bc544af55c0a987d0df426848cb06cb330fd52b
2024-03-07 18:16:23 +08:00
lb203 4a07be0b0e Merge pull request #82 from sennnnn/star_his
Rapidly increasing stars are the best honor of a great open-source project.

Former-commit-id: 6aeda408cc64765f00092e61b6ad9a3f0b48fe4d
2024-03-07 17:51:10 +08:00
sennnnn ff69d06080 Fix conflict.
Former-commit-id: 6d4a3950cb135718b3f122657733cb5f890db971
2024-03-07 17:46:31 +08:00
sennnnn 390d7f1896 complement emoji.
Former-commit-id: 50679cc482d8ecf4e3083c0f5a791f195a58a855
2024-03-07 17:39:51 +08:00
lb203 ca9bbce122 Update README.md
Former-commit-id: 7af6b82b29c265e53e9c48477083b4cff0e0a3d3
2024-03-07 17:37:24 +08:00
sennnnn 0fde7b4edd refine readme.
Former-commit-id: 391d9511b72aa9c09c7ad8992ae5df8c89c88c63
2024-03-07 17:37:22 +08:00
lb203 74115d65c4 Update README.md
Former-commit-id: f4c4b3b4b1ce3c1d0b56ba5a109cd6cc57da900d
2024-03-07 17:27:18 +08:00
lb203 a261d31f3a Update README.md
Former-commit-id: 2a286e5aab48aa68e7d3f04df23ec1cdfe71c6de
2024-03-07 17:25:52 +08:00
lb203 da2c84cf50 Update README.md
Former-commit-id: 4a04e71636dabccbef446623674f3a514395b7f9
2024-03-07 17:21:53 +08:00
lb203 3dc3c31fda Update README.md
Former-commit-id: da5507e570a27becab754ba24a0671d669b7e08e
2024-03-07 17:03:17 +08:00
lb203 d14563a9d4 Update README.md
Former-commit-id: 4d83f1feab7d6872aca1c6b41be95950ec3cf465
2024-03-07 17:00:33 +08:00
LinB203 784c246e9c videogpt inference
Former-commit-id: c5d617450a8271ad3b8a5e8f76ee2d77a822e31b
2024-03-07 16:44:19 +08:00
LinB203 11996dc9b4 fixed dit forward
Former-commit-id: 4cd537fab12231f7471107762b02b36217ca42d8
2024-03-07 16:41:31 +08:00
lb203 a03025a8f3 Merge pull request #80 from qqingzheng/videogpt
[docs] Modify README.md and add VQVAE documentation.

Former-commit-id: a92ddf62b7bbd15820401f52edf853dbbb513301
2024-03-07 16:39:28 +08:00
lb203 035214520e Update README.md
Former-commit-id: 89f9fc8eed3ca8cc2434ec227acbb845bfbcff82
2024-03-07 16:37:47 +08:00
lb203 93e98f1ef5 Update README.md
Former-commit-id: 4a893ff6177d59219b6755c4a89764ef54fc7f62
2024-03-07 16:36:16 +08:00
lb203 8b4ad5bf4f Update README.md
Former-commit-id: 0898e4b917f5e245c4394118511bcb7639973817
2024-03-07 16:33:17 +08:00
lb203 afdd297bd5 Update README.md
Former-commit-id: 8b553a4477c64b5452f8424ca6987c0fb8c111ce
2024-03-07 16:31:19 +08:00
lb203 8aa1181ab2 Update README.md
Former-commit-id: 8e9658ed993380abc43b518fcbac21a02d8ba9ba
2024-03-07 16:26:12 +08:00
Chestnut b64293c723 Merge branch 'PKU-YuanGroup:main' into videogpt
Former-commit-id: 339857e1b672a1a688f86c1b8e1d8dae29f10301
2024-03-07 15:03:43 +08:00
qqingzheng d681598e6e [docs] Modify README.md and add VQVAE documentation.
Former-commit-id: a5bfe70f8a8998db969d9c4938fda123aa153bc0
2024-03-07 06:59:40 +00:00
qqingzheng cb86e18190 [refactor] rename train_videogpt.sh
Former-commit-id: e2ea2caacaf5be046f48cf50a53ccafa7a3cc351
2024-03-07 06:15:14 +00:00
lb203 9d9bad199c Merge pull request #76 from luo3300612/dev-rec-video
[fix]: disable gradient computation to save GPU memory when reconstructing a video

Former-commit-id: b8cda6636444080d0cf00452be75a70c1e79f32a
2024-03-07 13:49:39 +08:00
lb203 b6bc9b1455 Update README.md
Former-commit-id: 36e69275598b0cff38fdc5006926e422cda069b3
2024-03-07 13:47:45 +08:00
lb203 5caa97e9a9 Update README.md
Former-commit-id: 4179ed075e68d48094655c38dbde784836fd6e0f
2024-03-07 13:46:58 +08:00
lb203 f507bb4947 Update README.md
Former-commit-id: a1cc0b26a3b156d7708e8df0cfe42aad33b1c946
2024-03-07 13:45:01 +08:00
luo3300612 0379dbb349 [fix]: disable gradient computation to save GPU memory when reconstructing a video
Former-commit-id: 06da17790b53cf593a0c264dda17c76f68d2a541
2024-03-07 13:39:56 +08:00
LinB203 c521999f80 attn_mask with bf16
Former-commit-id: d9667a8c15a5344e3c777197fd6badec15dcddbd
2024-03-07 13:35:01 +08:00
lb203 fcc8009570 Merge pull request #64 from khan-yin/main
[feat]: add sit model

Former-commit-id: acacfd670282b709e6c769717b5295e230861724
2024-03-07 13:29:24 +08:00
lb203 5a7e9c8822 Update README.md
Former-commit-id: 29f24a819fe2d4593c48afe6ca990659784bd776
2024-03-07 12:58:47 +08:00
khan-yin c716b3281e feat: update sample
Former-commit-id: e7dc492b9b4764db28aa31cc1eede914052b4ca6
2024-03-07 04:13:57 +00:00
khan-yin 94b71a5ea0 Merge remote-tracking branch 'upstream/main' into main
Former-commit-id: 7ef9b5e76a91b444f87955c3e0e7ea3d72315396
2024-03-07 04:05:06 +00:00
khan-yin e90373b161 feat: Incorporating SiT sample
Former-commit-id: 91e2931e412f8c96ef748cb2322ee956309fb6aa
2024-03-07 03:40:24 +00:00
lb203 7cf1bcc93d Update README.md
Former-commit-id: b3f6bb7f252505efb0ffd8195003c79206fac313
2024-03-07 11:38:23 +08:00
LinB203 cca84f1c6c denorm fun
Former-commit-id: 05707a07b14295f5e1932d0c10bb84a20a0cd328
2024-03-07 10:40:06 +08:00
lb203 1ebf531e1e Update README.md
Former-commit-id: 1a7178a61547c33a744d032302fb76ee0a3c6671
2024-03-07 10:33:20 +08:00
lb203 c9adc3e9a5 Update README.md
Former-commit-id: d097e40998a3396f74c2bfbcf26d6ace8080d0b6
2024-03-07 10:20:51 +08:00
Kehan Yin 3f19169f55 Merge branch 'PKU-YuanGroup:main' into main
Former-commit-id: a2dea6886ce90634fb7d1c2e0fbfce95a500eefc
2024-03-07 10:04:08 +08:00
lb203 006ad5353a Merge pull request #70 from qqingzheng/videogpt
[refactor] Performed code abstraction on the VideoGPT code and enabled support for accelerate training.

Former-commit-id: 822c1a173f5bc792868fae9e6e4cd7defe2ea183
2024-03-07 09:23:28 +08:00
Chestnut 589da07fec Merge branch 'main' into videogpt
Former-commit-id: 714cc434cea5823fc09ef6fe502b1fe83170cae2
2024-03-07 01:25:58 +08:00
qqingzheng b409ef399a [refactor] added more methods in abstract classes
Former-commit-id: a535e3a06eb766bebe7baa3d9df64b04d04d40e6
2024-03-06 17:23:40 +00:00
lb203 8944956249 Update README.md
Former-commit-id: 604a29d1de43e0fcb90ada4dd97f5f080e37caea
2024-03-07 01:11:28 +08:00
lb203 921d927a27 Update train.sh
Former-commit-id: 66aa3709f80d89a30d0a57c5e033ce1ae46f998f
2024-03-07 01:06:20 +08:00
lb203 06423d785d Update train.sh
Former-commit-id: 6c4957d715eaa489691be21dd0abdb9309660e0d
2024-03-07 00:57:14 +08:00
LinB203 38bf4afdfa support accelerate training
Former-commit-id: b32671c9991b9a9db66cb148822ccb3e8be7dc11
2024-03-07 00:55:35 +08:00
qqingzheng 7a78ef2a97 [refactor] configuration default values
Former-commit-id: d7ca7685d6a8c6d015082dda241bc98ab156770d
2024-03-06 15:59:13 +00:00
qqingzheng 4549eeeab3 Merge branch 'videogpt' of github.com:qqingzheng/Open-Sora-Plan into videogpt
Former-commit-id: ca5f92952c7a0daae6deaca40f6dab1d00a0c0b2
2024-03-06 15:48:26 +00:00
qqingzheng da08e767db [refactor] rename
Former-commit-id: ec317605d24223527b8c2d4b4d2ee876f86abbaf
2024-03-06 15:48:19 +00:00
lb203 0118dd8f1d Merge pull request #69 from luo3300612/dev-rec-video
[fix]: disable gradient computation to save GPU memory when reconstructing a video

Former-commit-id: f9afcda4130e1709d6f201ed310adcc8335480d9
2024-03-06 23:46:10 +08:00
lb203 4efdbc2aca Update README.md
Former-commit-id: 59a595a1e31b156500a944f244afec40988e872d
2024-03-06 23:39:17 +08:00
Chestnut 140eb267d9 Merge branch 'main' into videogpt
Former-commit-id: 0a45d76fdcc3faf167239dfe5dc055b30f3f4e05
2024-03-06 23:36:14 +08:00
LinB203 1b99cf2789 add sample script
Former-commit-id: 9ce847651bb6dba28c51ce5807e222ae2902a78a
2024-03-06 23:23:17 +08:00
lb203 ae2646e213 Merge pull request #74 from Linzy19/2307
[update]: update super_resolution

Former-commit-id: a2586e672bd5d60f1a5e937a796d5f6b6ac1da4d
2024-03-06 23:19:44 +08:00
ZongyingLin 36e0f8813a update
add RGT


Former-commit-id: a491111e48cec588095381eb70ffca94600ca4f7
2024-03-06 23:10:26 +08:00
qqingzheng 0622ea11dc [fix] fix a bug when using ucf101_stride4x4x4
Former-commit-id: f541e723d65ea44ae4560c68908123c7e6781bbc
2024-03-06 15:08:36 +00:00
qqingzheng df625f66cc [refactor] remove test.ipynb
Former-commit-id: e73c3a9e0d41e51a0fd755e57739424213a76312
2024-03-06 15:07:11 +00:00
qqingzheng 107f140b56 [fix] fix a bug when using ucf101_stride4x4x4
Former-commit-id: 9911403c688321db731450d9383bc02923a4be2d
2024-03-06 14:51:35 +00:00
Chestnut 2fd71909a4 Merge branch 'main' into videogpt
Former-commit-id: f8de10e627ce27f1511ddbbc5d6b29f3daf094ab
2024-03-06 22:13:26 +08:00
qqingzheng 2b272863b0 [refactor] adapt to old training code
Former-commit-id: 358d362ecbc100b21ea02895cb0426cf5b386a72
2024-03-06 13:58:13 +00:00
qqingzheng 1bed11aee3 [refactor] reformat videogpt, support training videogpt on accelerate
Former-commit-id: 75385ed47fe60008b61b1bcf89f61ea696d095bc
2024-03-06 13:45:16 +00:00
lb203 5dc0b3cb41 Update README.md
Former-commit-id: 673659495b7503ba402fbcb6c46204499761b4cf
2024-03-06 21:24:58 +08:00
root 80d394d64a [fix]: disable gradient computation to save GPU memory when reconstructing a video
Former-commit-id: b7bf763e7f8800959ed82bb89f41684db216dd8e
2024-03-06 21:04:29 +08:00
lb203 f58d14ba24 Update README.md
Former-commit-id: 398cfcd7b19cf1ba827eb49fc11223f72fa93a85
2024-03-06 20:46:10 +08:00
lb203 d3acf63c4e Update README.md
Former-commit-id: 9fd88eb8ac5122b0999ccb0071f195aaeb0979a0
2024-03-06 20:30:57 +08:00
LinB203 193d7ef7a0 support latte training
Former-commit-id: d471a9414842f422d1200c2fd3f3ddfcf9e97fd6
2024-03-06 20:29:31 +08:00
khan-yin e47f894f69 fix: rename sit model filename
Former-commit-id: f3b2cd7dcf644f680616eee0571db90478fede23
2024-03-06 12:08:04 +00:00
khan-yin 7cee5443e4 fix: remove replicated diffusion module in sit
Former-commit-id: 997cb049185421d303b14d59ab6c0603f1d1a66f
2024-03-06 12:06:55 +00:00
Kehan Yin edfe14e9fa Merge branch 'PKU-YuanGroup:main' into main
Former-commit-id: c7d4f9c2cfaa61e3babc711f13a095ed237978b1
2024-03-06 19:46:26 +08:00
lb203 31f5ce25d4 Update README.md
Former-commit-id: 898d35f15ce0fd124cd48078f0a8af2764bf55d1
2024-03-06 19:33:06 +08:00
LinB203 5a80da7ec1 add vae and fix bug in dit
Former-commit-id: 2dd4254c22ce671d5cd5e7d22a2386b7660fbdc4
2024-03-06 19:32:20 +08:00
lb203 ec0dd141e1 Merge pull request #63 from kabachuha/alternative-attentions
Support for less VRAM heavy alternative attentions (ReBased and Ring attention)

Former-commit-id: c90f9d3c780843edccc72d29f68f5ed7c655a721
2024-03-06 19:28:59 +08:00
khan-yin 54d3069702 [feat]: add sit model
Former-commit-id: 2eda854d6c25254b71d0105a3ed092953a9d6f68
2024-03-06 09:52:27 +00:00
kabachuha abffb358f9 support for ring attention
Former-commit-id: 0533c73b99217e88da121bd0f68e0949bc88016e
2024-03-06 12:24:10 +03:00
kabachuha e1459ff3d0 option to use rebased linear attention
Former-commit-id: f1b39b34df5fbdef14ef3f6a0e6d0972a8570445
2024-03-06 12:18:03 +03:00
Monius f2a826a2f8 add CI for docker autobuild
Former-commit-id: c4e2d5ff1dce8662a42f1ed6a7786855d40173f7
2024-03-06 16:58:03 +08:00
lb203 29409555f3 Update README.md
Former-commit-id: 10ee64c122f28efea39fc1a4542d278683a58ce1
2024-03-06 16:27:06 +08:00
lb203 d878709456 Update README.md
Former-commit-id: 0f2bdd2bcd819d528f2a41a96ba93f5c3c04dc8c
2024-03-06 16:23:53 +08:00
qqingzheng 3810227fd6 first commit
Former-commit-id: ec3481bc35a8120d519082dc76585428f0896470
2024-03-06 08:16:58 +00:00
lb203 8b3920706a Update README.md
Former-commit-id: f1aaeca19bc9e4304961c8c903f9bf03cb42fd9f
2024-03-06 16:13:26 +08:00
lb203 932c7fd80b Create LICENSE
Former-commit-id: 369fe085e91e55b9be6165da4a1eb7f9c0cf5a7f
2024-03-06 16:12:41 +08:00
lb203 cfb34d011b Delete LICENSE.txt
Former-commit-id: b10d8b2c6620e3a8412527bd03ea0adc26c981a6
2024-03-06 16:11:05 +08:00
lb203 0147265637 Update README.md
Former-commit-id: 437bc5c68b5c7a85865ad97d544c4818be45cb52
2024-03-06 15:38:04 +08:00
LinB203 cf40ce1e7a add latte
Former-commit-id: 01ba10e5d65c9d2d957472fbe7958e1b07a971e9
2024-03-06 15:28:20 +08:00
yanyang1024 4f608035e2 Update README.md
Former-commit-id: c5a79c7e993d9e6a77c6c70efa1c0ac2bb3a063b
2024-03-06 12:12:58 +08:00
yanyang1024 70ee26f0ec fix_readme_requirements
Former-commit-id: eb5a5c69d12cc412c12a4fea7f00baafd02701ca
2024-03-06 12:05:21 +08:00
lb203 e14c09b3da Update README.md
Former-commit-id: 1436d8feea7f82d3d747c632a68c47b3ea74258f
2024-03-05 22:11:38 +08:00
lb203 3ecbc2eab6 Merge pull request #50 from yunyangge/interpolation
[feat]: frame_interpolation

Former-commit-id: 59ac3fd0b48bcd89f2b1f816b74a4e17c83328b2
2024-03-05 22:10:18 +08:00
yunyangge 21b257c25e [feat]: frame_interpolation
Former-commit-id: 711fc4bba80d5f051f712e3460e2c0685117cf68
2024-03-05 21:53:22 +08:00
lb203 458d1d7770 Update README.md
Former-commit-id: 3e3d674134498512b1ef1b0dd72d6e19afd6ef65
2024-03-05 15:52:05 +08:00
lb203 34c1d8d2d0 Update README.md
Former-commit-id: 164e76cf496b1157eebea299ed58ca6c1799d70d
2024-03-05 15:50:45 +08:00
lb203 192abaeb9e Update README.md
Former-commit-id: be7ecc3d56f7289052fc66d535038e7111f7ac84
2024-03-05 15:46:32 +08:00
lb203 8b2f93325f Update Contribution_Guidelines.md
Former-commit-id: dc87e459ca5a8a090800331111118acb65214f65
2024-03-05 15:01:08 +08:00
lb203 2ca658809a Update README.md
Former-commit-id: 6ecaba8e03d7c7f99c068a57c923f11c5456a8c9
2024-03-05 14:54:38 +08:00
lb203 86332d9c0a Update README.md
Former-commit-id: 1dd9c945be2a170d87d568bb00e1ec30e7be592b
2024-03-05 13:46:09 +08:00
lb203 f26658b7de Merge pull request #42 from mio2333/patch-2
Update requirements.txt

Former-commit-id: 0c76835e4a4681ee317de45a18549dc0d9b9fef9
2024-03-05 13:39:23 +08:00
Junwu Zhang 5a56c37298 reformat code
Former-commit-id: ce23b280fbb419c08347ebcbf1ed4075bdf9af60
2024-03-05 05:38:23 +00:00
Junwu Zhang e01856047d Update README.md
Former-commit-id: 351be764d8648ac60cf170c3fe4e6ed9c544c90e
2024-03-05 13:34:35 +08:00
Junwu Zhang c14ef52685 Rename data.md to Data.md
Former-commit-id: 27484f7a2d8634bd1cd9781a2f10afa7f304f5cb
2024-03-05 13:31:50 +08:00
Junwu Zhang c62e40c802 Update README.md
Former-commit-id: c652acd1fc5a003b4d6eefcfb83dfb310f5b5e81
2024-03-05 13:31:23 +08:00
Junwu Zhang 370be675e8 Create Contribution_Guidelines.md
Former-commit-id: 897dbd63dc6f53095e1dda3f10a86dd40e4bd263
2024-03-05 13:28:32 +08:00
mio e72ef94a7f Update requirements.txt
As in previous step,the three packages have been installed, these packages should be deleted here because it can not be installed by " pip install -r requirements.txt" command and will abort the command

Former-commit-id: 0655fa28cf8bb761f1a3d2def62f44197b257b2c
2024-03-05 13:22:28 +08:00
Junwu Zhang e777c2e617 Update README.md
Former-commit-id: f295f4623448dd4ffa9be747a5cc6733cf48c663
2024-03-05 12:51:07 +08:00
Zhenyu Tang e4eee7bc20 Update README.md
Former-commit-id: 0ff9314bddef1dfbbf429726ed4240b823b221b3
2024-03-04 22:45:06 +08:00
Zhenyu Tang b2dbc2099e Update README.md
Former-commit-id: e27cc5d49a19e9580ba1a29d948051b918aede74
2024-03-04 22:41:37 +08:00
YuanLi 24006388be Update README.md
Former-commit-id: 1cd7d8b5df34ab9f61d9346b2647b643f78a8f5d
2024-03-04 22:41:04 +08:00
Zhenyu Tang cf5bafaee6 Update README.md
Former-commit-id: 5fd1e6afe1277ccffdfeaf507477d944a6a35f03
2024-03-04 22:28:08 +08:00
Zhenyu Tang 028c52fd9f Update README.md
Former-commit-id: fb46c80c1a2663c09849472e93b2da6cd068a98c
2024-03-04 22:24:21 +08:00
Zhenyu Tang b7dfe57c17 Update README.md
Former-commit-id: aee63a3a8422e029f1550e47825ca4beb916874b
2024-03-04 22:23:44 +08:00
Junwu Zhang a02832797a reformat code
Former-commit-id: 7a3dccf2c738eaf535e62996211c8bc5f3de5683
2024-03-04 14:06:39 +00:00
2343 changed files with 11548 additions and 423395 deletions
@@ -1,73 +0,0 @@
---
date: 2026-05-07
experiment: PR #1280 (daVinci-MagiHuman port), distill DiT parity bring-up
category: porting
severity: important
---
# Conversion `--cast-bf16` Needs an FP32-Keep Suffix Allowlist
## What Happened
`scripts/checkpoint_conversion/convert_magi_human_to_diffusers.py --cast-bf16`
produced a converted distill DiT checkpoint that loaded cleanly, ran end-to-
end, and emitted reasonable output — but `test_magi_human_distill_parity`
showed `diff_mean=0.114` against the upstream reference. The base DiT was
bit-exact with the same conversion script. Only the distill variant
regressed.
The error was small enough that visual quality looked normal, but large
enough to fail bit-exact parity. The MagiHuman base + distill DiTs share
most of their architecture, so a difference that affected only distill was
counterintuitive.
## Root Cause
`--cast-bf16` was downcasting **all** fp32 tensors to bf16 indiscriminately.
The base checkpoint and the FastVideo `final_linear` / adapter modules
require eight specific tensors to remain in fp32:
- LayerNorm `gamma` / `beta` weights for the final residual exit
- Adapter projection biases
- A handful of scale parameters in the output projection chain
These tensors participate in chains where bf16 precision causes accumulation
error large enough to drift the parity check. The base DiT happened to not
hit those specific chains in the path the test exercised (different
attention mask shape, different audio interleave); the distill variant did.
## Fix / Workaround
Added `_FP32_KEEP_SUFFIXES` allowlist to
`convert_magi_human_to_diffusers.py` (commit `829f70d3`) and gated `--cast-
bf16` on it. Tensors whose state-dict key ends with any allowlisted suffix
keep their original fp32 dtype regardless of the flag.
Distill DiT parity went from `diff_mean=0.114` (silently wrong) to bit-exact
in one commit.
## Prevention
1. **Treat `--cast-bf16` as opinionated, not blanket.** Any conversion
script that supports a global dtype downcast flag MUST own an explicit
allowlist of fp32-keep tensors, documented at the top of the file.
2. **The `add-model-conversion` skill** should enforce two checks for any
converter that ships a `--cast-bf16`-style flag:
- Run the parity test for **every** variant of the model (base, distill,
SR, etc.), not just the headline variant. Different variants exercise
different code paths.
- Diff the converted checkpoint's dtype map against the upstream
reference and assert the allowlist covers every fp32 tensor in the
reference.
3. **For MagiHuman specifically**: if you add or rename DiT modules that
touch `final_linear`, the adapter, or any LayerNorm in the residual exit
path, **check that any fp32-required tensors are covered by
`_FP32_KEEP_SUFFIXES`** in the conversion script and re-run
`test_magi_human_distill_parity` (it's the canary).
4. The lesson generalizes beyond MagiHuman: any DiT that uses bf16 mixed
precision but keeps specific tensors in fp32 (a common pattern with
flash-attn-style backends) needs this allowlist for any conversion that
downcasts.
@@ -1,77 +0,0 @@
---
date: 2026-05-07
experiment: PR #1280 (daVinci-MagiHuman port), DiT parity bring-up
category: porting
severity: important
---
# DiT Dtype Boundary Alignment with Flash-Attn-Style Backends
## What Happened
DiT bit-exact parity for daVinci-MagiHuman against the upstream reference
sat at `diff_max=0.5` after the architecture port was complete and weight
loading was correct. The error grew with depth (later layers diverged more
than earlier ones), suggesting an accumulating numerical drift rather than
a structural mismatch. None of the obvious culprits (RoPE, GQA expansion,
attention mask handling) accounted for the pattern.
## Root Cause
Four cumulative dtype-boundary mismatches, each individually small but
together pushing parity from `diff_max=0.5` to bit-exact (`diff_max=0.0`):
1. **SDPA inputs were not cast to bf16.** Upstream's `flash_attn_with_cp`
internally casts Q/K/V to bf16 at `dit_module.py:508` before the kernel.
FastVideo was passing fp32 tensors through, getting numerically different
intermediates even though the kernel accepts both.
2. **Post-attention output was kept in bf16 across the per-head gating
multiply.** Upstream upcasts to fp32 before the gating, FastVideo did the
gate in bf16 then upcast.
3. **A residual-stream cast at the block boundary.** FastVideo had a
`.to(bf16)` then `.to(fp32)` at the start of each block. Upstream keeps
the residual stream **continuously in fp32** across all 40 layers; only
the inputs to specific kernels are temporarily downcast.
4. **Parity test scheduler used a double-shift.** A separate per-block fix
(Wave 11 production migration) — single-shift schedule is what upstream
uses; the parity test was double-shifting.
## Fix / Workaround
Four cumulative changes in `fastvideo/models/dits/magi_human.py` (commit
`3a4816cb`), each with a comment at the call site explaining the upstream
parity rationale:
- Cast SDPA inputs to bf16 right before the attention call.
- Upcast attention output to fp32 before the per-head gate multiply.
- Drop the residual-stream `.to(bf16)`/`.to(fp32)` wrapper at the block
boundary; let the residual stay fp32 throughout.
- Single-shift schedule in the parity test fixture (matches upstream Wave 11).
## Prevention
1. **For any DiT port with a flash-attn-style backend**, treat the dtype of
the residual stream as a load-bearing invariant, not a performance knob.
Document it in the model's per-pipeline AGENTS.md. MagiHuman's invariant:
*residual stream stays fp32 across all blocks; only kernel inputs are
temporarily bf16*.
2. **Use layer-by-layer activation hooks** when DiT parity is close-but-not-
bit-exact and the gap grows with depth. The
`fastvideo/hooks/activation_trace.py` infra exists exactly for this case
(`add-model-trace` skill). In MagiHuman's case it would have localized the
first divergence point in one pass.
3. **The `add-model-port-dit` skill** should explicitly call out:
- SDPA input dtype must match the upstream kernel's internal cast.
- Post-attention upcast happens **before** any per-head gate, not after.
- Residual stream dtype across block boundaries is a parity invariant.
These rules apply to any DiT port whose upstream uses a flash-attn-style
backend (`flash_attn_with_cp`, `flex_flash_attn_func`, etc.).
4. **Add an "intermediate-layer parity" test** for new DiT ports — comparing
activations at layer 5, 10, 20, 30 — not just the final output. A growing-
with-depth pattern is otherwise indistinguishable from "almost right".
@@ -1,69 +0,0 @@
---
date: 2026-05-07
experiment: PR #1280 (daVinci-MagiHuman port), Wave 14
category: porting
severity: critical
---
# Silent Channel-Major Token-Packing Bugs
## What Happened
While porting daVinci-MagiHuman (`fastvideo/pipelines/basic/magi_human/`),
the pipeline-parity test passed bit-exactly but the E2E user-visible output
was **pure static noise**. Latent tensors compared identically against the
upstream reference at every checkpointed boundary, yet decoded videos showed
no recognizable content. The discrepancy reproduced on every variant
(base / distill / SR-540p / SR-1080p) with the same noise profile.
## Root Cause
Video tokens were being packed **spatial-major** instead of **channel-major**:
```python
# What we had (spatial-major, WRONG)
einops.rearrange(x, "b c (T pT) (H pH) (W pW) -> b (T H W) (pT pH pW C)", ...)
# What upstream's UnfoldNd produces (channel-major, CORRECT)
einops.rearrange(x, "b c (T pT) (H pH) (W pW) -> b (T H W) (C pT pH pW)", ...)
```
A single-character einops reorder. The pipeline-parity test used FastVideo's
own packer on **both** sides of the comparison, so the bug was invisible there
— both sides agreed on the wrong layout. The DiT consumed those tokens
without complaint because the channel dimension only matters at decode time,
when the VAE's first conv expects channel-major input. By that point the test
boundary was already passed.
The bug was load-bearing for any token-packed format that downstream feeds
into a `UnfoldNd`-shaped consumer. Wave 14 of the port took multiple bug-hunt
iterations and an Oracle consultation to localize.
## Fix / Workaround
Single-character einops change in `stages/latent_preparation.py:_img2tokens`
(commit `6d190693` of the original PR). After the fix, all four variants
produced expected E2E output and the pipeline-parity tests still passed
because both sides of the parity check are now correct.
## Prevention
1. **Never use the FastVideo-side packer on both sides of a parity test.**
At least one parity boundary must compare against an upstream tensor
produced by the upstream packer. For MagiHuman this means a separate
`_img2tokens` parity test that feeds upstream `UnfoldNd` output as the
reference, not FastVideo's reformatted equivalent.
2. **Add an E2E hash check** alongside latent-parity. The mp4 SHA was the
first signal that something was wrong; if it had been part of the standard
parity battery, the bug would have surfaced in Wave 1, not Wave 14. See
`fastvideo/tests/ssim/test_magi_human_similarity.py` for the CI version.
3. **For any new model port that involves explicit tensor reshaping into
tokens**, document the expected packing order (`(C pT pH pW)` vs
`(pT pH pW C)`) at the call site and assert the layout matches the
downstream consumer's expectation.
4. The `add-model-port-dit` skill's parity gate should require an E2E hash
check for any DiT that does video token packing, not just latent
bit-exactness.
@@ -1,41 +0,0 @@
---
date: 2026-05-22
experiment: PR #1386 DreamVerse app CI backend tests
category: infrastructure
severity: important
---
# DreamVerse App CI Streaming Imports Need GPU
## What Happened
DreamVerse app CI backend pytest collection imports FastVideo streaming surfaces.
When those tests run in a CPU-only Modal environment, collection can fail before
any app assertions run with Triton reporting:
```text
RuntimeError: 0 active drivers
```
## Root Cause
Some streaming import paths can import `fastvideo_kernel` at module import time.
Triton then probes for an active GPU driver during pytest collection. A CPU-only
Modal container has no active driver, so the failure appears as an import-time
collection error rather than a DreamVerse app behavior failure.
## Fix / Workaround
For PR #1386, use a surgical CI fix: allocate a GPU to
`run_dreamverse_app_tests`. Do not refactor core streaming/kernel imports just to
unstick this app CI path.
Keep `build_kernel=False` for this job. The DreamVerse app backend test imports
streaming surfaces but does not need to rebuild or exercise custom kernels.
## Prevention
When adding or modifying DreamVerse app CI jobs that import FastVideo streaming
modules, make the GPU requirement explicit if the import graph may touch
`fastvideo_kernel`. Prefer small CI resource fixes for app test collection issues
unless the product code genuinely requires lazy import cleanup.
-48
View File
@@ -1,48 +0,0 @@
# Lessons Learned Database
This directory stores documented mistakes, unexpected behaviors, and their fixes.
Each lesson is a permanent record that helps agents and humans avoid repeating
past errors.
## When to Create a Lesson
- An experiment failed for a non-obvious reason.
- A configuration or hyperparameter choice led to wasted compute.
- A porting, data, or infrastructure issue was discovered and resolved.
- A workaround was needed for a known framework/library bug.
## File Naming
`<YYYY-MM-DD>_<short-slug>.md` — e.g., `2026-03-02_lr-too-high-for-lora.md`
## Template
```markdown
---
date: <ISO-8601>
experiment: <reference to experiment_journal.md entry, if applicable>
category: hyperparameter | data | infrastructure | evaluation | porting | other
severity: critical | important | minor
---
# <Short Descriptive Title>
## What Happened
<Description of the problem and its symptoms.>
## Root Cause
<Analysis of why it happened.>
## Fix / Workaround
<What resolved the issue.>
## Prevention
<How to avoid this in the future — updated skills, SOPs, or checks.>
```
## Usage
- Before starting a task, **search this directory** for relevant lessons.
- After completing or failing a task, **check if a new lesson should be created**.
- Periodically review lessons for **patterns** — recurring themes may warrant
a new skill, SOP, or codebase fix.
-96
View File
@@ -1,96 +0,0 @@
#!/usr/bin/env bash
# Sync .agents/skills/ into .claude/skills/ via per-skill symlinks.
#
# Why: Claude Code only scans .claude/skills/ and ~/.claude/skills/ for
# user-invocable skills (no skillsPath config exists — see
# https://code.claude.com/docs/en/skills.md). This repo's skills live
# in .agents/skills/ so they travel with the repo and stay under git.
# Run this once after cloning (or after adding/removing a skill) to
# expose them to Claude Code without maintaining a parallel tree.
#
# Usage:
# .agents/scripts/sync-skills.sh
#
# Idempotent and safe to re-run. Prunes stale symlinks whose source
# has been removed from .agents/skills/. Leaves hand-written
# .claude/skills/<name>/ directories untouched (only symlinks are
# managed).
set -euo pipefail
REPO_ROOT="$(git -C "$(dirname "$0")" rev-parse --show-toplevel)"
SRC_DIR="$REPO_ROOT/.agents/skills"
DST_DIR="$REPO_ROOT/.claude/skills"
if [[ ! -d "$SRC_DIR" ]]; then
echo "Error: $SRC_DIR does not exist." >&2
exit 1
fi
mkdir -p "$DST_DIR"
linked=0
unchanged=0
skipped=0
pruned=0
link_skill() {
local name="$1"
local src="$SRC_DIR/$name"
local dst="$DST_DIR/$name"
# Relative target keeps symlinks portable across clones.
local rel="../../.agents/skills/$name"
if [[ -L "$dst" ]]; then
if [[ "$(readlink "$dst")" == "$rel" ]]; then
unchanged=$((unchanged + 1))
return
fi
rm "$dst"
elif [[ -e "$dst" ]]; then
echo "Skipped (not a symlink): .claude/skills/$name" >&2
skipped=$((skipped + 1))
return
fi
ln -s "$rel" "$dst"
echo "Linked: .claude/skills/$name -> $rel"
linked=$((linked + 1))
}
prune_stale() {
local link="$1"
local target
target="$(readlink "$link")"
case "$target" in
../../.agents/skills/*) ;;
*) return ;;
esac
local name="${target##*/}"
if [[ ! -d "$SRC_DIR/$name" ]]; then
rm "$link"
echo "Pruned stale: .claude/skills/$(basename "$link")"
pruned=$((pruned + 1))
fi
}
for src in "$SRC_DIR"/*/; do
[[ -d "$src" ]] || continue
name="$(basename "$src")"
# Only treat directories that actually contain a SKILL.md as skills.
[[ -f "$src/SKILL.md" ]] || continue
link_skill "$name"
done
shopt -s nullglob
for link in "$DST_DIR"/*; do
[[ -L "$link" ]] || continue
prune_stale "$link"
done
shopt -u nullglob
printf "\nSummary: %d linked, %d unchanged, %d pruned" "$linked" "$unchanged" "$pruned"
if [[ "$skipped" -gt 0 ]]; then
printf ", %d skipped (non-symlink collision)" "$skipped"
fi
printf "\n"
-55
View File
@@ -1,55 +0,0 @@
---
name: <skill-name>
description: <one-line description — Codex uses this for implicit invocation matching>
---
# <Skill Name>
## Purpose
<Why this skill exists and when to use it.>
## Prerequisites
- <What must be true before using this skill>
## Inputs
| Parameter | Required | Description |
|-----------|----------|-------------|
| `param1` | Yes | ... |
## Steps
1. **Step 1 title**
- Detail...
2. **Step 2 title**
- Detail...
## Outputs
- <What this skill produces>
## Example Usage
```
<Example invocation or prompt snippet>
```
## References
- <Links to relevant files in the codebase>
---
## Folder Structure
Each skill lives in its own directory under `.agents/skills/`:
```
.agents/skills/<skill-name>/
├── SKILL.md # Required: instructions + metadata (this file)
├── scripts/ # Optional: executable helper scripts
├── references/ # Optional: documentation, papers
└── assets/ # Optional: templates, resources
```
Skill discovery is directory-based; no hand-maintained registry entry is
required. Run `.agents/scripts/sync-skills.sh` if a local Claude Code checkout
needs refreshed `.claude/skills/` symlinks.
-173
View File
@@ -1,173 +0,0 @@
---
name: add-model-01-prep
description: Use at the start of a FastVideo model port to gather required inputs, inspect/download HF weights, clone and install the official reference repo in the current environment, create a local_tests README skeleton, and produce a handoff before conversion or implementation.
---
# Add Model Prep
## Goal
Prepare external assets and the shared parity-test environment for a FastVideo
model port. Stop before writing conversion scripts, model components, pipeline
code, registry entries, or executable parity tests.
## Ask First
Ask once, then proceed if the HF token is already exported:
```text
Before prep: (1) official reference repo or Diffusers pipeline URL, (2) HF repo
id or local weights path and whether it has a root model_index.json, (3) target
model_family, (4) workload types, (5) which token env var is exported:
HF_TOKEN, HUGGINGFACE_HUB_TOKEN, or HF_API_KEY, (6) may I stage clone and
weights under the FastVideo repo root, and (7) may I install official reference
dependencies into the current FastVideo conda/env for parity tests?
```
Useful optional inputs: `pipeline_class`, `reference_dir`, `hf_revision`,
`official_revision`, `reuse_hints`, `download_scope`.
## Rules
- Follow `../add-model/shared/common_rules.md` for token/auth safety, state files,
escape hatches, and skip/pass semantics.
- Run from the FastVideo repo root.
- Use repo-relative defaults: `<ReferenceDir>/`,
`official_weights/<model_family>/`, `converted_weights/<model_family>/`.
- Install official reference deps into the current FastVideo environment, not a
new venv/conda env, so parity tests run both implementations with one shared
numeric stack.
- If the reference is a Diffusers class/package instead of a cloneable repo,
record import path and version instead of cloning.
- Prep may create only the local-test README and `PORT_STATUS.md` skeletons;
executable `.py` parity tests belong to `../add-model-02-parity/SKILL.md`.
## Escape Hatches
Follow `../add-model/shared/common_rules.md`. Prep-specific ask cases include
overwriting an existing clone or weight directory, installing untrusted/private
deps, choosing between incompatible official references, large downloads outside
the agreed scope, or missing gated-repo auth setup by env var name.
## Workflow
1. Verify the repo:
```bash
git rev-parse --show-toplevel
```
Expected markers: `fastvideo/`, `scripts/checkpoint_conversion/`,
`scripts/huggingface/download_hf.py`, `fastvideo/registry.py`.
2. Inspect HF or local weight layout:
```bash
python ".agents/skills/add-model-01-prep/scripts/inspect_hf_layout.py" \
"Org/Model" \
--revision "<revision>" \
--json
```
For a local path, replace `Org/Model` with `/path/to/weights`. Record
`source_layout`, `needs_conversion`, `model_index_class`, and
`components_seen`.
3. Download HF weights if needed:
```bash
python ".agents/skills/add-model-01-prep/scripts/download_hf_weights.py" \
"Org/Model" \
"official_weights/<model_family>" \
--revision "<revision>"
```
For selected files, repeat `--file-name`. For partial snapshots, repeat
`--allow-pattern` or `--ignore-pattern`. If the user provided a local path,
record it instead of copying large weights by default.
4. Clone the official reference repo if applicable:
```bash
python ".agents/skills/add-model-01-prep/scripts/clone_reference_repo.py" \
"<official_repo_url>" \
"<ReferenceDir>" \
--branch "<tag-or-branch>" \
--commit "<commit-sha>" \
--update-gitignore
```
Omit `--branch`, `--commit`, or `--update-gitignore` when not needed. The
helper refuses to overwrite existing paths and prints remote/HEAD instead.
5. Keep prep assets ignored. Ensure `.gitignore` includes relevant entries:
```gitignore
/<ReferenceDir>/
/official_weights/
/converted_weights/
```
6. Follow the official repo's setup instructions in the current environment.
Inspect dependency files and README install docs before installing anything:
- `README*`, install docs, or model-card instructions.
- `requirements*.txt`, `pyproject.toml`, `setup.py`, `environment.yml`.
Use the current FastVideo conda/env. Do not create a new env even if upstream
docs recommend one; translate the needed install commands into the active env.
Prefer editable/no-deps first so the official source is importable without
changing shared pins:
```bash
uv pip install --no-deps -e ./<ReferenceDir>
```
Then install only missing official deps needed for parity imports. Stop before
installing requirements that would change FastVideo's core stack. If upstream
requires private/non-PyPI deps, record that parity needs a local stub helper
rather than pretending setup is complete.
7. Create the model-family local test skeleton and top-level port state file:
```bash
mkdir -p tests/local_tests/<model_family>
cp ".agents/skills/add-model-01-prep/templates/local_tests_readme.md" \
tests/local_tests/<model_family>/README.md
cp ".agents/skills/add-model-01-prep/templates/port_status.md" \
tests/local_tests/<model_family>/PORT_STATUS.md
```
Edit every placeholder in the README and `PORT_STATUS.md`. The README gives
later review agents enough information to reproduce the shared environment and
run/review parity work:
- official code URL or import path, local clone path, and commit/version;
- HF URL or local weight path, revision, access notes, and token env var name
only;
- commands already run and any blocked official dependency installs;
- shared-env install commands to re-run without changing core pins;
- expected local parity test paths and pytest commands;
- private-dependency stubs or known setup gaps;
- PR/review notes explaining which parity tests are required before handoff.
Do not include raw tokens, absolute cache paths that are not repo-reproducible,
or large generated outputs. If prep is blocked before imports work, still create
the README with `official_env_status=blocked` and the exact blocker.
`PORT_STATUS.md` must follow `../add-model/contracts/port_state.md`. Record open
questions and prep issues immediately, using stable IDs such as `Q001` and
`I001`. Keep resolved questions/issues in the table with a resolution instead of
deleting them.
## Handoff
End with the canonical prep handoff contract from
`../add-model/contracts/prep_handoff.md` and update the shared state files before
handoff.
## Helper Scripts
- `scripts/inspect_hf_layout.py`: classify HF/local layout.
- `scripts/download_hf_weights.py`: download HF snapshot or selected files.
- `scripts/clone_reference_repo.py`: clone reference repo safely.
@@ -1,119 +0,0 @@
#!/usr/bin/env python3
"""Clone an official reference repo without overwriting existing paths."""
from __future__ import annotations
import argparse
import subprocess
import sys
from pathlib import Path
def parse_args() -> argparse.Namespace:
parser = argparse.ArgumentParser(description="Clone a reference repo for FastVideo parity tests.")
parser.add_argument("repo_url", help="Official reference repository URL")
parser.add_argument("target_dir", help="Directory to clone into")
parser.add_argument("--branch", help="Branch or tag to clone")
parser.add_argument("--commit", help="Commit SHA to check out after clone")
parser.add_argument(
"--update-gitignore",
action="store_true",
help="Add the target directory to .gitignore if missing",
)
parser.add_argument(
"--gitignore",
default=".gitignore",
help="Path to gitignore file when --update-gitignore is used",
)
return parser.parse_args()
def run(command: list[str], check: bool = True) -> subprocess.CompletedProcess[str]:
return subprocess.run(
command,
check=check,
text=True,
capture_output=True,
)
def print_existing_repo_info(target: Path) -> int:
print(f"target_exists: {target}")
if not (target / ".git").exists():
print("error: target exists but is not a git repo", file=sys.stderr)
return 1
remote = run(["git", "-C", str(target), "remote", "-v"], check=False)
head = run(["git", "-C", str(target), "rev-parse", "HEAD"], check=False)
if remote.stdout:
print("remote_v:")
print(remote.stdout.rstrip())
if head.stdout:
print(f"head: {head.stdout.strip()}")
print("not_overwritten: true")
return 0
def gitignore_entry_for(target: Path) -> str:
root = Path.cwd().resolve()
resolved = target.resolve()
try:
relative = resolved.relative_to(root)
except ValueError as exc:
raise ValueError("--update-gitignore requires target_dir to be under the current directory") from exc
text = relative.as_posix().rstrip("/")
return "/" + text + "/"
def update_gitignore(path: Path, target: Path) -> bool:
entry = gitignore_entry_for(target)
existing = path.read_text().splitlines() if path.exists() else []
if entry in existing:
return False
new_text = "\n".join(existing).rstrip("\n")
if new_text:
new_text += "\n"
new_text += entry + "\n"
path.write_text(new_text)
return True
def main() -> int:
args = parse_args()
target = Path(args.target_dir)
if target.exists():
return print_existing_repo_info(target)
command = ["git", "clone", "--depth", "1"]
if args.branch:
command.extend(["--branch", args.branch])
command.extend([args.repo_url, str(target)])
try:
run(command)
if args.commit:
run(["git", "-C", str(target), "fetch", "--depth", "1", "origin", args.commit])
run(["git", "-C", str(target), "checkout", args.commit])
except subprocess.CalledProcessError as exc:
if exc.stdout:
print(exc.stdout, end="")
if exc.stderr:
print(exc.stderr, end="", file=sys.stderr)
return exc.returncode
head = run(["git", "-C", str(target), "rev-parse", "HEAD"])
print(f"cloned: {target}")
print(f"head: {head.stdout.strip()}")
if args.update_gitignore:
changed = update_gitignore(Path(args.gitignore), target)
print(f"gitignore_updated: {str(changed).lower()}")
return 0
if __name__ == "__main__":
raise SystemExit(main())
@@ -1,103 +0,0 @@
#!/usr/bin/env python3
"""Download HF weights using the standard FastVideo token env vars."""
from __future__ import annotations
import argparse
import os
import sys
from pathlib import Path
HF_TOKEN_ENV_KEYS = ("HF_TOKEN", "HUGGINGFACE_HUB_TOKEN", "HF_API_KEY")
def parse_args() -> argparse.Namespace:
parser = argparse.ArgumentParser(
description="Download a HF model snapshot or selected files into a local directory.")
parser.add_argument("repo_id", help="HF repo id, for example Org/Model")
parser.add_argument("local_dir", help="Destination directory")
parser.add_argument("--repo-type", default="model", help="HF repo type (default: model)")
parser.add_argument("--revision", help="HF branch, tag, or commit")
parser.add_argument(
"--file-name",
action="append",
default=[],
help="Download one file; may be repeated. If omitted, download full snapshot.",
)
parser.add_argument(
"--allow-pattern",
action="append",
default=[],
help="Snapshot allow pattern; may be repeated. Ignored when --file-name is used.",
)
parser.add_argument(
"--ignore-pattern",
action="append",
default=[],
help="Snapshot ignore pattern; may be repeated. Ignored when --file-name is used.",
)
return parser.parse_args()
def resolve_token() -> tuple[str | None, str | None]:
for key in HF_TOKEN_ENV_KEYS:
value = os.environ.get(key)
if value:
return key, value
return None, None
def main() -> int:
args = parse_args()
token_env, token = resolve_token()
local_dir = Path(args.local_dir).expanduser()
if token_env:
print(f"token_env: {token_env}")
else:
print("token_env: none", file=sys.stderr)
try:
if local_dir.exists() and not local_dir.is_dir():
print(
f"error: destination exists and is not a directory: {local_dir}",
file=sys.stderr,
)
return 1
local_dir.mkdir(parents=True, exist_ok=True)
if args.file_name:
from huggingface_hub import hf_hub_download
for file_name in args.file_name:
path = hf_hub_download(
repo_id=args.repo_id,
filename=file_name,
repo_type=args.repo_type,
revision=args.revision,
local_dir=str(local_dir),
token=token,
)
print(f"downloaded_file: {path}")
else:
from huggingface_hub import snapshot_download
path = snapshot_download(
repo_id=args.repo_id,
repo_type=args.repo_type,
revision=args.revision,
local_dir=str(local_dir),
token=token,
allow_patterns=args.allow_pattern or None,
ignore_patterns=args.ignore_pattern or None,
)
print(f"downloaded_snapshot: {path}")
except Exception as exc: # noqa: BLE001 - CLI should print concise failures.
print(f"error: {exc}", file=sys.stderr)
return 1
print(f"local_dir: {local_dir.resolve()}")
return 0
if __name__ == "__main__":
raise SystemExit(main())
@@ -1,260 +0,0 @@
#!/usr/bin/env python3
"""Inspect a Hugging Face repo or local weight directory layout."""
from __future__ import annotations
import argparse
import json
import os
import sys
from pathlib import Path
from typing import Any
HF_TOKEN_ENV_KEYS = ("HF_TOKEN", "HUGGINGFACE_HUB_TOKEN", "HF_API_KEY")
RAW_WEIGHT_SUFFIXES = (".safetensors", ".pt", ".pth", ".ckpt", ".bin")
KNOWN_COMPONENTS = {
"audio_vae",
"conditioner",
"feature_extractor",
"image_encoder",
"scheduler",
"text_encoder",
"text_encoder_2",
"tokenizer",
"tokenizer_2",
"transformer",
"transformer_2",
"unet",
"upsampler",
"vae",
"vocoder",
}
def parse_args() -> argparse.Namespace:
parser = argparse.ArgumentParser(
description="Classify a HF repo or local directory as Diffusers, raw, custom, or unknown.")
parser.add_argument("source", help="HF repo id or local weights directory")
parser.add_argument("--repo-type", default="model", help="HF repo type (default: model)")
parser.add_argument("--revision", help="HF revision to inspect")
parser.add_argument(
"--max-local-files",
type=int,
default=20000,
help="Maximum local files to scan recursively (default: 20000)",
)
parser.add_argument(
"--sample-limit",
type=int,
default=80,
help="Number of file paths to print in human output (default: 80)",
)
parser.add_argument("--json", action="store_true", help="Emit JSON only")
return parser.parse_args()
def resolve_token() -> tuple[str | None, str | None]:
for key in HF_TOKEN_ENV_KEYS:
value = os.environ.get(key)
if value:
return key, value
return None, None
def load_local_files(root: Path, max_files: int) -> tuple[list[str], bool]:
files: list[str] = []
truncated = False
for path in root.rglob("*"):
if not path.is_file():
continue
files.append(path.relative_to(root).as_posix())
if len(files) >= max_files:
truncated = True
break
return sorted(files), truncated
def load_local_model_index(root: Path) -> tuple[dict[str, Any] | None, str | None]:
index_path = root / "model_index.json"
if not index_path.is_file():
return None, None
try:
return json.loads(index_path.read_text()), None
except Exception as exc: # noqa: BLE001 - surface malformed JSON clearly.
return None, f"failed to parse local model_index.json: {exc}"
def load_remote_files(
repo_id: str,
repo_type: str,
revision: str | None,
token: str | None,
) -> list[str]:
from huggingface_hub import list_repo_files
return sorted(list_repo_files(
repo_id,
repo_type=repo_type,
revision=revision,
token=token,
))
def load_remote_model_index(
repo_id: str,
repo_type: str,
revision: str | None,
token: str | None,
) -> tuple[dict[str, Any] | None, str | None]:
from huggingface_hub import hf_hub_download
try:
path = hf_hub_download(
repo_id=repo_id,
filename="model_index.json",
repo_type=repo_type,
revision=revision,
token=token,
)
except Exception as exc: # noqa: BLE001 - missing/inaccessible file is data.
return None, f"failed to download model_index.json: {exc}"
try:
return json.loads(Path(path).read_text()), None
except Exception as exc: # noqa: BLE001 - surface malformed JSON clearly.
return None, f"failed to parse remote model_index.json: {exc}"
def root_file_names(files: list[str]) -> set[str]:
return {name for name in files if "/" not in name}
def component_names(files: list[str], model_index: dict[str, Any] | None) -> list[str]:
components: set[str] = set()
for name in files:
parts = name.split("/", 1)
if len(parts) != 2:
continue
top, rest = parts
if top in KNOWN_COMPONENTS or rest == "config.json":
components.add(top)
if model_index:
for key, value in model_index.items():
if key.startswith("_"):
continue
if isinstance(value, list) and len(value) == 2:
components.add(key)
return sorted(components)
def classify_layout(
files: list[str],
model_index: dict[str, Any] | None,
components: list[str],
) -> tuple[str, str]:
roots = root_file_names(files)
raw_weight_files = [name for name in roots if name.endswith(RAW_WEIGHT_SUFFIXES)]
has_model_index = "model_index.json" in roots or model_index is not None
if has_model_index and components:
return "diffusers", "no"
if has_model_index:
return "custom", "unknown"
if raw_weight_files:
return "raw_official", "yes"
if any(name.endswith(RAW_WEIGHT_SUFFIXES) for name in files):
return "custom", "yes"
return "unknown", "unknown"
def build_result(args: argparse.Namespace) -> dict[str, Any]:
token_env, token = resolve_token()
source_path = Path(args.source).expanduser()
is_local = source_path.exists()
if is_local:
root = source_path.resolve()
if not root.is_dir():
raise ValueError(f"local source is not a directory: {root}")
files, truncated = load_local_files(root, args.max_local_files)
model_index, model_index_error = load_local_model_index(root)
source_kind = "local"
source = str(root)
else:
files = load_remote_files(args.source, args.repo_type, args.revision, token)
truncated = False
model_index, model_index_error = load_remote_model_index(
args.source,
args.repo_type,
args.revision,
token,
)
source_kind = "hf"
source = args.source
components = component_names(files, model_index)
source_layout, needs_conversion = classify_layout(files, model_index, components)
return {
"source": source,
"source_kind": source_kind,
"repo_type": None if is_local else args.repo_type,
"revision": args.revision,
"token_env": token_env,
"source_layout": source_layout,
"needs_conversion": needs_conversion,
"model_index_class": (model_index or {}).get("_class_name"),
"model_index_diffusers_version": (model_index or {}).get("_diffusers_version"),
"model_index_error": model_index_error,
"components_seen": components,
"file_count": len(files),
"file_scan_truncated": truncated,
"files_sample": files[:args.sample_limit],
}
def print_human(result: dict[str, Any]) -> None:
for key in (
"source",
"source_kind",
"repo_type",
"revision",
"token_env",
"source_layout",
"needs_conversion",
"model_index_class",
"model_index_diffusers_version",
"model_index_error",
"file_count",
"file_scan_truncated",
):
value = result.get(key)
if value is not None:
print(f"{key}: {value}")
components = result["components_seen"]
print("components_seen: " + (", ".join(components) if components else "none"))
print("files_sample:")
for name in result["files_sample"]:
print(f" {name}")
def main() -> int:
args = parse_args()
try:
result = build_result(args)
except Exception as exc: # noqa: BLE001 - CLI should print concise failures.
print(f"error: {exc}", file=sys.stderr)
return 1
if args.json:
print(json.dumps(result, indent=2, sort_keys=True))
else:
print_human(result)
return 0
if __name__ == "__main__":
raise SystemExit(main())
@@ -1,122 +0,0 @@
# <Model Family> Local Tests
Local-only parity and smoke tests for the `<model_family>` FastVideo port. These
tests compare FastVideo against the official reference implementation and are
not expected to run in CI unless explicitly promoted later.
Port progress, open questions, issues, and handoff notes live in
`tests/local_tests/<model_family>/PORT_STATUS.md`.
## Reference Assets
| Field | Value |
|---|---|
| Model family | `<model_family>` |
| Workload types | `<T2V/I2V/V2V/T2I/or compatibility shim with rationale>` |
| Official reference | `<url or import path>` |
| Local reference dir | `<ReferenceDir or none>` |
| Official commit/version | `<sha, tag, package version, or unknown>` |
| HF weights | `<HF repo id/url or local path>` |
| HF revision | `<revision or default>` |
| Local weights dir | `<official_weights/model_family or local path>` |
| Source layout | `<diffusers/raw_official/monolithic/separate_components/mixed/custom/unknown>` |
| Needs conversion | `<yes/no/unknown>` |
Do not write token values in this file. Use only the token env var name:
`<HF_TOKEN or HUGGINGFACE_HUB_TOKEN or HF_API_KEY>`.
## Shared Environment Setup
Run from the FastVideo repo root in the same conda/env used for FastVideo.
Do not create a separate upstream environment for parity tests.
```bash
# Official reference source, if cloneable.
python ".agents/skills/add-model-01-prep/scripts/clone_reference_repo.py" \
"<official_repo_url>" \
"<ReferenceDir>" \
--commit "<commit-sha>" \
--update-gitignore
# Editable install without changing shared core pins.
uv pip install --no-deps -e ./<ReferenceDir>
# Additional official deps installed or required for imports:
# <package list or none>
```
Do not change core dependency versions (`torch`, `diffusers`, `transformers`,
`flash-attn`, `triton`, CUDA packages) without explicit approval.
## Official Environment Status
```text
dependency_changes: <none | installed no-deps editable | installed official deps in current env | blocked on user>
official_env_status: <imports_ok | private_deps_need_stubs | blocked>
private_dep_stubs: <none or tests/local_tests/helpers/<model_family>_upstream.py>
blocked_on: <none or exact blocker>
```
## Weight Setup
```bash
python ".agents/skills/add-model-01-prep/scripts/download_hf_weights.py" \
"<Org/Model>" \
"official_weights/<model_family>" \
--revision "<revision>"
```
If weights are local-only, record the local path and do not copy large files into
the repository.
## Prototype And Conversion Artifacts
State-dict key/shape dumps are generated after FastVideo native prototypes exist
and are used to build the conversion mapping.
```text
official_key_dumps:
<component>: converted_weights/<model_family>/_mapping/<component>_official_keys.json
fastvideo_key_dumps:
<component>: converted_weights/<model_family>/_mapping/<component>_fastvideo_keys.json
conversion_script: scripts/checkpoint_conversion/<model_family>_to_diffusers.py
conversion_source_layout: <diffusers | separate_components | monolithic | mixed | custom>
converted_weights_dir: converted_weights/<model_family>
strict_load_status: <not_run | pass | pass_with_documented_exclusions | blocked>
```
For monolithic official checkpoints, record the component prefix split here. For
example, a single checkpoint may contain transformer, VAE/pretransform,
conditioner, and scheduler/vocoder keys that the conversion script writes into
separate FastVideo component subfolders.
## Expected Parity Tests
Planned local tests for this family:
| Component | Official files / args | Test | Concerns | Status |
|---|---|---|---|---|
| `<component>` | `<definition path; instantiation path + args>` | `tests/local_tests/<bucket>/test_<model_family>_<component>_parity.py` | `<prototype or setup concerns>` | `<planned/scaffold_skip/debug_red/non_skip_pass/blocked>` |
| `pipeline` | `<official pipeline call>` | `tests/local_tests/pipelines/test_<model_family>_pipeline_parity.py` | `<pipeline concerns>` | `<planned/scaffold_skip/debug_red/non_skip_pass/blocked>` |
Include reused components in this table. Reuse is accepted only after the
FastVideo component definition and official instantiation arguments have both
been checked and the component parity test passes non-skip.
Run the relevant tests with:
```bash
pytest tests/local_tests/<bucket>/test_<model_family>_<component>_parity.py -v -s
pytest tests/local_tests/pipelines/test_<model_family>_pipeline_parity.py -v -s
```
## Review Notes
- Required before handoff: non-skip PASS for each required component parity
test, including reused components that own weights or numerical behavior.
- Pipeline parity may start as a scaffold, but final handoff requires non-skip
PASS or an explicit blocker accepted through the escape-hatch process.
- User decisions and pause points are tracked as `E###` rows in
`PORT_STATUS.md`; do not rely on chat history for escape-hatch context.
- Review agents should verify this README's setup commands still match the PR,
then run the listed parity tests or report the exact blocker.
@@ -1,69 +0,0 @@
# <Model Family> Port Status
## Summary
- model_family: `<model_family>`
- workload_types: `<T2V/I2V/V2V/T2I/or compatibility shim with rationale>`
- official_ref: `<url or import path>`
- official_ref_dir: `<ReferenceDir or none>`
- hf_weights_path: `<HF repo id/url or local path>`
- local_weights_dir: `<official_weights/model_family or local path>`
- source_layout: `<diffusers/raw_official/monolithic/separate_components/mixed/custom/unknown>`
- local_tests_readme: `tests/local_tests/<model_family>/README.md`
## Current Phase
- phase: `prep`
- status: `in_progress`
- owner: `prep`
- last_updated: `<YYYY-MM-DD>`
## Component Matrix
| Component | Type | Reuse/Port | Official Definition | Official Instantiation | FastVideo Target | Prototype | Conversion | Parity | Open Issues |
|---|---|---|---|---|---|---|---|---|---|
| `<component>` | `<dit/vae/encoder/generic>` | `<unknown/reuse/port>` | `<path + symbols>` | `<path + args>` | `<target files>` | `<not_started/in_progress/pass/blocked>` | `<not_started/pass/blocked>` | `<not_started/scaffold_skip/debug_red/non_skip_pass/blocked>` | `<none or IDs>` |
## Conversion State
- conversion_script: `scripts/checkpoint_conversion/<model_family>_to_diffusers.py`
- converted_weights_dir: `converted_weights/<model_family>`
- source_layout: `<diffusers/separate_components/monolithic/mixed/custom/unknown>`
- strict_load_status: `not_run`
- passthrough_components: `<none or list>`
- retry_history: `<none>`
## Parity Commands
| Scope | Command | Last Result | Notes |
|---|---|---|---|
| component | `pytest tests/local_tests/<bucket>/test_<model_family>_<component>_parity.py -v -s` | `not_run` | `<notes>` |
| pipeline | `pytest tests/local_tests/pipelines/test_<model_family>_pipeline_parity.py -v -s` | `not_run` | `<notes>` |
## Open Questions
| ID | Question | Owner | Needed By Phase | Status | Resolution |
|---|---|---|---|---|---|
| Q001 | `<question>` | `<owner>` | `<phase>` | `<open/resolved>` | `<resolution or blank>` |
## Issues And Blockers
| ID | Phase | Component | Severity | Issue | Evidence | Owner | Status | Resolution |
|---|---|---|---|---|---|---|---|---|
| I001 | `<phase>` | `<component or all>` | `<low/medium/high/blocker>` | `<issue>` | `<logs/paths/commands>` | `<owner>` | `<open/resolved>` | `<resolution or blank>` |
## Escape Hatches
| ID | Phase | Decision Type | Question | Recommended Option | Status | Resolution |
|---|---|---|---|---|---|---|
| E001 | `<phase>` | `<scope/dependency/auth/cost/destructive/ambiguity/blocker>` | `<one precise question>` | `<safe recommended option>` | `<open/resolved>` | `<resolution or blank>` |
## Decisions
| Date | Decision | Rationale | Impact |
|---|---|---|---|
| `<YYYY-MM-DD>` | `<decision>` | `<why>` | `<affected components/phases>` |
## Handoff Notes
- `<short notes for the next agent>`
-234
View File
@@ -1,234 +0,0 @@
---
name: add-model-02-parity
description: Use during /add-model after reference/architecture study to scaffold and later activate local FastVideo component parity tests. Emphasizes early test creation, official-reference loading, standardized FastVideo loading, and non-skip handoff gates.
---
# Add Model Parity
## Goal
Create parity tests as early as possible in a FastVideo port. The first pass can
land before conversion or component implementation as an executable scaffold;
handoff is blocked until the same tests become non-skip PASS with real weights.
## When To Run
Follow `../add-model/shared/common_rules.md` for token/auth safety, state files,
escape hatches, and skip/pass semantics.
Run immediately after `/add-model` Phase 1 has identified:
- official component classes and call signatures;
- FastVideo target component buckets/classes/configs;
- local reference clone or import path from `add-model-01-prep`;
- local raw or Diffusers weight path;
- `official_env_status=imports_ok`, or private deps that will be stubbed
locally in tests;
- `local_tests_readme` documenting setup and planned review/test commands;
- expected component inputs and output tensors.
Do not wait for all FastVideo components to be implemented. Write the tests
first, then let component-porting subagents make them pass.
## Outputs
- One component parity test per required component, including reused components:
`tests/local_tests/<bucket>/test_<family>_<component>_parity.py`.
- Optional helper for upstream private deps:
`tests/local_tests/helpers/<family>_upstream.py`.
- Pipeline parity is owned later by `../add-model-09-pipeline/SKILL.md` after all
component parity tests pass non-skip.
- A parity status block for the `/add-model` parity verification phase.
## Early Scaffold Rules
- A scaffold may skip while the FastVideo class, converted weights, or official
import is missing.
- A scaffold must already encode the real official load path, FastVideo load
path, deterministic inputs, expected output extraction, and tolerance target.
- Each parity test must declare its coverage scope in the file docstring or a
module constant: `production_loader`, `implementation_subcomponent`, or `both`.
Implementation/subcomponent parity may bypass production loaders deliberately,
but final handoff still needs production-loader coverage somewhere before the
pipeline depends on that component.
- Official reference imports must run in the current FastVideo environment; do
not create or assume a separate upstream venv/conda env.
- A scaffold is not evidence of correctness. It becomes evidence only after a
local non-skip PASS.
- Prefer env-var path overrides with repo-relative defaults.
- Keep tests local-only under `tests/local_tests/`; package/CI quality tests are
added later.
- Update shared state files as described in
`../add-model/shared/common_rules.md` whenever adding or activating parity
tests.
## Component Template
Copy `templates/component_parity_test.py` and fill every `TODO` marker. The
template is distilled from:
- `tests/local_tests/transformers/test_ltx2.py`
- `tests/local_tests/transformers/test_gamecraft_parity.py`
- `tests/local_tests/encoders/test_ltx2_gemma_parity.py`
- `tests/local_tests/vaes/test_oobleck_vae_parity.py`
- `tests/local_tests/sd35/test_sd35_component_parity.py`
The template supports three states:
| State | Meaning |
|---|---|
| Scaffold skip | Test is committed early, but official import, FastVideo class, or weights are not available yet. |
| Debug red | Both sides load and the test fails numerically. This is useful: porting can chase the first drift. |
| Non-skip pass | Required before `/add-model` handoff. |
## Subagent Dispatch Pattern
After Phase 1, dispatch one parity subagent per component before or alongside
component implementation:
```text
Create a local parity test scaffold for <family> <component>.
Use the prep handoff:
- official_ref_dir/import: <...>
- local_weights_dir: <...>
- source_layout: <...>
- needs_conversion: <yes/no>
- official_env_status: <imports_ok | private_deps_need_stubs>
- local_tests_readme: tests/local_tests/<model_family>/README.md
- port_state_file: tests/local_tests/<model_family>/PORT_STATUS.md
- official_definition_files: <paths + classes/functions>
- official_instantiation_files: <paths + factory/pipeline/config call sites + args>
- concerns_or_unknowns: <known ambiguous inputs, outputs, deps, or args>
The complete per-component packet must match
`../add-model/contracts/component_context.md`.
Read the official component call path and the planned FastVideo component API.
Add tests/local_tests/<bucket>/test_<family>_<component>_parity.py based on
add-model-02-parity/templates/component_parity_test.py.
The scaffold must load the official model with real weights when available,
load the FastVideo model through the standardized config/class/loader path when
available, create deterministic inputs, compare concrete outputs, and skip only
when a dependency is genuinely missing. Do not make an unconditional skip or a
shape-only test.
```
## FastVideo Load Patterns
Pick the narrowest load path that matches the component:
| Component | Preferred FastVideo load path |
|---|---|
| DiT / transformer | Bucket config + model class, or `TransformerLoader` when testing converted Diffusers component dirs. |
| VAE | VAE class `from_pretrained(...)` when implemented, or bucket config + class for local converted dirs. |
| Text/image encoder | Bucket config + model class; pass HF subpaths from `local_weights_dir` or converted component dirs. |
| Scheduler/conditioner | Native class/config plus exact official kwargs. |
For early scaffolds, an import of the planned FastVideo class may be inside a
helper that calls `pytest.skip` if the class does not exist yet. Replace that
skip with a real import once the component PR adds the class.
Direct class/config construction is allowed for implementation or subcomponent
parity, such as connector-only encoder checks or official monolithic-checkpoint
mapping tests. Label that scope explicitly and add separate production-loader
coverage when converted component dirs are available.
## Official Load Patterns
- Clone/reference repo path: add its source dir to `sys.path` before imports.
- HF/Diffusers reference: import only inside the test, not production code.
- Private deps: add a helper under `tests/local_tests/helpers/` to install
stubs before importing upstream modules; do not rely on an external upstream
environment.
- Gated HF repos: resolve `HF_TOKEN`, `HUGGINGFACE_HUB_TOKEN`, or `HF_API_KEY`
under the token rules in `../add-model/shared/common_rules.md`.
## Non-Skip Activation Checklist
Before `/add-model` handoff, each scaffolded test must be activated:
```text
[ ] Official side imports and loads real weights.
[ ] FastVideo side imports and loads the converted or original weights.
[ ] Test executes at least one real forward call on both sides.
[ ] Test compares output tensors, not only shapes or state-dict keys.
[ ] Local pytest output contains PASSED, not SKIPPED or XFAIL.
[ ] Tolerance is justified for the component scope and kernel alignment.
```
## Component Parity Details
Reference imports:
- Import from `official_ref_dir` or the recorded package/import path.
- If upstream has private deps, add a helper under
`tests/local_tests/helpers/<family>_upstream.py` that installs minimal stubs
before importing upstream modules.
- Common stubs: identity compile/op-registration decorators, CP world size set to
1, identity scatter/gather, and test-friendly custom-op kernels.
- Stub decorators that register `torch.ops.<ns>.<op>` must preserve the
`torch.library` registration side-effect. Identity decorators alone are not
enough.
- Delete stub helpers and every `install_stubs()` call as soon as the real deps
become required installs. No-op shims are dead code.
Kernel and wrapper pitfalls:
- If parity routes flash-attn GQA through SDPA, expand KV heads manually on the
SDPA side with `repeat_interleave` along the head axis.
- If upstream VAE `decode()` denormalizes internally but FastVideo/Diffusers
expects pre-denormalized latents, apply `z = z * std + mean` only on the
FastVideo side in the parity test.
- Per-channel VAE `latents_mean` / `latents_std` must be reshaped explicitly,
e.g. `.view(1, z_dim, 1, 1, 1)` for 5D video latents.
Tolerance guide:
| Scope | Start `atol` / `rtol` | Notes |
|---|---|---|
| Single block, same kernel | `1e-4` / `1e-4` | Tight default. |
| Full DiT, aligned kernels | `1e-2` / `1e-2` | Cross-layer accumulation. |
| Full DiT, cross-kernel bf16 | `0.1` / `0.1` | Also require abs-mean drift below 5% and per-modality diagnostics. |
| VAE decode fp32 | `5e-2` / `5e-2` | After normalization alignment. |
| Encoder wrapper around same HF class | `1e-3` / `1e-3` | Should be near-zero. |
Element-wise `assert_close` alone is not enough for deep full-DiT parity. Also
log global abs-mean drift and per-modality summaries.
When a non-skip component parity run is numerically red after weight/input
checks, invoke `../add-model-08-trace/SKILL.md` before adding bespoke forward
hooks. Use `docs/contributing/activation_trace.md` to keep
`FASTVIDEO_TRACE_LAYERS`, `FASTVIDEO_TRACE_STATS`, and `FASTVIDEO_TRACE_STEPS`
identical across FastVideo and upstream traces.
Useful local commands:
```bash
pytest tests/local_tests/<bucket>/test_<family>_*parity*.py -v -s
pytest tests/local_tests -k "<family> and parity" -v -s
```
## Escape Hatches
Follow `../add-model/shared/common_rules.md`. Parity-specific ask cases include
private dependency approval, choosing between incompatible official references,
accepting a shape-only substitute, or loosening required tolerances.
## Pipeline Parity
Pipeline parity is later than component parity because it needs stages, presets,
registry wiring, converted weights, and green component parity. Record official
pipeline call notes in `local_tests_readme`, but do not treat pipeline parity as
owned by this skill.
Use `../add-model-09-pipeline/SKILL.md` and its
`templates/pipeline_parity_test.py` for pipeline parity scaffolding and
debugging. Compare denoised latents or decoded media, not just successful
generation.
## Handoff Status Block
Return `../add-model/contracts/parity_status.md` to `/add-model` and update the
shared state files before handoff.
@@ -1,188 +0,0 @@
# SPDX-License-Identifier: Apache-2.0
"""Component parity scaffold for <FAMILY> <COMPONENT>.
This file is intended to be created early in a port. It may skip until the
official reference, FastVideo class, and real weights are available, but it must
never become an unconditional skip or shape-only test.
Fill every TODO before considering this test active.
"""
from __future__ import annotations
import importlib
import os
from pathlib import Path
import sys
import pytest
import torch
from torch.testing import assert_close
os.environ.setdefault("MASTER_ADDR", "localhost")
os.environ.setdefault("MASTER_PORT", "29519")
os.environ.setdefault("DISABLE_SP", "1")
os.environ.setdefault("FASTVIDEO_ATTENTION_BACKEND", "TORCH_SDPA")
REPO_ROOT = Path(__file__).resolve().parents[3]
FAMILY = "<family>" # TODO: snake_case family name.
COMPONENT = "<component>" # TODO: transformer | vae | encoder | conditioner | ...
PARITY_SCOPE = "implementation_subcomponent" # TODO: production_loader | implementation_subcomponent | both
OFFICIAL_MODULE = "<official.module>" # TODO: e.g. "ltx_core.model.transformer".
OFFICIAL_CLASS = "<OfficialClass>" # TODO: official class/factory name.
FASTVIDEO_CONFIG_MODULE = "fastvideo.configs.models.<bucket>" # TODO.
FASTVIDEO_CONFIG_CLASS = "<FastVideoConfig>" # TODO.
FASTVIDEO_MODEL_MODULE = "fastvideo.models.<bucket>.<module>" # TODO.
FASTVIDEO_MODEL_CLASS = "<FastVideoModel>" # TODO.
OFFICIAL_REF_DIR = Path(os.getenv("<FAMILY_UPPER>_OFFICIAL_REF_DIR", REPO_ROOT / "<ReferenceDir>"))
LOCAL_WEIGHTS_DIR = Path(os.getenv("<FAMILY_UPPER>_LOCAL_WEIGHTS_DIR", REPO_ROOT / "official_weights" / FAMILY))
CONVERTED_WEIGHTS_DIR = Path(os.getenv("<FAMILY_UPPER>_CONVERTED_WEIGHTS_DIR",
REPO_ROOT / "converted_weights" / FAMILY))
def _resolve_hf_token() -> str | None:
for key in ("HF_TOKEN", "HUGGINGFACE_HUB_TOKEN", "HF_API_KEY"):
value = os.environ.get(key)
if value:
return value
return None
def _add_official_to_path() -> None:
"""Add the official source path before importing upstream modules."""
# TODO: adjust for the official repo layout. Common examples:
# OFFICIAL_REF_DIR / "src"
# OFFICIAL_REF_DIR / "packages" / "<pkg>" / "src"
# OFFICIAL_REF_DIR
official_src = OFFICIAL_REF_DIR / "src"
if not official_src.exists():
official_src = OFFICIAL_REF_DIR
if official_src.exists() and str(official_src) not in sys.path:
sys.path.insert(0, str(official_src))
def _import_or_skip(module_name: str, attr_name: str | None = None):
if "<" in module_name or (attr_name is not None and "<" in attr_name):
pytest.skip(f"Template import placeholder not filled: {module_name}.{attr_name}")
try:
module = importlib.import_module(module_name)
except Exception as exc: # noqa: BLE001 - local parity should skip missing refs.
pytest.skip(f"Cannot import {module_name}: {exc}")
if attr_name is None:
return module
try:
return getattr(module, attr_name)
except AttributeError:
pytest.skip(f"{module_name} has no attribute {attr_name}")
def _load_official_model(device: torch.device, dtype: torch.dtype) -> torch.nn.Module:
"""Load the official component with real weights."""
_add_official_to_path()
if not OFFICIAL_REF_DIR.exists():
pytest.skip(f"Official reference missing: {OFFICIAL_REF_DIR}")
if not LOCAL_WEIGHTS_DIR.exists():
pytest.skip(f"Local weights missing: {LOCAL_WEIGHTS_DIR}")
# TODO: import official class/factory and load real weights strictly.
# Examples in-tree:
# - LTX2: SingleGPUModelBuilder(...).build(device=device, dtype=dtype)
# - GameCraft: torch.load(...)["module"] -> official_model.load_state_dict(...)
# - Oobleck: create_model_from_config(config) + ckpt state_dict
OfficialClass = _import_or_skip(OFFICIAL_MODULE, OFFICIAL_CLASS)
model = OfficialClass() # TODO: pass official config kwargs.
state_dict = {} # TODO: load official state dict from LOCAL_WEIGHTS_DIR.
missing, unexpected = model.load_state_dict(state_dict, strict=True)
assert not missing and not unexpected, (f"official load mismatch missing={missing[:5]} unexpected={unexpected[:5]}")
return model.to(device=device, dtype=dtype).eval()
def _load_fastvideo_model(device: torch.device, dtype: torch.dtype) -> torch.nn.Module:
"""Load the FastVideo component with the same tensor content."""
if not CONVERTED_WEIGHTS_DIR.exists() and not LOCAL_WEIGHTS_DIR.exists():
pytest.skip(f"No FastVideo loadable weights: {CONVERTED_WEIGHTS_DIR} or {LOCAL_WEIGHTS_DIR}")
# TODO: replace with the bucket-specific FastVideo config/class/loader.
# DiT examples:
# from fastvideo.configs.models.dits import <Config>
# from fastvideo.models.dits.<module> import <Model>
# VAE examples:
# from fastvideo.models.vaes.<module> import <VAE>
# model = <VAE>.from_pretrained(...)
FastVideoConfig = _import_or_skip(FASTVIDEO_CONFIG_MODULE, FASTVIDEO_CONFIG_CLASS)
FastVideoModel = _import_or_skip(FASTVIDEO_MODEL_MODULE, FASTVIDEO_MODEL_CLASS)
config = FastVideoConfig()
model = FastVideoModel(config=config)
state_dict = {} # TODO: load converted or directly mapped state dict.
missing, unexpected = model.load_state_dict(state_dict, strict=True)
assert not missing and not unexpected, (
f"FastVideo load mismatch missing={missing[:5]} unexpected={unexpected[:5]}")
return model.to(device=device, dtype=dtype).eval()
def _make_inputs(device: torch.device, dtype: torch.dtype) -> dict[str, torch.Tensor]:
"""Create deterministic inputs matching the official component call."""
torch.manual_seed(0)
# TODO: replace with component-specific tensors and metadata.
return {
"hidden_states": torch.randn(1, 4, 16, device=device, dtype=dtype),
"timestep": torch.tensor([10], device=device),
}
def _run_official(model: torch.nn.Module, inputs: dict[str, torch.Tensor]) -> torch.Tensor:
"""Run official component and return the tensor to compare."""
with torch.inference_mode():
output = model(**inputs) # TODO: adapt official call signature.
if isinstance(output, dict):
sample = output.get("sample")
output = sample if sample is not None else output.get("x")
elif hasattr(output, "sample"):
output = output.sample
elif isinstance(output, tuple):
output = output[0]
assert torch.is_tensor(output), f"official output is not tensor: {type(output)}"
return output.detach().float().cpu()
def _run_fastvideo(model: torch.nn.Module, inputs: dict[str, torch.Tensor]) -> torch.Tensor:
"""Run FastVideo component and return the tensor to compare."""
with torch.inference_mode():
output = model(**inputs) # TODO: adapt FastVideo call signature.
if isinstance(output, dict):
sample = output.get("sample")
output = sample if sample is not None else output.get("x")
elif hasattr(output, "sample"):
output = output.sample
elif isinstance(output, tuple):
output = output[0]
assert torch.is_tensor(output), f"FastVideo output is not tensor: {type(output)}"
return output.detach().float().cpu()
@pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA required for this parity test.")
def test_component_parity():
"""Compare official and FastVideo outputs on identical inputs."""
device = torch.device("cuda:0")
dtype = torch.bfloat16
official = _load_official_model(device, dtype)
fastvideo = _load_fastvideo_model(device, dtype)
inputs = _make_inputs(device, dtype)
official_out = _run_official(official, inputs)
fastvideo_out = _run_fastvideo(fastvideo, inputs)
assert official_out.shape == fastvideo_out.shape
diff = (official_out - fastvideo_out).abs()
print(f"official abs_mean={official_out.abs().mean().item():.6f} "
f"fastvideo abs_mean={fastvideo_out.abs().mean().item():.6f} "
f"diff_max={diff.max().item():.6f} diff_mean={diff.mean().item():.6f}")
# TODO: pick tolerance by scope:
# - single block / same kernel: 1e-4
# - full DiT aligned kernels: 1e-2
# - full DiT cross-kernel bf16: 1e-1 + abs_mean drift check
# - VAE decode fp32: 5e-2 after normalization alignment
assert_close(fastvideo_out, official_out, atol=1e-4, rtol=1e-4)
@@ -1,114 +0,0 @@
---
name: add-model-03-port-dit
description: Use during /add-model Phase 4 or Phase 6 to prototype or parity-debug one FastVideo-native DiT/transformer component.
---
# Add Model Port DiT
## Goal
Prototype or parity-debug one diffusion transformer in FastVideo-native code.
This skill is for one component only; do not work on the VAE, encoders,
pipeline, or unrelated conversion code unless the current component cannot load
without a minimal fix there.
## Inputs
Follow `../add-model/shared/component_skill_common.md` and require the complete
packet from `../add-model/contracts/component_context.md`.
DiT-specific packet fields:
- `component`: transformer or DiT name.
- `parity_test`: `tests/local_tests/<bucket>/test_<family>_<component>_parity.py`.
- `weights`: converted transformer dir or local official path.
- `target_files`: `fastvideo/models/dits/<family>.py` and
`fastvideo/configs/models/dits/<family>.py`.
## Modes
Use the common prototype and parity-debug modes from
`../add-model/shared/component_skill_common.md`.
DiT-specific prototype concerns include ambiguous official flags, shape
mismatches, missing FastVideo layer equivalents, and dedicated output heads.
## Reuse Proof
Apply the shared reuse proof. DiT-specific comparison must include attention
algorithm, positional embeddings, RoPE/patching, timestep/guidance embeddings,
scaling constants, dtype casts, state-dict names, and every output head.
## Existing FastVideo Patterns
- Base class: `fastvideo/models/dits/base.py::BaseDiT`.
- Config bases: `DiTConfig` and `DiTArchConfig` in
`fastvideo/configs/models/dits/base.py`.
- Use the matching DiT config bucket. Wrong bucket inheritance can typecheck but
fail during pipeline wiring.
- Config export: add the config to
`fastvideo/configs/models/dits/__init__.py`.
- Registry discovery: set `EntryClass = <ClassName>` in the model file.
- Loader path: `TransformerLoader` reads `transformer/config.json`, calls
`dit_config.update_model_arch(config)`, resolves `_class_name` through
`ModelRegistry`, and constructs the class with `config` and `hf_config`.
- Reference examples: `stable_audio.py`, `wanvideo.py`, `sd3.py`, `longcat.py`,
and `ltx2.py`.
- Layer guidance: `fastvideo/layers/AGENTS.md`.
## Implementation Rules
- Use FastVideo-native layers by default: `ReplicatedLinear` for DiT hot-path
linears, `DistributedAttention` for standard full-sequence attention, and
`LocalAttention` for local/window attention or simple single-GPU parity paths.
- Raw SDPA is acceptable for cross-modality flat streams when no FastVideo
distributed equivalent exists; document the SP gap in the module docstring.
- Mirror official tensor contracts exactly: latent packing, patch ordering,
timestep embedding scale, RoPE/positional embedding, guidance embedding,
cross-attention context order, output head order, and dtype casts.
- Preserve all output heads that the official DiT emits. Do not silently drop
audio, depth, pose, mask, or auxiliary heads.
- Put architecture fields on `DiTArchConfig`; keep inference steps, CFG scales,
FPS, flow shift, and sampling defaults out of the arch config.
- Define `_fsdp_shard_conditions`, `_compile_conditions`,
`param_names_mapping`, and `reverse_param_names_mapping` where needed.
- Follow the production import boundary in
`../add-model/shared/common_rules.md`.
## Prototype Checks
Follow the shared prototype success criteria. A useful one-off check is:
```bash
python - <<'PY'
# Import the target config/class, instantiate with random weights, and print
# state_dict names/shapes for the conversion mapping.
PY
```
## Parity-Debug Loop
Run the shared parity-debug loop. The component test command is:
```bash
pytest <parity_test> -v -s
```
For numerical drift, use `../add-model-08-trace/SKILL.md` before writing bespoke
hooks. Start with FastVideo's activation trace (`fastvideo/hooks/activation_trace.py`;
`docs/contributing/activation_trace.md`) and a block-level regex such as
`FASTVIDEO_TRACE_LAYERS="^block\.layers\.[0-9]+$"`. Only fall back to custom
per-block hooks if the needed boundary or statistic is not exposed by
`FASTVIDEO_TRACE_STATS`.
## Escape Hatches
Follow `../add-model/shared/common_rules.md` and the component-specific guidance
in `../add-model/shared/component_skill_common.md`. DiT-specific ask cases include
dropping an output head/modality, accepting an unsupported kernel/private op, or
choosing between incompatible official transformer definitions.
## Handoff
Return `../add-model/contracts/component_skill_handoff.md` following the common
handoff rules in `../add-model/shared/component_skill_common.md`.
@@ -1,100 +0,0 @@
---
name: add-model-04-port-vae
description: Use during /add-model Phase 4 or Phase 6 to prototype or parity-debug one FastVideo-native VAE component.
---
# Add Model Port VAE
## Goal
Prototype or parity-debug one VAE or autoencoder in FastVideo-native code. This
skill covers video, image, and audio VAEs.
## Inputs
Follow `../add-model/shared/component_skill_common.md` and require the complete
packet from `../add-model/contracts/component_context.md`.
VAE-specific packet fields:
- `component`: VAE or autoencoder name.
- `parity_test`: `tests/local_tests/vaes/test_<family>_<component>_parity.py`.
- `weights`: converted VAE dir, HF subfolder, or local official path.
- `target_files`: `fastvideo/models/vaes/<arch_or_family>.py` and
`fastvideo/configs/models/vaes/<arch_or_family>.py`.
## Modes
Use the common prototype and parity-debug modes from
`../add-model/shared/component_skill_common.md`.
VAE-specific prototype concerns include latent normalization, stochastic
posterior behavior, tiling incompatibility, temporal/spatial/audio layout, and
decode output containers.
## Reuse Proof
Apply the shared reuse proof. VAE-specific comparison must include latent layout,
temporal/spatial/audio compression, scaling factor, mean/std normalization,
posterior behavior, encode/decode output objects, tiling flags, and cropping.
## Existing FastVideo Patterns
- Shared tiling wrapper: `fastvideo/models/vaes/common.py::ParallelTiledVAE`.
- Config bases: `VAEConfig` and `VAEArchConfig` in
`fastvideo/configs/models/vaes/base.py`.
- Use the matching VAE config bucket. Wrong bucket inheritance can typecheck but
fail during pipeline wiring.
- Config export: add the config to
`fastvideo/configs/models/vaes/__init__.py`.
- Registry discovery: set `EntryClass = <ClassName>` in the model file.
- Loader path: VAE loaders resolve `_class_name` through `ModelRegistry` and
load converted component weights from the VAE subdir.
- Reference examples: `oobleck.py`, `autoencoder_kl.py`, `wanvae.py`,
`ltx2vae.py`, and `gamecraftvae.py`.
- Layer guidance: `fastvideo/layers/AGENTS.md`.
## Implementation Rules
- Name reusable VAE architectures by architecture (`oobleck.py`,
`autoencoder_kl.py`); name family-specific VAEs by family.
- Match official encode/decode contracts exactly: input layout, latent layout,
temporal/spatial/audio compression, scaling factor, mean/std normalization,
posterior sampling behavior, decode output object, and frame/sample cropping.
- Compare deterministic outputs in parity: decode outputs, encode mean/mode, or
round-trip tensors. Do not compare stochastic samples unless the RNG path is
explicitly controlled.
- Use FastVideo tiling only when it preserves official numerics for the tested
shape; disable it in config for audio or unsupported dimensions.
- Put architecture constants on `VAEArchConfig`; put `load_encoder`,
`load_decoder`, tiling, dtype, and pretrained path fields on `VAEConfig`.
- Follow the production import boundary in
`../add-model/shared/common_rules.md`.
## Prototype Checks
Follow the shared prototype success criteria.
## Parity-Debug Loop
Run the shared parity-debug loop. The component test command is:
```bash
pytest <parity_test> -v -s
```
For numerical drift, check normalization, latent scaling, posterior mode vs
sample, channel order, and temporal/spatial/audio cropping before changing
layers.
## Escape Hatches
Follow `../add-model/shared/common_rules.md` and the component-specific guidance
in `../add-model/shared/component_skill_common.md`. VAE-specific ask cases include
dropping an encode/decode path, accepting an unsupported private op, or choosing
between incompatible official VAE definitions.
## Handoff
Return `../add-model/contracts/component_skill_handoff.md` following the common
handoff rules in `../add-model/shared/component_skill_common.md`.
@@ -1,119 +0,0 @@
---
name: add-model-05-port-encoder
description: Use during /add-model Phase 4 or Phase 6 to prototype or parity-debug one FastVideo-native text, image, audio, or compound encoder component.
---
# Add Model Port Encoder
## Goal
Prototype or parity-debug one encoder or encoder-like conditioner in
FastVideo-native code. Use this for text encoders, image encoders, audio
encoders, and compound conditioners that fit the encoder config/loader bucket.
## Inputs
Follow `../add-model/shared/component_skill_common.md` and require the complete
packet from `../add-model/contracts/component_context.md`.
Encoder-specific packet fields:
- `component`: encoder or encoder-like conditioner name.
- `parity_test`: `tests/local_tests/encoders/test_<family>_<component>_parity.py`.
- `weights`: converted encoder dir, HF subfolder, or external HF id.
- `target_files`: `fastvideo/models/encoders/<arch_or_family>.py` and
`fastvideo/configs/models/encoders/<arch_or_family>.py`.
## Modes
Use the common prototype and parity-debug modes from
`../add-model/shared/component_skill_common.md`.
Encoder-specific prototype concerns include tokenizer kwargs, hidden-state
extraction, output packing, connector order, and external/passthrough weight
needs.
## Reuse Proof
Apply the shared reuse proof. Encoder-specific comparison must include tokenizer
contracts, hidden-state extraction, masks, positional IDs, output packing,
connector/projection ordering, passthrough paths, and returned dataclass shape.
## Existing FastVideo Patterns
- Base classes: `TextEncoder` and `ImageEncoder` in
`fastvideo/models/encoders/base.py`.
- Output type: `BaseEncoderOutput`.
- Config bases: `TextEncoderConfig`, `ImageEncoderConfig`,
`TextEncoderArchConfig`, and `ImageEncoderArchConfig` in
`fastvideo/configs/models/encoders/base.py`.
- Use the matching encoder config bucket. Wrong bucket inheritance can typecheck
but fail during pipeline wiring.
- Config export: add the config to
`fastvideo/configs/models/encoders/__init__.py`.
- Registry discovery: set `EntryClass = <ClassName>` or a list of class names in
the model file.
- Reference examples: native `t5.py`, `clip.py`, `siglip.py`, `llama.py`,
`qwen2_5.py`, `gemma.py`, and compound `stable_audio_conditioner.py`.
- Layer guidance: `fastvideo/layers/AGENTS.md`.
## Implementation Rules
- Reuse tokenizers and pure data utilities when needed, but do not add runtime
third-party model-class imports as a placeholder for a component that owns
weights or numerical behavior.
- For LLM-style encoders, follow existing tensor-parallel patterns such as
`QKVParallelLinear`, `MergedColumnParallelLinear`, `RowParallelLinear`,
`VocabParallelEmbedding`, and `RMSNorm` when matching native examples.
- Match official hidden-state extraction exactly: layer index, pooled output,
attention mask dtype, padding side, truncation, special tokens, final norm,
output_hidden_states, and returned tuple/dataclass shape.
- For connector or conditioner modules, preserve sub-conditioner order and the
exact packing of cross-attention tokens, masks, and global conditioning.
- Put tokenizer kwargs and architecture constants on the arch config when they
affect numerical behavior.
- If an external HF encoder is explicitly accepted as a lazy wrapper, keep it
isolated, document why it is not a native port, and still require parity for
the wrapper's output contract.
Hybrid external-HF encoder checklist:
- Put external model folders in passthrough subfolders such as
`text_encoder/<external_name>/`, or record a root `model_index.json` path field
that the loader resolves to a local directory.
- Keep external model parameters out of the FastVideo-owned state-dict surface
when the external model is loaded lazily from its own HF files.
- Convert and strict-check only the FastVideo-owned connector/projection weights;
document external model weights as passthrough.
- Add parity for the wrapper's final output contract and, when useful, a narrower
connector-only parity test that labels its scope as
`implementation_subcomponent`.
- Verify the production loader resolves the same external path used by the
pipeline, not just the direct class used in the parity test.
## Prototype Checks
Follow the shared prototype success criteria.
## Parity-Debug Loop
Run the shared parity-debug loop. The component test command is:
```bash
pytest <parity_test> -v -s
```
For numerical drift, check tokenization, masks, hidden-state selection,
positional IDs, dtype/autocast, and output packing before changing layers.
## Escape Hatches
Follow `../add-model/shared/common_rules.md` and the component-specific guidance
in `../add-model/shared/component_skill_common.md`. Encoder-specific ask cases
include accepting private model-code execution, choosing between incompatible
tokenizer/encoder references, or dropping a required conditioning stream.
## Handoff
Return `../add-model/contracts/component_skill_handoff.md` following the common
handoff rules in `../add-model/shared/component_skill_common.md`.
@@ -1,111 +0,0 @@
---
name: add-model-06-port-generic
description: Use during /add-model Phase 4 or Phase 6 to prototype or parity-debug one non-DiT, non-VAE, non-encoder FastVideo component.
---
# Add Model Port Generic
## Goal
Prototype or parity-debug one scheduler, conditioner, upsampler, vocoder,
adapter, preprocessor, or unknown component in FastVideo-native code.
## Inputs
Follow `../add-model/shared/component_skill_common.md` and require the complete
packet from `../add-model/contracts/component_context.md`.
Generic-component packet fields:
- `component`: component name.
- `component_type`: scheduler, conditioner, upsampler, vocoder, adapter,
preprocessor, or unknown.
- `parity_test`: `tests/local_tests/<bucket>/test_<family>_<component>_parity.py`.
- `weights`: converted component dir, HF subfolder, or none.
- `target_files`: matching `fastvideo/models/` and `fastvideo/configs/models/`
bucket files when applicable.
## Modes
Use the common prototype and parity-debug modes from
`../add-model/shared/component_skill_common.md`.
Generic-component prototype concerns include stateless/stateful ambiguity,
missing loader buckets, source prefixes, mutable scheduler state, and output
container shape.
## Reuse Proof
Apply the shared reuse proof. Generic-component comparison must include mutable
state, scaling constants, scheduler/conditioner semantics, output containers, and
whether the component owns state or is stateless.
## Existing FastVideo Patterns
- Schedulers live under `fastvideo/models/schedulers/` and expose `EntryClass`.
- Upsamplers use `fastvideo/models/upsamplers/` plus configs under
`fastvideo/configs/models/upsamplers/`; see `hunyuan15.py`.
- Vocoders and audio-specific modules can live under `fastvideo/models/audio/`
with configs under `fastvideo/configs/models/audio/`; see `ltx2_audio_vae.py`.
- Compound conditioners may fit the encoder bucket when the pipeline loader uses
`ConditionerLoader`; see `stable_audio_conditioner.py`.
- Registry discovery uses `EntryClass`; config bucket exports are required when
pipeline configs import them by bucket.
- Use the narrowest matching config bucket. Wrong bucket inheritance can typecheck
but fail during pipeline wiring.
- Layer guidance: `fastvideo/layers/AGENTS.md`.
## Bucket Decision
- If the component is a transformer/DiT, stop and use `add-model-03-port-dit`.
- If the component is a VAE/autoencoder, stop and use `add-model-04-port-vae`.
- If the component is a text/image/audio encoder or encoder-like conditioner,
stop and use `add-model-05-port-encoder` unless the loader requires a different
bucket.
- Otherwise choose the narrowest existing bucket. Add a new bucket only when no
existing loader/config shape can represent the component without misleading
names or unsafe runtime behavior.
## Implementation Rules
- Match official behavior, not just shapes: constructor args, default values,
runtime flags, RNG use, dtype/autocast, scaling constants, masks, and output
containers all matter.
- Keep the implementation minimal and native. Do not keep a runtime import of
the official implementation as the production component.
- For schedulers, compare timesteps, sigmas/noise levels, step outputs, shift
handling, prediction type, and any mutable internal state.
- For upsamplers, compare resize mode, align_corners, residual branches,
causal padding, normalization, and exact target-shape behavior.
- For vocoders/audio components, compare waveform shape, sample-rate contract,
channel order, hop length, normalization, and dtype.
- If private upstream deps are required only for tests, keep stubs under
`tests/local_tests/helpers/` and do not import them from production code.
## Prototype Checks
Follow the shared prototype success criteria.
## Parity-Debug Loop
Run the shared parity-debug loop. The component test command is:
```bash
pytest <parity_test> -v -s
```
For numerical drift, add targeted intermediate comparisons in the test to
identify the first divergent operation.
## Escape Hatches
Follow `../add-model/shared/common_rules.md` and the component-specific guidance
in `../add-model/shared/component_skill_common.md`. Generic-component ask cases
include creating a new loader bucket, accepting an unsupported private op,
choosing between incompatible official definitions, or dropping a required
component.
## Handoff
Return `../add-model/contracts/component_skill_handoff.md` following the common
handoff rules in `../add-model/shared/component_skill_common.md`.
@@ -1,186 +0,0 @@
---
name: add-model-07-conversion
description: Use during /add-model Phase 5 to write and verify a FastVideo checkpoint conversion script after native component prototypes expose FastVideo state-dict keys/shapes.
---
# Add Model Conversion
## Goal
Convert official weights into a FastVideo-loadable component layout after Phase 4
native prototypes exist. The conversion script owns parameter mapping, component
splitting, passthrough assets, config emission, and strict-load verification.
## Inputs
Follow `../add-model/shared/common_rules.md` for token/auth safety, state files,
escape hatches, production boundaries, and skip/pass semantics.
Require the initial request from
`../add-model/contracts/conversion_request.md`.
If the FastVideo key/shape dump is missing, return to `/add-model` Phase 4. Do
not write a final mapping against an unimplemented component.
For Phase 6 retry requests from component skills, also require the retry shape
from `../add-model/contracts/conversion_request.md`.
## Output
- `scripts/checkpoint_conversion/<family>_to_diffusers.py`.
- `converted_weights/<family>/` with `model_index.json` and per-component
subfolders.
- Updated `tests/local_tests/<model_family>/README.md` with conversion command,
source layout, output path, and strict-load status.
- Updated `tests/local_tests/<model_family>/PORT_STATUS.md` with conversion
state, retry history, open questions, and issues/blockers.
## Reference Scripts
- `scripts/checkpoint_conversion/convert_ltx2_weights.py`: component prefix
splitting, metadata config extraction, passthrough Gemma/tokenizer assets, and
optional component-only output.
- `scripts/checkpoint_conversion/stable_audio_to_diffusers.py`: monolithic
`model.safetensors` split into transformer/VAE/conditioner, plus copied
passthrough subfolders. Use this shape for single-checkpoint official repos.
- `scripts/checkpoint_conversion/convert_gamecraft_full.py`: separate official
sources for transformer, VAE, text encoders, tokenizers, scheduler, and root
`model_index.json`.
- `scripts/checkpoint_conversion/longcat_to_fastvideo.py`: fused QKV/KV split,
renamed native transformer weights, and copied existing Diffusers components.
- `scripts/checkpoint_conversion/pt_to_safetensors.py`: simple `.pt` extraction
helper for nested checkpoint dictionaries.
## Source Layout Decision
Choose exactly one primary layout:
| Layout | Conversion behavior |
|---|---|
| `diffusers` | Usually no tensor remap; verify configs/classes and copy or update `_class_name` only when needed. |
| `raw_official` | Convert a raw official checkpoint file or directory. Choose explicit component ownership before writing output. |
| `separate_components` | Convert/copy each component from its own file or directory. |
| `monolithic` | Load one model checkpoint and split state dict by authoritative prefixes into component buckets. |
| `mixed` | Convert some components and copy passthrough components such as tokenizers, text encoders, schedulers, or already-Diffusers VAE dirs. |
| `custom` | Document why none of the above fits before writing conversion code. |
Monolithic checkpoints need explicit prefix ownership. For example, Stable Audio
uses one `model.safetensors` with DiT, pretransform/VAE, and conditioner keys;
the converter splits those keys into FastVideo component subfolders and writes
per-component configs.
## Script Shape
Start from `templates/family_to_diffusers.py` or the closest reference script.
Keep the script explicit and reviewable:
- `COMPONENT_SPECS` or `COMPONENT_PREFIXES` declares component ownership.
- `PARAM_NAME_MAP` declares key renames.
- `SKIP_PATTERNS` declares intentionally dropped training-only keys.
- tensor split/fuse helpers are named by operation, e.g. `split_qkv`.
- `build_component_configs(...)` writes loader-compatible config files. Most
model components use `config.json`; schedulers use `scheduler_config.json`.
- `build_model_index(...)` writes a root `model_index.json` matching the target
FastVideo pipeline and component classes.
- verification reports missing, unexpected, skipped, unchanged, renamed, and
shape-mismatched keys.
`model_index.json` library tokens must match FastVideo loaders:
- standard native DiT/VAE/audio/vocoder/upsampler components loaded by existing
Diffusers-style loaders usually use `"diffusers"` with a FastVideo
`_class_name` in the component `config.json`;
- text encoders, tokenizers, image encoders, processors, and feature extractors
usually use `"transformers"`;
- `conditioner` currently expects `"fastvideo"`;
- use fully qualified `"fastvideo.<module>"` only when intentionally relying on
the custom fastvideo-library escape path;
- do not write bare `"fastvideo"` for transformer, VAE, or other loaders that
expect `"diffusers"` unless the loader explicitly expects it.
## Mapping Rules
- Use Phase 4 key/shape dumps to derive mappings. Do not guess from official key
names alone.
- Preserve each component's official file paths, parity test path, and prototype
concerns in comments or structured constants near the mapping that uses them.
- Every official inference parameter should be mapped, copied through, or listed
as intentionally skipped with a reason.
- Every FastVideo prototype parameter should receive a tensor or be listed as an
intentional external/passthrough parameter.
- Shape matches are necessary but not sufficient; check semantic pairing for
Q/K/V, gate/up/down, norm scale/bias, LoRA/base, and modality-specific heads.
- If official and FastVideo fuse or split tensors differently, convert tensors in
the script rather than changing production code to match checkpoint quirks.
## Verification
Run conversion locally, then verify before returning to Phase 6. For retry
requests, update the mapping, rerun conversion, and refresh only the implicated
converted component when safe; otherwise rerun the full conversion.
```bash
python scripts/checkpoint_conversion/<family>_to_diffusers.py \
--src <official_weights> \
--revision <hf_revision> \
--dst converted_weights/<model_family>
```
Omit `--revision` for local sources or when prep recorded `default` / `none`.
Minimum output layout:
```text
converted_weights/<family>/
model_index.json
transformer/config.json
transformer/*.safetensors
vae/config.json
vae/*.safetensors
scheduler/scheduler_config.json as needed
text_encoder/... as needed
```
Required checks:
- `model_index.json` exists and lists every required component.
- Each converted component has the config filename its loader expects and
safetensors weights when it owns weights. Scheduler dirs require
`scheduler_config.json`; most other native model dirs use `config.json`.
- Weight filenames may vary by loader: transformer and VAE loaders glob all
`*.safetensors`; text encoders may load `*.safetensors`, `*.bin`, and
sometimes `*.pt`; `conditioner` currently expects
`diffusion_pytorch_model.safetensors`. Use the loader's actual accepted layout
rather than assuming one global filename.
- Passthrough components are copied or referenced deliberately.
- Each emitted component config validates through the same path production
loaders use. Instantiate the relevant config and call `update_model_arch(...)`
or `update_model_config(...)` with the emitted JSON so unknown keys fail during
conversion, not at pipeline load time.
- Record production loader strictness for every stateful component. If the loader
intentionally uses non-strict loading, add explicit missing/unexpected-key
assertions in the parity test and document exactly which keys are allowed.
- Each new FastVideo component strict-loads converted weights where its production
loader is strict. If strict loading is impossible, record the exact allowed
missing/unexpected keys and why they are not inference weights.
- Retry fixes include the original component parity evidence and the new
strict-load result in `local_tests_readme` so the component subagent can resume
without rediscovering context.
- `local_tests_readme` records the command, output directory, and strict-load
result.
Do not chase numerical parity in this skill except to identify a conversion
mapping bug. Long parity-debug loops belong to `/add-model` Phase 6.
## Escape Hatches
Follow `../add-model/shared/common_rules.md`. Conversion-specific ask cases
include selecting between incompatible official checkpoints, publishing/uploading
weights, overwriting an existing converted repo not created by this run,
accepting non-strict missing inference weights, or dropping a component/output
from scope.
## Handoff
Return `../add-model/contracts/conversion_handoff.md` and update the shared state
files before handoff.
@@ -1,293 +0,0 @@
#!/usr/bin/env python3
# SPDX-License-Identifier: Apache-2.0
"""Convert <model_family> official weights to a FastVideo Diffusers-style tree.
This template supports both separate component sources and a monolithic pipeline
checkpoint that must be split by component prefix. Replace every TODO before
using it for a real port.
"""
from __future__ import annotations
import argparse
import json
import os
import re
import shutil
from collections import OrderedDict
from pathlib import Path
from typing import Any
import torch
from safetensors import safe_open
from safetensors.torch import load_file, save_file
try:
from huggingface_hub import snapshot_download
except ImportError: # pragma: no cover - optional local conversion dependency
snapshot_download = None
# TODO: fill with authoritative component prefixes for monolithic checkpoints.
# Example: {"model.model.": "transformer", "pretransform.model.": "vae"}
COMPONENT_PREFIXES: dict[str, str] = {}
# TODO: fill with component-specific source paths for separate-component repos.
# Example: {"transformer": "transformer/model.safetensors", "vae": "vae/"}
SEPARATE_COMPONENT_PATHS: dict[str, str] = {}
# TODO: copy passthrough dirs that are already loadable by FastVideo/Diffusers.
PASSTHROUGH_SUBFOLDERS: tuple[str, ...] = ("tokenizer", "scheduler")
# TODO: add regex renames derived from Phase 4 key/shape dumps.
PARAM_NAME_MAP: dict[str, str] = {}
# TODO: include training-only or dynamically-computed keys that must not load.
SKIP_PATTERNS: tuple[str, ...] = ()
def _hf_token() -> str | None:
return (os.environ.get("HF_TOKEN") or os.environ.get("HUGGINGFACE_HUB_TOKEN") or os.environ.get("HF_API_KEY"))
def resolve_src(src: str, revision: str | None) -> Path:
if os.path.exists(src):
return Path(src)
if snapshot_download is None:
raise RuntimeError("huggingface_hub is required when --src is a repo id")
return Path(snapshot_download(repo_id=src, revision=revision, token=_hf_token()))
def load_checkpoint(path: Path) -> dict[str, torch.Tensor]:
if path.is_dir():
weights: dict[str, torch.Tensor] = {}
for shard in sorted(path.glob("*.safetensors")):
weights.update(load_file(str(shard)))
if weights:
return weights
raise FileNotFoundError(f"No safetensors found in {path}")
if path.suffix == ".safetensors":
return load_file(str(path))
checkpoint = torch.load(path, map_location="cpu", weights_only=True)
if isinstance(checkpoint, dict):
for key in ("state_dict", "model_state_dict", "model", "module", "ema"):
if key in checkpoint and isinstance(checkpoint[key], dict):
return checkpoint[key]
return checkpoint
raise TypeError(f"Unsupported checkpoint type: {type(checkpoint)!r}")
def should_skip_key(key: str) -> bool:
return any(re.search(pattern, key) for pattern in SKIP_PATTERNS)
def apply_mapping(key: str) -> str | None:
if should_skip_key(key):
return None
for pattern, replacement in PARAM_NAME_MAP.items():
if re.match(pattern, key):
return re.sub(pattern, replacement, key)
return key
def split_monolithic(state: dict[str, torch.Tensor], ) -> dict[str, OrderedDict[str, torch.Tensor]]:
components: dict[str, OrderedDict[str, torch.Tensor]] = {
name: OrderedDict()
for name in set(COMPONENT_PREFIXES.values())
}
intentionally_skipped: list[str] = []
unowned: list[str] = []
for key, value in state.items():
if should_skip_key(key):
intentionally_skipped.append(key)
continue
for prefix, component in COMPONENT_PREFIXES.items():
if key.startswith(prefix):
mapped = apply_mapping(key[len(prefix):])
if mapped is not None:
components[component][mapped] = value
break
else:
unowned.append(key)
if unowned:
sample = ", ".join(unowned[:10])
raise ValueError(f"Unowned monolithic keys: {len(unowned)}. "
f"Add COMPONENT_PREFIXES or SKIP_PATTERNS entries. Sample: {sample}")
if intentionally_skipped:
print(f"Intentionally skipped {len(intentionally_skipped)} keys")
return {name: weights for name, weights in components.items() if weights}
def load_separate_components(src_dir: Path) -> dict[str, OrderedDict[str, torch.Tensor]]:
components: dict[str, OrderedDict[str, torch.Tensor]] = {}
for component, rel_path in SEPARATE_COMPONENT_PATHS.items():
state = load_checkpoint(src_dir / rel_path)
converted: OrderedDict[str, torch.Tensor] = OrderedDict()
for key, value in state.items():
mapped = apply_mapping(key)
if mapped is not None:
converted[mapped] = value
components[component] = converted
return components
def build_component_configs(_src_dir: Path) -> dict[str, dict[str, Any]]:
# TODO: emit config content accepted by FastVideo loaders. Most components use
# config.json; schedulers use scheduler_config.json.
return {
"transformer": {
"_class_name": "<FastVideoTransformerClass>"
},
"vae": {
"_class_name": "<FastVideoVAEClass>"
},
}
def config_filename(component: str) -> str:
if component == "scheduler":
return "scheduler_config.json"
return "config.json"
def source_label(src: str) -> str:
if os.path.exists(src):
return Path(src).name
return src
def build_model_index(
src: str,
revision: str | None,
available_components: set[str],
) -> dict[str, Any]:
# TODO: match the target pipeline and every required component.
index: dict[str, Any] = {
"_class_name": "<FastVideoPipelineClass>",
"_diffusers_version": "0.30.0",
"_fastvideo_converted_from": source_label(src),
# Existing transformer/VAE loaders expect "diffusers" even when
# _class_name names a FastVideo-native class registered in FastVideo.
"transformer": ["diffusers", "<FastVideoTransformerClass>"],
"vae": ["diffusers", "<FastVideoVAEClass>"],
}
if revision:
index["_fastvideo_converted_revision"] = revision
return {key: value for key, value in index.items() if key.startswith("_") or key in available_components}
def validate_component_configs(configs: dict[str, dict[str, Any]]) -> None:
# TODO: instantiate each FastVideo config and call update_model_arch(...) or
# update_model_config(...) with this JSON so unknown emitted keys fail here.
placeholder_configs = [name for name, config in configs.items() if "<" in json.dumps(config)]
if placeholder_configs:
raise ValueError(f"Replace config placeholders for: {placeholder_configs}")
def verify_conversion(
dst_dir: Path,
components: dict[str, OrderedDict[str, torch.Tensor]],
) -> None:
del dst_dir, components
# TODO: load each emitted stateful component through its production loader and
# assert strict load, or document exact allowed missing/unexpected keys.
raise NotImplementedError("Implement production config validation and strict-load checks")
def write_component(
dst_dir: Path,
name: str,
state: dict[str, torch.Tensor],
config: dict[str, Any] | None,
) -> None:
component_dir = dst_dir / name
if component_dir.exists() and any(component_dir.iterdir()):
shutil.rmtree(component_dir)
component_dir.mkdir(parents=True, exist_ok=True)
save_file(dict(state), str(component_dir / "diffusion_pytorch_model.safetensors"))
if config is not None:
config_path = component_dir / config_filename(name)
with config_path.open("w", encoding="utf-8") as f:
json.dump(config, f, indent=2)
f.write("\n")
print(f"Wrote {name}: {len(state)} tensors")
def copy_passthrough(src_dir: Path, dst_dir: Path) -> list[str]:
copied: list[str] = []
for subfolder in PASSTHROUGH_SUBFOLDERS:
src = src_dir / subfolder
if not src.is_dir():
continue
dst = dst_dir / subfolder
if dst.exists():
shutil.rmtree(dst)
shutil.copytree(src, dst)
copied.append(subfolder)
print(f"Copied {subfolder}/")
return copied
def default_monolithic_checkpoint(src_path: Path) -> Path:
if src_path.is_file():
return src_path
return src_path / "model.safetensors"
def convert(
src: str,
dst: str,
layout: str,
revision: str | None,
) -> None:
src_path = resolve_src(src, revision)
dst_dir = Path(dst)
dst_dir.mkdir(parents=True, exist_ok=True)
model_index_path = dst_dir / "model_index.json"
if layout in {"monolithic", "raw_official"}:
# TODO: replace model.safetensors with the official monolithic file name.
components = split_monolithic(load_checkpoint(default_monolithic_checkpoint(src_path)))
elif layout in {"separate_components", "mixed"}:
if not src_path.is_dir():
raise ValueError(f"{layout} layout requires a source directory: {src_path}")
components = load_separate_components(src_path)
else:
raise ValueError(f"Unsupported template layout: {layout}")
copied = (copy_passthrough(src_path, dst_dir) if src_path.is_dir() else [])
configs = build_component_configs(src_path if src_path.is_dir() else src_path.parent)
validate_component_configs(configs)
for name, state in components.items():
write_component(dst_dir, name, state, configs.get(name))
available = set(components) | set(copied)
with model_index_path.open("w", encoding="utf-8") as f:
json.dump(build_model_index(src, revision, available), f, indent=2)
f.write("\n")
print(f"Wrote {dst_dir / 'model_index.json'}")
verify_conversion(dst_dir, components)
def main() -> None:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--src", required=True, help="HF repo id, local dir, or checkpoint path")
parser.add_argument("--revision", help="HF branch, tag, or commit for repo sources")
parser.add_argument(
"--dst",
required=True,
help="Output converted_weights/<model_family> directory",
)
parser.add_argument(
"--layout",
choices=("raw_official", "monolithic", "separate_components", "mixed"),
required=True,
help="Official source layout",
)
args = parser.parse_args()
convert(args.src, args.dst, args.layout, args.revision)
if __name__ == "__main__":
main()
-214
View File
@@ -1,214 +0,0 @@
---
name: add-model-08-trace
description: Use during /add-model Phase 6 when component parity has failed and root cause requires layer-by-layer divergence analysis. Uses FastVideo activation trace first, falling back to custom hooks only for boundaries or stats the utility cannot observe.
---
# Add-Model Trace
## Manual Invocation
Load this skill when `/add-model` Phase 6 component parity has failed and the
root cause requires layer-by-layer divergence analysis. This skill is not
auto-fired. The calling subagent (DiT, VAE, encoder, or generic port skill)
loads it when its standard parity-debug loop hits a wall and cannot isolate
the divergence from end-to-end tensor comparisons alone.
Do not load this skill for first-pass parity failures. Try weight-diff and
end-to-end tensor comparison first. Load this skill only when those do not
isolate the cause.
## Goal
Find the first numerical divergence point between FastVideo's port and the
official reference, layer by layer, by instrumenting both sides at matching
tensor boundaries. The investigation must leave zero source residue in
production code when it closes.
## When To Run
After a component parity test FAILS at a bf16-noise-realistic tolerance AND
the calling subagent's first-pass debug (weight-diff, end-to-end tensor
compare) does not isolate the cause.
Required inputs before starting:
- A working FastVideo loader for the component under investigation.
- A working official loader, typically via
`tests/local_tests/helpers/<family>_upstream.py::load_upstream_<component>`.
- Shared deterministic test inputs (same tensors on both sides).
- The component parity test file path and its current failure output.
## Primary Path: FastVideo Activation Trace
Use FastVideo's first-class activation trace before writing custom hooks:
`fastvideo/hooks/activation_trace.py`, documented in
`docs/contributing/activation_trace.md`.
Pipeline runs attach trace to the transformer during pipeline initialization.
Component-only parity harnesses may call `attach_activation_trace(model)` from
local test/debug code; do not add trace calls to production model code.
Prefix the failing parity command with a tight layer regex:
```bash
FASTVIDEO_TRACE_ACTIVATIONS=1 \
FASTVIDEO_TRACE_LAYERS="^block\.layers\.[0-9]+$" \
FASTVIDEO_TRACE_STATS="abs_mean,sum,max,shape" \
FASTVIDEO_TRACE_STEPS="0" \
FASTVIDEO_TRACE_OUTPUT="/tmp/opencode/fv_trace.jsonl" \
pytest tests/local_tests -k "parity" -v -s
```
Match the layer regex to the actual `model.named_modules()` names. Empty or
broad regexes are expensive; prefer block-level names first, then narrow to
submodules after the first divergent block is known.
## Trace Compare Contract
One JSONL file per side. FastVideo output should use `FASTVIDEO_TRACE_OUTPUT`;
the upstream harness should emit the same JSONL shape:
```json
{"module":"block.layers.0","tensor":"out","step":0,"abs_mean":0.0123,"sum":1.0,"max":0.5,"shape":[1,16,32]}
```
Compare rows by `(module, step, tensor)`. The first row whose `shape`,
`abs_mean`, or `max` diverges beyond the component tolerance is the first broken
boundary. Keep `FASTVIDEO_TRACE_LAYERS`, `FASTVIDEO_TRACE_STATS`, and
`FASTVIDEO_TRACE_STEPS` identical between sides; if row order differs, sort or
normalize before diffing.
## Drill-Down Loop
**Initial run:** trace every top-level block (`^block\.layers\.[0-9]+$` or the
family's equivalent). Identify the first block index where `abs_mean` or `max`
drifts beyond tolerance while earlier blocks match.
**Drill run:** tighten `FASTVIDEO_TRACE_LAYERS` to submodules inside the first
divergent block: attention output, MLP projections, norm outputs, modality
adapters, or other named boundaries exposed by `named_modules()`.
**Iterate:** if the first divergent operation is a free function or tensor op not
visible as an `nn.Module`, use the fallback instrumentation hierarchy below.
The loop ends when the first divergent submodule or operation is identified with
a file:line citation in the official source.
## Fallback Instrumentation Hierarchy
Use these only when activation trace cannot observe the needed boundary or
statistic.
### (1) Custom forward hooks
`module.register_forward_hook(...)` and `register_forward_pre_hook(...)`.
Always within `try/finally` with `handle.remove()`. Zero source residue.
### (2) Runtime monkey-patch
`module.attr = wrapped_func` or `cls.method = wrapped_method`, restored via
`try/finally` (save original first). Use for free functions and non-Module sites
such as activation functions (`swiglu`, `apply_rotary_emb`).
### (3) Source edits in FastVideo's own code
Only when (1) and (2) are insufficient. Track all edits within a single named
`git stash` boundary OR a temporary branch. Run `git diff` before closing the
investigation to confirm cleanup.
### (4) Source edits in official repo source
Allowed only when hook and monkey-patch approaches cannot capture the site.
For git-tracked or editable official clones, use `git diff` in the clone path to
verify cleanup. For non-editable site-packages, back up the target file before
editing and restore it before handoff.
## Hypothesis Toggles
Use env-var-gated monkey-patches to A/B test suspect implementations without
source edits. Pattern: `<FAMILY>_DEBUG_PATCH_<HYPOTHESIS>=1`.
Example from the magi-human investigation:
```
MAGI_DEBUG_PATCH_LINEAR=1
```
This patched `PackedExpertLinear.forward` to mirror upstream's
`_BF16ComputeLinear` explicit-cast pattern, isolating a dtype-cast difference
as the root cause.
Document all toggles in the script docstring. Each toggle must:
- save the original before patching;
- restore the original in a `try/finally` block;
- print a `[debug] Patched <ClassName>.<method>` line to stdout when active.
## Cleanup Gate
The calling agent MUST report `[cleanup-gate] PASS` on all five items before
handoff. Do not hand off with any item unresolved.
1. `git diff` in the FastVideo repo: empty. No stray prints, hooks, or
monkey-patches in production code.
2. `git diff` in the official-repo clone (if used): empty. For non-editable
site-packages installs: `diff original.py original.py.trace-backup` is
empty OR `pip install --force-reinstall <pkg>` succeeded and the installed
file matches the original.
3. `git stash list`: only the named investigation stash (or empty). No
unnamed stashes left from this session.
4. No new untracked files outside `/tmp/opencode/` (logs) and the existing
debug script directory (`tests/local_tests/transformers/` or equivalent).
5. `mypy` clean on any production files touched during the investigation.
## Escape Hatches
Escalate to the calling bucket skill when:
- A forward hook on an official module raises because of a custom `forward`
signature or varlen handler args that the hook closure cannot satisfy. The
bucket skill has component-specific knowledge to work around this.
- The first divergent layer is `block[0]`, meaning the divergence is in the
adapter, modality dispatcher, coordinate embedding, or packing step before
any block runs. Check those sites first; the bug is not in attention or MLP.
- Per-block drift is never zero anywhere across all blocks. This usually means
the inputs are not bit-identical between sides. Verify with a state-dict
compare (weight-diff script) AND confirm the input tensors are the same
object or have identical values before the forward call.
## Handoff
Return to the calling subagent with:
- FastVideo trace JSONL path and upstream trace JSONL path.
- Trace settings used: `FASTVIDEO_TRACE_LAYERS`, `FASTVIDEO_TRACE_STATS`, and
`FASTVIDEO_TRACE_STEPS`.
- The first divergent `(module, step, tensor)` row and observed drift.
- The upstream file:line citation where the divergence originates.
- Fallback hook/patch verdict if activation trace could not observe the boundary.
- Hypothesis verdict if an A/B toggle was used, for example `PATCH_LINEAR=1`.
- Cleanup-gate status: `[cleanup-gate] PASS` or a list of unresolved items.
The calling agent uses this to scope the production fix in the FastVideo
component file.
## References
- `docs/contributing/activation_trace.md` for canonical activation-trace env vars,
JSONL output, cost model, and troubleshooting.
- `fastvideo/hooks/activation_trace.py` for the implementation and
`attach_activation_trace(model)` entry point.
- `templates/block_trace_debug.py` in this skill directory: fallback custom-hook
template when activation trace cannot observe the needed boundary or stat.
- `tests/local_tests/transformers/_debug_magi_human_block_parity.py` in the
FastVideo3 repo: historical worked example for custom hook/patch debugging.
- `add-model/SKILL.md` Phase 6: the calling context for this skill.
- `add-model-03-port-dit/SKILL.md`, `add-model-04-port-vae/SKILL.md`,
`add-model-05-port-encoder/SKILL.md`, `add-model-06-port-generic/SKILL.md`:
bucket-specific debug language and component-specific escape-hatch knowledge.
## Changelog
| Date | Change |
|---|---|
| 2026-05-01 | Initial skill extracted from `_debug_magi_human_block_parity.py` pattern. |
@@ -1,334 +0,0 @@
# SPDX-License-Identifier: Apache-2.0
"""Per-block divergence debugger template for FastVideo model ports.
Run directly (not a pytest test):
python tests/local_tests/transformers/_debug_<family>_<component>_parity.py
Generalizes: tests/local_tests/transformers/_debug_magi_human_block_parity.py
Fill FAMILY, COMPONENT, and the two loader functions. Run once for the initial
drift table, then set <FAMILY>_DEBUG_DRILL_LAYER=NN to drill into submodules.
Add <FAMILY>_DEBUG_PATCH_<HYPOTHESIS>=1 to A/B test a suspect implementation.
CLEANUP: all hooks removed in try/finally; monkey-patches restored in
try/finally; source edits tracked in a named git stash. Zero source residue.
See add-model-08-trace/SKILL.md for the full cleanup gate checklist.
"""
from __future__ import annotations
import gc
import os
import sys
from pathlib import Path
from typing import Any
import torch
FAMILY: str = "<family>" # e.g. "magi_human", "ltx2", "wan"
COMPONENT: str = "<component>" # e.g. "dit", "vae", "encoder"
DRILL_LAYER_ENV: str = "<FAMILY>_DEBUG_DRILL_LAYER"
HYPOTHESIS_ENV: str = "<FAMILY>_DEBUG_PATCH_<HYPOTHESIS>"
REL_THRESHOLD: float = 0.005 # 0.5% abs_mean drift flags a block as divergent
LOG_DIR: Path = Path("/tmp/opencode")
REPO_ROOT = Path(__file__).resolve().parents[3]
sys.path.insert(0, str(REPO_ROOT))
def load_official(device: torch.device) -> torch.nn.Module:
"""Load the official upstream model. TODO: implement for your family.
Example (magi-human):
from tests.local_tests.helpers.magi_human_upstream import install_stubs, load_upstream_dit
install_stubs()
return load_upstream_dit(base_shard_dir, device=device, dtype=None)
"""
raise NotImplementedError(f"Fill load_official() for {FAMILY}/{COMPONENT}.")
def load_fastvideo(device: torch.device) -> torch.nn.Module:
"""Load the FastVideo-native model. TODO: implement for your family.
Example (magi-human):
from fastvideo.configs.models.dits.magi_human import MagiHumanVideoConfig
from fastvideo.models.dits.magi_human import MagiHumanDiT
from safetensors.torch import load_file; import glob
fv = MagiHumanDiT(MagiHumanVideoConfig())
state = {}
for shard in sorted(glob.glob(str(transformer_dir / "*.safetensors"))): state.update(load_file(shard))
fv.load_state_dict(state, strict=False); return fv.to(device).eval()
"""
raise NotImplementedError(f"Fill load_fastvideo() for {FAMILY}/{COMPONENT}.")
def build_inputs(device: torch.device) -> dict[str, Any]:
"""Return deterministic inputs shared by both sides. TODO: replace.
Both sides must receive the SAME tensors (clone before each forward call).
Non-identical inputs cause non-zero drift everywhere.
"""
torch.manual_seed(0)
return {"x": torch.randn(64, 1024, dtype=torch.bfloat16, device=device)}
def _stat(name: str, t: torch.Tensor) -> dict:
f = t.detach().float()
return {
"name": name,
"shape": tuple(t.shape),
"abs_mean": f.abs().mean().item(),
"sum": f.sum().item(),
"min": f.min().item(),
"max": f.max().item(),
}
def _attach_block_hooks(
model: torch.nn.Module,
label: str,
log: list[dict],
tensors: dict[str, torch.Tensor] | None = None,
drill_layer: int | None = None,
) -> list[Any]:
"""Return hook handles. Caller MUST remove them in try/finally."""
handles: list[Any] = []
def _hook(name: str):
def fn(_module, _inputs, outputs):
t = outputs[0] if isinstance(outputs, tuple) else outputs
if not torch.is_tensor(t):
return
log.append({"side": label, **_stat(name, t)})
if tensors is not None:
tensors[name] = t.detach().float().cpu()
return fn
def _pre_hook(name: str):
# Pre-hooks observe a free function's output by intercepting the next
# module's input (useful when the activation is not an nn.Module).
def fn(_module, inputs):
t = inputs[0] if isinstance(inputs, tuple) else inputs
if not torch.is_tensor(t):
return
key = f"{name}<in>"
log.append({"side": label, **_stat(key, t)})
if tensors is not None:
tensors[key] = t.detach().float().cpu()
return fn
# TODO: adapt attribute paths to your model. Remove adapter block if absent.
if hasattr(model, "adapter"):
handles.append(model.adapter.register_forward_hook(_hook("adapter")))
# TODO: adapt model.block.layers to your block container.
# Alternatives: model.transformer.layers, model.blocks, model.layers
block_layers = model.block.layers # type: ignore[attr-defined]
for i, layer in enumerate(block_layers):
handles.append(layer.register_forward_hook(_hook(f"block[{i:02d}]")))
if drill_layer is not None and i == drill_layer:
tag = f"L{i:02d}"
# TODO: adapt submodule names to your layer's attributes.
# magi-human uses: attention, mlp.pre_norm, mlp.up_gate_proj,
# mlp.down_proj (pre+post), mlp, attn_post_norm, mlp_post_norm.
if hasattr(layer, "attention"):
handles.append(layer.attention.register_forward_hook(_hook(f"{tag}.attention")))
if hasattr(layer, "mlp"):
mlp = layer.mlp
if hasattr(mlp, "pre_norm"):
handles.append(mlp.pre_norm.register_forward_hook(_hook(f"{tag}.mlp.pre_norm")))
if hasattr(mlp, "up_gate_proj"):
handles.append(mlp.up_gate_proj.register_forward_hook(_hook(f"{tag}.mlp.up_gate_proj")))
if hasattr(mlp, "down_proj"):
handles.append(mlp.down_proj.register_forward_pre_hook(_pre_hook(f"{tag}.mlp.down_proj")))
handles.append(mlp.down_proj.register_forward_hook(_hook(f"{tag}.mlp.down_proj")))
handles.append(mlp.register_forward_hook(_hook(f"{tag}.mlp")))
if hasattr(layer, "attn_post_norm"):
handles.append(layer.attn_post_norm.register_forward_hook(_hook(f"{tag}.attn_post_norm")))
if hasattr(layer, "mlp_post_norm"):
handles.append(layer.mlp_post_norm.register_forward_hook(_hook(f"{tag}.mlp_post_norm")))
return handles
def _apply_hypothesis_patch() -> bool:
"""Apply an optional monkey-patch gated by HYPOTHESIS_ENV. TODO: implement.
Pattern: save original on the class, patch, restore in _restore_hypothesis_patch().
"""
if os.getenv(HYPOTHESIS_ENV) != "1":
return False
# TODO: import FastVideo class, save original, apply patch.
print(f"[debug] Hypothesis patch {HYPOTHESIS_ENV}=1 applied.")
return True
def _restore_hypothesis_patch() -> None:
if os.getenv(HYPOTHESIS_ENV) != "1":
return
# TODO: restore original, e.g.: _mod.TargetClass.method = _mod._ORIGINAL_METHOD
def _write_log(entries: list[dict], path: Path) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
with open(path, "w") as f:
for e in entries:
f.write(f"{e['name']} {e['shape']} "
f"{e['abs_mean']:.8f} {e['sum']:.4f} "
f"{e['min']:.6f} {e['max']:.6f}\n")
def _sort_key(name: str, drill_layer: int) -> tuple:
if name == "adapter":
return (0, "")
if name.startswith(f"L{drill_layer:02d}."):
sub_order = {
"attention": 0,
"attn_post_norm": 1,
"mlp.pre_norm": 2,
"mlp.up_gate_proj": 3,
"mlp.down_proj<in>": 4,
"mlp.down_proj": 5,
"mlp": 6,
"mlp_post_norm": 7,
}.get(name.split(".", 1)[1], 9)
return (1, f"block[{drill_layer:02d}]", sub_order)
if name.startswith("block["):
return (1, name, 99)
return (2, name, 0)
def _print_table(by_name: dict[str, dict], drill_layer: int) -> int | None:
hdr = (f"{'name':<18} {'up_shape':<22} {'up_absmean':>12} {'fv_absmean':>12} "
f"{'absmean_diff':>14} {'rel%':>8} {'up_sum':>14} {'fv_sum':>14} {'sum_diff':>12}")
print(f"\n{hdr}\n{'-' * len(hdr)}")
first_div: int | None = None
for name in sorted(by_name.keys(), key=lambda n: _sort_key(n, drill_layer)):
d = by_name[name]
up, fv = d.get("up"), d.get("fv")
if up is None or fv is None:
continue
am_diff = abs(up["abs_mean"] - fv["abs_mean"])
am_rel = am_diff / max(up["abs_mean"], 1e-9)
sum_diff = abs(up["sum"] - fv["sum"])
flag = ""
if name.startswith("block[") and am_rel > REL_THRESHOLD:
flag = " <<< DIVERGE"
if first_div is None:
first_div = int(name[len("block["):-1])
print(f"{name:<18} {str(up['shape']):<22} {up['abs_mean']:>12.6f} "
f"{fv['abs_mean']:>12.6f} {am_diff:>14.6f} {am_rel * 100:>7.3f}% "
f"{up['sum']:>14.4f} {fv['sum']:>14.4f} {sum_diff:>12.4f}{flag}")
return first_div
def _print_elementwise(up_t: dict[str, torch.Tensor], fv_t: dict[str, torch.Tensor], drill_layer: int) -> None:
common = set(up_t.keys()) & set(fv_t.keys())
if not common:
return
hdr = f"{'name':<30} {'shape':<22} {'diff_max':>12} {'diff_mean':>12} {'diff_rel%':>10}"
print(f"\nElement-wise diffs for drilled L{drill_layer:02d} submodules:\n{hdr}\n{'-' * len(hdr)}")
for name in sorted(common):
a, b = up_t[name], fv_t[name]
if a.shape != b.shape:
continue
diff = (a - b).abs()
rel = (diff.mean().item() / max(a.abs().mean().item(), 1e-9)) * 100
print(f"{name:<30} {str(tuple(a.shape)):<22} "
f"{diff.max().item():>12.6f} {diff.mean().item():>12.6f} {rel:>9.4f}%")
def main() -> None:
if not torch.cuda.is_available():
print("Need CUDA. Skipping.")
return
# TODO: add precondition checks (official clone present, weights available).
drill_layer = int(os.getenv(DRILL_LAYER_ENV, "0"))
device = torch.device("cuda:0")
patched = _apply_hypothesis_patch()
try:
inputs = build_inputs(device)
print("Loading official model...")
official = load_official(device)
up_log: list[dict] = []
up_t: dict[str, torch.Tensor] = {}
up_handles = _attach_block_hooks(official, "up", up_log, up_t, drill_layer)
print("Running official forward (with hooks)...")
try:
with torch.inference_mode():
# TODO: adapt forward call signature to your component.
ref_out = official(**{k: v.clone() for k, v in inputs.items()})
if isinstance(ref_out, dict):
sample = ref_out.get("sample")
ref_out = sample if sample is not None else ref_out.get("x")
elif hasattr(ref_out, "sample"):
ref_out = ref_out.sample
elif isinstance(ref_out, tuple):
ref_out = ref_out[0]
assert torch.is_tensor(ref_out), f"official output is not tensor: {type(ref_out)}"
ref_out = ref_out.detach().float().cpu()
finally:
for h in up_handles:
h.remove()
del official
gc.collect()
torch.cuda.empty_cache()
print("Loading FastVideo model...")
fv = load_fastvideo(device)
fv_log: list[dict] = []
fv_t: dict[str, torch.Tensor] = {}
fv_handles = _attach_block_hooks(fv, "fv", fv_log, fv_t, drill_layer)
print("Running FastVideo forward (with hooks)...")
try:
with torch.inference_mode():
# TODO: adapt forward call signature to your component.
fv_out = fv(**{k: v.clone() for k, v in inputs.items()})
if isinstance(fv_out, dict):
sample = fv_out.get("sample")
fv_out = sample if sample is not None else fv_out.get("x")
elif hasattr(fv_out, "sample"):
fv_out = fv_out.sample
elif isinstance(fv_out, tuple):
fv_out = fv_out[0]
assert torch.is_tensor(fv_out), f"FastVideo output is not tensor: {type(fv_out)}"
fv_out = fv_out.detach().float().cpu()
finally:
for h in fv_handles:
h.remove()
finally:
_restore_hypothesis_patch()
LOG_DIR.mkdir(parents=True, exist_ok=True)
up_path = LOG_DIR / f"{FAMILY}_{COMPONENT}_up_layers.log"
fv_path = LOG_DIR / f"{FAMILY}_{COMPONENT}_fv_layers.log"
_write_log(up_log, up_path)
_write_log(fv_log, fv_path)
print(f"\nLogs: {up_path} {fv_path}\nDiff: diff {up_path} {fv_path}")
by_name: dict[str, dict] = {}
for entry in up_log + fv_log:
by_name.setdefault(entry["name"], {})[entry["side"]] = entry
first_div = _print_table(by_name, drill_layer)
print()
if first_div is not None:
print(f"First block exceeding {REL_THRESHOLD * 100:.2f}% drift: block[{first_div:02d}]")
print(f"Re-run with {DRILL_LAYER_ENV}={first_div} to drill submodules.")
else:
print(f"No block exceeded {REL_THRESHOLD * 100:.2f}% -- divergence is amortized or pre-block.")
diff = (ref_out - fv_out).abs()
print(f"\nFinal ref_abs={ref_out.abs().mean():.6f} fv_abs={fv_out.abs().mean():.6f} "
f"diff_max={diff.max():.6f} diff_mean={diff.mean():.6f}")
_print_elementwise(up_t, fv_t, drill_layer)
if patched:
print(f"\n[debug] Hypothesis {HYPOTHESIS_ENV}=1 was active this run.")
if __name__ == "__main__":
main()
@@ -1,255 +0,0 @@
---
name: add-model-09-pipeline
description: Use during /add-model Phase 7 after all required component parity tests pass to define FastVideo pipeline wiring, configs, presets, registry entries, examples, smoke tests, and pipeline parity tests.
---
# Add Model Pipeline
## Goal
Implement and verify the end-to-end FastVideo pipeline after the native
components and converted weights have passed non-skip component parity. This
skill owns pipeline class/stage wiring, pipeline configs, presets, registry
entries, examples, smoke tests, and pipeline parity-debug.
FastVideo has one pipeline architecture: stage-based composition through
`ComposedPipelineBase`. Add or specialize stages only when existing stages cannot
represent the official behavior safely.
## Hard Gate
Do not start pipeline work until every required component, including reused
components, has a non-skip local parity PASS.
If any component row is missing, skipped, red, or blocked, return to `/add-model`
Phase 6. Pipeline parity cannot distinguish stage wiring mistakes from broken
component numerics when component parity is still unresolved.
## Inputs
Follow `../add-model/shared/common_rules.md` for token/auth safety, state files,
escape hatches, production boundaries, and skip/pass semantics.
Require a complete packet matching
`../add-model/contracts/pipeline_context.md`.
The packet must include:
- official pipeline files and official call/default sources;
- workload types, input/output modalities, and output contract;
- converted or source `model_index.json` path;
- component parity rows, all `non_skip_pass`;
- target FastVideo pipeline/config/preset/registry/example/test paths;
- `local_tests_readme` and `port_state_file` paths.
## Outputs
- Pipeline package under `fastvideo/pipelines/basic/<family>/`.
- Pipeline config under `fastvideo/configs/pipelines/<family>.py` or a documented
family-local config file when that matches existing project style.
- Presets under `fastvideo/pipelines/basic/<family>/presets.py`.
- Registry updates in `fastvideo/registry.py`.
- Basic example under `examples/inference/basic/basic_<family>*.py`.
- Local smoke and parity tests under `tests/local_tests/pipelines/`.
- Updated `tests/local_tests/<model_family>/README.md`.
- Updated `tests/local_tests/<model_family>/PORT_STATUS.md`.
- Handoff matching `../add-model/contracts/pipeline_handoff.md`.
## Mode: Pipeline Definition
Use this mode first.
1. Read the official pipeline call path before editing FastVideo code.
2. Compare official defaults against the planned FastVideo config and presets:
steps, CFG scales, secondary CFG, flow shift, schedulers, sigmas, seed/RNG,
resolution, frames, FPS, duration, VAE scaling, decode slicing, negative
prompt defaults, and output heads.
3. Create or update the pipeline class with `_required_config_modules` matching
the emitted `model_index.json` and `ComposedPipelineBase.load_modules`.
Runtime pipeline resolution is exact: `model_index.json["_class_name"]` must
match a registered `EntryClass.__name__`, or a wrapper/alias class in
`EntryClass`. Registry detectors do not select the executable pipeline class.
4. Add new public generation kwargs to `fastvideo/api/sampling_param.py` before
examples or presets use them. `SamplingParam.update()` ignores unknown keys
except for logging, and preset defaults apply only to declared fields. Add CLI
args when the option should be available from command-line entrypoints.
5. Put loader-time changes in `load_modules()` or earlier, not
`initialize_pipeline()`. `ComposedPipelineBase.__init__` loads modules before
`post_init()` calls `initialize_pipeline()`, so process-global flags, loader
path rewrites, dtype overrides, and tokenizer path changes needed for loading
cannot be introduced there.
6. Use `self.get_module("transformer_2", None)` and similar optional accessors
for truly optional modules. Do not hard-require optional modules by accident.
7. Avoid mutating class-level `_required_config_modules` in custom code. If a
pipeline needs dynamic modules, copy the list to an instance-owned value or
pass `required_config_modules` explicitly so one pipeline instance cannot leak
module requirements into another.
8. Create the stage chain in official execution order. Prefer existing shared
stages for standard text encoding, timestep preparation, latent preparation,
denoising, and decoding.
9. Add model-specific stages only for family-specific behavior that does not fit
the shared stage contracts.
10. Add pipeline config classes for wiring and runtime defaults. Do not duplicate
component architecture fields unless a loader requires them in the subconfig.
Family-local config files such as
`fastvideo/pipelines/basic/<family>/pipeline_configs.py` are valid only when
`fastvideo/registry.py` imports and registers the classes explicitly.
11. Add `InferencePreset` objects with `model_family`, `name`, `version`,
`defaults`, optional validation-only `stage_schemas`, and an `ALL_PRESETS`
tuple. `stage_schemas` validates user-facing `stage_overrides` names; it does
not drive `create_pipeline_stages()` execution.
12. Register config classes and presets in `fastvideo/registry.py`: add
`register_configs(...)`, import the family's `ALL_PRESETS`, and append it to
`_register_presets()`. Detectors should cover HF paths and `_class_name`
strings for config/preset lookup, but not as a replacement for exact pipeline
class-name resolution.
13. Add a basic example with a user-story docstring and normal file-path inputs
for image, audio, or video references. Keep orchestration glue in the
pipeline or a helper, not in the example.
14. Add a separate smoke test
`tests/local_tests/pipelines/test_<family>_pipeline_smoke.py` that proves
imports, `EntryClass`, registry, presets, config defaults, and at least one
real load/generate path when weights are local. Older local tests sometimes
colocate smoke checks in parity files; new ports should use the separate file
convention.
15. Add or update pipeline parity test scaffolding with
`templates/pipeline_parity_test.py`.
16. Update `local_tests_readme` and `port_state_file` with commands, statuses,
default sources, decisions, and blockers.
Production import boundaries are defined in
`../add-model/shared/common_rules.md`.
## Mode: Pipeline Parity Debug
Run after pipeline definition and after smoke can execute far enough to load the
pipeline. Loop until pipeline parity is a non-skip PASS or a precise blocker is
returned.
Mandatory order:
```bash
pytest tests/local_tests/pipelines/test_<family>_pipeline_smoke.py -v -s
DISABLE_SP=1 pytest tests/local_tests/pipelines/test_<family>_pipeline_parity.py -v -s
python examples/inference/basic/basic_<family>.py
```
Pipeline parity must compare real outputs, not only successful generation:
- denoised latents when decode parity is expensive or nondeterministic;
- decoded videos/images when visual output should be deterministic enough;
- decoded waveform or audio features for audio pipelines;
- separate video and audio targets for joint AV pipelines unless a validated
joint metric exists.
Debug pipeline drift in this order:
1. Confirm both sides use the same component weights and component parity PASS
results are still valid.
2. Align official and FastVideo call arguments, presets, and default values.
3. Align scheduler timesteps, sigmas/noise levels, prediction type, flow shift,
guidance math, and secondary-guidance branches.
4. Align RNG: initial latents/noise, generator device, seed, per-step noise, VAE
sampling, and any official `+1 frame` or crop/slice behavior.
5. Align conditioning: prompt templates, negative prompts, masks, image/audio
preprocessing, modality packing, text truncation, and dtype/autocast.
6. Align decode: latent scaling, per-channel mean/std, tiling flags, output
channel order, sample rate, FPS, and final slicing.
7. Add targeted stage-level diagnostics to identify the first divergent stage.
If stage diagnostics show the first bad stage is transformer/denoising or a
mid-DiT block, enable activation trace before adding ad hoc pipeline prints; see
`docs/contributing/activation_trace.md` and `../add-model-08-trace/SKILL.md`.
Keep `FASTVIDEO_TRACE_LAYERS`, `FASTVIDEO_TRACE_STATS`, and
`FASTVIDEO_TRACE_STEPS` identical across reruns so pipeline parity traces diff
one-to-one.
If the first divergence belongs to component implementation, strict loading, or
conversion mapping, stop pipeline edits and return `next_step=return_to_phase_6`
with the exact failing evidence. Do not patch conversion from this skill.
## Stage And Variant Rules
- Canonical video T2V order: `InputValidationStage`, `TextEncodingStage`,
`ConditioningStage`, `TimestepPreparationStage`, `LatentPreparationStage`,
`DenoisingStage`, `DecodingStage`.
- Canonical video I2V delta adds image loading/encoding and image VAE encoding in
the official order, commonly: `TextEncodingStage`, `ImageEncodingStage`,
`ConditioningStage`, `TimestepPreparationStage`, `LatentPreparationStage`,
`ImageVAEEncodingStage`, `DenoisingStage`, `DecodingStage`.
- Treat `ConditioningStage` as default-present for Wan-style pipelines, but still
follow the reference if another family truly skips or replaces it.
- T2V video pipelines usually use validation, text encoding, conditioning,
timestep preparation, latent preparation, denoising, and decoding.
- I2V adds image loading/encoding and image-latent preparation according to the
official pipeline, not by assuming CLIP or Wan-specific branches.
- Pick image, audio, and video encoders from the reference. Do not assume CLIP or
any other common encoder unless the reference uses it.
- Cross-attention class names are not prescribed; match the family style and
preserve the official tensor contract.
- `WorkloadType` currently has no `T2A`, `A2A`, or `AV` values. Until that enum
is extended, audio-only pipelines may register with `WorkloadType.T2V` and
preset `workload_type="t2v"` as a compatibility shim, but must document the
rationale in code and `PORT_STATUS.md`.
- Audio-only pipelines should not force real video semantics into presets. Use
minimal video-shaped placeholders such as small `height`/`width` and
`num_frames=1` only when shared `VideoGenerator`/validation paths require them,
and document that the real output is audio.
- Record modality-specific shape knobs and output contract in the pipeline
handoff: video uses `height`, `width`, `num_frames`, and `fps`; audio uses
`audio_seconds` and `sampling_rate`; joint AV records both plus whether output
is muxed or paired files.
- Use sibling pipeline classes/configs when required modules, HF repo layout,
stage chains, or inputs differ materially.
- Use one kwargs-driven pipeline class only when variants share weights,
modules, stage chain, and safe call semantics.
- Split later if components diverge, workload tags require separate discovery,
signatures become unsafe, or stage branches become substantial.
- If the DiT branches on `added_kv_proj_dim`, document the T2V/I2V split.
- If the reference uses `transformer_2`, `boundary_ratio`, `guidance_scale_2`, or
DMD step lists, keep those on config, presets, or stages deliberately.
- Support every official output head in scope. If a head is out of scope, record
explicit user approval in `PORT_STATUS.md`.
Pipeline verification order:
```bash
pytest tests/local_tests/pipelines/test_<family>_pipeline_smoke.py -v -s
DISABLE_SP=1 pytest -v -s tests/local_tests/pipelines/test_<family>_pipeline_parity.py
python examples/inference/basic/basic_<family>.py
```
Smoke tests prove loadability only. They are not a substitute for numerical
component or pipeline parity.
## Escape Hatches
Follow `../add-model/shared/common_rules.md`. Pipeline-specific ask cases include
dropping a public mode, modality, or output head; adding a new workload enum;
changing official defaults for user-facing behavior; accepting a known pipeline
parity blocker; running GPU-heavy quality work outside the agreed scope; or
publishing/uploading generated references or converted weights.
## Handoff
Return `../add-model/contracts/pipeline_handoff.md` and update the shared state
files before handoff.
Do not hand back a green pipeline if smoke or parity skipped locally. A skip is a
setup gap, not a pass.
## References
- `fastvideo/pipelines/composed_pipeline_base.py` for module loading and stage
execution.
- `fastvideo/pipelines/basic/wan/` for standard video T2V/I2V/DMD variants.
- `fastvideo/pipelines/basic/stable_audio/` for audio-specific stage composition.
- `fastvideo/configs/pipelines/stable_audio.py` and
`fastvideo/pipelines/basic/stable_audio/presets.py` for config/preset shape.
- `fastvideo/registry.py` for `register_configs(...)` and preset registration.
- `tests/local_tests/pipelines/test_gamecraft_pipeline_parity.py` for latent
parity structure.
- `tests/local_tests/pipelines/test_stable_audio_pipeline_parity.py` for audio
parity structure.
- `tests/local_tests/pipelines/test_stable_audio_pipeline_smoke.py` for no-GPU
import/registry/preset preflight shape.
@@ -1,147 +0,0 @@
# SPDX-License-Identifier: Apache-2.0
"""Pipeline parity scaffold for TODO_MODEL_FAMILY.
Copy this file to
`tests/local_tests/pipelines/test_<family>_pipeline_parity.py` and replace every
TODO before treating it as an executable scaffold.
The filled test should compare denoised latents, decoded media, audio waveform,
or another concrete output from the official pipeline against FastVideo. A
successful generation without tensor/media comparison is not parity.
"""
from __future__ import annotations
import os
import sys
from pathlib import Path
from typing import Any
import pytest
import torch
from torch.testing import assert_close
_REPO_ROOT = Path(__file__).resolve().parents[3]
_MODEL_FAMILY = "TODO_MODEL_FAMILY"
_OFFICIAL_REF_ENV = "TODO_OFFICIAL_REF_PATH"
_OFFICIAL_REF_DEFAULT = _REPO_ROOT / "TODO_OFFICIAL_REF_DIR"
_FASTVIDEO_MODEL_ENV = "TODO_FASTVIDEO_MODEL_PATH"
_FASTVIDEO_MODEL_DEFAULT = _REPO_ROOT / "converted_weights" / _MODEL_FAMILY
def _path_from_env(env_name: str, default: Path) -> Path:
return Path(os.getenv(env_name, str(default))).expanduser()
def _add_official_to_path() -> Path:
official_path = _path_from_env(_OFFICIAL_REF_ENV, _OFFICIAL_REF_DEFAULT)
if not official_path.exists():
pytest.skip(f"Official reference not found at {official_path}")
if str(official_path) not in sys.path:
sys.path.insert(0, str(official_path))
return official_path
def _log_tensor_stats(label: str, tensor: torch.Tensor) -> None:
value = tensor.detach().float()
print(f"[{_MODEL_FAMILY} PIPELINE] {label}: shape={tuple(tensor.shape)} "
f"dtype={tensor.dtype} device={tensor.device} "
f"min={value.min().item():.6f} max={value.max().item():.6f} "
f"mean={value.mean().item():.6f} std={value.std().item():.6f}")
def _extract_tensor(output: Any, key: str) -> torch.Tensor:
if isinstance(output, dict):
value = output.get(key)
else:
value = getattr(output, key, None)
if value is None:
raise AssertionError(f"Pipeline output did not contain {key!r}")
if not torch.is_tensor(value):
try:
import numpy as np
value = torch.from_numpy(np.asarray(value))
except Exception as exc: # pragma: no cover - scaffold guard
raise AssertionError(f"Could not convert {key!r} to tensor") from exc
return value.detach().float().cpu()
def _run_official_pipeline(
official_path: Path,
params: dict[str, Any],
device: torch.device,
) -> Any:
del official_path, params, device
pytest.skip("TODO: import the official pipeline/factory, load official weights, "
"run with params, and return the comparison target.")
def _run_fastvideo_pipeline(model_path: Path, params: dict[str, Any]) -> Any:
from fastvideo import VideoGenerator
generator = VideoGenerator.from_pretrained(
str(model_path),
num_gpus=1,
use_fsdp_inference=False,
dit_cpu_offload=False,
vae_cpu_offload=False,
text_encoder_cpu_offload=False,
)
try:
return generator.generate_video(
prompt=params["prompt"],
negative_prompt=params.get("negative_prompt"),
output_path=f"outputs_{_MODEL_FAMILY}/pipeline_parity",
save_video=False,
height=params.get("height"),
width=params.get("width"),
num_frames=params.get("num_frames"),
fps=params.get("fps"),
num_inference_steps=params["num_inference_steps"],
guidance_scale=params.get("guidance_scale"),
seed=params["seed"],
)
finally:
generator.shutdown()
@pytest.mark.skipif(
not torch.cuda.is_available(),
reason="TODO_MODEL_FAMILY pipeline parity requires CUDA.",
)
def test_todo_model_family_pipeline_official_parity() -> None:
official_path = _add_official_to_path()
fastvideo_model_path = _path_from_env(
_FASTVIDEO_MODEL_ENV,
_FASTVIDEO_MODEL_DEFAULT,
)
if not fastvideo_model_path.exists():
pytest.skip(f"FastVideo model path not found at {fastvideo_model_path}")
device = torch.device("cuda:0")
params = {
"prompt": "TODO: stable parity prompt",
"negative_prompt": "",
"height": 64,
"width": 64,
"num_frames": 9,
"fps": 8,
"num_inference_steps": 4,
"guidance_scale": 1.0,
"seed": 0,
}
official_output = _run_official_pipeline(official_path, params, device)
fastvideo_output = _run_fastvideo_pipeline(fastvideo_model_path, params)
comparison_key = "TODO_COMPARISON_KEY"
official_tensor = _extract_tensor(official_output, comparison_key)
fastvideo_tensor = _extract_tensor(fastvideo_output, comparison_key)
_log_tensor_stats("official", official_tensor)
_log_tensor_stats("fastvideo", fastvideo_tensor)
assert official_tensor.shape == fastvideo_tensor.shape
diff = (official_tensor - fastvideo_tensor).abs()
print(f"diff max={diff.max().item():.6f} "
f"mean={diff.mean().item():.6f} median={diff.median().item():.6f}")
assert_close(fastvideo_tensor, official_tensor, atol=1e-2, rtol=1e-2)
@@ -1,141 +0,0 @@
---
name: add-model-10-pr-review
description: Review rubric for FastVideo PRs that add or modify model families, variants, first-class components, checkpoint conversion, pipelines, parity coverage, or generated-media quality baselines. Use when reviewing a PR whose diff touches fastvideo/models/, fastvideo/pipelines/basic/, fastvideo/registry.py, scripts/checkpoint_conversion/, fastvideo/tests/ssim/, or related model-port surfaces. Pairs with review-pr-link as a project-scoped review pass; produces findings, not fixes.
---
# Add-Model PR Review
Use this skill when a reviewed PR appears to add, port, or substantially modify
a FastVideo model family, model variant, first-class model component,
checkpoint conversion, model pipeline, or local parity coverage.
This is a review skill, not an implementation workflow. Do not run `/add-model`
or start writing missing port code during review. Use the add-model skill stack
as a rubric for findings.
## Trigger Paths
Trigger this skill if `git diff --name-only <base>...HEAD` includes any of:
- `fastvideo/models/dits/`, `fastvideo/configs/models/dits/`
- `fastvideo/models/vaes/`, `fastvideo/configs/models/vaes/`
- `fastvideo/models/encoders/`, `fastvideo/configs/models/encoders/`
- `fastvideo/models/schedulers/`, `fastvideo/configs/models/schedulers/`
- `fastvideo/models/upsamplers/`, `fastvideo/configs/models/upsamplers/`
- `fastvideo/models/audio/`, `fastvideo/configs/models/audio/`
- `fastvideo/pipelines/basic/`, `fastvideo/configs/pipelines/`
- `fastvideo/registry.py`, `fastvideo/api/sampling_param.py`
- `scripts/checkpoint_conversion/`
- `examples/inference/basic/`
- `tests/local_tests/`, especially component or pipeline parity tests
- `fastvideo/tests/ssim/` or other quality-regression tests for generated media
Also trigger when the PR title/body claims a new model, model variant, VAE,
encoder, scheduler, conditioner, pipeline, conversion script, or generated-media
quality baseline even if the path list is incomplete.
## Review Inputs
Read these add-model references as review checklists:
- `../add-model/SKILL.md`: phase gates and final handoff requirements.
- `../add-model/shared/common_rules.md`: token/auth safety, production import
boundaries, state files, and skip/pass semantics.
- `../add-model/contracts/final_handoff.md`: final evidence expected from a
complete port.
- `../add-model/contracts/component_context.md` and
`../add-model/contracts/component_skill_handoff.md`: component evidence and
parity-debug expectations.
- `../add-model/contracts/conversion_request.md` and
`../add-model/contracts/conversion_handoff.md`: conversion evidence,
strict-load status, config validation, and retry context.
- `../add-model/contracts/pipeline_context.md` and
`../add-model/contracts/pipeline_handoff.md`: pipeline class/stage/config/
preset/registry/example evidence.
Then read only the satellite skill(s) that match touched areas:
- DiT/transformer changes: `../add-model-03-port-dit/SKILL.md`.
- VAE changes: `../add-model-04-port-vae/SKILL.md`.
- Encoder/conditioner changes: `../add-model-05-port-encoder/SKILL.md`.
- Scheduler/upsampler/vocoder/other components:
`../add-model-06-port-generic/SKILL.md`.
- Component parity tests: `../add-model-02-parity/SKILL.md`.
- Checkpoint conversion: `../add-model-07-conversion/SKILL.md`.
- Pipeline/config/presets/registry/examples:
`../add-model-09-pipeline/SKILL.md`.
- Prep/state docs: `../add-model-01-prep/SKILL.md`.
## Required Review Lanes
For a full model-family or model-variant PR, cover all lanes. For a
component-only PR, cover the component, conversion/parity as applicable, and the
documented downstream consumer.
1. Scope and source-of-truth lane:
Verify the PR clearly identifies the official reference, weights/revision,
supported variants, modalities, output heads, and any approved scope cuts.
2. Component lane:
Verify each required component is FastVideo-native or has a documented and
accepted lazy-wrapper exception. Check bucket/config inheritance, `EntryClass`,
state-dict surface, reused-component evidence, and output heads.
3. Conversion lane:
Verify mappings are derived from prototype key/shape dumps, source layout is
supported, skipped keys are intentional, emitted configs validate through
production paths, component strict-load status is recorded, `model_index.json`
library tokens match loaders, and revisions are pinned when converting from
HF.
4. Component parity lane:
Verify local parity tests exist for every required component, including reused
components. Scaffolds may skip in CI, but the PR must provide local non-skip
PASS evidence or an explicit accepted blocker.
5. Pipeline lane:
Verify stage order, required modules, `_class_name` / `EntryClass.__name__`
resolution, config defaults, presets, `SamplingParam` fields, registry
registration, examples, smoke tests, and pipeline parity.
6. Quality and evidence lane:
Verify media quality regression is added or explicitly deferred, examples run,
generated outputs are non-corrupt, `tests/local_tests/<family>/README.md` and
`PORT_STATUS.md` are current, and final blockers are surfaced in the review.
## Findings To Prioritize
Prioritize review findings in this order:
- Missing or skipped required component parity without accepted blocker.
- Pipeline parity/smoke/example missing or skipped for a pipeline PR.
- Conversion emits unloadable or unvalidated configs/weights.
- Wrong `model_index.json` `_class_name`, component library token, or registry
class resolution.
- Runtime diffusers/transformers model-class imports for components that own
weights or numerical behavior.
- Dropped modalities, output heads, variants, or conditioning streams without
explicit approval.
- Reused FastVideo component lacks exact definition/instantiation proof or
non-skip parity.
- Public generation kwargs/preset defaults missing from `SamplingParam`.
- Tests only check shapes, importability, or successful generation without
numerical/media comparison.
- Tokens, credentials, reference clones, staged weights, or generated bulk assets
committed to the PR.
## Output Format
Write normal code-review findings first, ordered by severity. Include file and
line references from the PR diff when possible.
Use this phrasing for missing add-model evidence:
```text
This PR does not satisfy the add-model <component|conversion|pipeline|final>
gate because <specific required evidence> is missing. The risk is <runtime load,
numerical parity, dropped output, registry resolution, etc.>.
```
Keep the summary short. Mention which lanes were reviewed and which could not be
verified because assets, GPU time, or external credentials were unavailable.
-438
View File
@@ -1,438 +0,0 @@
---
name: add-model
description: Manual /add-model workflow for implementing a FastVideo model or first-class component port after add-model-01-prep has staged reference code and weights. Organizes the port into numbered phases with conversion rules, component policies, parity gates, and handoff checks.
---
# Add Model
## Manual Invocation
This skill is for explicit `/add-model` use only. Do not auto-start it from a
casual model-port mention. The setup-only workflow is
`../add-model-01-prep/SKILL.md`.
## Goal
Port a new FastVideo model family, model variant, or first-class reusable
component so it can be loaded through FastVideo's native model, config, stage,
registry, preset, and test infrastructure.
FastVideo has one pipeline architecture: stage-based composition via
`ComposedPipelineBase`. Vary the stages and modules, not the architecture.
## Scope Shapes
Use this skill for either shape:
| Shape | Required output |
|---|---|
| Full model family or variant | Native components, conversion if needed, pipeline config/class, presets, registry, smoke test, local parity tests, example, quality regression. |
| First-class component contribution | Native component class/config, bucket export, component parity test, and a documented downstream pipeline that will consume it. Skip pipeline/preset/registry rows only when the contribution is intentionally component-only. |
If upstream ships many variants, lock scope before coding. "Base model" means
checkpoint variant, not a modality subset. If the base checkpoint produces
audio, pose, depth, masks, or other output heads, either support those outputs
or get explicit user agreement to drop them.
## Required Input
Start from an `add-model-01-prep` handoff, or equivalent fields matching
`contracts/prep_handoff.md`.
Before Phase 0, read the shared rules and all relevant schemas:
- `shared/common_rules.md`
- `contracts/prep_handoff.md`
- `contracts/port_state.md`
- `contracts/escape_hatch.md`
- `contracts/component_context.md`
- `contracts/parity_status.md`
- `contracts/conversion_request.md`
- `contracts/conversion_handoff.md`
- `contracts/component_skill_handoff.md`
- `contracts/pipeline_context.md`
- `contracts/pipeline_handoff.md`
- `contracts/final_handoff.md`
## Hard Rules
- Follow `shared/common_rules.md` for token/auth safety, state files, escape
hatches, production import boundaries, and skip/pass semantics.
- If the prep handoff is missing or ambiguous, stop and run
`../add-model-01-prep/SKILL.md`.
- If a needed component is not ported, do not ship the pipeline that needs it.
- Wan is grandfathered for missing local parity; do not copy its missing-test
precedent for new work.
## Escape Hatches
Follow `shared/common_rules.md` and `contracts/escape_hatch.md`. The main
orchestrator should ask only when no phase skill can safely continue under the
shared rules.
## Files Map
| Area | Paths |
|---|---|
| DiT | `fastvideo/models/dits/<family>.py`, `fastvideo/configs/models/dits/<family>.py`, bucket `__init__.py`. |
| VAE | `fastvideo/models/vaes/<arch_or_family>.py`, `fastvideo/configs/models/vaes/<arch_or_family>.py`, bucket `__init__.py`. Name by shared arch when reusable (`oobleck.py`, `autoencoder_kl.py`), otherwise by family (`wanvae.py`). |
| Encoder / conditioner / scheduler / upsampler | Native class/config in the matching `fastvideo/models/<bucket>/` and `fastvideo/configs/models/<bucket>/` bucket. |
| Lazy loader wrapper | Optional `fastvideo/models/<bucket>/<family>_loader.py` or similar thin `nn.Module` wrapper when a component is fetched from an external HF repo and should be hidden from host-pipeline state-dict matching. |
| Conversion | `scripts/checkpoint_conversion/<family>_to_diffusers.py` only when `needs_conversion=yes`. |
| Pipeline | `fastvideo/pipelines/basic/<family>/<family>_pipeline.py` plus sibling files for variants whose components or required modules differ. |
| Pipeline config | `fastvideo/configs/pipelines/<family>.py` or `fastvideo/pipelines/basic/<family>/pipeline_configs.py`. |
| Stages | `fastvideo/pipelines/basic/<family>/stages/` only for model-specific stage subclasses. |
| Presets / registry | `fastvideo/pipelines/basic/<family>/presets.py`, `fastvideo/registry.py`. |
| Tests | Component parity under `tests/local_tests/<bucket>/`; pipeline smoke/parity under `tests/local_tests/pipelines/`; CI-backed quality tests under `fastvideo/tests/`. |
| Example | `examples/inference/basic/basic_<family>*.py`, one per public mode/variant. |
## Phase 0: Scope And Handoff Gate
1. Validate every required handoff field.
2. Resolve `needs_conversion=unknown` before component work:
```bash
python ".agents/skills/add-model-01-prep/scripts/inspect_hf_layout.py" \
"<hf-or-local-path>" \
--json
```
3. List first-PR scope across both axes:
- Variant axis: base, distill, SR/refine, causal, DMD, I2V, V2V, etc.
- Modality axis: video, image, audio, pose, depth, masks, text, etc.
4. For component-only work, explicitly name the downstream full-pipeline PR or
planned consumer.
5. Confirm `official_env_status` is `imports_ok` or
`private_deps_need_stubs`. If it is `blocked`, return to
`../add-model-01-prep/SKILL.md` before parity scaffolding.
6. Confirm `local_tests_readme` exists and records official setup, HF weights,
dependency changes, and planned parity commands for reviewers.
7. Confirm `port_state_file` exists, follows `contracts/port_state.md`, and has
rows for open questions/issues found during prep.
8. If there are multiple official implementations, choose the one whose
architecture matches the published weights. A blessed library port can be a
better parity reference than a highly configurable research repo; document
the choice in tests.
## Phase 1: Reference And Architecture Study
Read the official pipeline call path before writing code.
Record:
- Required modules from `model_index.json` or equivalent: transformer, VAE,
text encoders, tokenizers, scheduler, image encoders, audio VAE, vocoder,
conditioners, upsamplers.
- Input/output modalities and every dedicated DiT output head.
- Text/image/audio encoding flow, latent shape, dtype, scaling, packing,
scheduler/timestep math, guidance math, VAE normalization, and decode flow.
- Whether the official code relies on private deps, custom ops, or special
kernels that parity tests must stub.
Arch config rule:
- `ArchConfig` fields must match the emitted per-component config, especially
`transformer/config.json`, one-to-one.
- Pipeline knobs do not belong on the DiT arch config: inference steps, CFG
scales, flow shift, FPS, VAE stride, text target length, data-proxy knobs,
eval defaults, and sampling defaults go on `PipelineConfig`, presets, or
stages.
- If the HF repo is raw or has empty configs, synthesize
`transformer/config.json` from the official Python model-config class, not
from data/eval config classes.
## Phase 2: Early Parity Scaffolding
Create component parity tests before or alongside implementation. Use
`../add-model-02-parity/SKILL.md` and its `templates/component_parity_test.py`.
The official reference must import in the current FastVideo environment, or the
prep handoff must identify private deps that will be stubbed locally for tests.
Use `local_tests_readme` as the reviewer-facing source for setup commands and
update its planned test table as parity scaffolds are added.
This phase is early by design:
- Official loading can be implemented from the reference study.
- FastVideo loading can target planned standardized class/config/loader paths.
- Tests may initially skip because the FastVideo class or converted weights do
not exist yet.
- The scaffold must still contain real official loading, deterministic inputs,
output extraction, and concrete tensor comparisons. No unconditional skips,
no shape-only tests.
Use subagents here: dispatch one parity-test subagent per required component,
including components that may be reused. Their output becomes the red/skip
target that porting or reuse-verification subagents make pass later.
## Phase 3: Reuse Gate And Component Dispatch
Build a component inventory before implementation:
| Field | Meaning |
|---|---|
| Component | transformer, VAE, text encoder, image encoder, scheduler, conditioner, upsampler, vocoder, etc. |
| Official definition | Repo-relative source file, class/function name, and relevant line/range if known. |
| Official instantiation | Repo-relative pipeline/config/factory call site plus constructor args and runtime flags. |
| FastVideo target | Existing class to reuse or new bucket/file/config to add. |
| Parity test | Required local test path, including reused components. |
| Status | `reuse_pending`, `reuse_proven`, `port_pending`, `non_skip_pass`, or `blocked`. |
Reuse is allowed only from the checked-out FastVideo tree. Do not wait for or
depend on an open PR adding a native class; add the native port directly in this
PR if the current tree cannot be reused.
Reuse decision:
1. Record exact official definition and instantiation evidence for every
component.
2. If an existing FastVideo class and config match both definition and
instantiation, pass that reused target to the bucket-specific skill in
`mode=prototype` and require reuse evidence plus key/shape dumps.
3. If either definition or instantiation differs, port the component directly as
FastVideo-native code through the bucket-specific skill.
4. Reused components still require non-skip component parity against the exact
official instantiation used by the target pipeline.
Porting subagent dispatch:
- Dispatch one subagent per component after Phase 2 parity scaffolds exist.
- Use `../add-model-03-port-dit/SKILL.md` for DiTs/transformers.
- Use `../add-model-04-port-vae/SKILL.md` for VAEs.
- Use `../add-model-05-port-encoder/SKILL.md` for text, image, audio, or compound
encoders/conditioners that fit the encoder config bucket.
- Use `../add-model-06-port-generic/SKILL.md` for schedulers, upsamplers,
vocoders, adapters, preprocessors, or unknown components.
- Each subagent owns one component only and must loop on that component's local
parity test until it produces a non-skip PASS or returns a precise blocker.
Every component subagent must receive a complete packet matching
`contracts/component_context.md`. If any required path is unknown, pass `unknown`
plus the exact search already performed. Do not silently omit ambiguous official
files or prototype concerns.
Bucket, layer, and attention rules live in the bucket-specific skills and
`fastvideo/layers/AGENTS.md`.
## Phase 4: Native Component Prototype
Conversion needs a FastVideo state-dict surface. Use the Phase 3
bucket-specific skill in `mode=prototype` for every required component, including
reused components.
Prototype success criteria:
- the FastVideo-native or reused class/config can import and instantiate with the
exact official architecture args;
- official and FastVideo key/shape dumps exist for every stateful component;
- `local_tests_readme` and `port_state_file` record prototype status and concerns;
- the returned handoff matches `contracts/component_skill_handoff.md`.
Do not chase numerical parity in Phase 4. Prototype mode ends when conversion has
the key/shape surface it needs, or when the component skill returns a precise
blocker or escape hatch.
## Phase 5: Param Mapping And Weight Conversion
Use `../add-model-07-conversion/SKILL.md` after Phase 4 prototypes exist.
Send a request matching `contracts/conversion_request.md`; consume the returned
`contracts/conversion_handoff.md` update before Phase 6.
Use the prep handoff's `needs_conversion` value:
- `no`: verify the source already has the component layout FastVideo loaders can
consume, then record any passthrough components.
- `yes`: write `scripts/checkpoint_conversion/<family>_to_diffusers.py` and
output `converted_weights/<family>/`.
- `unknown`: return to Phase 0.
The conversion skill owns source-layout handling, mapping derivation, config and
`model_index.json` emission, passthrough assets, strict-load verification, and
Phase 6 retry requests. Component skills must not patch conversion scripts or
converted weights ad hoc.
## Phase 6: Component Parity Debug
This is the expected expensive loop. Dispatch one subagent per required
component, including reused components, using the bucket-specific skill in
`mode=parity-debug`.
Each subagent gets:
- the complete component context packet from Phase 3/4;
- updated conversion mapping notes and strict-load result from Phase 5;
- any prototype concerns or unknowns that were not resolved before conversion.
The bucket-specific skills own parity-debug tactics. If a failure belongs to
conversion, route it through `../add-model-07-conversion/SKILL.md` with a retry
request matching `contracts/conversion_request.md`, then resume the component
skill with the updated conversion handoff.
When a component failure narrows to layer-by-layer numerical drift, load
`../add-model-08-trace/SKILL.md` before writing custom hooks. It uses
`fastvideo/hooks/activation_trace.py`; canonical env vars and JSONL format are
documented in `docs/contributing/activation_trace.md`.
Phase 6 ends only when every required component handoff reports
`parity_status=non_skip_pass`, or when a precise blocker or escape hatch is
recorded in `port_state_file`.
## Phase 7: Pipeline, Stages, And Variants
Do not start Phase 7 until every required component, reused or ported, has a
non-skip local parity PASS from Phase 6. If any component parity test is still
`scaffold_skip`, `debug_red`, `blocked`, or missing, resume Phase 6 first.
Use `../add-model-09-pipeline/SKILL.md` for pipeline definition and parity-debug.
Send a complete packet matching `contracts/pipeline_context.md`; consume the
returned `contracts/pipeline_handoff.md` before moving to quality regression or
final handoff.
The pipeline skill owns:
- pipeline class, stage chain, and optional model-specific stages;
- pipeline config, presets, registry updates, and examples;
- official args/defaults/presets comparison before setting FastVideo defaults;
- pipeline smoke and parity tests;
- continuous pipeline parity-debug until non-skip PASS or precise blocker;
- updates to `local_tests_readme` and `port_state_file`.
The pipeline handoff must explicitly cover stage order, variants, modality and
output-head handling, config/preset/registry/example status, smoke/parity tests,
and any return-to-Phase-6 evidence.
## Phase 8: PipelineConfig, Presets, Registry, Examples
This phase is implemented through `../add-model-09-pipeline/SKILL.md` after the
Phase 7 component-parity gate passes. Accept the pipeline handoff only if it
covers configs, presets, registry detection/exact class resolution, examples,
new `SamplingParam` fields for public kwargs/defaults, and local smoke/parity
status. Detailed rules live in `../add-model-09-pipeline/SKILL.md`.
## Phase 9: Parity Activation And Local Verification
Local parity is author-run, not CI-enforced. CI may only run package-level
quality tests later. Before handoff, Phase 2 scaffolds must be activated into
non-skip PASS results.
Order is mandatory:
1. Run conversion if needed.
2. Run component parity for every required component, including reused ones.
3. Run pipeline smoke.
4. Run pipeline parity.
5. Run the basic example.
If pipeline smoke or parity points back to component implementation,
strict-load, or conversion mapping, return to Phase 6 or Phase 5 rather than
patching around the issue in the pipeline.
Skip policy:
- Follow `shared/common_rules.md`: a committed local test may skip for absent
clones/weights, but a local skip is not a verified pass.
Use the commands and tolerance guidance from `../add-model-02-parity/SKILL.md` for
component checks and from `../add-model-09-pipeline/SKILL.md` for pipeline smoke,
pipeline parity, and examples. Record exact commands, status, and blockers in
`local_tests_readme` and `port_state_file`.
## Phase 10: Quality Regression
Video outputs:
- Add `fastvideo/tests/ssim/test_<family>_similarity.py` when output video
quality must be preserved.
- Seed references through `seed-ssim-references` after the test exists.
Audio outputs:
- SSIM does not apply. Use an audio-specific regression metric such as
mel-spectrogram L1, multi-resolution STFT, CLAP cosine, or a project-approved
learned metric.
- Document the metric and hardware/runtime assumptions in the test.
Joint AV outputs:
- Keep video and audio regression checks separate unless there is a validated
joint metric.
## Phase 11: Post-Parity Review And Handoff
After parity is green, run a hot-path review before handoff:
- Hoist constant tensor allocations out of sampler/denoising loops.
- Replace per-step `randn_like` churn with preallocated buffers plus
`.normal_()` when safe.
- Move `torch.backends.*` flag changes to one-shot setup/load paths.
- Delete `batch.extra` writes that nothing reads.
- Derive magic constants from configs when possible.
Pre-handoff checklist:
```text
[ ] Prep handoff is complete and committed nowhere with token values.
[ ] Conversion was run if needed and output loads with real weights.
[ ] Every required component, reused or newly ported, has a non-skip local parity PASS.
[ ] `local_tests_readme` lists every component parity test, command, status, and blocker if any.
[ ] `port_state_file` has every open question/issue either resolved or listed as an explicit blocker.
[ ] Any `next_step=ask_user` has a matching `escape_hatch` block and `E###` row.
[ ] Pipeline smoke has a non-skip local PASS.
[ ] Pipeline parity has a non-skip local PASS against the official reference.
[ ] Basic example runs and writes a non-corrupt output.
[ ] Video SSIM or audio-specific quality regression is added or explicitly deferred.
[ ] Runtime production code has no diffusers/transformers model-class imports.
[ ] Production comments are WHY-focused; examples have user-story docstrings.
[ ] Post-parity hot-path pass is complete.
```
Ask before deleting any reference clone or staged weights created by
`add-model-01-prep`. Leave `.gitignore` entries so future parity assets stay
untracked. Never commit the clone, weights, `.env`, credentials, or anything
matching `*secret*`.
## References
- `../add-model-01-prep/SKILL.md` for user-input collection, HF inspection,
weight staging, reference cloning, and setup handoff.
- `contracts/` for canonical handoff schemas used by prep, parity, conversion,
component porting, escape hatches, and final handoff.
- `../add-model-02-parity/SKILL.md` for early component parity scaffolds and
activation templates.
- `../add-model-07-conversion/SKILL.md` for Phase 5 mapping, conversion scripts,
monolithic checkpoint splitting, and strict-load checks.
- `../add-model-03-port-dit/SKILL.md`, `../add-model-04-port-vae/SKILL.md`,
`../add-model-05-port-encoder/SKILL.md`, and
`../add-model-06-port-generic/SKILL.md` for component subagent implementation
and parity-debug loops.
- `../add-model-09-pipeline/SKILL.md` for pipeline definition, config/preset/
registry/example wiring, smoke tests, and pipeline parity-debug.
- `fastvideo/layers/AGENTS.md` for native layer selection and state-dict surface
guidance.
- `docs/contributing/coding_agents.md` for narrative context.
- `docs/design/overview.md` for pipeline/config/registry architecture.
- `fastvideo/pipelines/basic/wan/` for standard T2V/I2V/DMD/Causal variants.
- `fastvideo/pipelines/basic/ltx2/` for non-standard stages and audio/video
patterns.
- `tests/local_tests/pipelines/test_gamecraft_pipeline_parity.py` for pipeline
parity shape.
- `tests/local_tests/transformers/test_ltx2.py`,
`tests/local_tests/vaes/test_ltx2_vae.py`, and
`tests/local_tests/encoders/test_ltx2_gemma_parity.py` for component parity.
- `scripts/checkpoint_conversion/convert_ltx2_weights.py` for modern conversion
script shape.
- `scripts/checkpoint_conversion/wan_to_diffusers.py` for legacy regex mapping
reference only.
## Changelog
| Date | Change |
|---|---|
| 2026-04-24 | Initial FastVideo add-model workflow. |
| 2026-04-30 | Split external setup into `add-model-01-prep`. |
| 2026-04-30 | Rewrote as manual `/add-model` phase workflow and incorporated prior review decisions. |
| 2026-04-30 | Extracted early parity scaffolding into `add-model-02-parity` and moved it before conversion/component implementation. |
| 2026-04-30 | Added component reuse proof gate, bucket-specific porting skills, and parity PASS requirement for reused components. |
| 2026-04-30 | Split prototype, conversion, and parity-debug phases; added conversion skill for monolithic and separate checkpoint layouts. |
| 2026-04-30 | Extracted handoff schemas into `contracts/` for shared use across skills. |
| 2026-04-30 | Added pipeline skill contract and Phase 7 component-parity gate. |
| 2026-04-30 | Added escape-hatch contract for user decisions and `ask_user` handoffs. |
@@ -1,29 +0,0 @@
# Add Model Contracts
Canonical handoff schemas for the `/add-model` workflow. When a skill needs to
send or receive structured context, use these files instead of inventing a local
schema.
| Contract | Use |
|---|---|
| `prep_handoff.md` | `add-model-01-prep` output and `/add-model` Phase 0 input. |
| `port_state.md` | Per-port `PORT_STATUS.md` file tracking progress, open questions, and issues. |
| `escape_hatch.md` | Shared pause-and-ask schema for user decisions the workflow cannot safely choose. |
| `component_context.md` | Per-component packet passed to parity, prototype, conversion, and parity-debug subagents. |
| `parity_status.md` | `add-model-02-parity` scaffold/activation status returned to `/add-model`. |
| `conversion_request.md` | Phase 5 conversion input and Phase 6 conversion retry request. |
| `conversion_handoff.md` | `add-model-07-conversion` output back to `/add-model` and component subagents. |
| `component_skill_handoff.md` | Component porting skill output in prototype or parity-debug mode. |
| `pipeline_context.md` | Phase 7 packet passed to `add-model-09-pipeline` after component parity is green. |
| `pipeline_handoff.md` | `add-model-09-pipeline` output back to `/add-model` after pipeline definition or parity-debug. |
| `final_handoff.md` | Final `/add-model` pre-handoff checklist summary. |
Rules:
- Do not omit required fields. Use `unknown` plus the search already performed
when the value is not known yet.
- Do not include raw token values. Use env var names only.
- Keep model-specific mapping details in conversion scripts and the local tests
README/status notes, not in generic skill docs.
- Use `next_step=ask_user` only with an `escape_hatch` block matching
`escape_hatch.md`.
@@ -1,55 +0,0 @@
# Component Context Contract
Canonical per-component packet passed from `/add-model` to parity, prototype,
conversion, and parity-debug subagents.
```text
component_context:
model_family: <snake_case>
component: <name>
component_type: <dit|vae|encoder|scheduler|conditioner|upsampler|vocoder|generic>
mode: parity-scaffold | prototype | parity-debug
official_ref_dir: <path or import path>
official_definition_files:
- path: <repo-relative or absolute path in official repo>
symbols: <class/function names>
notes: <layer graph, output contract, state-dict owner>
official_instantiation_files:
- path: <repo-relative or absolute path in official repo>
symbols: <factory/pipeline/config names>
args: <constructor args, config values, runtime flags>
official_weight_source: <checkpoint file, subfolder, prefix, or passthrough source>
fastvideo_target_files:
- fastvideo/models/<bucket>/<file>.py
- fastvideo/configs/models/<bucket>/<file>.py
local_tests_readme: tests/local_tests/<model_family>/README.md
port_state_file: tests/local_tests/<model_family>/PORT_STATUS.md
parity_test: tests/local_tests/<bucket>/test_<family>_<component>_parity.py
prototype_key_dumps:
official: converted_weights/<family>/_mapping/<component>_official_keys.json | planned | unknown
fastvideo: converted_weights/<family>/_mapping/<component>_fastvideo_keys.json | planned | unknown
conversion:
script: scripts/checkpoint_conversion/<family>_to_diffusers.py | not_created | not_needed | unknown
converted_component_dir: converted_weights/<family>/<component> | not_created | not_needed | unknown
model_index_library: <diffusers|transformers|fastvideo|fastvideo.*|unknown|none>
config_file: <config.json|scheduler_config.json|none|unknown>
mapping_notes: <key prefixes, split/fuse concerns, skipped keys, not_created, not_needed, or unknown>
production_loader_strictness: <strict|non_strict_with_allowed_keys|stateless|unknown>
strict_load: <not_run | pass | pass_with_documented_exclusions | blocked>
concerns_or_unknowns:
- <prototype mismatch, ambiguous arg, missing op, dtype concern, output head, etc.>
```
Rules:
- If any required path is unknown, pass `unknown` plus the exact search already
performed.
- Do not silently omit ambiguous official files, instantiation args, or prototype
concerns.
- For reused components, still fill every field and set `fastvideo_target_files`
to the reused class/config.
- In `mode=parity-scaffold`, prototype and conversion fields may be `planned`,
`not_created`, `not_needed`, or `unknown`; do not invent paths or statuses that
do not exist yet.
- Update `port_state_file` when concerns, issues, conversion status, or parity
status change.
@@ -1,41 +0,0 @@
# Component Skill Handoff Contract
Returned by `add-model-03-port-dit`, `add-model-04-port-vae`,
`add-model-05-port-encoder`, and `add-model-06-port-generic`.
```text
component: <name>
mode: prototype | parity-debug
files_changed: <model/config/export/test/readme paths>
official_files_used: <definition files, instantiation files>
prototype_key_dumps: <official path, fastvideo path, or none>
port_state_file: tests/local_tests/<model_family>/PORT_STATUS.md
concerns_or_unknowns: <remaining or newly discovered concerns>
parity_test: <path>
parity_status: scaffold_skip | debug_red | non_skip_pass | blocked
production_loader_strictness: strict | non_strict_with_allowed_keys | stateless
strict_load: pass | pass_with_documented_exclusions | blocked | not_run
pytest_output: <command + short result>
blocker: <none or exact missing dependency/weights/numeric mismatch>
conversion_retry_request: <none or failing keys/shapes/prefixes/evidence for add-model-07-conversion>
readme_updated: yes | no
next_step: phase_5_conversion | phase_5_conversion_retry | phase_6_continue | ask_user | blocked
escape_hatch: <none or block matching contracts/escape_hatch.md>
```
Rules:
- In `mode=prototype`, parity may be `scaffold_skip` or `blocked`; key dumps are
the required artifact. Successful prototype handoff should use
`next_step=phase_5_conversion`.
- In `mode=parity-debug`, final success requires `parity_status=non_skip_pass`.
- If conversion is implicated, return `conversion_retry_request` and do not edit
conversion scripts or converted weights directly.
- If production loading is non-strict, list allowed missing/unexpected keys in
the parity test or handoff and mark `strict_load=pass_with_documented_exclusions`.
- Update `port_state_file` before returning: component row, open questions,
issues/blockers, decisions, and handoff notes.
- Return an `escape_hatch` only for user decisions, not for normal component
implementation or parity-debug failures.
- Use `next_step=ask_user` only with an `escape_hatch` block and a matching
`PORT_STATUS.md` row.
@@ -1,41 +0,0 @@
# Conversion Handoff Contract
Returned by `../add-model-07-conversion/SKILL.md` to `/add-model` and component
parity-debug subagents.
```text
conversion_script: scripts/checkpoint_conversion/<family>_to_diffusers.py
source_layout: <diffusers|raw_official|separate_components|monolithic|mixed|custom>
converted_weights_dir: converted_weights/<model_family>
port_state_file: tests/local_tests/<model_family>/PORT_STATUS.md
components_written: <list>
passthrough_components: <list>
strict_load: pass | pass_with_documented_exclusions | blocked
component_context_updates:
- component: <name>
converted_component_dir: <path>
model_index_library: <diffusers|transformers|fastvideo|fastvideo.*>
config_file: <path or none>
config_validation: pass | blocked | not_applicable
mapping_notes: <prefixes, split/fuse ops, skipped keys>
production_loader_strictness: strict | non_strict_with_allowed_keys | stateless
strict_load: pass | pass_with_documented_exclusions | blocked | not_run
retry_resolved: <yes | no | not_a_retry>
concerns_or_unknowns: <remaining list>
blocked_on: <none or exact blocker>
next_step: phase_6_component_parity_debug | ask_user
escape_hatch: <none or block matching contracts/escape_hatch.md>
```
Rules:
- Include strict-load evidence for every stateful converted component.
- If a component intentionally loads non-strictly, list the exact missing or
unexpected keys and why they are safe.
- Include the actual `model_index.json` library token, config filename, and config
validation result for every emitted component.
- Preserve retry evidence so the requesting component subagent can resume with
updated context.
- Keep `port_state_file` synchronized with `component_context_updates`.
- Use `next_step=ask_user` only with an `escape_hatch` block and a matching
`PORT_STATUS.md` row.
@@ -1,54 +0,0 @@
# Conversion Request Contract
Consumed by `../add-model-07-conversion/SKILL.md` in Phase 5 and during Phase 6
conversion retries.
Initial conversion request:
```text
model_family: <snake_case>
source_layout: diffusers | raw_official | monolithic | separate_components | mixed | custom
official_weights: <HF repo, local dir, or checkpoint file>
hf_revision: <revision | default | none>
converted_weights_dir: converted_weights/<model_family>
local_tests_readme: tests/local_tests/<model_family>/README.md
port_state_file: tests/local_tests/<model_family>/PORT_STATUS.md
components:
- name: <transformer|vae|text_encoder|conditioner|scheduler|...>
component_type: <dit|vae|encoder|scheduler|conditioner|upsampler|vocoder|generic>
official_definition_files: <paths + symbols>
official_instantiation_files: <paths + call sites + args>
official_weight_source: <checkpoint file, prefix, subfolder, or passthrough source>
official_keys: <path to official key/shape dump>
fastvideo_keys: <path to FastVideo prototype key/shape dump>
fastvideo_class: <class name>
model_index_library: <diffusers|transformers|fastvideo|fastvideo.*>
config_filename: <config.json|scheduler_config.json|none>
production_loader_strictness: <strict|non_strict_with_allowed_keys|stateless>
source_prefix_or_path: <prefix or path>
parity_test: <component parity test path>
prototype_concerns_or_unknowns: <short list>
```
Retry request from a component skill:
```text
conversion_retry_request:
component: <name>
parity_test: <path>
failing_keys: <official and FastVideo keys, if known>
expected_actual_shapes: <expected vs actual shapes, if known>
source_prefix_or_path: <prefix/path implicated by the failure>
evidence: <strict-load error, first divergent tensor, parity log excerpt>
suspected_fix: <rename | split | fuse | skip | component bucket | config | unknown>
```
Rules:
- Phase 5 conversion requires Phase 4 official/FastVideo key dumps.
- Component skills must use the retry request instead of editing conversion
scripts or converted weights directly.
- Conversion must update `port_state_file` with conversion status, retry history,
strict-load status, new issues, and resolved issues.
- Conversion must validate emitted config keys through the production config
update path and record the config filename expected by each loader.
@@ -1,51 +0,0 @@
# Escape Hatch Contract
Canonical pause-and-ask schema for `/add-model` skills. Use this when the next
action requires user input instead of autonomous debugging.
```text
escape_hatch:
needs_user_input: yes | no
decision_type: scope | dependency | auth | cost | destructive | ambiguity | blocker
question: <one precise question>
recommended_option: <safe recommended choice>
options:
- <option + consequence>
safe_default: <what the agent will do after approval, or none>
blocked_until_answered: yes | no
state_snapshot:
phase: <phase or skill mode>
files_changed:
- <paths>
command_or_test: <last relevant command, or not_run>
evidence: <short logs, paths, error text, or blocker ID>
```
Use `needs_user_input=no` when the handoff is green or the next step is already
specified by the workflow.
Ask the user only for decisions the workflow cannot safely choose:
- product or PR scope changes, including dropping a modality, output head, or
variant;
- core dependency changes, version pin changes, or installing untrusted/private
dependencies;
- auth setup for gated repos, using env var names only and never token values;
- large downloads, publishing weights, SSIM/reference uploads, or GPU-heavy work
where cost/runtime approval is needed;
- destructive file/git operations, overwriting existing clones/weights, or
deleting staged assets;
- ambiguous official sources of truth with incompatible behavior;
- accepting a known blocker, loosening parity/quality tolerances, or shipping
without required non-skip parity.
Do not ask for normal recoverable failures:
- missing imports, missing local paths, skipped tests, failing parity, conversion
mapping errors, strict-load failures, format/lint failures, or implementation
bugs covered by the skill workflow.
Before returning `next_step=ask_user`, update
`tests/local_tests/<model_family>/PORT_STATUS.md` with the blocker/question ID,
include the exact evidence, and provide one recommended option plus at most three
alternatives.
@@ -1,38 +0,0 @@
# Final Handoff Contract
Completed by `/add-model` before handing work back to the user or opening a PR.
```text
final_handoff:
prep_handoff_complete: yes | no
conversion_status: not_needed | pass | blocked
components:
- name: <component>
reuse_or_port: reused | ported
parity_test: <path>
parity_status: non_skip_pass | blocked
concerns_or_unknowns: <none or list>
pipeline_smoke: pass | blocked | not_run
pipeline_parity: pass | blocked | not_run
example_status: pass | blocked | not_run
quality_regression: added | deferred_with_reason | not_applicable
local_tests_readme: tests/local_tests/<model_family>/README.md
port_state_file: tests/local_tests/<model_family>/PORT_STATUS.md
token_values_committed: no
runtime_third_party_model_imports: none | listed_with_rationale
blockers: <none or list>
escape_hatch: <none or block matching contracts/escape_hatch.md>
```
Required before handoff:
- Every required component, reused or ported, has non-skip local parity PASS.
- Pipeline smoke and pipeline parity are non-skip PASS, or a blocker is explicit.
- Basic example runs and writes a non-corrupt output.
- `local_tests_readme` lists every component parity command/status/blocker.
- `port_state_file` has no unresolved blocker that is omitted from the final
response or PR notes.
- No raw HF token values, credentials, `.env`, reference clone, or staged weight
blobs are committed.
- If final handoff is blocked on user input, include an `escape_hatch` block and
matching `PORT_STATUS.md` row.
@@ -1,44 +0,0 @@
# Parity Status Contract
Returned by `../add-model-02-parity/SKILL.md` to `/add-model` and later updated by
component parity-debug subagents.
```text
component_parity:
- component: <name>
test: tests/local_tests/<bucket>/test_<family>_<component>_parity.py
status: scaffold_skip | debug_red | non_skip_pass | blocked
missing: <none | fastvideo_class | converted_weights | official_import | ...>
coverage_scope: production_loader | implementation_subcomponent | both
official_definition_files: <paths>
official_instantiation_files: <paths>
concerns_or_unknowns: <short list>
pipeline_parity:
test: <path or not-created>
status: not_started | scaffold_skip | debug_red | non_skip_pass | blocked
local_tests_readme: tests/local_tests/<model_family>/README.md
port_state_file: tests/local_tests/<model_family>/PORT_STATUS.md
notes: <short list>
escape_hatch: <none or block matching contracts/escape_hatch.md>
```
Status meanings:
- `scaffold_skip`: test is present but skips for a specific missing dependency,
FastVideo class, or weights.
- `debug_red`: both sides load and the test fails numerically.
- `non_skip_pass`: required before final handoff for every required component,
including reused components.
- `blocked`: a precise missing dependency, weight, official call path, or
component/conversion regression prevents local activation.
Coverage meanings:
- `production_loader`: FastVideo side loads through the same loader/path used by
a pipeline.
- `implementation_subcomponent`: FastVideo side constructs classes or remaps
tensors directly to isolate implementation behavior.
- `both`: the test covers both production loading and implementation behavior.
Use `escape_hatch` only when blocked status requires a user decision. Normal
skips or red parity should be debugged by the workflow without asking.
@@ -1,82 +0,0 @@
# Pipeline Context Contract
Canonical packet passed from `/add-model` to `add-model-09-pipeline` for pipeline
definition and pipeline parity-debug work.
```text
pipeline_context:
model_family: <snake_case>
mode: pipeline-definition | pipeline-parity-debug
workload_types:
- <T2V|I2V|V2V|T2I|compatibility-shim-with-rationale>
modalities:
inputs: <text/image/video/audio/pose/depth/mask/etc.>
outputs: <video/image/audio/joint-av/latents/etc.>
official_ref_dir: <path or import path>
official_pipeline_files:
- path: <repo-relative or absolute path in official repo>
symbols: <pipeline/factory/sample functions>
notes: <stage order, mutable state, output contract>
official_call:
command_or_api: <official CLI, Python call, or package entrypoint>
args_and_defaults: <height, width, frames, fps, duration, steps, CFG, scheduler, seeds, etc.>
preset_source: <model card, config file, official script, or unknown>
scheduler_and_rng: <timestep/sigma/noise/generator behavior>
output_contract: <decoded media, denoised latents, waveform, dict keys, etc.>
model_index:
class_name: <FastVideo pipeline class name to emit in model_index.json>
entry_class_names: <registered EntryClass.__name__ values that must include class_name>
required_modules: <text_encoder, tokenizer, vae, transformer, scheduler, etc.>
passthrough_modules: <tokenizer, scheduler, processor, external HF dirs, or none>
sampling_param:
new_fields: <none or list of public kwargs/preset defaults to add to SamplingParam>
cli_fields: <none or list of fields that need CLI args>
placeholder_fields: <none or video-shaped compatibility placeholders with rationale>
components:
- name: <component>
component_type: <dit|vae|encoder|scheduler|conditioner|upsampler|vocoder|generic>
parity_test: tests/local_tests/<bucket>/test_<family>_<component>_parity.py
parity_status: non_skip_pass
fastvideo_target_files: <model/config/export files>
converted_component_dir: converted_weights/<family>/<component>
conversion:
converted_weights_dir: converted_weights/<family>
source_layout: <diffusers|raw_official|monolithic|separate_components|mixed|custom>
model_index_path: converted_weights/<family>/model_index.json
fastvideo_targets:
pipeline_files:
- fastvideo/pipelines/basic/<family>/<family>_pipeline.py
stage_files:
- fastvideo/pipelines/basic/<family>/stages/<stage>.py
pipeline_config_files:
- fastvideo/configs/pipelines/<family>.py
preset_file: fastvideo/pipelines/basic/<family>/presets.py
registry_file: fastvideo/registry.py
example_files:
- examples/inference/basic/basic_<family>.py
smoke_test: tests/local_tests/pipelines/test_<family>_pipeline_smoke.py
parity_test: tests/local_tests/pipelines/test_<family>_pipeline_parity.py
local_tests_readme: tests/local_tests/<model_family>/README.md
port_state_file: tests/local_tests/<model_family>/PORT_STATUS.md
concerns_or_unknowns:
- <pipeline branch, unsupported workload, output head, preset ambiguity, etc.>
```
Rules:
- Start only after every required component, reused or ported, has
`parity_status=non_skip_pass`. If any row is missing or skipped, return to
`/add-model` Phase 6.
- Record official call arguments and default sources before writing FastVideo
presets. Do not invent inference defaults from memory.
- `model_index.class_name` must match a registered pipeline `EntryClass.__name__`;
registry detectors are not sufficient for executable pipeline resolution.
- Every public generation kwarg or preset default must be represented in
`SamplingParam`, or documented as an intentional internal-only field.
- `T2A`, `A2A`, and `AV` may be used only after `WorkloadType` supports them;
otherwise record the compatibility shim and rationale explicitly.
- Keep token values out of the packet. Use only token environment variable names.
- If a target path is unknown, use `unknown` plus the exact search already
performed.
- Update `local_tests_readme` and `port_state_file` whenever pipeline smoke,
parity, presets, registry, examples, or blockers change.
@@ -1,71 +0,0 @@
# Pipeline Handoff Contract
Returned by `add-model-09-pipeline` to `/add-model` after pipeline definition or
pipeline parity-debug work.
```text
pipeline_handoff:
model_family: <snake_case>
mode: pipeline-definition | pipeline-parity-debug
files_changed:
- <pipeline/config/preset/registry/stage/example/test/readme/status paths>
official_files_used:
- <definition/call/default source paths>
required_config_modules:
emitted: <list from pipeline class>
model_index: <list from converted or source model_index.json>
status: match | mismatch | blocked
pipeline_class_resolution:
model_index_class_name: <_class_name>
entry_class_names: <registered EntryClass.__name__ values>
status: exact_match | alias_added | blocked
sampling_param:
fields_added: <none or list>
cli_fields_added: <none or list>
unknown_kwargs_checked: yes | no | blocked
stage_chain:
- <stage names in execution order>
pipeline_config:
file: <path>
classes: <class names>
official_defaults_checked: yes | no | blocked
presets:
file: <path>
names: <preset names>
status: pass | blocked | not_run
registry:
status: pass | blocked | not_run
detectors: <HF paths and model_index _class_name strings covered>
smoke_test:
path: tests/local_tests/pipelines/test_<family>_pipeline_smoke.py
status: non_skip_pass | blocked | not_run
pytest_output: <command + short result>
pipeline_parity:
path: tests/local_tests/pipelines/test_<family>_pipeline_parity.py
status: scaffold_skip | debug_red | non_skip_pass | blocked
pytest_output: <command + short result>
comparison_target: <latents|decoded video|audio|joint outputs>
example:
path: examples/inference/basic/basic_<family>.py
status: pass | blocked | not_run
output: <path or none>
readme_updated: yes | no
port_state_updated: yes | no
blockers: <none or exact blocker list>
next_step: phase_10_quality_regression | return_to_phase_6 | ask_user
escape_hatch: <none or block matching contracts/escape_hatch.md>
```
Rules:
- `pipeline-definition` may return with parity still `scaffold_skip` only if the
exact missing dependency, weight, or call-path blocker is recorded.
- Final `/add-model` handoff requires `smoke_test.status=non_skip_pass` and
`pipeline_parity.status=non_skip_pass`, unless the user explicitly accepts a
documented blocker.
- If parity failure traces to a component, conversion, or strict-load issue,
return `next_step=return_to_phase_6` and include the exact failing evidence.
- Keep `local_tests_readme` and `port_state_file` synchronized with this
handoff before returning.
- Use `next_step=ask_user` only with an `escape_hatch` block and a matching
`PORT_STATUS.md` row.
@@ -1,89 +0,0 @@
# Port State Contract
Canonical per-port state file created during prep and updated by every
`/add-model` phase.
Path:
```text
tests/local_tests/<model_family>/PORT_STATUS.md
```
Purpose:
- Single source of truth for resumable port progress.
- Tracks component status, conversion status, parity status, open questions,
blockers, escape hatches, and issue history.
- Lets review agents run the same setup/tests without reconstructing handoffs
from conversation history.
Required sections:
```text
# <Model Family> Port Status
## Summary
- model_family:
- workload_types:
- official_ref:
- official_ref_dir:
- hf_weights_path:
- local_weights_dir:
- source_layout:
- local_tests_readme:
## Current Phase
- phase:
- status: not_started | in_progress | blocked | complete
- owner: orchestrator | prep | parity | conversion | component:<name> | pipeline
- last_updated:
## Component Matrix
| Component | Type | Reuse/Port | Official Definition | Official Instantiation | FastVideo Target | Prototype | Conversion | Parity | Open Issues |
|---|---|---|---|---|---|---|---|---|---|
## Conversion State
- conversion_script:
- converted_weights_dir:
- source_layout:
- strict_load_status:
- passthrough_components:
- retry_history:
## Parity Commands
| Scope | Command | Last Result | Notes |
|---|---|---|---|
## Open Questions
| ID | Question | Owner | Needed By Phase | Status | Resolution |
|---|---|---|---|---|---|
## Issues And Blockers
| ID | Phase | Component | Severity | Issue | Evidence | Owner | Status | Resolution |
|---|---|---|---|---|---|---|---|---|
## Escape Hatches
| ID | Phase | Decision Type | Question | Recommended Option | Status | Resolution |
|---|---|---|---|---|---|---|
## Decisions
| Date | Decision | Rationale | Impact |
|---|---|---|---|
## Handoff Notes
- <short notes for the next agent>
```
Rules:
- Update this file whenever a phase starts, blocks, resolves an issue, or hands
off to another skill.
- Record open questions and issues immediately. Do not leave blockers only in
chat history or subagent responses.
- Use stable IDs: `Q001`, `Q002`, `I001`, `I002`, etc.
- Use stable escape-hatch IDs: `E001`, `E002`, etc. Link them from handoff
`escape_hatch.state_snapshot.evidence` when returning `next_step=ask_user`.
- Do not include raw token values, machine-local cache internals, or large output
dumps. Use repo-relative paths when possible.
- If a question or issue is resolved, keep the row and fill `Resolution` instead
of deleting it.
@@ -1,42 +0,0 @@
# Prep Handoff Contract
Produced by `../add-model-01-prep/SKILL.md` and consumed by `/add-model` Phase 0.
```text
model_family: <snake_case>
workload_types: <T2V/I2V/V2V/T2I/or compatibility shim with rationale>
official_ref: <url or import path>
official_ref_dir: <ReferenceDir or none>
official_ref_commit: <sha or unknown>
hf_weights_path: <HF id or local path>
hf_revision: <revision or default>
local_weights_dir: official_weights/<model_family> or <local path>
source_layout: diffusers | raw_official | monolithic | separate_components | mixed | custom | unknown
model_index_class: <_class_name or none>
components_seen: <components>
needs_conversion: yes | no | unknown
hf_token_env: <env var name only>
dependency_changes: none | installed no-deps editable | installed official deps in current env | blocked on user
official_env_status: imports_ok | private_deps_need_stubs | blocked
local_tests_readme: tests/local_tests/<model_family>/README.md
port_state_file: tests/local_tests/<model_family>/PORT_STATUS.md
gitignore_entries_added: <list>
next_step: add-model | ask_user
open_questions: <short list>
escape_hatch: <none or block matching contracts/escape_hatch.md>
```
Validation:
- `official_env_status` must be `imports_ok` or `private_deps_need_stubs` before
component parity scaffolding.
- `local_tests_readme` must exist and describe official setup, HF weights,
dependency changes, planned parity commands, and review notes.
- `port_state_file` must exist and follow `contracts/port_state.md`.
- Prep does not go directly to conversion; `/add-model` must run component
prototype/key-dump Phase 4 before Phase 5 conversion.
- `T2A`, `A2A`, and `AV` may be used only after `WorkloadType` supports them;
otherwise record the compatibility shim and rationale explicitly.
- Never include HF token values.
- Use `next_step=ask_user` only with an `escape_hatch` block and a matching
`PORT_STATUS.md` row.
@@ -1,96 +0,0 @@
# Shared Add-Model Rules
These rules apply to every `add-model` related skill: prep, parity, conversion,
component porting, pipeline, and the main `/add-model` orchestrator.
## Token And Auth Safety
- Never accept, print, echo, log, hard-code, or commit raw HF token values.
- Refer only to token environment variable names: `HF_TOKEN`,
`HUGGINGFACE_HUB_TOKEN`, or `HF_API_KEY`.
- Scripts may read those environment variables but must not print their values.
- Ask for auth setup only by env var name. Do not ask the user to paste a token.
- Read scope is needed for gated repos during conversion/load. Write scope is
needed for publishing converted weights or seeding generated references.
## Shared State Files
- `tests/local_tests/<model_family>/README.md` is the reviewer-facing setup and
verification log. Keep it current with setup commands, dependency blockers,
parity commands, conversion commands, and pass/blocker status.
- `tests/local_tests/<model_family>/PORT_STATUS.md` is the per-port state file.
It must follow `../contracts/port_state.md` and keep stable `Q###`, `I###`, and
`E###` IDs.
- Keep resolved questions/issues in `PORT_STATUS.md` with the resolution instead
of deleting them.
- Before returning a handoff, update both state files when the skill changed
setup, tests, conversion, parity status, blockers, or decisions.
- Do not include raw tokens, non-reproducible absolute cache paths, large
generated outputs, `.env`, credentials, or anything matching `*secret*`.
## Escape Hatches
Continue autonomously for recoverable setup, implementation, conversion,
strict-load, smoke, parity-debug, lint, or test failures. Stop and ask the user
only when the next action requires a product, cost, safety, auth, dependency, or
scope decision the workflow cannot safely choose.
Use `../contracts/escape_hatch.md` whenever returning `next_step=ask_user`.
Ask for user input only for:
- scope changes, such as dropping a modality, output head, variant, component, or
public mode;
- core dependency changes, untrusted/private dependency installs, or version pin
changes;
- auth setup for gated repos, using env var names only;
- large downloads, publishing weights, SSIM/reference uploads, or GPU-heavy work
where cost/runtime approval is needed;
- destructive file/git operations, overwriting existing clones/weights, or
deleting staged assets;
- incompatible official sources of truth where no reference can be chosen from
published weights and docs;
- accepting a blocker, loosening tolerances, using shape-only substitutes, or
shipping without required non-skip parity.
Do not ask for normal recoverable failures: missing imports, missing local paths,
skipped tests, failing parity, conversion mapping bugs, strict-load errors,
format/lint failures, smoke failures, registry import issues, example failures,
or implementation bugs covered by the phase workflow.
Before asking:
- update `PORT_STATUS.md` with an `E###` escape-hatch row plus any linked `Q###`
or `I###` row;
- include exact evidence: command, path, short error text, parity or strict-load
excerpt, or blocker ID;
- provide one recommended option and at most three alternatives;
- set the relevant handoff `next_step=ask_user` and include the `escape_hatch`
block.
Skill-specific escape-hatch sections may add extra examples, but they must not
weaken these shared rules.
## Production Boundary
- No runtime `from diffusers import <model class>` or
`from transformers import <model class>` in `fastvideo/` production code.
- Components that own weights or numerical behavior must be FastVideo-native
unless the user explicitly accepts a documented lazy-wrapper exception.
- Allowed third-party runtime exceptions are tokenizers and pure data utilities
when they match existing project patterns.
- Tests may import diffusers/transformers as parity references.
- Production comments explain why, not what or provenance. Avoid narrative
comments like `vendored from`, `matches upstream`, `REVIEW`, or session-history
commentary.
## Verification Semantics
- A committed local test may skip when clones, weights, or private deps are absent
so CI and other contributors are not blocked.
- On the porter's machine, a skip is not a pass. Fix the missing import, weights,
or path before claiming verification.
- New ports require local non-skip parity for required components and pipeline
parity when a pipeline is in scope.
- Smoke tests prove loadability only. They are not a substitute for numerical
component or pipeline parity.
@@ -1,117 +0,0 @@
# Component Skill Common Instructions
These instructions apply to `add-model-03-port-dit`, `add-model-04-port-vae`,
`add-model-05-port-encoder`, and `add-model-06-port-generic`. Bucket-specific skills
add target paths, implementation patterns, drift checks, and scope questions.
## Required Context
Require the complete packet from `../contracts/component_context.md`.
Do not start if the official definition files, official instantiation files, or
parity test path are missing. Ask the `/add-model` orchestrator for the complete
component context packet instead of rediscovering broad scope silently.
If the parity scaffold is missing, create it first with
`../../add-model-02-parity/templates/component_parity_test.py`.
## Prototype Mode
Prototype mode runs before conversion:
- implement or prove reuse for the minimal native component, config, export, and
`EntryClass` surface needed by the relevant loader;
- instantiate with random weights using the exact official architecture args, or
instantiate/document stateless components with no weights;
- dump official and FastVideo `state_dict()` names/shapes for every stateful
component so conversion can derive mappings from real surfaces;
- return concerns discovered during prototype work, such as ambiguous official
flags, shape mismatches, private ops, missing loader buckets, passthrough
weights, or output heads;
- update `local_tests_readme` with prototype status and key-dump paths;
- update `port_state_file` with prototype status, open questions, issues, and
handoff notes;
- do not chase numerical parity and do not block on converted weights.
Prototype mode succeeds when the component imports, instantiates with official
args, and required key/shape dumps exist. Converted weights and parity PASS are
not required yet.
## Parity-Debug Mode
Parity-debug mode runs after conversion:
- strict-load converted weights through the same path the pipeline will use, or
document that the component is stateless or an approved passthrough;
- use `conversion_context` and `concerns_or_unknowns` to decide whether a failure
belongs to mapping, loading, implementation, tokenization, normalization,
scheduler semantics, or the parity test;
- run only the component parity test first with `pytest <parity_test> -v -s`;
- if it skips, fix the missing official import, FastVideo class, tokenizer,
converted weights, or path;
- if it fails numerically, add targeted intermediate comparisons to identify the
first divergent operation or tensor;
- update component implementation only when the failure is a component
layer/config/forward/contract bug;
- update `local_tests_readme` with the command, result, and blocker or PASS;
- update `port_state_file` with parity status, resolved/new issues, open
questions, and handoff notes;
- keep iterating until the test is a non-skip PASS or return a precise blocker.
## Conversion Boundary
Component skills must not patch conversion scripts or converted weights ad hoc.
If the first drift or strict-load failure points to wrong keys, missing tensors,
shape mismatches, component prefixes, split/fuse logic, skipped-key policy, or
config emission, return a conversion retry request for
`../../add-model-07-conversion/SKILL.md` matching
`../contracts/conversion_request.md`.
Resume parity-debug only after conversion returns an updated handoff.
## Reuse Proof
When `fastvideo_target_files` point to existing FastVideo code instead of a new
port:
- compare the official definition against the FastVideo target: graph/operation
structure, parameter or state shapes, normalization, activation, positional or
temporal behavior, scaling constants, dtype behavior, state-dict names, output
containers, and output tensors;
- compare the official instantiation against the FastVideo config and loader
args: constructor args, config values, defaults, variant flags, optional
submodules, checkpoint metadata, tokenizer/media paths, and loader path;
- treat a matching class instantiated with different args as not reusable;
- record reuse evidence in `local_tests_readme` and keep the reused component in
`prototype_key_dumps` when it owns state so conversion and parity-debug use the
same surface;
- still run parity-debug to a non-skip PASS. If mismatch is found, return the
concern so `/add-model` can switch the component to a native port.
## Handoff
Return `../contracts/component_skill_handoff.md`.
Mode-specific expectations:
- In `mode=prototype`, `parity_status` may be `scaffold_skip` or `blocked`; key
dumps are the required artifact and successful prototype handoff should use
`next_step=phase_5_conversion`.
- In `mode=parity-debug`, final success requires
`parity_status=non_skip_pass`.
- If conversion is implicated, return `conversion_retry_request` and leave
conversion edits to `add-model-07-conversion`.
- If production loading is non-strict, list allowed missing/unexpected keys in
the parity test or handoff and mark
`strict_load=pass_with_documented_exclusions`.
## Escape Hatches
Follow `common_rules.md`. Do not ask for normal prototype or parity-debug
failures such as missing imports, tokenizer/path issues, red parity,
strict-load failures, key mismatches, shape mismatches, or implementation bugs.
Return conversion retry requests or precise blockers as directed by the workflow.
Ask only when component work requires a scope or safety decision, such as
dropping a required stream/output/path, changing core dependencies, accepting
private model code or unsupported private ops, choosing between incompatible
official definitions, creating a new loader bucket, or loosening required parity.
@@ -1,338 +0,0 @@
---
name: decompose-pipeline-pr
description: Decompose an oversized FastVideo pipeline PR into a stack of independently-reviewable PRs. Tiers the diff by blast radius (invisible / dead code / cross-cutting infra / activation), produces a branch graph and worktree bootstrap, drafts the AGENTS.md manifest, flags missing tests on cross-cutting infra changes, and extracts lessons from the PR body.
---
# Decompose Pipeline PR
## Purpose
When a PR adds a new pipeline (or first-class component port) and crosses
~3,000 LOC, single-shot review converges to rubber-stamping. This skill
decomposes such a PR into a stack of independently-reviewable PRs without
disturbing `main`.
It is the inverse of `add-model`: where `add-model` walks adding a new
pipeline as a fresh PR, this skill walks decomposing an existing oversized
pipeline PR.
**Worked example:** PR #1280 (daVinci-MagiHuman, 9,812 LOC, 56 files) →
2 prerequisite PRs off main + 8-PR stack:
- #1293 `will/activation-trace` (prerequisite)
- #1294 `will/loader-infra` (prerequisite)
- #1295 (1/8) housekeeping
- #1296 (2/8) t5gemma encoder
- #1297 (3/8) DiT
- #1298 (4/8) pipeline stages
- #1299 (5/8) pipeline orchestrator
- #1300 (6/8) provenance (AGENTS.md, JOURNAL.md, lessons)
- #1301 (7/8) conversion scripts
- #1302 (8/8) registry activation
## Prerequisites
- Open PR number on `hao-ai-lab/FastVideo` (or any FastVideo fork)
- `gh` CLI authenticated against the target remote
- Local git worktree support (`git worktree`)
- Git config `user.name` / `user.email` set
- Pre-commit installed (`pre-commit install --hook-type pre-commit --hook-type commit-msg`)
- The target PR's branch fetched locally as `origin/<feature-branch>`
## Inputs
| Parameter | Required | Description |
|-----------|----------|-------------|
| PR number or URL | Yes | E.g. `1280` or `https://github.com/hao-ai-lab/FastVideo/pull/1280` |
| Max desired PR size | No | Defaults to ~2,500 LOC of code per stack PR (excluding generated/journal files) |
| Output directory | No | Defaults to `.agents/tmp/decompose-<pr-number>/` (gitignored) |
## Steps
### 1. Verify ground truth (do not trust `gh pr diff --name-only`)
`gh pr diff <N> --name-only` has been observed to emit phantom file entries.
Always cross-check against the authoritative `git diff`:
```bash
mkdir -p .agents/tmp/decompose-<N>
git fetch origin pull/<N>/head:<feature-branch>
git diff origin/main..origin/<feature-branch> --name-status \
> .agents/tmp/decompose-<N>/files.txt
git diff origin/main..origin/<feature-branch> --stat
```
Use the `--name-status` output as the authoritative file list. If it
disagrees with `gh pr diff --name-only`, trust the git diff.
### 2. Tier the diff by blast radius
Classify every changed file into one of four tiers:
| Tier | Description | Examples |
|---|---|---|
| **Tier 0 — Invisible** | Lint/style/CI configs that don't affect runtime | `.gitignore`, `pyproject.toml` (codespell only), agent documentation |
| **Tier 1 — Dead code** | New files in their own dirs; aggregator one-liners | `fastvideo/models/dits/<new>/`, `fastvideo/pipelines/basic/<new>/`, `examples/inference/basic/basic_<new>*.py`, `tests/local_tests/<new>/`, `__init__.py` exports |
| **Tier 2 — Cross-cutting infra** | Modifications to files used by every pipeline | See protected-paths list below |
| **Tier 3 — Activation switch** | `register_configs(...)` calls + the example scripts that demo them | `fastvideo/registry.py` |
**FastVideo Tier 2 protected paths:**
```
fastvideo/utils.py
fastvideo/pipelines/composed_pipeline_base.py
fastvideo/models/loader/component_loader.py
fastvideo/configs/models/dits/__init__.py
fastvideo/configs/models/encoders/__init__.py
fastvideo/configs/models/vaes/__init__.py
fastvideo/envs.py
fastvideo/fastvideo_args.py
fastvideo/distributed/**
fastvideo/layers/**
fastvideo/attention/**
fastvideo/registry.py # treat as Tier 3 if change is the activation
```
Tier 3 detection (mechanical):
```bash
git diff origin/main..origin/<feature-branch> -- fastvideo/registry.py | \
grep -E "^\+.*register_configs\("
```
If `registry.py` only contains `register_configs` additions, treat it as
Tier 3. If it modifies existing behavior, treat it as Tier 2 (rare).
### 3. Identify reusable Tier-1 components
Within Tier 1, look for sub-trees that are **not** model-specific and could
land separately:
- Encoders matching a known multi-model base (T5/T5-Gemma/Llama/Gemma/CLIP variants)
- New stage classes that subclass shared bases without referencing the new model
- Hook/profiler/debug infra under `fastvideo/hooks/`
- New helpers that have no model-specific dependencies
These get split into their own PRs (e.g. PR 4 `t5gemma-encoder` in the
MagiHuman example).
### 4. Hunt for missing test coverage on Tier 2 changes
For every Tier-2 file modified, check whether the original PR added unit
tests for the new behavior:
```bash
for f in <list-of-tier-2-files>; do
echo "=== Tests for $f ==="
git diff origin/main..origin/<feature-branch> -- \
"$(echo $f | sed 's|fastvideo/|fastvideo/tests/|; s|\.py|*|')"
done
```
If a Tier-2 PR has no accompanying tests, **emit a "must-add tests" list**
with a sketch of the case grid. Tier-2 PRs do not ship without those tests.
The MagiHuman example required this for PR-B (`utils.py`): the original PR
shipped no `test_utils_loader.py`, so the decomposition added 9 unit-test
cases covering the umbrella-detector boundary, the optional-component-dirs
relaxation, and regression coverage on every existing 2-segment HF id.
### 5. Build the dependency DAG and topo-sort
Edges:
- Tier 2 infra → Tier 1 code that imports it
- Reusable Tier 1 components → model-specific Tier 1 code that uses them
(encoder before DiT before pipeline)
- Tier 1 → Tier 3 (activation always last)
- Tier 0 has no dependents (lands first as a freebie)
Topo-sort produces the stack ordering. Pull Tier-2 PRs **out of the stack**
when they have no model-specific dependency — they should land off main
with their own focused review, not buried in a model port.
Render as a tree (markdown):
```
main
├─ <prereq-A>
│ └─ <prereq-B>
│ ├─ <stack-01-housekeeping>
│ │ └─ <stack-02-encoder>
│ │ └─ <stack-03-dit>
│ │ └─ ...
│ │ └─ <stack-N-activate>
│ └─ (parallel) <skill-pr> off main
```
### 6. Detect mis-shelved docs and debug scratch
Two categories to flag:
- **Mis-shelved docs**: Markdown files under `tests/local_tests/` are
journals, not tests. Flag for relocation to the package dir as
`JOURNAL.md`.
- **Debug scratch**: files starting with `_debug_`, `_scratch_`, or
`_explore_`. Flag for drop (do not carry into any output PR).
For MagiHuman: `tests/local_tests/magi-human.md` → relocate. Two
`_debug_magi_human_*.py` files → drop.
### 7. Author the AGENTS.md manifest skeleton
For the new pipeline package, generate a 6-section `AGENTS.md` scaffold
with the file table pre-populated from the diff:
1. **Manifest** — file table by role
2. **Parity invariants** — load-bearing rules with one-paragraph each + lesson refs
3. **Cross-refs** — "If you change X, re-run Y" matrix
4. **Run book** — single pytest command + prereqs (HF tokens, GPU, wall-time)
5. **Open questions** — known issues (e.g. tolerance carve-outs)
6. **Provenance** — PR table with branch names and source SHA
The provenance section is filled incrementally during stack execution and
finalized in the activation PR.
### 8. Extract lessons from the PR body
Scan the PR body for sections titled "Key implementation work", "Bug hunt",
"Lessons", or sentences with patterns like "took N waves to localize",
"silent regression", "investigation revealed". Each becomes a candidate
`.agents/lessons/<YYYY-MM-DD>_<slug>.md` draft.
Lessons MUST follow the existing template in
`.agents/lessons/README.md`:
- YAML frontmatter: `date`, `experiment`, `category`, `severity`
- Sections: What Happened, Root Cause, Fix / Workaround, Prevention
- Filename: `<YYYY-MM-DD>_<short-slug>.md`
Lessons co-locate with the code they concern: a conversion-script lesson
lands in the same PR as the conversion script, not in the docs PR.
### 9. Emit the commit-footer convention
Every commit in the stack ends with:
```
<Feature>-Stack: N/M
```
E.g. `Magi-Stack: 5/8`. Use the package directory name as the feature key.
After all PRs squash-merge, `git log --grep='^<Feature>-Stack:'` reconstructs
the lineage even if PR numbers later get renumbered.
### 10. Produce the worktree bootstrap
Generate a runnable bash script:
```bash
#!/bin/bash
set -euo pipefail
REPO=/home/<user>/FastVideo
WORKTREE=/home/<user>/FastVideoMagi # NB: directory name must be a valid
# Python identifier (no hyphens) so
# mypy doesn't choke
SOURCE_PR=<N>
SOURCE_BRANCH=will/<feature>
SOURCE_SHA=$(git -C "$REPO" rev-parse "origin/$SOURCE_BRANCH")
git -C "$REPO" fetch origin main:main
git -C "$REPO" fetch "origin/$SOURCE_BRANCH"
git -C "$REPO" worktree add "$WORKTREE" origin/main
# Capture baseline for provenance. Everything under .agents/tmp is transient
# and ignored by git.
OUTPUT_DIR="$REPO/.agents/tmp/decompose-$SOURCE_PR"
mkdir -p "$OUTPUT_DIR"
cat > "$OUTPUT_DIR/<feature>-baseline-${SOURCE_SHA:0:8}.txt" <<EOF
Source PR: <repo>#$SOURCE_PR
Source SHA: $SOURCE_SHA
Authoritative file count: $(git -C "$REPO" diff origin/main..origin/$SOURCE_BRANCH --name-only | wc -l)
Date captured: $(date -u +%Y-%m-%dT%H:%M:%SZ)
EOF
```
### 11. Author preserve via `git checkout`, not `cherry-pick`
For each stack PR:
```bash
git -C "$WORKTREE" switch -c <new-branch> <base-branch>
git -C "$WORKTREE" checkout origin/<source-branch> -- <file1> <file2> ...
git -C "$WORKTREE" commit -m "[<scope>]: <subject>
<body>
<Feature>-Stack: N/M"
git -C "$WORKTREE" push -u origin <new-branch>
gh pr create --base <base-branch> --head <new-branch> --title "..." --body "$(cat <<EOF ... EOF)"
```
Notes:
- `git checkout origin/<source> -- <files>` extracts only the named files,
preserving the diff. The original PR's author is **not** preserved on the
new commit (it's authored by whoever runs the script). Reference the
original PR + source SHA in every commit body and PR description for
authorship attribution.
- **Never use `git cherry-pick`** for this workflow — cherry-pick applies
whole commits, which mixes concerns across PR boundaries.
## Outputs
The skill produces all transient planning artifacts under
`.agents/tmp/decompose-<pr>/`:
1. A markdown decomposition plan (`plan.md`)
2. A proposed branch graph
3. A worktree-bootstrap script (`bootstrap.sh`)
4. Per-PR file allocation lists (under `stack/`)
5. AGENTS.md scaffolds for any new pipeline packages
6. Draft lesson files (placed alongside the PR that owns the code they concern)
7. A finalized provenance table for the package AGENTS.md
## Anti-Patterns
The skill should warn against:
- **"Just rebase the megaPR into smaller commits."** Doesn't help review;
reviewer still sees one PR.
- **Co-locating tests under the new package.** FastVideo's convention is
by-kind under `fastvideo/tests/` and `tests/local_tests/<family>/`. Don't
invent a new layout per pipeline.
- **Splitting Tier 2 changes into "one file per PR."** Tier 2 PRs are
about semantic units (e.g., "loader umbrella + optional component dirs"
together because they jointly define the new diffusers-format contract),
not file-count.
- **Landing the activation switch first** ("just register, the code can
be empty"). The skill enforces activation-last so every intermediate
state is dead code, not broken code.
- **Trusting `gh pr diff --name-only`.** Cross-check against
`git diff origin/main..origin/<feature-branch> --name-status` —
`gh`'s output has been observed to include phantom entries.
- **Worktree dir names with hyphens.** mypy interprets them as invalid
Python package names and refuses to run. Use CamelCase or underscores.
- **Skipping the lesson-extraction step.** PR bodies contain the most
expensive learnings of the original implementation. Losing them to a
squash-merge is the silent decay of institutional knowledge.
## Example Usage
```
User: split PR 1280
Agent: [invokes decompose-pipeline-pr]
→ produces .agents/tmp/decompose-1280/plan.md with:
- tiered file table (56 files: 3 tier-0, 35 tier-1, 9 tier-2,
9 tier-3)
- branch graph (PR-A + PR-B + 8-PR stack)
- worktree bootstrap script
- per-PR file lists
- AGENTS.md scaffold for fastvideo/pipelines/basic/magi_human/
- 3 draft lessons extracted from the PR body
→ asks user to confirm before opening branches
```
## References
- The MagiHuman decomposition (worked example):
`fastvideo/pipelines/basic/magi_human/AGENTS.md` (after PR #1302 merges)
- Existing skill: `.agents/skills/add-model/SKILL.md` (the inverse — adding
a new pipeline as a fresh PR)
- Lesson template: `.agents/lessons/README.md`
- Skill template: `.agents/skills/SKILL_TEMPLATE.md`
-196
View File
@@ -1,196 +0,0 @@
---
name: dreamverse-deploy
description: Use when redeploying the migrated Dreamverse app backend and frontend on a chosen local GPU; tears down existing ports, launches services, and waits for readiness checks.
---
# dreamverse-deploy — redeploy migrated Dreamverse on a chosen GPU
**Scope:** project (lives in this repo at `.agents/skills/dreamverse-deploy/`)
**When to use:** you want to (re)launch the migrated `apps/dreamverse/` backend
and frontend on this dev node, pinned to a specific physical GPU. Tears down
any existing deploy on the same ports first, then boots fresh and waits for
both `/readyz` and the FE root to return 200.
## Prerequisites
- Working tree containing `apps/dreamverse/`
- `dreamverse-server` installed from this checkout; if missing, run
`uv pip install -e ".[dreamverse]"`
- Local conda env at `~/miniconda3/envs/fv-main/` with `flashinfer-python`,
`cerebras-cloud-sdk`, `openai` installed (override the default path with
`DREAMVERSE_PYTHON=/path/to/python`)
- `~/.env` exporting `CEREBRAS_API_KEY`, `GROQ_API_KEY`, etc.
- npm available in `$PATH` (or set `NPM=/path/to/npm`)
- `gcc-13` + `g++-13` at `/usr/bin/` (workaround for nvcc gcc-15 rejection)
- **Recommended:** native ffmpeg at `$HOME/opt/ffmpeg-native/bin/ffmpeg`, built
via `bash apps/dreamverse/scripts/install_native_ffmpeg.sh`. The deploy
detects that binary directly and exports it for the backend. The installer's
generated `apps/dreamverse/scripts/ffmpeg-env.sh` is for manual launches.
When the binary is missing, the deploy falls back to system ffmpeg with a
warning. Set
`DREAMVERSE_REQUIRE_NATIVE_FFMPEG=true` to make the missing binary a hard
failure.
If any required prereq is missing, the script fails fast with a clear message.
## Usage
```bash
# Deploy on GPU 4 with the current web port. The legacy helper default remains
# 5274, so pass 5299 explicitly. Torch compile and warmup are both off.
./.agents/skills/dreamverse-deploy/scripts/dreamverse-deploy.sh 4 8009 5299
# Deploy on GPU 6 with custom ports
./.agents/skills/dreamverse-deploy/scripts/dreamverse-deploy.sh 6 8089 5275
# Deploy on GPU 0 with warmup enabled
./.agents/skills/dreamverse-deploy/scripts/dreamverse-deploy.sh --warmup 0 8009 5299
# Deploy with torch.compile enabled (max-autotune; first segment ~3-4min,
# subsequent segments save ~3s — only worth it for benchmarking)
./.agents/skills/dreamverse-deploy/scripts/dreamverse-deploy.sh --torch-compile 4 8009 5299
# Deploy with both warmup AND torch.compile enabled
./.agents/skills/dreamverse-deploy/scripts/dreamverse-deploy.sh --warmup --torch-compile 4 8009 5299
# Flags can appear before, between, or after positional args
./.agents/skills/dreamverse-deploy/scripts/dreamverse-deploy.sh 4 8089 5275 --warmup
```
### Arguments
| Position | Name | Default | Notes |
|---|---|---|---|
| 1 | `GPU` | (required) | Physical GPU index, e.g. `4` |
| 2 | `BACKEND_PORT` | `8009` | TCP port for the FastAPI server |
| 3 | `FRONTEND_PORT` | `5274` | TCP port for the Next.js dev server |
### Flags
| Flag | Default | Notes |
|---|---|---|
| `--warmup` / `--no-warmup` | off | Run GPU warmup at boot (~minutes). Overrides `DREAMVERSE_WARMUP` |
| `--torch-compile` / `--no-torch-compile` | off | Enable max-autotune `torch.compile`. First segment ~3-4min when on, ~45s when off. Overrides `DREAMVERSE_TORCH_COMPILE` |
| `--nvenc` / `--no-nvenc` | off | Use `h264_nvenc` hardware encoder instead of `libx264` software. Eliminates ~1100ms/segment of CPU encoding cost (raises realtime ratio from ~0.78x → ≥1.0x, eliminating inter-segment buffer-drain stutter). Requires native ffmpeg built with `--enable-nvenc` (the install script's default since the NVENC update). Hard-fails up-front if the binary is missing or lacks NVENC. Overrides `DREAMVERSE_NVENC` |
| `-h` / `--help` | — | Show usage |
Flags can appear in any position relative to the positional args. Explicit flag values always win over env-var defaults.
### Environment variables (used when no flag is given)
| Var | Default | Purpose |
|---|---|---|
| `DREAMVERSE_WARMUP` | `false` | Same as `--warmup`/`--no-warmup`. Flag takes precedence |
| `DREAMVERSE_TORCH_COMPILE` | `false` | Same as `--torch-compile`/`--no-torch-compile`. Flag takes precedence |
| `DREAMVERSE_NVENC` | `false` | Same as `--nvenc`/`--no-nvenc`. Flag takes precedence |
| `DREAMVERSE_PYTHON` | `~/miniconda3/envs/fv-main/bin/python` | Conda environment used for the flashinfer prerequisite probe; `dreamverse-server` itself is resolved from `PATH` |
| `DREAMVERSE_REPO_ROOT` | git rev-parse | Repo root override |
| `DREAMVERSE_LOG_DIR` | `/tmp/opencode/dreamverse-deploy` | Directory for the per-GPU backend and per-port frontend logs |
| `DREAMVERSE_REQUIRE_NATIVE_FFMPEG` | `false` | If `true`, fail when `$HOME/opt/ffmpeg-native/bin/ffmpeg` is absent |
## What it does
1. Validates prereqs.
2. Kills any process on the target backend/frontend ports + waits for the
target GPU to release memory (allows up to 30s for cleanup).
3. Sources `~/.env`.
4. Exports the env recipe required for boot:
- `CUDA_VISIBLE_DEVICES=<gpu>`
- `FASTVIDEO_ENABLE_DEVTOOLS=1`
- `FASTVIDEO_ENABLE_STARTUP_WARMUP=<DREAMVERSE_WARMUP>`
- `FASTVIDEO_GPU_COUNT=1`
- `ENABLE_TORCH_COMPILE=<0|1 derived from DREAMVERSE_TORCH_COMPILE>`
- `CC=/usr/bin/gcc-13 CXX=/usr/bin/g++-13 CUDAHOSTCXX=/usr/bin/g++-13`
- `NVCC_PREPEND_FLAGS="-ccbin /usr/bin/gcc-13 -allow-unsupported-compiler"`
- `FASTVIDEO_FFMPEG_BIN=$HOME/opt/ffmpeg-native/bin/ffmpeg` +
`FASTVIDEO_VIDEO_CODEC=<libx264|h264_nvenc>` (when the native binary exists)
5. Launches the installed `dreamverse-server` console command in a detached
`setsid` session and captures its PID.
6. Polls `/readyz` until 200. The budget is 5 minutes by default, 8 minutes
with one startup optimization enabled, and 15 minutes with both warmup and
`torch.compile` enabled.
7. Launches the devtools frontend through npm in a detached session and
captures its PID.
8. Polls FE `/` until 200 (max 60s).
9. Prints URLs, PIDs, and log paths.
## What it does NOT do
- Does not modify `~/.env` or the FastVideo `.venv`.
- Does not push code or commit anything.
- Does not run Playwright. Use the e2e wrapper separately:
```bash
cd apps/dreamverse/web
PLAYWRIGHT_SKIP_WEBSERVER=1 BACKEND_HOST=127.0.0.1 BACKEND_PORT=8009 \
PLAYWRIGHT_BASE_URL=http://127.0.0.1:5299 \
NEXT_PUBLIC_INCLUDE_DEVTOOLS=1 \
npm exec -- playwright test
```
The standard suite runs by default; the long-running two-segment
audio-continuation spec is gated behind
`PLAYWRIGHT_LONG_RUNNING=1` (see below).
## Long-running e2e (paired with `--warmup --torch-compile`)
[`apps/dreamverse/web/e2e/long-running-segments.spec.ts`](../../../apps/dreamverse/web/e2e/long-running-segments.spec.ts)
drives a real two-segment session through the FE, captures every WS
frame, and asserts segments 1 AND 2 both reach `media_segment_complete`
with at least one binary fMP4 chunk per segment. It guards against the
BrokenPipe regression previously caused by dropped LTX-2 audio continuation
kwargs.
Skipped by default. Enable with:
```bash
./.agents/skills/dreamverse-deploy/scripts/dreamverse-deploy.sh \
--warmup --torch-compile 4 8009 5299
cd apps/dreamverse/web
PLAYWRIGHT_SKIP_WEBSERVER=1 \
BACKEND_HOST=127.0.0.1 \
BACKEND_PORT=8009 \
PLAYWRIGHT_BASE_URL=http://127.0.0.1:5299 \
NEXT_PUBLIC_INCLUDE_DEVTOOLS=1 \
PLAYWRIGHT_LONG_RUNNING=1 \
npm exec -- playwright test e2e/long-running-segments.spec.ts
```
Expected runtime: ~7-9 minutes on a B200 (torch.compile max-autotune
warm-up dominates the cold start; per-test timeout is 900s). The spec
hard-fails on any WS `error`/`step_error` frame so the BrokenPipe
regression surfaces with the actual ffmpeg/audio diagnostics rather
than an opaque "test timed out".
## Teardown
Stop both services without redeploying:
```bash
# Stop services on default ports (port-pattern based)
./.agents/skills/dreamverse-deploy/scripts/dreamverse-deploy.sh --stop
# Stop AND nuke any process holding GPU N
./.agents/skills/dreamverse-deploy/scripts/dreamverse-deploy.sh --stop 4
```
The redeploy path (`<GPU>` mode) automatically nukes any process holding the
target GPU before launching — including orphan `multiproc_executor` worker
subprocesses left over from a parent backend that was killed without grace.
This was the failure mode of an earlier naive port-only kill: parent dies,
children survive, GPU stays full, next deploy OOMs.
## Notes
- The installed `dreamverse-server` console command enters
`apps/dreamverse/dreamverse/server_entry.py`, which loads the current
Dreamverse runtime from `apps/dreamverse/dreamverse/`.
- The B200 / sm_100a NVCC flags are mandatory on this dev node because the
conda toolchain ships gcc-15, which nvcc rejects. The script requires the
configured gcc-13 and g++-13 binaries during preflight.
## Deployment boundary
This skill is for a local checkout on a directly attached GPU. For a container
image, use `apps/dreamverse/docker/README.md`. For Modal, follow
`apps/dreamverse/scripts/modal/README.md`; do not adapt this process-killing
workflow to a remote deployment.
@@ -1,460 +0,0 @@
#!/usr/bin/env bash
# See ../SKILL.md for full usage.
set -euo pipefail
is_pid_alive() {
kill -0 "$1" 2>/dev/null
}
terminate_pid() {
local pid="$1"
local label="${2:-pid=${pid}}"
[[ -n "${pid}" ]] && [[ "${pid}" != "$$" ]] || return 0
is_pid_alive "${pid}" || return 0
kill "${pid}" 2>/dev/null || true
for _ in $(seq 1 10); do
is_pid_alive "${pid}" || return 0
sleep 0.5
done
if is_pid_alive "${pid}"; then
kill -9 "${pid}" 2>/dev/null && echo " force-killed ${label}" || true
fi
}
terminate_pattern() {
local pattern="$1"
local pid
if ! command -v pgrep >/dev/null 2>&1; then
pkill -TERM -f "${pattern}" 2>/dev/null || true
sleep 2
pkill -KILL -f "${pattern}" 2>/dev/null || true
return 0
fi
for pid in $(pgrep -f -- "${pattern}" 2>/dev/null || true); do
terminate_pid "${pid}" "pattern='${pattern}' pid=${pid}"
done
}
list_port_pids() {
local port="$1"
if command -v lsof >/dev/null 2>&1; then
lsof -t -iTCP:"${port}" -sTCP:LISTEN 2>/dev/null || true
return 0
fi
ss -tlnp 2>/dev/null | awk -v port=":${port}" '
$0 ~ port {
while (match($0, /pid=[0-9]+/)) {
print substr($0, RSTART + 4, RLENGTH - 4)
$0 = substr($0, RSTART + RLENGTH)
}
}
' || true
}
if [[ "${1:-}" == "--stop" ]]; then
for pat in 'apps/dreamverse/dreamverse/main.py' 'dreamverse-server --host 0.0.0.0 --port' 'next dev --port' 'next-server (v'; do
terminate_pattern "${pat}"
done
if [[ -n "${2:-}" ]] && [[ "${2}" =~ ^[0-9]+$ ]]; then
gpu_uuid="$(nvidia-smi --query-gpu=index,uuid --format=csv,noheader 2>/dev/null | awk -F', ' -v g="${2}" '$1==g {print $2}')"
if [[ -n "${gpu_uuid}" ]]; then
for pid in $(nvidia-smi --query-compute-apps=pid,gpu_uuid --format=csv,noheader 2>/dev/null \
| awk -F', ' -v u="${gpu_uuid}" '$2==u {print $1}'); do
terminate_pid "${pid}" "GPU${2} pid=${pid}"
done
fi
fi
sleep 2
echo "stopped: ports may take a few seconds to free"
exit 0
fi
# ---------------------------------------------------------------------------
# Args
# ---------------------------------------------------------------------------
usage() {
cat <<USAGE
Usage: $(basename "$0") [FLAGS] <GPU> [BACKEND_PORT] [FRONTEND_PORT]
$(basename "$0") --stop [GPU]
Positional:
GPU Physical GPU index (required), e.g. 4
BACKEND_PORT default 8009
FRONTEND_PORT default 5274
Flags (override env vars when both set):
--warmup / --no-warmup run GPU warmup at boot (default off)
--torch-compile / --no-torch-compile
enable max-autotune torch.compile
(default off — first segment ~3-4min
when on, ~45s when off)
--nvenc / --no-nvenc use h264_nvenc hardware encoder (default
off — uses libx264 software encoder).
Requires native ffmpeg built with NVENC.
-h, --help show this help
Env overrides:
DREAMVERSE_WARMUP 'true'|'false' (default false)
DREAMVERSE_TORCH_COMPILE 'true'|'false' (default false)
DREAMVERSE_NVENC 'true'|'false' (default false)
DREAMVERSE_REPO_ROOT default: \$(git rev-parse --show-toplevel)
DREAMVERSE_LOG_DIR default: /tmp/opencode/dreamverse-deploy
DREAMVERSE_REQUIRE_NATIVE_FFMPEG 'true'|'false' (default false)
USAGE
}
WARMUP_OVERRIDE=""
TORCH_COMPILE_OVERRIDE=""
NVENC_OVERRIDE=""
POSITIONAL=()
while [[ $# -gt 0 ]]; do
case "$1" in
-h|--help) usage; exit 0 ;;
--warmup) WARMUP_OVERRIDE=true; shift ;;
--no-warmup) WARMUP_OVERRIDE=false; shift ;;
--torch-compile) TORCH_COMPILE_OVERRIDE=true; shift ;;
--no-torch-compile) TORCH_COMPILE_OVERRIDE=false; shift ;;
--nvenc) NVENC_OVERRIDE=true; shift ;;
--no-nvenc) NVENC_OVERRIDE=false; shift ;;
--) shift; while [[ $# -gt 0 ]]; do POSITIONAL+=("$1"); shift; done ;;
-*) echo "error: unknown flag '$1'" >&2; usage >&2; exit 2 ;;
*) POSITIONAL+=("$1"); shift ;;
esac
done
set -- "${POSITIONAL[@]+"${POSITIONAL[@]}"}"
if [[ $# -lt 1 ]]; then
usage >&2
exit 2
fi
GPU="${1}"
BACKEND_PORT="${2:-8009}"
FRONTEND_PORT="${3:-5274}"
if ! [[ "${GPU}" =~ ^[0-9]+$ ]]; then
echo "error: GPU must be a non-negative integer (got '${GPU}')" >&2
exit 2
fi
WARMUP="${WARMUP_OVERRIDE:-${DREAMVERSE_WARMUP:-false}}"
case "${WARMUP}" in
true|false) ;;
*) echo "error: warmup must be 'true' or 'false' (got '${WARMUP}')" >&2; exit 2 ;;
esac
TORCH_COMPILE="${TORCH_COMPILE_OVERRIDE:-${DREAMVERSE_TORCH_COMPILE:-false}}"
case "${TORCH_COMPILE}" in
true|false) ;;
*) echo "error: torch-compile must be 'true' or 'false' (got '${TORCH_COMPILE}')" >&2; exit 2 ;;
esac
TORCH_COMPILE_FLAG=$([[ "${TORCH_COMPILE}" == "true" ]] && echo 1 || echo 0)
NVENC="${NVENC_OVERRIDE:-${DREAMVERSE_NVENC:-false}}"
case "${NVENC}" in
true|false) ;;
*) echo "error: nvenc must be 'true' or 'false' (got '${NVENC}')" >&2; exit 2 ;;
esac
REPO_ROOT="${DREAMVERSE_REPO_ROOT:-$(git rev-parse --show-toplevel 2>/dev/null || pwd)}"
LOG_DIR="${DREAMVERSE_LOG_DIR:-/tmp/opencode/dreamverse-deploy}"
# ---------------------------------------------------------------------------
# Prereq checks
# ---------------------------------------------------------------------------
bail() { echo "error: $*" >&2; exit 3; }
[[ -d "${REPO_ROOT}/apps/dreamverse" ]] \
|| bail "REPO_ROOT '${REPO_ROOT}' does not contain apps/dreamverse/. Are you on a migration branch?"
DREAMVERSE_SERVER="$(command -v dreamverse-server 2>/dev/null || true)"
[[ -n "${DREAMVERSE_SERVER}" ]] && [[ -x "${DREAMVERSE_SERVER}" ]] \
|| bail "dreamverse-server not executable or not in PATH (run: uv pip install -e \".[dreamverse]\")"
CONDA_ENV_PYTHON="${DREAMVERSE_PYTHON:-${HOME}/miniconda3/envs/fv-main/bin/python}"
[[ -x "${CONDA_ENV_PYTHON}" ]] \
|| bail "conda env python missing at ${CONDA_ENV_PYTHON} (set DREAMVERSE_PYTHON to override)"
"${CONDA_ENV_PYTHON}" -c 'import flashinfer' 2>/dev/null \
|| bail "flashinfer-python not installed in ${CONDA_ENV_PYTHON} (run: ${CONDA_ENV_PYTHON} -m pip install flashinfer-python --no-build-isolation)"
NPM="${NPM:-npm}"
NPM_REQUESTED="${NPM}"
NPM="$(command -v "${NPM}" 2>/dev/null || true)"
[[ -n "${NPM}" ]] && [[ -x "${NPM}" ]] || bail "npm not executable or not in PATH: ${NPM_REQUESTED} (set NPM to override)"
GCC13="$(command -v "${GCC13:-gcc-13}" 2>/dev/null || true)"
GPP13="$(command -v "${GPP13:-g++-13}" 2>/dev/null || true)"
[[ -n "${GCC13}" ]] && command -v "${GCC13}" >/dev/null 2>&1 \
|| bail "gcc-13 not found or not executable (needed for nvcc workaround). Set GCC13 or install gcc-13 in PATH"
[[ -n "${GPP13}" ]] && command -v "${GPP13}" >/dev/null 2>&1 \
|| bail "g++-13 not found or not executable (needed for nvcc workaround). Set GPP13 or install g++-13 in PATH"
[[ -f "${HOME}/.env" ]] || echo "warn: ${HOME}/.env missing — provider API keys may be unset" >&2
NATIVE_FFMPEG_BIN="${HOME}/opt/ffmpeg-native/bin/ffmpeg"
if [[ "${NVENC}" == "true" ]]; then
NATIVE_VIDEO_CODEC=h264_nvenc
else
NATIVE_VIDEO_CODEC=libx264
fi
REQUIRE_NATIVE_FFMPEG="${DREAMVERSE_REQUIRE_NATIVE_FFMPEG:-false}"
case "${REQUIRE_NATIVE_FFMPEG}" in
true|false) ;;
*) bail "DREAMVERSE_REQUIRE_NATIVE_FFMPEG must be 'true' or 'false' (got '${REQUIRE_NATIVE_FFMPEG}')" ;;
esac
if [[ -x "${NATIVE_FFMPEG_BIN}" ]]; then
if [[ "${NVENC}" == "true" ]]; then
encoder_list="$("${NATIVE_FFMPEG_BIN}" -hide_banner -encoders 2>/dev/null || true)"
if [[ "${encoder_list}" != *h264_nvenc* ]]; then
bail "--nvenc requested but ${NATIVE_FFMPEG_BIN} was not built with NVENC. Rebuild: bash apps/dreamverse/scripts/install_native_ffmpeg.sh (with ENABLE_NVENC=1, the default)"
fi
if ! "${NATIVE_FFMPEG_BIN}" -hide_banner -loglevel error -y \
-f lavfi -i 'color=red:size=64x64:rate=24:duration=0.2' \
-c:v h264_nvenc -f null - >/dev/null 2>&1; then
bail "--nvenc requested but the GPU on this host has no NVENC silicon (probe failed: 'OpenEncodeSessionEx unsupported device'). Datacenter Blackwell (B200) and some H100 SKUs ship without NVENC; --nvenc only works on hosts with NVENC-capable GPUs (RTX 50-series, T4, A10, etc.)."
fi
fi
echo " native ffmpeg: ${NATIVE_FFMPEG_BIN} (codec=${NATIVE_VIDEO_CODEC})"
elif [[ "${REQUIRE_NATIVE_FFMPEG}" == "true" ]] || [[ "${NVENC}" == "true" ]]; then
bail "${NATIVE_FFMPEG_BIN} missing (required by --nvenc or DREAMVERSE_REQUIRE_NATIVE_FFMPEG=true). Run: bash apps/dreamverse/scripts/install_native_ffmpeg.sh"
else
echo "warn: ${NATIVE_FFMPEG_BIN} missing — backend will fall back to system ffmpeg (\$(command -v ffmpeg))." >&2
echo " Build native ffmpeg with: bash apps/dreamverse/scripts/install_native_ffmpeg.sh" >&2
fi
echo " python: ${CONDA_ENV_PYTHON}"
mkdir -p "${LOG_DIR}"
# ---------------------------------------------------------------------------
# Teardown anything on target ports
# ---------------------------------------------------------------------------
echo "[1/8] killing any existing deploy on ports ${BACKEND_PORT}/${FRONTEND_PORT} and GPU ${GPU}..."
kill_port_pid() {
local port="$1"
local pid
for pid in $(list_port_pids "${port}"); do
terminate_pid "${pid}" "port=${port} pid=${pid}"
done
}
for pat in "dreamverse-server --host 0.0.0.0 --port ${BACKEND_PORT}" "next dev --port ${FRONTEND_PORT}" "NEXT_PUBLIC_INCLUDE_DEVTOOLS=1 next dev --port ${FRONTEND_PORT}"; do
terminate_pattern "${pat}"
done
kill_port_pid "${BACKEND_PORT}"
kill_port_pid "${FRONTEND_PORT}"
gpu_uuid="$(nvidia-smi --query-gpu=index,uuid --format=csv,noheader 2>/dev/null | awk -F', ' -v g="${GPU}" '$1==g {print $2}')"
if [[ -n "${gpu_uuid}" ]]; then
for pid in $(nvidia-smi --query-compute-apps=pid,gpu_uuid --format=csv,noheader 2>/dev/null \
| awk -F', ' -v u="${gpu_uuid}" '$2==u {print $1}'); do
if [[ -n "${pid}" ]] && [[ "${pid}" != "$$" ]]; then
cmd="$(ps -p "${pid}" -o comm= 2>/dev/null || true)"
terminate_pid "${pid}" "GPU${GPU} pid=${pid} (${cmd:-?})"
fi
done
fi
for i in $(seq 1 30); do
free_be=true
free_fe=true
ss -tln 2>/dev/null | grep -qE ":${BACKEND_PORT}\b" && free_be=false
ss -tln 2>/dev/null | grep -qE ":${FRONTEND_PORT}\b" && free_fe=false
gpu_mem="$(nvidia-smi --query-gpu=memory.used --format=csv,noheader,nounits 2>/dev/null | sed -n "$((GPU + 1))p" || echo 99999)"
if "${free_be}" && "${free_fe}" && [[ "${gpu_mem}" -lt 1000 ]]; then
break
fi
sleep 1
done
gpu_mem="$(nvidia-smi --query-gpu=memory.used --format=csv,noheader,nounits 2>/dev/null | sed -n "$((GPU + 1))p" || echo 0)"
echo " ports cleared; GPU${GPU} at ${gpu_mem} MiB"
# ---------------------------------------------------------------------------
# Launch backend
# ---------------------------------------------------------------------------
echo "[2/8] launching backend on GPU ${GPU} port ${BACKEND_PORT} (warmup=${WARMUP} torch_compile=${TORCH_COMPILE} nvenc=${NVENC})..."
backend_log="${LOG_DIR}/backend-gpu${GPU}.log"
: > "${backend_log}"
setsid bash -c "
set -a
if [[ -f \"${HOME}/.env\" ]]; then
source \"${HOME}/.env\"
fi
set +a
if [[ -x \"${NATIVE_FFMPEG_BIN}\" ]]; then
export FASTVIDEO_FFMPEG_BIN=\"${NATIVE_FFMPEG_BIN}\"
export FASTVIDEO_VIDEO_CODEC=\"${NATIVE_VIDEO_CODEC}\"
fi
export DREAMVERSE_PYTHON=\"${CONDA_ENV_PYTHON}\"
export CUDA_VISIBLE_DEVICES=${GPU}
export FASTVIDEO_ENABLE_DEVTOOLS=1
export FASTVIDEO_ENABLE_STARTUP_WARMUP=${WARMUP}
export FASTVIDEO_GPU_COUNT=1
export ENABLE_TORCH_COMPILE=${TORCH_COMPILE_FLAG}
export CC=${GCC13}
export CXX=${GPP13}
export CUDAHOSTCXX=${GPP13}
export NVCC_PREPEND_FLAGS=\"-ccbin ${GCC13} -allow-unsupported-compiler\"
cd \"${REPO_ROOT}\"
exec \"${DREAMVERSE_SERVER}\" --host 0.0.0.0 --port ${BACKEND_PORT}
" > "${backend_log}" 2>&1 < /dev/null &
disown
# Wait briefly, then resolve actual python PID (the inner process, not the
# wrapper bash).
sleep 4
backend_pid="$(pgrep -f "dreamverse-server --host 0.0.0.0 --port ${BACKEND_PORT}" | head -1 || true)"
if [[ -z "${backend_pid}" ]]; then
echo "error: backend failed to spawn. Last 30 lines of log:" >&2
tail -30 "${backend_log}" >&2
exit 4
fi
echo " backend pid=${backend_pid} log=${backend_log}"
# Poll /readyz. Deadline scales with warmup + torch.compile flags
# because warmup runs two synthetic segments before /readyz=200, and
# torch.compile max-autotune adds ~3-4min cold start to the first
# segment. Empirical worst case (warmup=true, torch_compile=true):
# ~7 min on B200; we budget 15 min for safety.
if [[ "${WARMUP}" == "true" ]] && [[ "${TORCH_COMPILE}" == "true" ]]; then
READYZ_BUDGET_SECONDS=900
elif [[ "${WARMUP}" == "true" ]] || [[ "${TORCH_COMPILE}" == "true" ]]; then
READYZ_BUDGET_SECONDS=480
else
READYZ_BUDGET_SECONDS=300
fi
READYZ_POLL_INTERVAL=6
READYZ_MAX_ITERS=$(( READYZ_BUDGET_SECONDS / READYZ_POLL_INTERVAL ))
echo "[3/8] polling http://127.0.0.1:${BACKEND_PORT}/readyz (budget=${READYZ_BUDGET_SECONDS}s) ..."
ready=0
for i in $(seq 1 ${READYZ_MAX_ITERS}); do
code="$(curl -s -o /dev/null -w '%{http_code}' --max-time 2 "http://127.0.0.1:${BACKEND_PORT}/readyz" 2>/dev/null || echo 000)"
if [[ "${code}" == "200" ]]; then
ready=1
break
fi
if ! kill -0 "${backend_pid}" 2>/dev/null; then
echo "error: backend pid ${backend_pid} died. Last 50 lines:" >&2
tail -50 "${backend_log}" >&2
exit 5
fi
sleep ${READYZ_POLL_INTERVAL}
done
if [[ "${ready}" != "1" ]]; then
echo "error: backend did not become /readyz=200 within ${READYZ_BUDGET_SECONDS}s. Last 50 lines:" >&2
tail -50 "${backend_log}" >&2
exit 5
fi
echo "[4/8] backend /readyz OK"
# ---------------------------------------------------------------------------
# Launch frontend
# ---------------------------------------------------------------------------
echo "[5/8] launching frontend on port ${FRONTEND_PORT}..."
frontend_log="${LOG_DIR}/frontend-port${FRONTEND_PORT}.log"
: > "${frontend_log}"
# Resolve dev script: dev:devtools forces port 5274 + devtools env. If the
# requested port differs, run `next dev --port` directly with devtools env.
fe_cmd="run dev:devtools"
if [[ "${FRONTEND_PORT}" != "5274" ]]; then
fe_cmd="exec -- next dev --port ${FRONTEND_PORT}"
fi
setsid bash -c "
cd \"${REPO_ROOT}/apps/dreamverse/web\"
export NEXT_PUBLIC_INCLUDE_DEVTOOLS=1
export BACKEND_URL=http://127.0.0.1:${BACKEND_PORT}
export BACKEND_HOST=127.0.0.1
export BACKEND_PORT=${BACKEND_PORT}
exec '${NPM}' ${fe_cmd}
" > "${frontend_log}" 2>&1 < /dev/null &
disown
sleep 4
frontend_pid="$(pgrep -f "next dev --port ${FRONTEND_PORT}" | head -1 || true)"
if [[ -z "${frontend_pid}" ]]; then
echo "error: frontend failed to spawn. Last 30 lines:" >&2
tail -30 "${frontend_log}" >&2
exit 6
fi
echo " frontend pid=${frontend_pid} log=${frontend_log}"
# Poll FE root
echo "[6/8] polling http://127.0.0.1:${FRONTEND_PORT}/ ..."
fe_ready=0
for i in $(seq 1 30); do
code="$(curl -s -o /dev/null -w '%{http_code}' --max-time 2 "http://127.0.0.1:${FRONTEND_PORT}/" 2>/dev/null || echo 000)"
if [[ "${code}" == "200" ]]; then
fe_ready=1
break
fi
if ! kill -0 "${frontend_pid}" 2>/dev/null; then
echo "error: frontend pid ${frontend_pid} died. Last 30 lines:" >&2
tail -30 "${frontend_log}" >&2
exit 7
fi
sleep 2
done
if [[ "${fe_ready}" != "1" ]]; then
echo "error: frontend did not respond 200 within 60s. Last 30 lines:" >&2
tail -30 "${frontend_log}" >&2
exit 7
fi
echo "[7/8] frontend / OK"
# ---------------------------------------------------------------------------
# Print summary
# ---------------------------------------------------------------------------
cwd="$(readlink "/proc/${backend_pid}/cwd" 2>/dev/null || echo unknown)"
gpu_mem_now="$(nvidia-smi --query-gpu=memory.used --format=csv,noheader,nounits 2>/dev/null | sed -n "$((GPU + 1))p" || echo 0)"
ffmpeg_in_use="$(tr '\0' '\n' < "/proc/${backend_pid}/environ" 2>/dev/null | sed -n 's/^FASTVIDEO_FFMPEG_BIN=//p' | head -1)"
[[ -z "${ffmpeg_in_use}" ]] && ffmpeg_in_use="$(command -v ffmpeg 2>/dev/null || echo '<not found>') (system fallback)"
cat <<SUMMARY
[8/8] redeploy OK
Frontend : http://localhost:${FRONTEND_PORT} (PID ${frontend_pid})
Backend : http://localhost:${BACKEND_PORT} (PID ${backend_pid})
cwd=${cwd}
gpu=${GPU} mem=${gpu_mem_now} MiB
ffmpeg=${ffmpeg_in_use}
Logs : ${backend_log}
${frontend_log}
Stop : ./.agents/skills/dreamverse-deploy/scripts/dreamverse-deploy.sh --stop
E2E : cd apps/dreamverse/web && \\
PLAYWRIGHT_SKIP_WEBSERVER=1 \\
BACKEND_URL=http://127.0.0.1:${BACKEND_PORT} \\
PLAYWRIGHT_BASE_URL=http://127.0.0.1:${FRONTEND_PORT} \\
NEXT_PUBLIC_INCLUDE_DEVTOOLS=1 \\
npm exec -- playwright test
SUMMARY
@@ -1,642 +0,0 @@
---
name: reseed-performance-baseline
description: Re-seed the HF performance-tracking baseline for an intentional runtime, dependency, environment-caused benchmark shift, or reviewed v2 calibration using one or more reviewed normalized performance JSONs. Use when performance CI fails because metrics such as latency, throughput, component time, or peak memory changed for an accepted reason and the rolling median baseline in FastVideo/performance-tracking must be advanced, or when a new v2 exact comparable identity needs its first approved baseline. The workflow backs up existing history under /tmp, validates all source JSONs for the same legacy (model_id, gpu_type) target or the same v2 exact identity, rejects internally inconsistent source batches, uploads one success=true baseline record per accepted source JSON, and offers to clean local temp state after a successful upload.
---
# Re-seed Performance Baseline
## Purpose
Replace or advance the rolling performance baseline in the HF dataset
`FastVideo/performance-tracking`. Legacy targets are scoped by
`(model_id, gpu_type)`. V2 targets are scoped by exact comparable identity:
`workload_id`, `variant_id`, `benchmark_version`, `hardware_profile_id`,
`software_profile_id`, and `recipe_fingerprint`.
Performance comparison uses the median of up to the last 5 successful,
baseline-eligible records for the same target. Failed or calibration-only
records are useful audit history, but they do not move the future baseline
because `compare_baseline.py` loads records with `successful_only=True` and
`baseline_eligible_only=True`.
This skill now reseeds from a reviewed batch of one or more source performance
JSONs. It uploads one new `success=true` record per accepted source JSON; it
does not blindly replicate one measurement into 3 or 5 records. The effective
reseed size is therefore dynamic and equals the number of provided, validated,
internally consistent source JSONs.
For baseline shifts with existing history, if the operator provides fewer than
3 records, call out that the last-5 rolling median may not move immediately. If
the operator provides 3 consistent shifted records, the rolling median usually
moves immediately. If the operator provides 5 consistent shifted records, the
last-5 window is effectively reset to the new runtime profile. For the first
approved v2 baseline of a new exact identity, one reviewed calibration seed is
enough for the next comparable run to leave `CALIBRATION_NEEDED`.
These records are intentional operator-approved baseline resets, not ordinary
independent main-branch persistence. Mark them clearly with provenance fields
so the HF history remains auditable.
Use this skill when a performance test fails for an intentional and reviewed
reason, such as a torch/runtime/container upgrade that legitimately increases
peak memory or changes timings. This is the performance equivalent of
`reseed-ssim-references`: backup first, scope tightly, require explicit human
approval, then upload reviewed accepted baseline records.
## When to use
- A PR or main run failed the rolling performance comparison by more than the
allowed regression threshold, and maintainers agree the shift is caused by
an intentional runtime, dependency, hardware image, or benchmark environment
change rather than a FastVideo logic regression.
- One or more shifted source result JSONs have been reviewed and accepted, and
the operator wants to use those exact reviewed results to advance the rolling
baseline.
- The source batch is internally consistent: no provided source JSON regresses
against the batch median by more than the configured tolerance.
## When not to use
- The benchmark failure might be a real code regression. Fix or investigate
the code path first.
- The fixed benchmark thresholds in
`.buildkite/performance-benchmarks/tests/*.json` are too low. Those are a
separate gate from the rolling HF baseline and may need a code review change.
- There is no clear source run, commit, and rationale. Baseline history is a
production signal; do not edit it without provenance.
- The provided source JSONs disagree materially with each other. Rerun or
investigate instead of uploading a noisy reseed batch.
## Inputs
| Parameter | Required | Description |
|-----------|----------|-------------|
| `model_id` | Legacy required; v2 inferred | Benchmark id, e.g. `wan-t2v-1.3b-2gpu`. This maps to the HF subdirectory after `sanitize(model_id)`. For v2 records, use the `model_id` from each source artifact only as the upload directory; comparison is by exact identity. |
| `gpu_type` | Legacy required; v2 inferred | Exact GPU device string from the performance record, e.g. the L40S device name emitted by CI. V2 hardware matching uses `hardware_profile_id`; preserve `gpu_type` as display metadata. |
| `source_results` | Yes | One or more local paths or Buildkite artifact URLs for accepted shifted performance JSONs. Prefer normalized `normalized_perf_*.json` artifacts emitted by `compare_baseline.py`. Accept `source_result` as an alias only for a single JSON. |
| `max_intra_batch_regression` | No | Maximum allowed regression of any source JSON against the source batch median. Default: `0.05` (5%). |
| `intent_rationale` | Yes | One-line explanation for why the baseline shift is legitimate. This is written into provenance and should be reused in the PR. |
Hardcoded defaults:
- HF repo: `FastVideo/performance-tracking` (`HF_REPO_ID` override is
supported by the code, but use the default unless the user explicitly asks).
- Local sync root: `/tmp/perf-tracking` (`PERFORMANCE_TRACKING_ROOT` override
is supported).
- Prepared-record staging root: `/tmp/performance_reseed_prepared`
(`PERFORMANCE_RESEED_STAGING_ROOT` override is supported). Keep it separate
and non-nested from the sync root.
- Backup root: `/tmp/performance_reseed_backup`.
- Download scratch root for source artifact URLs: `/tmp/performance_reseed_source`.
- Baseline window: last 5 `success=true`, `baseline_eligible=true` records
for the same legacy `(model_id, gpu_type)` target or the same v2 exact
comparable identity.
- Reseed count: dynamic. Upload exactly one accepted seed record per validated
source JSON.
## Steps
### 1. Validate the target and source results
Normalize `source_results` to a list. If the user passes a single
`source_result`, treat it as a one-element `source_results` list and report
that a single record may not move the last-5 median immediately.
If any source result is a Buildkite artifact URL, download it first into a
local scratch directory under `/tmp/performance_reseed_source/` and use the
downloaded JSON path for the rest of the workflow. If the agent cannot access
the artifact because Buildkite authentication is missing, ask the user to
download the artifact manually and provide the local path.
Prefer the normalized Buildkite artifact emitted by `compare_baseline.py`:
```text
perf_reports/results/normalized_perf_*.json
```
That file is already in the HF tracking schema. Load each normalized JSON
directly:
```python
import json
with open(source_result, encoding="utf-8") as f:
record = json.load(f)
```
Classify the source batch before continuing:
- **Legacy source records** have no v2 exact identity fields. Stop if any
normalized record's `model_id` or `gpu_type` does not match the requested
`model_id` and `gpu_type`.
- **V2 source records** have exact identity fields. Stop unless every source
record has all six comparable identity fields and they are identical across
the batch: `workload_id`, `variant_id`, `benchmark_version`,
`hardware_profile_id`, `software_profile_id`, and `recipe_fingerprint`.
Do not fall back to legacy `(model_id, gpu_type)` matching for v2 records.
The source records may have `success: false` when they came from failed
rolling baseline comparisons. That is expected; only the reviewed reseed
records become new `success: true` baseline records after explicit approval.
For a first v2 baseline seed, the source records must instead be successful
scheduled-main full-suite `CALIBRATION_NEEDED` normalized artifacts. Reject PR,
local, direct-run, non-main-branch, or non-full-suite calibration artifacts as
seed sources.
Sort validated source records by their original `timestamp` ascending before
preparing the seed records. If a source timestamp is missing or unparsable,
preserve input order for those records and print a warning. This makes the
fresh reseed timestamps deterministic and makes it clear which records enter
the last-5 window when more than 5 source JSONs are provided.
Check that `HF_API_KEY` is exported. The sync path may be public, but the
upload path requires write access.
### 1a. Check source batch consistency
Before syncing or preparing uploads, reject source batches that are internally
inconsistent. Use the same metric direction as `compare_baseline.py`:
- Lower is better: `latency`, `memory`, `text_encoder_time_s`, `dit_time_s`,
`vae_decode_time_s`.
- Higher is better: `throughput`.
For each metric with at least two non-null source values:
1. Compute the source batch median.
2. For lower-is-better metrics, compute `(source_value - batch_median) / batch_median`.
3. For `throughput`, compute `(batch_median - source_value) / batch_median`.
4. Stop if any source record regresses against the batch median by more than
`max_intra_batch_regression`.
Default `max_intra_batch_regression` to `0.05`. Print a table with per-source values, batch median, and
worst intra-batch regression.
This check prevents uploading a mixed batch where one JSON is materially
slower or faster than the others. If the batch fails this check, ask the user
to provide a cleaner batch or explicitly investigate the variance. Do not
silently drop outliers unless the user gives a concrete reviewed reason and a
new source list.
### 1b. How to obtain source results from CI
The performance CI exports normalized source results for failed rolling
baseline comparisons when `compare_baseline.py` ran. The preferred artifacts
come from:
```text
perf_reports/results/normalized_perf_*.json
```
The normal operator flow is:
1. Open the failed Buildkite performance job or several reruns of the same
benchmark after the accepted environment shift.
2. Download the `normalized_perf_*.json` artifacts for the target benchmark.
3. Pass all reviewed local paths or artifact URLs as `source_results`.
Do not scrape the Markdown performance summary to reconstruct JSON. The
normalized JSON artifacts are the only supported source of truth for reseed
metrics and provenance. Raw `fastvideo/tests/performance/results/perf_*.json`
artifacts are not accepted by this skill. If no normalized JSON artifact is
present, that run is not a valid source for baseline reseeding.
### 2. Sync and back up existing HF records under /tmp
Use `fastvideo/performance/hf_store.py` helpers directly. Do **not** use
`compare_baseline.py` as a sync shortcut; on full main runs it can persist
records, while this step must only fetch and back up existing history.
The sync command pattern is:
```bash
export PERFORMANCE_TRACKING_ROOT="${PERFORMANCE_TRACKING_ROOT:-/tmp/perf-tracking}"
export HF_REPO_ID="${HF_REPO_ID:-FastVideo/performance-tracking}"
python -c 'from fastvideo.performance.hf_store import sync_from_hf; import os; sync_from_hf(os.environ["PERFORMANCE_TRACKING_ROOT"], strict=True)'
```
For legacy records, back up the sanitized model directory under `/tmp`:
```bash
SHORT_COMMIT=$(git rev-parse --short=12 HEAD)
TIMESTAMP=$(date -u +%Y%m%d_%H%M%S)
MODEL_SAFE=$(python - <<'PY'
from fastvideo.performance.hf_store import sanitize
print(sanitize("<model_id>"))
PY
)
BACKUP_DIR="/tmp/performance_reseed_backup/${TIMESTAMP}_${SHORT_COMMIT}_${MODEL_SAFE}"
mkdir -p "$BACKUP_DIR"
cp -R "${PERFORMANCE_TRACKING_ROOT}/${MODEL_SAFE}" "$BACKUP_DIR/" 2>/dev/null || true
```
For v2 records, back up the full local tracking root after sync. Exact identity
lookup scans across model directories, so a benchmark rename may have relevant
history outside the current source artifact's `model_id` directory:
```bash
BACKUP_DIR="/tmp/performance_reseed_backup/${TIMESTAMP}_${SHORT_COMMIT}_v2_exact_identity"
mkdir -p "$BACKUP_DIR"
cp -R "${PERFORMANCE_TRACKING_ROOT}" "$BACKUP_DIR/tracking-root"
```
Write provenance next to the backup:
```bash
cat > "$BACKUP_DIR/PROVENANCE.txt" <<EOF
model_id: <model_id>
gpu_type: <gpu_type>
source_results:
- <source_result_1>
- <source_result_2>
reseed_record_count: <len(source_results)>
max_intra_batch_regression: <threshold>
head_commit: $(git rev-parse HEAD)
timestamp_utc: $(date -u +%FT%TZ)
reason: <intent_rationale>
EOF
```
If the backup has no prior records, this is not a destructive reseed; it is a
first baseline seed. Continue, but report that baseline history was empty.
### 3. Compute old baseline and candidate shift
Load the last 5 successful baseline records for the target.
For legacy targets:
```python
from fastvideo.performance.hf_store import load_records_for_model
records = load_records_for_model(
"/tmp/perf-tracking",
"<model_id>",
"<gpu_type>",
last_n=5,
successful_only=True,
baseline_eligible_only=True,
)
```
For v2 exact-identity targets:
```python
from fastvideo.performance.hf_store import load_records_for_identity
records = load_records_for_identity(
"/tmp/perf-tracking",
{
"workload_id": "<workload_id>",
"variant_id": "<variant_id>",
"benchmark_version": "<benchmark_version>",
"hardware_profile_id": "<hardware_profile_id>",
"software_profile_id": "<software_profile_id>",
"recipe_fingerprint": "<recipe_fingerprint>",
},
last_n=5,
successful_only=True,
baseline_eligible_only=True,
)
```
Print a small table showing old medians, source batch medians, candidate
medians after appending the proposed seed records, and source batch spread for:
- `latency`
- `throughput`
- `memory`
- `text_encoder_time_s`
- `dit_time_s`
- `vae_decode_time_s`
Also print how many successful old records exist. Make clear:
- 1 seed record usually does not move an existing last-5 median by itself, but
it is enough to establish the first v2 baseline for a new exact identity.
- 3 consistent seed records usually move the last-5 median immediately.
- 5 consistent seed records effectively reset the last-5 window.
- The records are intentional approved baseline resets and must be labeled
that way.
### 4. Confirm intent
Require an explicit confirmation phrase before preparing the upload:
> About to RE-SEED performance baseline for `<target description>`.
> This will upload `<N>` new `success=true` records to
> `FastVideo/performance-tracking/<sanitize(model_id)>/` or the source
> artifact's v2 model directory, one per accepted source JSON.
>
> Reason: `<intent_rationale>`
> Source results: `<source_results>`
> Reseed record count: `<N>`
> Max intra-batch regression: `<threshold>`
> Note: these records come from a reviewed source batch and are intended to
> move the rolling median to the accepted runtime profile. They are not
> ordinary main-branch persistence.
> HEAD: `<git rev-parse --short=12 HEAD>`
> Backup: `<BACKUP_DIR>`
>
> Reply `confirm performance reseed` to proceed, anything else to abort.
Do not continue unless the user types exactly `confirm performance reseed`.
### 5. Create the accepted seed records
Create one seed record from each normalized source result.
For first v2 baseline seeds, use the scoped utility. It validates exact
identity, requires successful scheduled-main full-suite `CALIBRATION_NEEDED`
source artifacts, preserves the normalized v2 identity and metadata fields,
and writes seed records with `success=true`, `baseline_eligible=true`, and
`comparison_status=PASS`:
```bash
python fastvideo/tests/performance/seed_baseline.py \
--source-result <normalized_perf_1.json> \
--source-result <normalized_perf_2.json> \
--intent-rationale "<intent_rationale>" \
--max-intra-batch-regression 0.05 \
--tracking-root "${PERFORMANCE_TRACKING_ROOT}" \
--staging-root "${PERFORMANCE_RESEED_STAGING_ROOT:-/tmp/performance_reseed_prepared}"
```
The utility is prepare-only and intentionally has no upload option. Upload the
scoped records only after the separate confirmation in step 6.
The utility validates against an isolated fresh HF snapshot and leaves
`PERFORMANCE_TRACKING_ROOT` untouched; that argument only proves the staging
root is separate from the operator's tracking mirror. Before writing, it stops
if the exact identity already has a successful baseline-eligible record or if
the workload/variant/version already trusts another recipe. It atomically
reserves the exact identity and writes a digest-protected upload manifest bound
to the current HF endpoint, repository id, and repository type. Keep the
prepared records, manifest, source files, and reservation unchanged until the
operation is uploaded or explicitly cleaned up.
If the prepared seed records look correct, upload only those scoped records in
step 7. Do not rerun the utility with a different source list after approval.
For legacy reseeds or accepted v2 baseline shifts from regression artifacts,
create one seed record from each normalized source result. Do not copy the
source JSON wholesale.
Infer the baseline field allowlist from all existing HF records for the target
after syncing, including both `success=true` and `success=false` records. For
legacy targets the target is `(model_id, gpu_type)`. For v2 baseline-shift
reseeds the target is the exact comparable identity. Use the union of
non-provenance keys present in those target records, preserving only fields
that also exist in the normalized source record or are explicitly set by the
reseed workflow. Always include `model_id`, `timestamp`, `success`,
`baseline_eligible`, and `comparison_status` because the upload path and
baseline loader depend on them. Always set `timestamp` to a fresh reseed
timestamp, `success` to `true`, `baseline_eligible` to `true`, and
`comparison_status` to `PASS`. Do not include unrelated source-only fields
that are absent from existing HF records.
Exclude existing provenance or operator metadata from the inferred baseline
field allowlist. At minimum, exclude keys prefixed with `baseline_reseed` and
any fields known to be local-only audit metadata.
If there are no previous HF records for the target, fall back to this default
baseline field list:
- `model_id`
- `timestamp`
- `commit_sha`
- `gpu_type`
- `latency`
- `throughput`
- `memory`
- `text_encoder_time_s`
- `dit_time_s`
- `vae_decode_time_s`
- `success`
- `baseline_eligible`
- `comparison_status`
For v2 baseline-shift reseeds with no previous HF records for the exact
identity, also preserve:
- `workload_id`
- `variant_id`
- `benchmark_version`
- `recipe_fingerprint`
- `hardware_profile_id`
- `software_profile_id`
- `recipe`
- `hardware_profile`
- `software_profile`
- `software_comparison_profile`
Do not upload extra fields from the source artifact.
Optional provenance fields are allowed and useful:
- `baseline_reseed: true`
- `baseline_reseed_reason`
- `baseline_reseed_source_result`
- `baseline_reseed_source_timestamp`
- `baseline_reseed_source_success`
- `baseline_reseed_batch_size`
- `baseline_reseed_batch_index`
- `baseline_reseed_operator`
- `baseline_reseed_max_intra_batch_regression`
The v2 calibration seed utility writes analogous first-seed provenance:
- `baseline_seed: true`
- `baseline_seed_reason`
- `baseline_seed_source_result`
- `baseline_seed_source_status`
- `baseline_seed_source_timestamp`
- `baseline_seed_source_success`
- `baseline_seed_source_run_source`
- `baseline_seed_source_branch`
- `baseline_seed_source_test_scope`
- `baseline_seed_source_pr_number`
- `baseline_seed_batch_size`
- `baseline_seed_batch_index`
- `baseline_seed_operator`
Use a fresh reseed timestamp for each seed record, not the original source
result timestamp. This is required because
`load_records_for_model(..., last_n=5)` keeps the last records after loading
the model directory; stale filenames/timestamps may not enter the last-5
window and therefore may not move the median. Preserve the original source
timestamp in `baseline_reseed_source_timestamp`.
Use the existing filename convention from `_write_tracking_record()`:
`<sanitize(timestamp)>_<sanitize(commit_sha)>.json` under the sanitized model
directory, but include a deterministic suffix such as `_reseed_01`,
`_reseed_02`, and so on before `.json` so multiple records from the same
batch do not overwrite each other.
If a source record already exists on HF with `success=false`, do not edit it
in place unless the user explicitly asked for an audit-preserving correction.
Prefer uploading new accepted seed records so failed history remains visible.
### 6. Pause before upload
Print:
- Backup directory path under `/tmp`.
- Prepared local record paths under `PERFORMANCE_RESEED_STAGING_ROOT`.
- Prepared upload-manifest path under the identity reservation.
- HF paths that will receive the new records.
- Old rolling medians.
- Source batch medians, source batch spread, reseed count, and candidate
medians.
- Rationale.
Ask the user to reply exactly `upload`. Anything else aborts and leaves the
prepared records plus backup on disk.
### 7. Upload only the scoped records
For a first v2 calibration seed, use the manifest uploader after the user
replies exactly `upload`:
```bash
python -c 'from fastvideo.tests.performance.seed_baseline import upload_prepared_seed_manifest; print(upload_prepared_seed_manifest("<prepared_manifest>"))'
```
The uploader verifies the source and prepared-record digests, pins and scans
the current HF revision, rechecks exact-identity and recipe-cohort conflicts,
and writes the entire batch in one commit whose `parent_commit` must still be
current. A concurrent Hub update makes the commit fail. Do not retry
automatically: preserve staging, refresh/review remote state, and request a new
explicit `upload` after the conflict is understood. Each record goes to:
```text
FastVideo/performance-tracking/<sanitize(model_id)>/<record_filename>.json
```
Never call `upload_record()` once per first-seed record: that can partially
land the batch and has no compare-and-swap guard.
For a legacy reseed or an accepted v2 baseline shift, the first-seed manifest
validator does not apply because an eligible baseline already exists. Upload
only the individually reviewed records prepared in step 5 with the shared
`upload_record(local_path, record, strict=True)` helper. Stop on the first
failure and report exactly which records reached HF; do not silently rerun or
replicate the remainder.
Never bulk upload the tracking or staging root, and never modify another
model's directory in the same operation.
### 8. Report outcome and offer cleanup
Report:
- Uploaded HF paths.
- Backup directory under `/tmp`.
- Local tracking root, usually `/tmp/perf-tracking`.
- Old baseline window count and medians.
- Source batch medians, source batch spread, reseed count, and candidate
medians.
- Expected effect based on reseed count.
- Any separate threshold changes still needed in
`.buildkite/performance-benchmarks/tests/*.json`.
Include the `intent_rationale` in the PR or follow-up comment so reviewers can
distinguish an accepted baseline shift from a hidden regression.
After the upload is verified, ask whether the user wants to clear temporary
local state. Explain what each directory is for:
- `PERFORMANCE_TRACKING_ROOT`, usually `/tmp/perf-tracking`: read-only local
synced mirror used for operator review and reporting. First-v2 preparation
independently proves remote state from a fresh temporary HF snapshot.
- `PERFORMANCE_RESEED_STAGING_ROOT`, usually
`/tmp/performance_reseed_prepared`: prepared local seed records used for the
scoped upload, plus the identity reservation and digest manifest. Keeping
this separate prevents aborted preparations from appearing in later
baseline reads.
- `/tmp/performance_reseed_backup/<...>`: local backup of the target model's
pre-reseed HF history plus `PROVENANCE.txt`, kept so a bad reseed can be
audited or corrected.
- `/tmp/performance_reseed_source/<...>` when used: downloaded source JSON
artifacts from Buildkite URLs.
Ask:
> Reseed succeeded. Do you want me to delete the local temp tracking mirror,
> this reseed's prepared staging records, source downloads, and reseed backup
> under `/tmp`? These files are local safety/audit artifacts only; HF already
> has the uploaded records.
>
> Reply `cleanup reseed temp` to delete them, anything else to keep them.
Do not delete anything unless the user replies exactly
`cleanup reseed temp`. If cleanup is requested, remove only the specific
directories and prepared record paths created for this reseed. Do not remove
the shared staging root when it contains other records. Remove this operation's
identity reservation only with its prepared records and manifest, and never
remove unrelated `/tmp` contents.
## Failure modes and handling
- **`HF_API_KEY` unset.** Stop before upload. Do not create an untracked
process that appears to have reseeded but never reached HF.
- **Source result does not match target.** Stop. The wrong benchmark or GPU
would poison a separate baseline.
- **Source batch is internally inconsistent.** Stop if any source regresses
against the source batch median by more than `max_intra_batch_regression`.
Ask for cleaner sources or a reviewed explanation before continuing.
- **Too few source records to move the median.** Continue only after making
clear that one or two records may not immediately move an existing last-5
median. This warning does not block a first v2 calibration seed for an exact
identity with no eligible baseline yet.
- **The source results are noisy or suspicious.** Stop. Reseeding amplifies
those measurements into the baseline, so they must be reviewed first.
- **HF sync fails.** Stop for destructive reseeds. A stale or empty sync can
make the old baseline look missing.
- **The exact v2 identity already has an eligible baseline.** Stop. The
`CALIBRATION_NEEDED` artifact is stale; use the reviewed baseline-shift path
instead of the first-seed utility.
- **The workload/variant/version trusts another recipe.** Stop. The source is
stale relative to the current recipe cohort and must not bypass
`RECIPE_MISMATCH` by creating a second trusted recipe.
- **The staging root already has a prepared seed for the exact identity.**
Stop and reuse, upload, or explicitly clean that preparation. Do not prepare
another copy of the same measurement.
- **The conditional Hub commit loses its parent race.** Stop without retrying.
Keep the preparation, refresh and review the new remote state, then request
a new explicit `upload` only if the seed is still valid.
- **Candidate still violates fixed thresholds.** Report that this skill only
handles the rolling HF baseline; update benchmark JSON thresholds in code
review if maintainers accept the new absolute limit.
- **The user aborts at either confirmation.** Leave the backup and prepared
records on disk. Nothing should be uploaded.
- **The user declines cleanup.** Keep `/tmp/perf-tracking`, the prepared seed
records under `/tmp/performance_reseed_prepared`, the source download
directory if any, and `/tmp/performance_reseed_backup/<...>` in place for
audit/debugging.
- **A bad seed was uploaded.** Use the backup and HF history to identify the
uploaded file, then remove or supersede it with an explicitly reviewed
corrective record. Do not silently rewrite unrelated history.
## References
- `.agents/skills/reseed-ssim-references/SKILL.md` — safety pattern for
intentional baseline replacement.
- `fastvideo/tests/performance/compare_baseline.py` — normalization, rolling
median comparison, and persistence rules.
- `fastvideo/performance/hf_store.py` — HF sync and record loading helpers.
- `fastvideo/tests/performance/seed_baseline.py` — first-seed preparation,
staging reservation, manifest validation, and conditional batch upload.
- `fastvideo/tests/performance/test_inference_performance.py` — source result
JSON schema.
- `.buildkite/performance-benchmarks/tests/*.json` — fixed absolute benchmark
thresholds, separate from rolling baseline comparisons.
## Changelog
| Date | Change |
|------|--------|
| 2026-05-03 | Initial version. Sister workflow to `reseed-ssim-references`, scoped to one performance `(model_id, gpu_type)` baseline seed with backup, confirmation, provenance, and `success=true` upload. |
| 2026-05-03 | Previous policy: replicate one approved shifted source result into 3 success records by default, or 5 only when explicitly requested. Add provenance marker for replicated-source reseeds. Superseded by the 2026-05-08 dynamic multi-source policy. |
| 2026-05-08 | Replace fixed 3/5 replication with dynamic multi-source reseeding: upload one seed record per reviewed source JSON, validate intra-batch consistency, move backup/source scratch under `/tmp`, and ask whether to clean temp state after successful upload. |
| 2026-07-13 | Keep first-v2-seed preparation outside the canonical mirror, reserve staging identities atomically, reject stale or replayed calibration seeds, and upload reviewed manifests with a single parent-guarded Hub commit. |
@@ -1,343 +0,0 @@
---
name: reseed-ssim-references
description: Re-seed HF reference videos for a single existing SSIM test on Modal L40S. Always backs up current refs locally first, regenerates on Modal, pauses for the user to eyeball before-vs-after quality, then overwrites the targeted `<model_id>` subtree on `FastVideo/ssim-reference-videos` with `--force`. Use when an intentional code change (model port fix, attention backend swap, kernel upgrade, hyperparameter change) has invalidated existing refs and they need to be regenerated. Pairs with `seed-ssim-references`, which is for first-time seeding only.
---
# Re-seed SSIM Reference Videos
## Purpose
Replace the existing SSIM reference videos for a single `(test_file, model_id)`
pair on the HF dataset (`FastVideo/ssim-reference-videos`). This is **destructive**
on HF — the old refs are overwritten — so the skill always:
1. Confirms intent with a one-liner the user has to type.
2. Downloads the existing refs as a local, timestamped backup.
3. Regenerates on Modal L40S (same code path that CI uses).
4. Pauses for a side-by-side eyeball of backup vs new mp4s.
5. Uploads with `--force`, scoped to the single `--model-id`.
6. Reminds the user to keep the backup until the PR lands.
Pairs with `seed-ssim-references`, which is the inverse (first-time seeding
only, refuses to overwrite). Re-seeding is intentionally a separate, more
ceremonial operation because mistakenly clobbering production refs is much
harder to recover from than failing closed.
## When to use
- An intentional code change (model port fix, kernel upgrade, attention
backend swap, hyperparameter change in the test itself) has shifted the
expected SSIM output and the existing refs no longer represent the new
ground truth.
- A test is failing in CI **for the right reason** (the new code is correct,
the old refs are stale).
## When not to use
- A test is failing for the **wrong** reason (the port is buggy, not the
refs). Fix the port; re-seeding hides the bug.
- A brand-new test that has no refs on HF yet. Use `seed-ssim-references`.
- "Just to clean up drift" without a concrete code change to point at. The
PR description has to justify *why* refs changed; without a concrete
change, there's nothing to write.
## Inputs
| Parameter | Required | Description |
|-----------|----------|-------------|
| `test_file` | Yes | Path to the SSIM test, e.g. `fastvideo/tests/ssim/test_matrixgame_similarity.py`. Validated against `fastvideo/tests/ssim/test_*_similarity.py`. |
| `model_id` | Yes | Single model id from the test's `*_MODEL_TO_PARAMS`, e.g. `Matrix-Game-2.0-Diffusers-Base`. Re-seed runs are **per model**. For multi-model tests, invoke the skill once per model. |
| `intent_rationale` | Yes | One-line explanation of *why* refs are being regenerated (e.g. "Relax FA-2 head_size whitelist to include 80 — matrix_game now uses FLASH_ATTN instead of TORCH_SDPA"). Recorded in the backup directory and reused in the PR description. |
Hardcoded:
- Modal GPU: **L40S** (matches CI; re-seeding from another SKU produces refs
that L40S CI cannot match).
- Quality tier: **`default`**. `full_quality` is a separate, deliberate
operation.
- HF repo: `FastVideo/ssim-reference-videos` (override via
`FASTVIDEO_SSIM_REFERENCE_HF_REPO`).
- Device folder: `L40S_reference_videos`.
## Prerequisites
The user has confirmed:
- `modal` CLI authenticated.
- `hf` CLI authenticated, **and** `HF_API_KEY` (or `HUGGINGFACE_HUB_TOKEN` /
`HF_TOKEN`) exported with **write** access to
`FastVideo/ssim-reference-videos`.
- The current branch's code is the change that motivated the re-seed (i.e.
`git rev-parse HEAD` is the commit that intentionally invalidated refs).
Fail fast if any of these are missing.
## Steps
### 1. Validate inputs and confirm intent
- Verify `test_file` exists and matches `fastvideo/tests/ssim/test_*_similarity.py`.
- Grep the file for `*_MODEL_TO_PARAMS` and assert `model_id` is one of its
keys. If the file has only a single hardcoded model, accept that model id
as the only valid value.
- Print the rationale and ask the user to type **`confirm reseed`** (not just
`y` — make it deliberate):
> About to RE-SEED references for model `<model_id>` from test `<test_file>`.
> This will OVERWRITE existing refs on
> `FastVideo/ssim-reference-videos/reference_videos/default/L40S_reference_videos/<model_id>/`
> after backup + Modal regen + eyeball.
>
> Reason: `<intent_rationale>`
> HEAD: `<git rev-parse --short=12 HEAD>`
>
> Reply `confirm reseed` to proceed, anything else to abort.
Stop until the user types exactly `confirm reseed`. Anything else aborts
with no side effects.
### 2. Back up existing refs
Always required. The backup is the only graceful path back if anything goes
wrong later.
```bash
SHORT_COMMIT=$(git rev-parse --short=12 HEAD)
TIMESTAMP=$(date -u +%Y%m%d_%H%M%S)
MODEL_SAFE=$(echo "<model_id>" | tr '/' '_')
BACKUP_DIR="ssim_reseed_backup/${TIMESTAMP}_${SHORT_COMMIT}_${MODEL_SAFE}"
mkdir -p "$BACKUP_DIR"
hf download \
--repo-type dataset FastVideo/ssim-reference-videos \
--include "reference_videos/default/L40S_reference_videos/<model_id>/**" \
--local-dir "$BACKUP_DIR"
mp4_count=$(find "$BACKUP_DIR" -name "*.mp4" | wc -l)
echo "Backup mp4 count: $mp4_count"
[ "$mp4_count" -gt 0 ] || {
echo "ERROR: backup is empty for <model_id>. Either the model id is wrong"
echo "or there are no existing refs (use seed-ssim-references instead)."
exit 1
}
# Provenance — used in the PR description
cat > "$BACKUP_DIR/PROVENANCE.txt" <<EOF
test_file: <test_file>
model_id: <model_id>
head_commit: $(git rev-parse HEAD)
timestamp_utc: $(date -u +%FT%TZ)
reason: <intent_rationale>
EOF
```
If the `hf download` produces zero mp4s, abort — the user has either picked a
non-existent `model_id` or there are no refs yet (in which case
`seed-ssim-references` is the right tool).
### 3. Regenerate on Modal L40S
Mirror CI's exact env recipe so the regenerated refs are byte-comparable to
what CI will produce on the same commit. Two differences from CI:
1. **Pass the same env prefix CI uses** (`IMAGE_VERSION`, `BUILDKITE_*`) — see
`.buildkite/pipeline.yml:1-3` and `.buildkite/scripts/pr_test.sh:62-83`.
Without this, `ssim_test.py:17-18` resolves a different GHCR image tag
(default is `latest`, CI is `py3.12-latest`), and `ssim_test.py:38-46`
bakes different values into the image's frozen env block. **Mismatched
image or env is the most common source of SSIM drift between reseed and
CI runs.**
2. **Do not pass `--skip-reference-download`**. Letting the test fetch the
existing refs and run the full SSIM compare gives "before" SSIM numbers
for the PR description, and the test still produces the new mp4s
regardless of whether the comparison passes or fails.
```bash
SUBDIR="${TIMESTAMP}_${SHORT_COMMIT}"
IMAGE_VERSION="py3.12-latest" \
BUILDKITE_REPO="$(git config --get remote.origin.url)" \
BUILDKITE_COMMIT="$(git rev-parse HEAD)" \
BUILDKITE_PULL_REQUEST="${BUILDKITE_PULL_REQUEST:-false}" \
modal run fastvideo/tests/modal/ssim_test.py \
--git-repo="$(git config --get remote.origin.url)" \
--git-commit="$(git rev-parse HEAD)" \
--hf-api-key="$HF_API_KEY" \
--test-files="<test_file>" \
--sync-generated-to-volume \
--generated-volume-subdir="$SUBDIR" \
--no-fail-fast
```
Capture the printed `modal volume get ...` hint — its `<SUBDIR>` matches
`$SUBDIR` and is needed for step 4. Capture the SSIM numbers from the test
output (or from the JSON next to the generated mp4) for the PR description.
### 4. Download generated videos
```bash
modal volume get --force hf-model-weights \
ssim_generated_videos/default/"$SUBDIR"/generated_videos \
./generated_videos_modal/default
```
After this, the new mp4s live at:
```
./generated_videos_modal/default/generated_videos/L40S_reference_videos/<model_id>/<backend>/<prompt>.mp4
```
`--force` is required when `./generated_videos_modal/default` already exists
from a prior run; safe on the first run too.
### 5. PAUSE — user reviews quality side-by-side
Print the diff and the comparison:
```bash
echo "=== File list diff (backup vs new) ==="
diff -u \
<(find "$BACKUP_DIR/reference_videos/default/L40S_reference_videos/<model_id>" -name "*.mp4" \
| sed "s|$BACKUP_DIR/reference_videos/default/L40S_reference_videos/||" | sort) \
<(find ./generated_videos_modal/default/generated_videos/L40S_reference_videos/<model_id> -name "*.mp4" \
| sed "s|./generated_videos_modal/default/generated_videos/L40S_reference_videos/||" | sort) \
|| true
echo
echo "=== SSIM numbers from this run (paste into PR) ==="
find ./generated_videos_modal/default/generated_videos/L40S_reference_videos/<model_id> -name "*_ssim.json" -exec cat {} \;
```
Then stop and tell the user:
> Old refs backed up to `$BACKUP_DIR`.
> New videos in `./generated_videos_modal/default/generated_videos/L40S_reference_videos/<model_id>/`.
>
> Open both in a video player. Confirm the new videos:
> 1. Look correct (no obvious artifacts, no black/static frames).
> 2. Are *intentionally* different from the backup in the way described
> in `<intent_rationale>` (e.g. slight numerical drift only, not a
> different scene / different motion / corrupted output).
>
> Reply **`upload`** to overwrite HF, anything else to abort.
> Aborting leaves the backup and new videos on disk for inspection — nothing
> on HF changes.
Do not proceed until the user types exactly `upload`. If they abort, leave
everything on disk and stop here.
### 6. Copy into the local reference layout
Same as `seed-ssim-references` step 5:
```bash
python fastvideo/tests/ssim/reference_videos_cli.py copy-local \
--quality-tier default \
--device-folder L40S_reference_videos \
--generated-dir ./generated_videos_modal/default/generated_videos/L40S_reference_videos
```
Result: `fastvideo/tests/ssim/reference_videos/default/L40S_reference_videos/<model_id>/<backend>/<prompt>.mp4`.
### 7. Upload with `--force`, scoped to `--model-id`
The `--force` flag is what makes this skill different from `seed-ssim-references`.
Always pair it with `--model-id` so a typo cannot accidentally overwrite a
neighboring model's refs.
```bash
python fastvideo/tests/ssim/reference_videos_cli.py upload \
--quality-tier default \
--device-folder L40S_reference_videos \
--model-id "<model_id>" \
--force
```
The CLI's overwrite guard refuses without `--force`; with `--force` it
overwrites only files under
`reference_videos/default/L40S_reference_videos/<model_id>/`.
### 8. Report success and retention guidance
Print:
- The HF path that was overwritten (`<repo>/reference_videos/default/L40S_reference_videos/<model_id>/`).
- The local backup directory path.
- The new SSIM numbers from step 5.
- This restore command, in case the PR review surfaces a problem after
upload:
```bash
python fastvideo/tests/ssim/reference_videos_cli.py upload \
--quality-tier default \
--device-folder L40S_reference_videos \
--model-id "<model_id>" \
--reference-dir "$BACKUP_DIR/reference_videos/default/L40S_reference_videos" \
--force
```
- This PR-description checklist (see `fastvideo/tests/ssim/AGENTS.md` →
*Updating Reference Videos*):
1. Source commit that produced the new refs (HEAD at re-seed time).
2. Test command and GPU SKU (`L40S`).
3. Before/after SSIM numbers.
4. The `<intent_rationale>` from step 1.
5. A note that the backup lives at `$BACKUP_DIR` and should be retained
until CI on the PR is green.
Do **not** auto-rerun the SSIM test — the user does that as part of the PR.
## Failure modes and how to handle them
- **`HF_API_KEY` unset.** Stop before step 2.
- **Backup is empty (zero mp4s).** Stop before step 3 — the model id is
wrong or the refs don't exist yet (use `seed-ssim-references`).
- **Modal run fails before generation.** No mp4s on the volume. Don't
upload. Investigate the failure (test crash, OOM, partition exhaustion),
fix, then retry from step 3. Backup is still intact.
- **Quality regressed (visual or metric).** User aborts at step 5. Backup
retained. New videos retained on disk for inspection. Nothing on HF
changed. Either fix the underlying code change or abandon the re-seed.
- **User confirmed `upload` but later realized the new refs are wrong.**
Run the restore command from step 8 with the backup `--reference-dir`.
This is exactly why the backup exists.
- **Multi-model test, only one model is being re-seeded.** Run the skill
once per model id. The `--model-id` scope on upload guarantees the others
are untouched.
## Design notes (for future skill maintainers)
- Per-`model_id` scope is mandatory. The dataset houses many model subtrees;
re-seeding the wrong one is hard to undo without backup.
- `default` tier only; `full_quality` is a separate, deliberate operation
with different params and ~doubled runtime, and isn't what CI gates on.
- The skill deliberately does **not** pass `--skip-reference-download` to
Modal so we get pre-reseed SSIM numbers for the PR. The `seed`-skill
passes it because no refs exist yet; for re-seed, refs do exist and
exposing the comparison is informative.
- The two-token confirm (`confirm reseed`, then `upload`) is intentional.
Re-seeding is high-blast-radius and should not be one-keystroke.
- The backup directory is plain mp4s + `PROVENANCE.txt`. No HF metadata is
preserved; the restore path uses `reference_videos_cli.py upload
--reference-dir` which doesn't need it.
## References
- `.agents/skills/seed-ssim-references/SKILL.md` — the first-time seed
skill this one parallels. Read it for the Modal flag rationale shared
between the two flows.
- `fastvideo/tests/ssim/AGENTS.md` — directory rules, including the PR
expectations for any reference-video change (rationale, before/after
SSIM, source commit/model/backend).
- `fastvideo/tests/ssim/reference_videos_cli.py` — `copy-local`, `upload`
(with `--model-id`, `--force`), `download`. The overwrite guard at
`upload_reference_videos` is the safety net this skill leans on.
- `fastvideo/tests/modal/ssim_test.py` — Modal orchestrator;
`--sync-generated-to-volume`, `--generated-volume-subdir`,
`--skip-reference-download`, `--no-fail-fast`.
## Changelog
| Date | Change |
|------|--------|
| 2026-05-02 | Initial version. Sister skill to `seed-ssim-references`, scoped to single `(test_file, model_id)` re-seeds, with mandatory backup and two-token confirm. |
@@ -1,378 +0,0 @@
---
name: seed-ssim-references
description: Seed HF reference artefacts for a single newly-added SSIM test (pixel `.mp4` for `run_text_to_video_similarity_test`-style tests, or latent `.pt` for `run_text_to_latent_similarity_test`-style tests). Runs the test on Modal L40S, downloads the generated artefacts via `modal volume get`, pauses for the user to verify (visual eyeball for mp4, numerics dump for pt), then uploads only that test's files to `FastVideo/ssim-reference-videos`. Use when a new `fastvideo/tests/ssim/test_*_similarity.py` has just been added and has no references on HF yet.
---
# Seed SSIM Reference Artefacts (mp4 or pt)
## Purpose
A brand-new SSIM test in `fastvideo/tests/ssim/` fails forever until its
reference artefacts exist on the HF dataset
(`FastVideo/ssim-reference-videos`). The dataset hosts two kinds of artefacts
side-by-side per `(model_id, backend, prompt)`:
- **`.mp4`** — pixel ground-truth for tests that call
`run_text_to_video_similarity_test` / `run_image_to_video_similarity_test`
in `inference_similarity_utils.py`. Compared via SSIM.
- **`.pt`** — pre-VAE latent bundle (fp16 full latent + fp32 slice +
metadata + `slice_spec` + `format_version`) for tests that call
`run_text_to_latent_similarity_test` in `latent_similarity_utils.py`.
Compared via cosine distance on the slice and the full tensor.
This skill:
1. Detects which artefact type the test produces (pixel vs latent).
2. Runs the test on Modal's L40S pool to generate the artefacts.
3. Downloads them to the local repo via `modal volume get`.
4. Pauses so the user can verify quality:
- **mp4**: visual eyeball in a video player.
- **pt**: numerics dump (shape, slice stats, NaN/Inf check, metadata).
5. Uploads only the new test's files to HF, with a guard that refuses to
overwrite anything already present.
The skill is run **manually**, once per new test. Before invoking it, the user
has already sanity-tested the new test locally — it launches `VideoGenerator`
and writes an artefact without crashing (the missing-reference assertion at
the end is expected). The skill does not re-test locally; it goes straight
to Modal L40S (which is what CI uses).
## When to use
- A new `test_*_similarity.py` file has been added in `fastvideo/tests/ssim/`
and the HF dataset has no `reference_videos/default/L40S_reference_videos/<model_id>/`
subtree for it yet.
## When not to use
- Regular CI runs — once refs exist, `pytest fastvideo/tests/ssim/` downloads
them automatically.
- Re-seeding an existing test. That requires `--force` on the upload step, and
is out of scope here; treat as a separate, deliberate operation.
## Inputs
The skill has **one required input**: the path to the new SSIM test file.
Prompt the user for it if they didn't supply it.
| Parameter | Required | Description |
|-----------|----------|-------------|
| `test_file` | Yes | e.g. `fastvideo/tests/ssim/test_ltx2_similarity.py`. The skill's first action is to ask for this if missing. |
Everything else is fixed:
- Modal runner GPU: **L40S** (hardcoded in `fastvideo/tests/modal/ssim_test.py`).
- Device folder: `L40S_reference_videos`.
- Quality tier: `default` (the tier CI runs). The `full_quality` tier is not
seeded by this skill.
- HF repo: `FastVideo/ssim-reference-videos` (dataset).
- Multi-model test files: all model ids in `*_MODEL_TO_PARAMS` are seeded
together; the Modal run produces one mp4 per (model, prompt, backend) and
the upload scopes by `--model-id`, looping if there is more than one.
## Prerequisites
The user has confirmed:
- `modal` CLI authenticated.
- `HF_API_KEY` (or `HUGGINGFACE_HUB_TOKEN` / `HF_TOKEN`) exported with write
access to `FastVideo/ssim-reference-videos`.
- The test file runs locally end-to-end (generates an mp4; SSIM assertion
failure due to missing reference is expected and fine).
Fail fast if the token env var is missing.
## Steps
### 1. Ask for the test file, then detect artefact type
If the user didn't name one, ask: *"Which SSIM test file do you want to seed
references for? (e.g. `fastvideo/tests/ssim/test_ltx2_similarity.py`)"*.
Validate:
- Path exists and matches `fastvideo/tests/ssim/test_*_similarity.py`.
- File defines a `*_MODEL_TO_PARAMS` dict — grep it to extract the set of
model ids. Those ids drive step 5.
Detect artefact type by inspecting the file's imports / helper call:
- **latent** (`.pt`) — file imports `run_text_to_latent_similarity_test`
from `fastvideo.tests.ssim.latent_similarity_utils` (or any other helper
that ends with `_latent_similarity_test`).
- **pixel** (`.mp4`) — file imports
`run_text_to_video_similarity_test` / `run_image_to_video_similarity_test`
from `fastvideo.tests.ssim.inference_similarity_utils`, OR uses the
legacy custom-inline helper pattern (see `test_gamecraft`,
`test_longcat`, etc.). Default to pixel when both heuristics fail.
Record `ARTEFACT_TYPE ∈ {pixel, latent}` for use in step 4. Steps 2, 3, 5,
and 6 are artefact-type-agnostic — `_iter_reference_files`,
`copy_generated_to_reference`, and `upload_reference_videos` already walk
both `.mp4` and `.pt` (see `reference_videos_cli.py`).
If either check fails, stop and tell the user what's wrong.
### 2. Run the test on Modal L40S
Pick a subdir name so repeated runs don't collide:
```bash
SHORT_COMMIT=$(git rev-parse --short=12 HEAD)
TIMESTAMP=$(date -u +%Y%m%d_%H%M%S)
SUBDIR="${TIMESTAMP}_${SHORT_COMMIT}"
```
Then launch the Modal run. The `IMAGE_VERSION` and `BUILDKITE_*` env-prefix
**must** match what CI exports in `.buildkite/scripts/pr_test.sh`, otherwise
`fastvideo/tests/modal/ssim_test.py` resolves a different GHCR image tag
(default is `latest`, CI is `py3.12-latest`) and bakes different values into
the image's frozen env block (`ssim_test.py:17-18, 38-46`). Mismatched image
or env produces SSIM drift that doesn't show up until the same commit runs
in CI.
```bash
IMAGE_VERSION="py3.12-latest" \
BUILDKITE_REPO="$(git config --get remote.origin.url)" \
BUILDKITE_COMMIT="$(git rev-parse HEAD)" \
BUILDKITE_PULL_REQUEST="${BUILDKITE_PULL_REQUEST:-false}" \
modal run fastvideo/tests/modal/ssim_test.py \
--git-repo="$(git config --get remote.origin.url)" \
--git-commit="$(git rev-parse HEAD)" \
--hf-api-key="$HF_API_KEY" \
--test-files="<test_file>" \
--sync-generated-to-volume \
--generated-volume-subdir="$SUBDIR" \
--skip-reference-download \
--no-fail-fast
```
Env prefix rationale (parity with CI; see `.buildkite/pipeline.yml:1-3` and
`.buildkite/scripts/pr_test.sh:62-83`):
- `IMAGE_VERSION=py3.12-latest`: pins the Modal image tag to the same one CI
uses. The published `py3.12-latest` and `latest` tags point at Python 3.12 /
CUDA 12.6.3 / cu126; `py3.12-cuda12.6.3-latest` is the explicit alias for the
same image. CUDA 13 / cu130 is available under the explicit
`py3.12-cuda13.0.0-latest` tag. This tag policy comes from
`infra-build-image.yml`; the unparameterized `docker/Dockerfile` build itself
still defaults to CUDA 13 / cu130.
- `BUILDKITE_REPO`/`BUILDKITE_COMMIT`/`BUILDKITE_PULL_REQUEST`: mirror what
Buildkite exports. `ssim_test.py:38-46` bakes these into the image's
`.env(...)` block; mismatched values can perturb in-container code paths
that branch on PR-vs-non-PR. `false` for `BUILDKITE_PULL_REQUEST` matches
Buildkite's "non-PR build" sentinel.
Flag rationale:
- `--skip-reference-download`: no refs exist yet, so conftest must not try to
pull them.
- `--no-fail-fast`: lets the test finish generation before `_assert_similarity`
raises `FileNotFoundError: Reference video folder does not exist`. The
expected failure is what we want — the mp4 has already been written.
- `--sync-generated-to-volume` + `--generated-volume-subdir`: copies the
generated mp4s to the `hf-model-weights` Modal volume under
`ssim_generated_videos/default/<SUBDIR>/generated_videos/` so we can pull
them locally.
The Modal run will end with a nonzero exit (expected) and print a
`modal volume get hf-model-weights ssim_generated_videos/default/<SUBDIR>/generated_videos ./generated_videos_modal/default`
command. Capture that `<SUBDIR>` — you need it for step 3.
### 3. Download generated videos locally
```bash
modal volume get --force hf-model-weights \
ssim_generated_videos/default/"$SUBDIR"/generated_videos \
./generated_videos_modal/default
```
`--force` is required when the parent `./generated_videos_modal/default`
already exists; without it, `modal volume get` errors with `[Errno 21] Is a
directory`. Safe to pass on the first run too.
After this, the mp4s live at
`./generated_videos_modal/default/generated_videos/L40S_reference_videos/<model_id>/<backend>/<prompt>.mp4`.
The extra `generated_videos/` level comes from the volume layout in
`_sync_generated_videos_to_volume` (`ssim_test.py`) — the command copies
`<repo>/fastvideo/tests/ssim/generated_videos/<tier>` to
`ssim_generated_videos/<tier>/<SUBDIR>/generated_videos/`, and `modal volume
get` preserves that trailing `generated_videos/` segment.
### 4. PAUSE — user reviews quality
Type-aware verification.
**For `ARTEFACT_TYPE = pixel`** — list the downloaded mp4s and ask the user to
open them in a video player:
> "Generated videos downloaded to `./generated_videos_modal/default/generated_videos/L40S_reference_videos/`. Please open them and confirm the quality looks correct. Reply **`upload`** to continue, or anything else to abort."
**For `ARTEFACT_TYPE = latent`** — `.pt` files are not human-watchable. Print
a numerics dump for each `.pt` so the user can sanity-check shape, distribution,
and metadata:
```python
import torch
from pathlib import Path
ROOT = Path("./generated_videos_modal/default/generated_videos/L40S_reference_videos")
for p in sorted(ROOT.rglob("*.pt")):
d = torch.load(p, map_location="cpu", weights_only=False)
s = d["expected_slice"]
L = d["latent"].float()
print(f"=== {p.relative_to(ROOT)} ===")
print(f" format_version: {d['format_version']}")
print(f" shape: {d['shape']}")
print(f" dtype_original: {d['dtype_original']}")
print(f" slice_spec: {d['slice_spec']}")
print(f" slice shape={tuple(s.shape)} mean={s.mean():+.4f} std={s.std():.4f} min={s.min():+.4f} max={s.max():+.4f}")
print(f" latent shape={tuple(L.shape)} mean={L.mean():+.4f} std={L.std():.4f} min={L.min():+.4f} max={L.max():+.4f}")
print(f" finite: latent NaN={torch.isnan(L).any().item()} Inf={torch.isinf(L).any().item()}; "
f"slice NaN={torch.isnan(s).any().item()} Inf={torch.isinf(s).any().item()}")
print(f" metadata: {d['metadata']}\n")
```
Sanity criteria:
- `format_version == 1` (matches `LATENT_REFERENCE_FORMAT_VERSION`).
- `shape` matches what the model produces (e.g. LTX-2 distilled =
`[1, 128, T_lat, H_lat, W_lat]`; Stable Audio Open 1.0 = `[1, 64, 1024]`).
- `slice_spec.kind` matches a registered kind (`corner_3x3_first_frame`
for video, `audio_first_8_timesteps` for audio).
- No `NaN`/`Inf`. `mean ≈ 0`, `std ≈ 1` (denoised latents stay close to
the initial Gaussian distribution; very wide deviations suggest
numerical drift).
- `metadata.prompt` matches the test's prompt.
Then ask:
> "Numerics look right? Reply **`upload`** to continue, or anything else to abort."
Do not proceed until the user explicitly says `upload`. If they abort, leave
everything on disk so they can inspect further — no cleanup.
### 5. Copy into the local reference layout
Scoped copy — only the new test's artefacts. Single command works for both
artefact types because `_iter_reference_files` walks `.mp4` and `.pt`:
```bash
python fastvideo/tests/ssim/reference_videos_cli.py copy-local \
--quality-tier default \
--device-folder L40S_reference_videos \
--generated-dir ./generated_videos_modal/default/generated_videos/L40S_reference_videos
```
(The `--generated-dir` points at the device-folder root inside the
downloaded tree; `copy-local` walks all `<model>/<backend>/*.{mp4,pt}`
underneath it. Since the Modal run was scoped to a single test file via
`--test-files`, only that test's model(s) are present — so the copy is
implicitly per-test.)
Result for pixel: `fastvideo/tests/ssim/reference_videos/default/L40S_reference_videos/<model_id>/<backend>/<prompt>.mp4`.
Result for latent: same path with `.pt` extension.
### 6. Upload to HF — scoped per model_id, with overwrite guard
For each `<model_id>`:
```bash
python fastvideo/tests/ssim/reference_videos_cli.py upload \
--quality-tier default \
--device-folder L40S_reference_videos \
--model-id "<model_id>"
```
The upload command:
- Uploads **only** `reference_videos/default/L40S_reference_videos/<model_id>/`.
- **Refuses** if any file already exists at that path on HF (this is the
guard — seeding a new test should never clobber existing refs). To override,
the user must re-run with `--force`. If the guard fires, stop and report
exactly which files exist; do not silently `--force`.
Reads the HF token from `HF_API_KEY` / `HUGGINGFACE_HUB_TOKEN` / `HF_TOKEN`.
### 7. Report success
List what was uploaded (paths in repo) and remind the user to push any
related code changes. Do **not** auto-verify by re-running Modal — the user
can run `pytest fastvideo/tests/ssim/<test_file>` later to confirm end-to-end;
it will auto-download the refs they just uploaded.
## Failure modes and how to handle them
- **`HF_API_KEY` unset.** Stop before step 2. The Modal run needs it (passed
via `--hf-api-key`), and step 6 needs it for upload. If the user
ran `hf auth login` instead of exporting an env var, read the cached
token via `huggingface_hub.get_token()` and forward it to Modal as
`--hf-api-key="$CACHED_TOKEN"`.
- **Modal run fails before generation.** No artefacts on the volume — nothing
to download. Fix the test locally (`pytest fastvideo/tests/ssim/<test_file>`)
and retry from step 2.
- **`./generated_videos_modal/default/L40S_reference_videos/` missing after
`modal volume get`.** The run didn't produce artefacts (most likely the
test crashed before writing, or `REQUIRED_GPUS` exceeded the partition
capacity — see Modal logs).
- **Latent test crashed with FSDP / inference_mode error
(`RuntimeError: Inference tensors do not track version counter`).** The
test must pass `init_kwargs_override={"use_fsdp_inference": False}` when
`sp_size == 1` — see `test_stable_audio_similarity.py` for the pattern.
Fix in the test, push, retry.
- **Upload guard fires (files already exist).** The test name / model id
collides with something already on HF. Verify the user actually wants to
replace existing refs; if so, re-run the upload with `--force`. If not,
rename the model id in `*_MODEL_TO_PARAMS` and re-seed.
- **Quality looks wrong in step 4.** Abort. The artefacts stay on disk for
inspection. The fix is usually in the test's params (resolution, steps,
seed) — edit the test, then re-run the skill.
- For latent: also check `slice_spec.kind` matches the latent rank
(`corner_3x3_first_frame` requires 5-D, `audio_first_8_timesteps`
requires 3-D); a rank/kind mismatch raises in `_extract_expected_slice`.
## Design notes (for future skill maintainers)
- The skill deliberately runs on Modal, **not** locally, because the CI
runner is L40S. Seeding from a different GPU SKU produces refs that CI's
L40S runs can't match (pixel SSIM drifts across SKUs; latent cosine has
tighter cross-SKU bf16 drift but the configured tolerances assume
same-SKU seed → same-SKU verify).
- The skill is default-tier only. `full_quality` refs are seeded by a
separate, deliberate operation — they double runtime and aren't what CI
gates on.
- The overwrite guard in `reference_videos_cli.py upload` is default-on
specifically because this skill exists. Re-seeding is a distinct operation
that requires explicit `--force`.
- Both artefact types share the same Modal flow: the orchestrator sets
`--skip-reference-download` + `--no-fail-fast`, runs pytest, the test's
helper writes the artefact (`.mp4` via `imageio` for pixel,
`save_latent_reference` → `torch.save` for latent) BEFORE the
missing-reference assertion raises. `_sync_generated_videos_to_volume` in
`ssim_test.py` does a `shutil.copytree` of the whole `generated_videos/`
tree, picking up `.mp4`, `.pt`, and the `*_ssim.json` / `*_latent.json`
metric files alongside.
## References
- `fastvideo/tests/modal/ssim_test.py` — Modal orchestrator; see
`--sync-generated-to-volume`, `--generated-volume-subdir`,
`--skip-reference-download`, `--no-fail-fast`.
- `fastvideo/tests/ssim/reference_videos_cli.py` — `copy-local`, `upload`
(with `--model-id`, `--force`), `download`, `ensure` subcommands.
Extension allowlist is `REFERENCE_EXTENSIONS = VIDEO_EXTENSIONS +
LATENT_EXTENSIONS` (`.pt`).
- `fastvideo/tests/ssim/README.md` — reference layout, HF repo conventions.
- `fastvideo/tests/ssim/inference_similarity_utils.py` — pixel helpers
(`run_text_to_video_similarity_test`,
`run_image_to_video_similarity_test`, `build_init_kwargs`).
- `fastvideo/tests/ssim/latent_similarity_utils.py` — latent helper
(`run_text_to_latent_similarity_test`), slice spec dispatch
(`_extract_expected_slice`), reference schema
(`save_latent_reference` / `load_latent_reference`),
`LATENT_REFERENCE_FORMAT_VERSION`.
## Changelog
| Date | Change |
|------|--------|
| 2026-04-17 | Initial version (Modal sync-to-volume flow). |
| 2026-04-21 | Rewrite: single-test scope, explicit user-review pause, per-`model_id` upload, HF overwrite guard. Dropped `scripts/seed_ssim.sh`. |
| 2026-04-21 | Post-first-run fixes: `modal volume get` needs `--force` when parent exists; download tree has an extra `generated_videos/` level so `--generated-dir` must reflect it. |
| 2026-05-01 | Latent (`*.pt`) artefact support: artefact-type detection in step 1, type-aware verification (visual eyeball for mp4, numerics dump for pt) in step 4, FSDP+inference_mode failure-mode added, design notes for the unified Modal flow. Triggered by PR #1253 (LTX-2 latent migration + Stable Audio latent test). |
@@ -1,51 +0,0 @@
{
"benchmark_id": "wan-t2v-1.3b-1gpu-gb10",
"config_schema_version": 2,
"workload_id": "wan-t2v",
"variant_id": "1.3b-sp1",
"benchmark_version": 3,
"description": "Wan2.1 T2V 1.3B single-GPU inference performance on NVIDIA DGX Spark (GB10). Single-GPU variant of wan-t2v-1.3b (same workload_id for dashboard comparability). Gated to the GB10 via run_config.gpu_types so it does not run on the shared H100/L40S lanes.",
"model": {
"model_path": "Wan-AI/Wan2.1-T2V-1.3B-Diffusers",
"model_short_name": "Wan2.1-T2V-1.3B"
},
"init_kwargs": {
"num_gpus": 1,
"flow_shift": 7.0,
"sp_size": 1,
"tp_size": 1,
"vae_sp": false,
"vae_tiling": true,
"text_encoder_precisions": ["fp32"]
},
"generation_kwargs": {
"height": 480,
"width": 832,
"num_frames": 45,
"num_inference_steps": 4,
"guidance_scale": 3,
"embedded_cfg_scale": 6,
"seed": 1024,
"fps": 24,
"neg_prompt": "Bright tones, overexposed, static, blurred details, subtitles, style, works, paintings, images, static, overall gray, worst quality, low quality, JPEG compression residue, ugly, incomplete, extra fingers, poorly drawn hands, poorly drawn faces, deformed, disfigured, misshapen limbs, fused fingers, still picture, messy background, three legs, many people in the background, walking backwards"
},
"test_prompts": [
"Will Smith casually eats noodles, his relaxed demeanor contrasting with the energetic background of a bustling street food market. The scene captures a mix of humor and authenticity. Mid-shot framing, vibrant lighting."
],
"run_config": {
"num_warmup_runs": 2,
"num_measurement_runs": 5,
"required_gpus": 1,
"gpu_types": ["GB10"]
},
"thresholds": {
"GB10": {
"max_generation_time_s": 55.0,
"max_peak_memory_mb": 12000.0
},
"default": {
"max_generation_time_s": 120.0,
"max_peak_memory_mb": 40000.0
}
}
}
@@ -1,53 +0,0 @@
{
"benchmark_id": "wan-t2v-1.3b-2gpu",
"config_schema_version": 2,
"workload_id": "wan-t2v",
"variant_id": "1.3b-sp2",
"benchmark_version": 3,
"description": "Wan2.1 T2V 1.3B inference performance",
"model": {
"model_path": "Wan-AI/Wan2.1-T2V-1.3B-Diffusers",
"model_short_name": "Wan2.1-T2V-1.3B"
},
"init_kwargs": {
"num_gpus": 2,
"flow_shift": 7.0,
"sp_size": 2,
"tp_size": 1,
"vae_sp": true,
"vae_tiling": true,
"text_encoder_precisions": ["fp32"]
},
"generation_kwargs": {
"height": 480,
"width": 832,
"num_frames": 45,
"num_inference_steps": 4,
"guidance_scale": 3,
"embedded_cfg_scale": 6,
"seed": 1024,
"fps": 24,
"neg_prompt": "Bright tones, overexposed, static, blurred details, subtitles, style, works, paintings, images, static, overall gray, worst quality, low quality, JPEG compression residue, ugly, incomplete, extra fingers, poorly drawn hands, poorly drawn faces, deformed, disfigured, misshapen limbs, fused fingers, still picture, messy background, three legs, many people in the background, walking backwards"
},
"test_prompts": [
"Will Smith casually eats noodles, his relaxed demeanor contrasting with the energetic background of a bustling street food market. The scene captures a mix of humor and authenticity. Mid-shot framing, vibrant lighting."
],
"run_config": {
"num_warmup_runs": 2,
"num_measurement_runs": 5,
"required_gpus": 2
},
"thresholds": {
"L40S": {
"max_generation_time_s": 34.0,
"max_peak_memory_mb": 11000.0,
"max_text_encoder_time_s": 5.0,
"max_dit_time_s": 10.0,
"max_vae_decode_time_s": 10.0
},
"default": {
"max_generation_time_s": 120.0,
"max_peak_memory_mb": 30000.0
}
}
}
-548
View File
@@ -1,548 +0,0 @@
env:
IMAGE_VERSION: "py3.12-latest"
BUILDKITE_CLEAN_CHECKOUT: true
# Buildkite only launches Modal; remote jobs initialize their own submodules.
BUILDKITE_GIT_SUBMODULES: false
notify:
- github_commit_status:
context: "fastcheck-passed"
if: build.env("TEST_SCOPE") == "fastcheck" || build.env("TEST_SCOPE") == null
- github_commit_status:
context: "full-suite-passed"
if: build.env("TEST_SCOPE") == "full"
- github_commit_status:
context: "direct-test-completed"
if: build.env("TEST_SCOPE") == "direct"
steps:
# ============================================================
# Direct test: triggered by /test <name> slash command.
# Labels match fastcheck/full-suite counterparts so the GitHub
# check status overwrites the original failed check.
# Only ONE step executes per build (gated by TEST_TYPE).
# ============================================================
# --- Fastcheck-scope direct tests ---
- label: ":microscope: Encoder Tests"
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "encoder"
command: "timeout 90m .buildkite/scripts/pr_test.sh"
retry:
automatic:
- exit_status: 128
limit: 3
- exit_status: -1
limit: 2
agents:
queue: "default"
- label: ":microscope: VAE Tests"
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "vae"
command: "timeout 90m .buildkite/scripts/pr_test.sh"
retry:
automatic:
- exit_status: 128
limit: 3
- exit_status: -1
limit: 2
agents:
queue: "default"
- label: ":microscope: Transformer Tests"
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "transformer"
command: "timeout 90m .buildkite/scripts/pr_test.sh"
retry:
automatic:
- exit_status: 128
limit: 3
- exit_status: -1
limit: 2
agents:
queue: "default"
- label: ":microscope: Kernel Tests"
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "kernel_tests"
command: "timeout 90m .buildkite/scripts/pr_test.sh"
retry:
automatic:
- exit_status: 128
limit: 3
- exit_status: -1
limit: 2
agents:
queue: "default"
- label: ":vertical_traffic_light: Golden-Gate Tests"
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "golden_gate"
command: "timeout 90m .buildkite/scripts/pr_test.sh"
retry:
automatic:
- exit_status: 128
limit: 3
- exit_status: -1
limit: 2
agents:
queue: "default"
- label: ":microscope: Unit Tests"
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "unit_test"
command: "timeout 90m .buildkite/scripts/pr_test.sh"
retry:
automatic:
- exit_status: 128
limit: 3
- exit_status: -1
limit: 2
agents:
queue: "default"
- label: ":microscope: DreamVerse App Tests"
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "dreamverse_app"
command: "timeout 90m .buildkite/scripts/pr_test.sh"
retry:
automatic:
- exit_status: 128
limit: 3
- exit_status: -1
limit: 2
agents:
queue: "default"
# --- Full-suite-scope direct tests ---
- label: ":bar_chart: SSIM Tests"
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "ssim"
command: "timeout 90m .buildkite/scripts/pr_test.sh"
retry:
automatic:
- exit_status: 128
limit: 3
- exit_status: -1
limit: 2
- exit_status: 1
limit: 2
agents:
queue: "default"
- label: ":test_tube: LoRA Inference Tests"
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "inference_lora"
command: "timeout 90m .buildkite/scripts/pr_test.sh"
retry:
automatic:
- exit_status: 128
limit: 3
- exit_status: -1
limit: 2
agents:
queue: "default"
- label: ":test_tube: LoRA Extraction Tests"
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "lora_extraction"
command: "timeout 90m .buildkite/scripts/pr_test.sh"
retry:
automatic:
- exit_status: 128
limit: 3
- exit_status: -1
limit: 2
agents:
queue: "default"
- label: ":test_tube: Training Tests"
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "training"
command: "timeout 90m .buildkite/scripts/pr_test.sh"
retry:
automatic:
- exit_status: 128
limit: 3
- exit_status: -1
limit: 2
agents:
queue: "default"
- label: ":test_tube: Distillation DMD Tests"
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "distillation_dmd"
command: "timeout 90m .buildkite/scripts/pr_test.sh"
retry:
automatic:
- exit_status: 128
limit: 3
- exit_status: -1
limit: 2
agents:
queue: "default"
- label: ":test_tube: Self-Forcing Tests"
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "self_forcing"
command: "timeout 90m .buildkite/scripts/pr_test.sh"
retry:
automatic:
- exit_status: 128
limit: 3
- exit_status: -1
limit: 2
agents:
queue: "default"
- label: ":test_tube: LoRA Training Tests"
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "training_lora"
command: "timeout 90m .buildkite/scripts/pr_test.sh"
retry:
automatic:
- exit_status: 128
limit: 3
- exit_status: -1
limit: 2
- exit_status: 1
limit: 2
agents:
queue: "default"
- label: ":test_tube: Training Tests VSA"
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "training_vsa"
command: "timeout 90m .buildkite/scripts/pr_test.sh"
retry:
automatic:
- exit_status: 128
limit: 3
- exit_status: -1
limit: 2
- exit_status: 1
limit: 2
agents:
queue: "default"
- label: ":test_tube: Inference Tests VMoBA"
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "inference_vmoba"
command: "timeout 90m .buildkite/scripts/pr_test.sh"
retry:
automatic:
- exit_status: 128
limit: 3
- exit_status: -1
limit: 2
agents:
queue: "default"
- label: ":test_tube: Performance Tests"
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "performance"
command: "timeout 90m .buildkite/scripts/pr_test.sh"
retry:
automatic:
- exit_status: 128
limit: 3
- exit_status: -1
limit: 2
agents:
queue: "default"
- label: ":test_tube: API Server Tests"
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "api_server"
command: "timeout 90m .buildkite/scripts/pr_test.sh"
retry:
automatic:
- exit_status: 128
limit: 3
- exit_status: -1
limit: 2
agents:
queue: "default"
- label: ":test_tube: Train Framework Tests"
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "train_framework"
command: "timeout 90m .buildkite/scripts/pr_test.sh"
retry:
automatic:
- exit_status: 128
limit: 3
- exit_status: -1
limit: 2
agents:
queue: "default"
- label: ":test_tube: Eval Metrics Tests"
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "eval"
command: "timeout 90m .buildkite/scripts/pr_test.sh"
retry:
automatic:
- exit_status: 128
limit: 3
- exit_status: -1
limit: 2
agents:
queue: "default"
# ============================================================
# Fastcheck: Runs on every PR (~10-15 min parallel)
# Core component validation: encoders, VAEs, transformers,
# CUDA kernels, and unit tests.
# ============================================================
- label: "Trigger Fastcheck"
if: build.env("TEST_SCOPE") == "fastcheck" || build.env("TEST_SCOPE") == null
retry:
automatic:
- exit_status: 128
limit: 3
- exit_status: -1
limit: 2
plugins:
- monorepo-diff#v1.4.0:
diff: 'git fetch origin "${BUILDKITE_PULL_REQUEST_BASE_BRANCH:-main}" && git diff --name-only "origin/${BUILDKITE_PULL_REQUEST_BASE_BRANCH:-main}...HEAD"'
watch:
- path:
- "fastvideo/models/encoders/**"
- "fastvideo/models/loader/**"
- "fastvideo/tests/encoders/**"
- "pyproject.toml"
- "docker/Dockerfile"
config:
command: "timeout 20m .buildkite/scripts/pr_test.sh"
label: ":microscope: Encoder Tests"
env:
- TEST_TYPE=encoder
agents:
queue: "default"
- path:
- "fastvideo/models/vaes/**"
- "fastvideo/models/loader/**"
- "fastvideo/tests/vaes/**"
- "pyproject.toml"
- "docker/Dockerfile"
config:
command: "timeout 20m .buildkite/scripts/pr_test.sh"
label: ":microscope: VAE Tests"
env:
- TEST_TYPE=vae
agents:
queue: "default"
- path:
- "fastvideo/models/dits/**"
- "fastvideo/models/loader/**"
- "fastvideo/tests/transformers/**"
- "fastvideo/layers/**"
- "fastvideo/attention/**"
- "pyproject.toml"
- "docker/Dockerfile"
config:
command: "timeout 15m .buildkite/scripts/pr_test.sh"
label: ":microscope: Transformer Tests"
env:
- TEST_TYPE=transformer
agents:
queue: "default"
- path:
- "fastvideo-kernel/**"
- "pyproject.toml"
- "docker/Dockerfile"
config:
command: "timeout 15m .buildkite/scripts/pr_test.sh"
label: ":microscope: Kernel Tests"
env:
- TEST_TYPE=kernel_tests
agents:
queue: "default"
- path:
- "fastvideo/**"
- ".buildkite/**"
- ".github/**"
- "pyproject.toml"
- "docker/Dockerfile"
config:
command: "timeout 15m .buildkite/scripts/pr_test.sh"
label: ":microscope: Unit Tests"
env:
- TEST_TYPE=unit_test
agents:
queue: "default"
- path:
- "apps/dreamverse/**"
- "pyproject.toml"
config:
command: "timeout 30m .buildkite/scripts/pr_test.sh"
label: ":microscope: DreamVerse App Tests"
env:
- TEST_TYPE=dreamverse_app
agents:
queue: "default"
# ============================================================
# Full Suite: Runs when TEST_SCOPE=full
# Triggered by adding the 'ready' label (via ci-trigger-full-suite.yml)
# or on-demand via /test full slash command.
# Includes integration tests, SSIM regression, training pipelines,
# and performance benchmarks.
# ============================================================
- label: "Trigger Full Suite"
if: build.env("TEST_SCOPE") == "full"
retry:
automatic:
- exit_status: 128
limit: 3
- exit_status: -1
limit: 2
plugins:
- monorepo-diff#v1.4.0:
diff: 'git fetch origin "${BUILDKITE_PULL_REQUEST_BASE_BRANCH:-main}" && git diff --name-only "origin/${BUILDKITE_PULL_REQUEST_BASE_BRANCH:-main}...HEAD"'
watch:
- path:
- "fastvideo/**/*.py"
- "pyproject.toml"
- "docker/Dockerfile"
config:
command: "timeout 90m .buildkite/scripts/pr_test.sh"
label: ":bar_chart: SSIM Tests"
env:
- TEST_TYPE=ssim
retry:
automatic:
- exit_status: 1
limit: 2
agents:
queue: "default"
- path:
- "fastvideo/tests/lora/**"
- "fastvideo/models/loader/**"
- "fastvideo/tests/transformers/**"
- "fastvideo/pipelines/**"
- "fastvideo/layers/lora/**"
- "pyproject.toml"
- "docker/Dockerfile"
config:
command: "timeout 20m .buildkite/scripts/pr_test.sh"
label: ":test_tube: LoRA Inference Tests"
env:
- TEST_TYPE=inference_lora
agents:
queue: "default"
- path:
- "scripts/lora_extraction/**"
- "fastvideo/tests/lora_extraction/**"
- "fastvideo/models/loader/**"
- "fastvideo/training/training_utils.py"
- "fastvideo/layers/lora/**"
- "pyproject.toml"
- "docker/Dockerfile"
config:
command: "timeout 90m .buildkite/scripts/pr_test.sh"
label: ":test_tube: LoRA Extraction Tests"
env:
- TEST_TYPE=lora_extraction
agents:
queue: "default"
- path:
- "fastvideo/**"
- "pyproject.toml"
- "docker/Dockerfile"
config:
command: "timeout 25m .buildkite/scripts/pr_test.sh"
label: ":test_tube: Training Tests"
env:
- TEST_TYPE=training
agents:
queue: "default"
- path:
- "fastvideo/training/*distillation_pipeline.py"
- "pyproject.toml"
- "docker/Dockerfile"
config:
command: "timeout 25m .buildkite/scripts/pr_test.sh"
label: ":test_tube: Distillation DMD Tests"
env:
- TEST_TYPE=distillation_dmd
agents:
queue: "default"
- path:
- "fastvideo/training/*self_forcing_distillation_pipeline.py"
- "fastvideo/tests/training/self-forcing/**"
- "pyproject.toml"
- "docker/Dockerfile"
config:
command: "timeout 30m .buildkite/scripts/pr_test.sh"
label: ":test_tube: Self-Forcing Tests"
env:
- TEST_TYPE=self_forcing
agents:
queue: "default"
- path:
- "fastvideo/**"
- "pyproject.toml"
- "docker/Dockerfile"
config:
command: "timeout 25m .buildkite/scripts/pr_test.sh"
label: ":test_tube: LoRA Training Tests"
env:
- TEST_TYPE=training_lora
retry:
automatic:
- exit_status: 1
limit: 2
agents:
queue: "default"
- path:
- "fastvideo/**"
- "fastvideo-kernel/**"
- "pyproject.toml"
- "docker/Dockerfile"
config:
command: "timeout 15m .buildkite/scripts/pr_test.sh"
label: ":test_tube: Training Tests VSA"
env:
- TEST_TYPE=training_vsa
retry:
automatic:
- exit_status: 1
limit: 2
agents:
queue: "default"
- path:
- "fastvideo-kernel/**"
- "fastvideo/attention/backends/vmoba.py"
- "pyproject.toml"
- "docker/Dockerfile"
config:
command: "timeout 15m .buildkite/scripts/pr_test.sh"
label: ":test_tube: Inference Tests VMoBA"
env:
- TEST_TYPE=inference_vmoba
agents:
queue: "default"
- path:
- "fastvideo/models/dits/**"
- "fastvideo/pipelines/**"
- "fastvideo/attention/**"
- "fastvideo/layers/**"
- "fastvideo/worker/**"
- "fastvideo/entrypoints/**"
- "fastvideo/performance/**"
- "fastvideo/tests/performance/**"
- ".buildkite/performance-benchmarks/**"
- "pyproject.toml"
- "docker/Dockerfile"
config:
command: "timeout 30m .buildkite/scripts/pr_test.sh"
label: ":test_tube: Performance Tests"
env:
- TEST_TYPE=performance
agents:
queue: "default"
- path:
- "fastvideo/entrypoints/openai/**"
- "fastvideo/entrypoints/cli/serve.py"
- "fastvideo/tests/entrypoints/test_openai_api_integration.py"
- "pyproject.toml"
- "docker/Dockerfile"
config:
command: "timeout 30m .buildkite/scripts/pr_test.sh"
label: ":test_tube: API Server Tests"
env:
- TEST_TYPE=api_server
agents:
queue: "default"
- path:
- "fastvideo/train/**"
- "fastvideo/tests/train/models/**"
- "fastvideo/tests/train/fixtures/**"
- "fastvideo/models/dits/**"
- "fastvideo/models/loader/**"
- "pyproject.toml"
- "docker/Dockerfile"
config:
command: "timeout 30m .buildkite/scripts/pr_test.sh"
label: ":test_tube: Train Framework Tests"
env:
- TEST_TYPE=train_framework
agents:
queue: "default"
- path:
- "fastvideo/eval/**"
- "fastvideo/tests/eval/**"
- "pyproject.toml"
- "docker/Dockerfile"
config:
command: "timeout 90m .buildkite/scripts/pr_test.sh"
label: ":test_tube: Eval Metrics Tests"
env:
- TEST_TYPE=eval
agents:
queue: "default"
-287
View File
@@ -1,287 +0,0 @@
#!/bin/bash
set -uo pipefail
log() {
echo "[$(date '+%Y-%m-%d %H:%M:%S')] $1"
}
log "=== Starting Modal test execution ==="
# Change to the project directory
cd "$(dirname "$0")/../.."
PROJECT_ROOT=$(pwd)
log "Project root: $PROJECT_ROOT"
# Install Modal if not available
if ! python3 -m modal --version &> /dev/null; then
log "Modal not found, installing..."
if ! command -v uv &> /dev/null; then
log "uv not found, bootstrapping..."
if ! curl -LsSf https://astral.sh/uv/install.sh | sh; then
log "Error: Failed to bootstrap uv via astral.sh installer."
exit 1
fi
export PATH="$HOME/.local/bin:$PATH"
if ! command -v uv &> /dev/null; then
log "Error: uv still not on PATH after bootstrap."
exit 1
fi
fi
# --break-system-packages preserves prior `pip install --user` semantics on PEP 668 agents.
uv pip install --system --break-system-packages modal
# Verify installation
if ! python3 -m modal --version &> /dev/null; then
log "Error: Failed to install modal. Please install it manually."
exit 1
fi
fi
log "modal version: $(python3 -m modal --version)"
# Set up Modal authentication using Buildkite secrets
log "Setting up Modal authentication from Buildkite secrets..."
MODAL_TOKEN_ID=$(buildkite-agent secret get modal_token_id)
MODAL_TOKEN_SECRET=$(buildkite-agent secret get modal_token_secret)
# Retrieve other secrets
WANDB_API_KEY=$(buildkite-agent secret get wandb_api_key)
HF_API_KEY=$(buildkite-agent secret get hf_api_key)
if [ -n "$MODAL_TOKEN_ID" ] && [ -n "$MODAL_TOKEN_SECRET" ]; then
log "Retrieved Modal credentials from Buildkite secrets"
python3 -m modal token set --token-id "$MODAL_TOKEN_ID" --token-secret "$MODAL_TOKEN_SECRET" --profile buildkite-ci --activate --verify
if [ $? -eq 0 ]; then
log "Modal authentication successful"
else
log "Error: Failed to set Modal credentials"
exit 1
fi
else
log "Error: Could not retrieve Modal credentials from Buildkite secrets."
log "Please ensure 'modal_token_id' and 'modal_token_secret' secrets are set in Buildkite."
exit 1
fi
MODAL_TEST_FILE="fastvideo/tests/modal/pr_test.py"
MODAL_SSIM_TEST_FILE="fastvideo/tests/modal/ssim_test.py"
if [ -z "${TEST_TYPE:-}" ]; then
log "Error: TEST_TYPE environment variable is not set"
exit 1
fi
log "Test type: $TEST_TYPE"
EFFECTIVE_PR=${BUILDKITE_PULL_REQUEST:-false}
if [ "$EFFECTIVE_PR" = "false" ] && [ -n "${PR_NUMBER:-}" ]; then
EFFECTIVE_PR=$PR_NUMBER
fi
MODAL_ENV="BUILDKITE_REPO=$BUILDKITE_REPO BUILDKITE_COMMIT=$BUILDKITE_COMMIT BUILDKITE_PULL_REQUEST=$EFFECTIVE_PR BUILDKITE_BRANCH=${BUILDKITE_BRANCH:-} BUILDKITE_SOURCE=${BUILDKITE_SOURCE:-} TEST_SCOPE=${TEST_SCOPE:-} BUILDKITE_BUILD_URL=${BUILDKITE_BUILD_URL:-} BUILDKITE_BUILD_ID=${BUILDKITE_BUILD_ID:-} BUILDKITE_JOB_ID=${BUILDKITE_JOB_ID:-} IMAGE_VERSION=$IMAGE_VERSION"
POST_RUN_HOOK=""
is_truthy() {
case "${1:-}" in
1|true|TRUE|yes|YES|on|ON) return 0 ;;
*) return 1 ;;
esac
}
ssim_bootstrap_args() {
local title="${PR_TITLE:-}"
local message="${BUILDKITE_MESSAGE:-}"
if is_truthy "${FASTVIDEO_SSIM_BOOTSTRAP_MODE:-}" \
|| [[ "$title" == *"[new-model]"* ]] \
|| [[ "$message" == *"[new-model]"* ]]; then
printf ' --bootstrap-mode'
fi
}
upload_performance_artifacts() {
SHORT_SHA=${BUILDKITE_COMMIT:0:7}
LOCAL_DIR="downloaded_reports"
_download_reports() {
log "Downloading perf_reports/ from Modal Volume..."
mkdir -p "$LOCAL_DIR"
if ! modal volume get hf-model-weights "perf_reports/" "$LOCAL_DIR"; then
log "Error: Failed to download perf_reports/ from Modal Volume."
return 1
fi
}
_upload_dashboard() {
local target
target=$(find "$LOCAL_DIR" -name "dashboard_${SHORT_SHA}_*" | head -n 1)
log "TARGET dashboard: '$target'"
if [ -n "$target" ]; then
log "Found dashboard: $target. Uploading to Buildkite..."
buildkite-agent artifact upload "$target"
buildkite-agent annotate --style info --context "perf-dashboard" < "$target"
else
log "Warning: Could not find a dashboard file matching $SHORT_SHA"
fi
}
_upload_perf_summary() {
local target
target=$(find "$LOCAL_DIR" -name "perf_${SHORT_SHA}_*" | head -n 1)
log "TARGET perf summary: '$target'"
if [ -n "$target" ]; then
log "Found perf summary: $target. Uploading to Buildkite..."
buildkite-agent artifact upload "$target"
buildkite-agent annotate --style info --context "perf-summary" < "$target"
else
log "Warning: Could not find a perf summary file matching $SHORT_SHA"
fi
}
_upload_normalized_perf_results() {
local found=0
while IFS= read -r -d '' target; do
found=1
log "Found normalized performance result: $target. Uploading to Buildkite..."
buildkite-agent artifact upload "$target"
done < <(find "$LOCAL_DIR" -path "*/results/normalized_perf_*.json" -print0)
if [ "$found" -eq 0 ]; then
log "No normalized performance result artifacts found. This is expected when the rolling performance comparison did not run."
fi
}
_cleanup_modal_volume() {
log "Cleaning up perf_reports/ from Modal Volume..."
if modal volume rm hf-model-weights "perf_reports/" --recursive; then
log "Successfully deleted perf_reports/ from Modal Volume."
else
log "Warning: Failed to delete perf_reports/ from Modal Volume. Manual cleanup may be required."
fi
}
_cleanup_local() {
log "Cleaning up local download directory..."
rm -rf "$LOCAL_DIR"
}
# --- Main flow ---
_download_reports || { _cleanup_local; return 1; }
_upload_dashboard
_upload_perf_summary
_upload_normalized_perf_results
_cleanup_modal_volume
_cleanup_local
}
case "$TEST_TYPE" in
"encoder")
log "Running encoder tests..."
MODAL_COMMAND="$MODAL_ENV HF_API_KEY=$HF_API_KEY python3 -m modal run $MODAL_TEST_FILE::run_encoder_tests"
;;
"vae")
log "Running VAE tests..."
MODAL_COMMAND="$MODAL_ENV HF_API_KEY=$HF_API_KEY python3 -m modal run $MODAL_TEST_FILE::run_vae_tests"
;;
"transformer")
log "Running transformer tests..."
MODAL_COMMAND="$MODAL_ENV HF_API_KEY=$HF_API_KEY python3 -m modal run $MODAL_TEST_FILE::run_transformer_tests"
;;
"golden_gate")
log "Running golden-gate tests..."
MODAL_COMMAND="$MODAL_ENV HF_API_KEY=$HF_API_KEY python3 -m modal run $MODAL_TEST_FILE::run_golden_gate_tests"
;;
"ssim")
log "Running SSIM tests..."
SSIM_BOOTSTRAP_ARGS=$(ssim_bootstrap_args)
if [ -n "$SSIM_BOOTSTRAP_ARGS" ]; then
log "SSIM bootstrap mode enabled for new-model reference draft generation"
fi
MODAL_COMMAND="$MODAL_ENV HF_API_KEY=$HF_API_KEY python3 -m modal run "
MODAL_COMMAND+="$MODAL_SSIM_TEST_FILE::run_ssim_tests$SSIM_BOOTSTRAP_ARGS"
;;
"training")
log "Running training tests..."
MODAL_COMMAND="$MODAL_ENV WANDB_API_KEY=$WANDB_API_KEY python3 -m modal run $MODAL_TEST_FILE::run_training_tests"
;;
"training_lora")
log "Running LoRA training tests..."
MODAL_COMMAND="$MODAL_ENV WANDB_API_KEY=$WANDB_API_KEY python3 -m modal run $MODAL_TEST_FILE::run_training_lora_tests"
;;
"training_vsa")
log "Running training VSA tests..."
MODAL_COMMAND="$MODAL_ENV WANDB_API_KEY=$WANDB_API_KEY python3 -m modal run $MODAL_TEST_FILE::run_training_tests_VSA"
;;
"kernel_tests")
log "Running kernel tests..."
MODAL_COMMAND="$MODAL_ENV python3 -m modal run $MODAL_TEST_FILE::run_kernel_tests"
;;
"inference_lora")
log "Running LoRA tests..."
MODAL_COMMAND="$MODAL_ENV python3 -m modal run $MODAL_TEST_FILE::run_inference_lora_tests"
;;
"distillation_dmd")
log "Running distillation DMD tests..."
MODAL_COMMAND="$MODAL_ENV WANDB_API_KEY=$WANDB_API_KEY python3 -m modal run $MODAL_TEST_FILE::run_distill_dmd_tests"
;;
# run_inference_tests_vmoba
"self_forcing")
log "Running self-forcing tests..."
MODAL_COMMAND="$MODAL_ENV WANDB_API_KEY=$WANDB_API_KEY python3 -m modal run $MODAL_TEST_FILE::run_self_forcing_tests"
;;
"inference_vmoba")
log "Running V-MoBA inference tests..."
MODAL_COMMAND="$MODAL_ENV python3 -m modal run $MODAL_TEST_FILE::run_inference_tests_vmoba"
;;
"unit_test")
log "Running unit tests..."
MODAL_COMMAND="$MODAL_ENV python3 -m modal run $MODAL_TEST_FILE::run_unit_test"
;;
"dreamverse_app")
log "Running DreamVerse app tests..."
MODAL_COMMAND="$MODAL_ENV python3 -m modal run $MODAL_TEST_FILE::run_dreamverse_app_tests"
;;
"train_framework")
log "Running fastvideo.train framework tests..."
MODAL_COMMAND="$MODAL_ENV HF_API_KEY=$HF_API_KEY python3 -m modal run $MODAL_TEST_FILE::run_train_framework_tests"
;;
"eval")
log "Running eval metric tests..."
MODAL_COMMAND="$MODAL_ENV HF_API_KEY=$HF_API_KEY python3 -m modal run $MODAL_TEST_FILE::run_eval_tests"
;;
"lora_extraction")
log "Running LoRA extraction tests..."
MODAL_COMMAND="$MODAL_ENV HF_API_KEY=$HF_API_KEY python3 -m modal run $MODAL_TEST_FILE::run_lora_extraction_tests"
;;
"performance")
log "Running performance tests on Modal..."
MODAL_COMMAND="$MODAL_ENV HF_API_KEY=$HF_API_KEY python3 -m modal run $MODAL_TEST_FILE::run_performance_tests"
POST_RUN_HOOK="upload_performance_artifacts"
;;
"api_server")
log "Running API server integration tests..."
MODAL_COMMAND="$MODAL_ENV HF_API_KEY=$HF_API_KEY python3 -m modal run $MODAL_TEST_FILE::run_api_server_tests"
;;
*)
log "Error: Unknown test type: $TEST_TYPE"
exit 1
;;
esac
log "Executing: $MODAL_COMMAND"
eval "$MODAL_COMMAND"
TEST_EXIT_CODE=$?
if [ $TEST_EXIT_CODE -eq 0 ]; then
log "Modal test completed successfully"
else
log "Error: Modal test failed with exit code: $TEST_EXIT_CODE"
fi
if [ -n "$POST_RUN_HOOK" ]; then
log "Executing post-run hook: $POST_RUN_HOOK"
"$POST_RUN_HOOK"
fi
log "=== Test execution completed with exit code: $TEST_EXIT_CODE ==="
exit $TEST_EXIT_CODE
-53
View File
@@ -1,53 +0,0 @@
#!/bin/bash
set -uo pipefail
log() {
echo "[$(date '+%Y-%m-%d %H:%M:%S')] $1"
}
log "=== Starting pre-commit checks ==="
cd "$(dirname "$0")/../.."
PROJECT_ROOT=$(pwd)
log "Project root: $PROJECT_ROOT"
if ! python3 -m pre_commit --version &> /dev/null; then
log "pre-commit not found, installing..."
if ! command -v uv &> /dev/null; then
log "uv not found, bootstrapping..."
if ! curl -LsSf https://astral.sh/uv/install.sh | sh; then
log "Error: Failed to bootstrap uv via astral.sh installer."
exit 1
fi
export PATH="$HOME/.local/bin:$PATH"
if ! command -v uv &> /dev/null; then
log "Error: uv still not on PATH after bootstrap."
exit 1
fi
fi
# --break-system-packages preserves prior `pip install --user` semantics on PEP 668 agents.
uv pip install --system --break-system-packages pre-commit==4.0.1
if ! python3 -m pre_commit --version &> /dev/null; then
log "Error: Failed to install pre-commit."
exit 1
fi
fi
log "Pre-commit version: $(python3 -m pre_commit --version)"
log "Installing/updating pre-commit hooks..."
python3 -m pre_commit install --install-hooks
log "Running pre-commit checks on all files..."
python3 -m pre_commit run --all-files
PRE_COMMIT_EXIT_CODE=$?
if [ $PRE_COMMIT_EXIT_CODE -eq 0 ]; then
log "Pre-commit checks completed successfully"
else
log "Error: Pre-commit checks failed with exit code: $PRE_COMMIT_EXIT_CODE"
fi
log "=== Pre-commit checks completed with exit code: $PRE_COMMIT_EXIT_CODE ==="
exit $PRE_COMMIT_EXIT_CODE
-31
View File
@@ -1,31 +0,0 @@
# Build-context excludes: keep the context small and the `COPY . .` layer cache
# stable. Docker uploads everything here to the daemon and bakes it into a layer;
# without this, the 6.5 GB host .venv alone is shipped + cached on every build.
#
# IMPORTANT: do NOT ignore .git — fastvideo-kernel's build runs
# `git submodule update --init --recursive`, which needs the repo metadata.
# Virtualenvs — the image builds its own /opt/venv
.venv/
venv/
env/
# Python caches & build/test/lint artifacts
**/__pycache__/
*.py[cod]
*.egg-info/
.eggs/
.pytest_cache/
.mypy_cache/
.ruff_cache/
.cache/
# Local run outputs / logs (not needed in the image)
outputs/
wandb/
*.log
# Editor / OS cruft
.DS_Store
.idea/
.vscode/
-29
View File
@@ -1,29 +0,0 @@
name: 🐞 Bug report
description: Create a report to help us reproduce and fix the bug
title: "[Bug] "
labels: ['Bug']
body:
- type: textarea
attributes:
label: Describe the bug
description: A clear and concise description of what the bug is.
validations:
required: true
- type: textarea
attributes:
label: Reproduction
description: |
What command or script did you run? Which **model** are you using?
placeholder: |
A placeholder for the command.
validations:
required: true
- type: textarea
attributes:
label: Environment
description: |
Please share your environment with us. You can run the command **python collect_env.py** and copy-paste its output below.
placeholder: FastVideo version, platform, python version, cuda version...
validations:
required: true
@@ -1,17 +0,0 @@
name: 🚀 Feature request
description: Suggest an idea for this project
title: "[Feature] "
body:
- type: textarea
attributes:
label: Motivation
description: |
A clear and concise description of the motivation of the feature.
validations:
required: true
- type: textarea
attributes:
label: Related resources
description: |
If there is an official code release or third-party implementations, please also provide the information here, which would be very helpful.
-56
View File
@@ -1,56 +0,0 @@
name: 💬 Request for comments (RFC).
description: Ask for feedback on major architectural changes or design choices.
title: "[RFC]: "
labels: ["RFC"]
body:
- type: markdown
attributes:
value: >
#### Please take a look at previous [RFCs](https://github.com/hao-ai-lab/FastVideo/issues?q=label%3ARFC+sort%3Aupdated-desc) for reference.
- type: textarea
attributes:
label: Motivation.
description: >
The motivation of the RFC.
validations:
required: true
- type: textarea
attributes:
label: Proposed Change.
description: >
The proposed change of the RFC.
validations:
required: true
- type: textarea
attributes:
label: Feedback Period.
description: >
The feedback period of the RFC. Usually at least one week.
validations:
required: false
- type: textarea
attributes:
label: CC List.
description: >
The list of people you want to CC.
validations:
required: false
- type: textarea
attributes:
label: Any Other Things.
description: >
Any other things you would like to mention.
validations:
required: false
- type: markdown
attributes:
value: >
Thanks for contributing 🎉!
- type: checkboxes
id: askllm
attributes:
label: Before submitting a new issue...
options:
- label: Make sure you already searched for relevant issues.
required: true
-1
View File
@@ -1 +0,0 @@
blank_issues_enabled: false
-63
View File
@@ -1,63 +0,0 @@
<!--
PR TITLE: Must start with a type tag, e.g.:
[feat] Add new model [bugfix] Fix VAE tiling [refactor] Restructure pipeline
[perf] Optimize kernel [ci] Update tests [docs] Add guide
[misc] Cleanup configs [new-model] Port Flux2 [infra] Add trace hooks
[skill] Add agent skill
MERGE WORKFLOW:
1. Ensure pre-commit passes and you have at least 1 approval
2. Comment /merge (or add the "ready" label) to enter the Merge Queue
3. Full Test Suite runs automatically on a staging branch → auto-merge on success
ON-DEMAND TESTING (write access required):
/test full — Full Test Suite /test ssim — SSIM regression
/test training — Training pipeline /test encoder — Encoder tests
/test transformer — Transformer tests /test vae — VAE tests
/test kernel — CUDA kernel tests /test unit — Unit tests
See docs/contributing/pull_requests.md for all 17 test commands
-->
## Purpose
<!-- What does this PR do? Link the related issue if applicable. -->
Fixes #
## Changes
<!-- Describe your changes concisely. What approach did you take? -->
-
## Test Plan
<!-- How did you verify your changes? Paste exact commands and output. -->
```bash
# Commands you ran
```
## Test Results
<!-- Paste test output, before/after comparisons, or SSIM scores for model changes. -->
<details>
<summary>Test output</summary>
```
# Paste output here
```
</details>
## Checklist
- [ ] I ran `pre-commit run --all-files` and fixed all issues
- [ ] I added or updated tests for my changes
- [ ] I updated documentation if needed
- [ ] I considered GPU memory impact of my changes
**For model/pipeline changes, also check:**
- [ ] I verified SSIM regression tests pass
- [ ] I updated the support matrix if adding a new model
-324
View File
@@ -1,324 +0,0 @@
merge_protections:
- name: PR merge requirements
if:
- base = main
success_conditions:
- "title~=(?i)^\\[(feat|feature|bugfix|fix|refactor|perf|ci|doc|docs|misc|chore|kernel|new.?model|skill|skills|infra)\\]"
- "#approved-reviews-by>=1"
- check-success~=pre-commit
- check-success=fastcheck-passed
- check-success=full-suite-passed
pull_request_rules:
# ============================================================
# Type labels (from PR title prefix)
# ============================================================
- name: "label type: feat"
conditions:
- "title~=(?i)^\\[(feat|feature)\\]"
- -closed
actions:
label:
add: ["type: feat"]
- name: "label type: bugfix"
conditions:
- "title~=(?i)^\\[(bug)?fix\\]"
- -closed
actions:
label:
add: ["type: bugfix"]
- name: "label type: refactor"
conditions:
- "title~=(?i)^\\[refactor\\]"
- -closed
actions:
label:
add: ["type: refactor"]
- name: "label type: perf"
conditions:
- "title~=(?i)^\\[perf\\]"
- -closed
actions:
label:
add: ["type: perf"]
- name: "label type: ci"
conditions:
- "title~=(?i)^\\[ci\\]"
- -closed
actions:
label:
add: ["type: ci"]
- name: "label type: docs"
conditions:
- "title~=(?i)^\\[(doc|docs)\\]"
- -closed
actions:
label:
add: ["type: docs"]
- name: "label type: misc"
conditions:
- "title~=(?i)^\\[(misc|chore)\\]"
- -closed
actions:
label:
add: ["type: misc"]
- name: "label type: new-model"
conditions:
- "title~=(?i)^\\[new.?model\\]"
- -closed
actions:
label:
add: ["type: new-model"]
- name: "label type: infra"
conditions:
- "title~=(?i)^\\[infra\\]"
- -closed
actions:
label:
add: ["type: infra"]
- name: "label type: skill"
conditions:
- "title~=(?i)^\\[skills?\\]"
- -closed
actions:
label:
add: ["type: skill"]
# ============================================================
# Scope labels (from changed files)
# ============================================================
- name: "label scope: training"
conditions:
- or:
- files~=^fastvideo/train/
- files~=^fastvideo/training/
- files~=^fastvideo/distillation/
- files~=^examples/train/
- files~=^examples/training/
- files~=^examples/distill/
- -closed
actions:
label:
add: ["scope: training"]
- name: "label scope: inference"
conditions:
- or:
- files~=^fastvideo/pipelines/basic/
- files~=^fastvideo/pipelines/stages/
- files~=^fastvideo/pipelines/samplers/
- files~=^fastvideo/entrypoints/
- files~=^fastvideo/worker/
- files~=^fastvideo/api/sampling_param
- files~=^fastvideo/configs/pipelines/
- files~=^examples/inference/
- -closed
actions:
label:
add: ["scope: inference"]
- name: "label scope: attention"
conditions:
- files~=^fastvideo/attention/
- -closed
actions:
label:
add: ["scope: attention"]
- name: "label scope: kernel"
conditions:
- or:
- files~=^fastvideo-kernel/
- files~=^csrc/
- -closed
actions:
label:
add: ["scope: kernel"]
- name: "label scope: data"
conditions:
- or:
- files~=^fastvideo/dataset/
- files~=^fastvideo/pipelines/preprocess/
- files~=^examples/preprocessing/
- -closed
actions:
label:
add: ["scope: data"]
- name: "label scope: infra"
conditions:
- or:
- files~=^\.github/
- files~=^\.buildkite/
- files~=^fastvideo/tests/
- files~=^docker/
- -closed
actions:
label:
add: ["scope: infra"]
- name: "label scope: distributed"
conditions:
- files~=^fastvideo/distributed/
- -closed
actions:
label:
add: ["scope: distributed"]
- name: "label scope: docs"
conditions:
- files~=^docs/
- -closed
actions:
label:
add: ["scope: docs"]
- name: "label scope: studio"
conditions:
- files~=^apps/fastvideo_studio/
- -closed
actions:
label:
add: ["scope: studio"]
- name: "label scope: model"
conditions:
- or:
- files~=^fastvideo/models/
- files~=^fastvideo/layers/
- files~=^fastvideo/configs/models/
- -closed
actions:
label:
add: ["scope: model"]
# ============================================================
# Pre-commit failure help comment
# ============================================================
- name: comment on pre-commit failure
conditions:
- check-failure~=pre-commit
- -closed
actions:
comment:
message: |
## Pre-commit checks failed
Hi @{{author}}, the pre-commit checks have failed. To fix them locally:
```bash
# Install pre-commit if you haven't already
uv pip install pre-commit
pre-commit install
# Run all checks and auto-fix what's possible
pre-commit run --all-files
```
Common fixes:
- **yapf**: `yapf -i <file>` (formatting)
- **ruff**: `ruff check --fix <file>` (linting)
- **codespell**: `codespell --write-changes <file>` (spelling)
After fixing, commit and push the changes. The checks will re-run automatically.
For future commits, `pre-commit` will run automatically on changed files before each commit.
# ============================================================
# Merge conflict detection
# ============================================================
- name: label conflicting PRs
conditions:
- conflict
- -closed
- label!=stale
actions:
label:
add: [needs-rebase]
comment:
message: |
This PR has merge conflicts with the base branch. Please rebase:
```bash
git fetch origin main
git rebase origin/main
# Resolve any conflicts, then:
git push --force-with-lease
```
- name: remove conflict label when resolved
conditions:
- -conflict
- -closed
- label=needs-rebase
actions:
label:
remove: [needs-rebase]
# ============================================================
# Auto-merge
# ============================================================
- name: auto-merge when ready and all checks pass
conditions:
- label=ready
- "title~=(?i)^\\[(feat|feature|bugfix|fix|refactor|perf|ci|doc|docs|misc|chore|kernel|new.?model|skill|skills|infra)\\]"
- "#approved-reviews-by>=1"
- check-success~=pre-commit
- check-success=fastcheck-passed
- check-success=full-suite-passed
- -conflict
- -closed
- -draft
actions:
merge:
method: squash
# ============================================================
# PR title format help
# ============================================================
- name: comment on invalid PR title format
conditions:
- -closed
- -draft
- "-title~=(?i)^\\[(feat|feature|bugfix|fix|refactor|perf|ci|doc|docs|misc|chore|kernel|new.?model|skill|skills|infra)\\]"
actions:
comment:
message: |
## ⚠️ PR title format required
Your PR title must start with a type tag in brackets. Examples:
- `[feat] Add new model support`
- `[bugfix] Fix VAE tiling corruption`
- `[refactor] Restructure training pipeline`
- `[perf] Optimize attention kernel`
- `[ci] Update test infrastructure`
- `[infra] Add activation trace hooks`
- `[docs] Add inference guide`
- `[misc] Clean up configs`
- `[new-model] Port Flux2 to FastVideo`
- `[skill] Add add-model agent skill`
Valid tags: `feat`, `feature`, `bugfix`, `fix`, `refactor`, `perf`, `ci`, `infra`, `doc`, `docs`, `misc`, `chore`, `kernel`, `new-model`, `skill`, `skills`
Please update your PR title and the merge protection check will pass automatically.
merge_protections_settings:
reporting_method: check-runs
-133
View File
@@ -1,133 +0,0 @@
#!/usr/bin/env bash
# Gate the expensive Buildkite full suite on the cheap GitHub checks.
#
# Polls the workflow runs for the PR head commit and only exits 0 once the
# watched cheap workflows (pre-commit, docs build) have succeeded, so the
# 'ready' label cannot burn ~20 GPU lanes on a head that a cheap check has
# already doomed.
#
# Semantics:
# - watched run completed with a bad conclusion -> exit 1 (fail CLOSED:
# no full suite; the next push re-arms via the 'synchronize' trigger)
# - watched run cancelled -> still pending: the docs
# workflow's repo-global 'pages' concurrency group cancels runs superseded
# by unrelated pushes, so 'cancelled' is not a verdict on this PR
# - watched runs pending -> poll until done
# - docs run absent -> not applicable after a
# short grace period ('Deploy Documentation' is path-filtered on PRs)
# - pre-commit run absent -> keep polling: pre-commit
# is never path-filtered, so its absence is always anomalous
# - 'ready' label removed while waiting -> exit 1 (fail CLOSED:
# un-labeling is a deliberate maintainer action)
# - GitHub API unreachable or timeout -> exit 0 (fail OPEN,
# loud warning: never brick CI on a GitHub outage)
#
# Required env: PR_SHA (PR head commit), PR_NUMBER, GITHUB_REPOSITORY, GH_TOKEN.
set -euo pipefail
: "${PR_SHA:?PR_SHA (PR head commit) is required}"
: "${PR_NUMBER:?PR_NUMBER (pull request number) is required}"
: "${GITHUB_REPOSITORY:?GITHUB_REPOSITORY is required}"
# Workflow-level `name:` values that must be green before the full suite
# may start. "Deploy Documentation" is path-filtered on PRs, so its run may
# legitimately never exist; pre-commit always runs, so it must appear.
WATCHED_NAMES='["pre-commit", "Deploy Documentation"]'
WATCHED_REGEX='^(pre-commit|Deploy Documentation)$'
POLL_SECS="${POLL_SECS:-20}"
GRACE_SECS="${GRACE_SECS:-60}"
MAX_WAIT_SECS="${MAX_WAIT_SECS:-1500}"
# Bound each API call so a hung connection hits the 3-strike fail-open path
# instead of pinning the loop until the job timeout (which would fail closed
# on exactly the GitHub-outage case this script is meant to survive).
if command -v timeout >/dev/null 2>&1; then
gh_api() { timeout 30 gh api "$@"; }
else
gh_api() { gh api "$@"; } # macOS dev boxes; CI always has coreutils timeout
fi
# The workflow checked the label before starting the gate, but the wait can
# last ~25 min: re-check once before any exit 0 and fail closed if 'ready'
# was removed in the meantime. An API error here proceeds (the label was
# present when the gate started; never brick CI on an outage).
recheck_ready_label() {
local pr_json
if pr_json=$(gh_api "repos/${GITHUB_REPOSITORY}/pulls/${PR_NUMBER}" 2>/dev/null); then
if ! jq -e '[.labels[]?.name] | index("ready")' <<<"$pr_json" >/dev/null 2>&1; then
echo "::error::PR #${PR_NUMBER} no longer has the 'ready' label —" \
"NOT triggering the Buildkite full suite. Re-add the label to re-arm."
exit 1
fi
else
echo "::warning::Could not re-check the 'ready' label on PR #${PR_NUMBER}; proceeding (it was present when the gate started)."
fi
}
start=$(date +%s)
api_fails=0
missing=""
while true; do
elapsed=$(( $(date +%s) - start ))
if runs_json=$(gh_api "repos/${GITHUB_REPOSITORY}/actions/runs?head_sha=${PR_SHA}&per_page=100" 2>/dev/null) \
&& state=$(jq --arg re "$WATCHED_REGEX" '
[.workflow_runs[]? | select(.name // "" | test($re))]
| group_by(.name) | map(max_by(.id))
| map({name, status, conclusion})' <<<"$runs_json" 2>/dev/null); then
api_fails=0
echo "t+${elapsed}s watched checks: $(jq -c . <<<"$state")"
failed=$(jq -r '[.[] | select(.status == "completed"
and (.conclusion | IN("success", "skipped", "neutral", "cancelled") | not))]
| map(.name) | join(", ")' <<<"$state")
if [ -n "$failed" ]; then
echo "::error::Cheap check(s) failed on ${PR_SHA}: ${failed}." \
"NOT triggering the Buildkite full suite. Push a fix (the 'ready'" \
"label re-arms on every push), or re-run the failed check and then" \
"re-run this workflow."
exit 1
fi
# 'cancelled' counts as pending: wait for a re-run to reach a real verdict
# (bounded by MAX_WAIT, then the fail-open below).
pending=$(jq '[.[] | select(.status != "completed" or .conclusion == "cancelled")] | length' <<<"$state")
missing=$(jq -r --argjson watched "$WATCHED_NAMES" '($watched - map(.name)) | join(", ")' <<<"$state")
if [ "$pending" -eq 0 ]; then
if [ -z "$missing" ]; then
recheck_ready_label
echo "All watched cheap checks are green — full suite may proceed."
exit 0
fi
case "$missing" in
*pre-commit*)
echo "pre-commit run not found for ${PR_SHA} yet; waiting (pre-commit is never path-filtered, so its absence is anomalous)."
;;
*)
if [ "$elapsed" -ge "$GRACE_SECS" ]; then
recheck_ready_label
echo "::warning::Watched run(s) never appeared for ${PR_SHA}: ${missing} (path-filtered, likely not applicable). Proceeding on the checks that did run."
exit 0
fi
echo "Waiting up to ${GRACE_SECS}s grace for path-filtered run(s) to appear: ${missing}."
;;
esac
fi
else
api_fails=$(( api_fails + 1 ))
echo "::warning::GitHub API error querying workflow runs for ${PR_SHA} (attempt ${api_fails}/3)."
if [ "$api_fails" -ge 3 ]; then
recheck_ready_label
echo "::warning::FAILING OPEN: cannot query GitHub check status — triggering the full suite WITHOUT the cheap-check gate."
exit 0
fi
fi
if [ "$elapsed" -ge "$MAX_WAIT_SECS" ]; then
recheck_ready_label
echo "::warning::FAILING OPEN: watched checks still pending after $(( MAX_WAIT_SECS / 60 )) min${missing:+ (never appeared: ${missing})} — triggering the full suite anyway."
exit 0
fi
sleep "$POLL_SECS"
done
-122
View File
@@ -1,122 +0,0 @@
#!/usr/bin/env bash
# Self-test for gate_full_suite.sh using a mocked `gh`. No network, runs on
# any dev box: bash .github/scripts/test_gate_full_suite.sh
set -u
here=$(cd "$(dirname "$0")" && pwd)
tmp=$(mktemp -d)
trap 'rm -rf "$tmp"' EXIT
# Mock gh. Asserts the exact endpoint (including head_sha) it is called
# with — an endpoint typo in the gate script fails the test rather than
# silently serving canned data. On the runs endpoint it serves
# $MOCK_DIR/response_<call#>.json, sticking on the highest existing file,
# and exits 1 if none exist (simulates a GitHub API outage). On the pulls
# endpoint it serves $MOCK_DIR/pr.json, defaulting to a 'ready'-labeled PR.
cat > "$tmp/gh" <<'EOF'
#!/usr/bin/env bash
if [ "${1:-}" != "api" ]; then
echo "unexpected gh invocation: $*" >> "$MOCK_DIR/endpoint_error"
exit 2
fi
case "${2:-}" in
"repos/o/r/actions/runs?head_sha=deadbeef&per_page=100")
n=$(( $(cat "$MOCK_DIR/count" 2>/dev/null || echo 0) + 1 ))
echo "$n" > "$MOCK_DIR/count"
while [ "$n" -gt 0 ]; do
if [ -f "$MOCK_DIR/response_$n.json" ]; then
cat "$MOCK_DIR/response_$n.json"
exit 0
fi
n=$(( n - 1 ))
done
echo "api outage" >&2
exit 1
;;
"repos/o/r/pulls/42")
if [ -f "$MOCK_DIR/pr.json" ]; then
cat "$MOCK_DIR/pr.json"
else
echo '{"labels": [{"name": "ready"}]}'
fi
;;
*)
echo "unexpected gh endpoint: $2" >> "$MOCK_DIR/endpoint_error"
exit 2
;;
esac
EOF
chmod +x "$tmp/gh"
PC_OK='{"name": "pre-commit", "id": 1, "status": "completed", "conclusion": "success"}'
PC_BAD='{"name": "pre-commit", "id": 1, "status": "completed", "conclusion": "failure"}'
PC_PENDING='{"name": "pre-commit", "id": 1, "status": "in_progress", "conclusion": null}'
DOCS_OK='{"name": "Deploy Documentation", "id": 2, "status": "completed", "conclusion": "success"}'
DOCS_BAD='{"name": "Deploy Documentation", "id": 2, "status": "completed", "conclusion": "failure"}'
DOCS_CANCELLED='{"name": "Deploy Documentation", "id": 2, "status": "completed", "conclusion": "cancelled"}'
OTHER='{"name": "Trigger Full Suite", "id": 3, "status": "in_progress", "conclusion": null}'
NULL_NAME='{"name": null, "id": 4, "status": "completed", "conclusion": "failure"}'
PC_OK_RERUN='{"name": "pre-commit", "id": 5, "status": "completed", "conclusion": "success"}'
fails=0
want_log="" # optional: expect() also greps out.log for this regex, then resets
pr_json="" # optional: served for the pulls (label re-check) endpoint, then resets
raw_body="" # optional: serve responses verbatim instead of wrapping in workflow_runs
expect() { # <name> <expected-exit> <response json>...
local name=$1 want=$2 dir i=1
shift 2
dir=$(mktemp -d "$tmp/test_XXXXXX")
for body in "$@"; do
if [ -n "$raw_body" ]; then
printf '%s' "$body" > "$dir/response_$i.json"
else
printf '{"workflow_runs": [%s]}' "$body" > "$dir/response_$i.json"
fi
i=$(( i + 1 ))
done
[ -n "$pr_json" ] && printf '%s' "$pr_json" > "$dir/pr.json"
( export PATH="$tmp:$PATH" MOCK_DIR="$dir" PR_SHA=deadbeef PR_NUMBER=42 \
GITHUB_REPOSITORY=o/r POLL_SECS=0 GRACE_SECS=1 MAX_WAIT_SECS=3
bash "$here/gate_full_suite.sh" > "$dir/out.log" 2>&1 )
local rc=$?
if [ "$rc" -ne "$want" ]; then
echo "FAIL: $name (exit $rc, want $want)"
cat "$dir/out.log"
fails=1
elif [ -f "$dir/endpoint_error" ]; then
echo "FAIL: $name (mock gh got an unexpected call)"
cat "$dir/endpoint_error"
fails=1
elif [ -n "$want_log" ] && ! grep -Eq "$want_log" "$dir/out.log"; then
echo "FAIL: $name (log does not match: $want_log)"
cat "$dir/out.log"
fails=1
else
echo "ok: $name"
fi
want_log="" pr_json="" raw_body=""
}
expect "both green -> proceed" 0 "$PC_OK, $DOCS_OK, $OTHER, $NULL_NAME"
expect "docs build failed -> blocked" 1 "$PC_OK, $DOCS_BAD"
expect "pre-commit failed -> blocked" 1 "$PC_BAD"
expect "pending then green -> proceed" 0 "$PC_PENDING" "$PC_OK, $DOCS_OK"
want_log="never appeared.*Deploy Documentation"
expect "docs run absent (path-filtered) -> proceed after grace" 0 "$PC_OK"
expect "API outage -> fail open" 0
want_log="FAILING OPEN"
expect "pending past MAX_WAIT -> fail open" 0 "$PC_PENDING"
want_log="FAILING OPEN"
expect "unrelated runs only -> no grace, fail open at MAX_WAIT" 0 "$OTHER"
expect "cancelled docs then green -> proceed" 0 \
"$PC_OK, $DOCS_CANCELLED" "$PC_OK, $DOCS_OK"
want_log="FAILING OPEN"
expect "cancelled docs forever -> fail open at MAX_WAIT" 0 "$PC_OK, $DOCS_CANCELLED"
want_log="FAILING OPEN"
expect "pre-commit absent -> no grace, fail open at MAX_WAIT" 0 "$DOCS_OK"
expect "duplicate run names -> latest wins" 0 "$PC_BAD, $PC_OK_RERUN, $DOCS_OK"
raw_body=1
expect "garbage response body -> fail open" 0 "this is not json"
pr_json='{"labels": [{"name": "other"}]}'
expect "ready label removed mid-gate -> blocked" 1 "$PC_OK, $DOCS_OK"
exit "$fails"
-200
View File
@@ -1,200 +0,0 @@
name: Build Image Template
on:
workflow_call:
inputs:
python_version:
required: true
type: string
dockerfile_path:
required: true
type: string
tag_suffix:
required: true
type: string
image_name:
required: false
type: string
default: fastvideo-dev
build_args:
required: false
type: string
default: ''
include_latest_tags:
required: false
type: boolean
default: true
mark_as_latest:
required: false
type: boolean
default: false
runner:
required: false
type: string
default: ubuntu-latest
architecture:
required: false
type: string
default: amd64
push_by_digest:
required: false
type: boolean
default: false
digest_artifact_name:
required: false
type: string
default: ''
jobs:
build-and-push:
runs-on: ${{ inputs.runner }}
permissions:
contents: read
packages: write
steps:
- name: Checkout code
uses: actions/checkout@v4
# The Docker context intentionally includes .git so the kernel build can
# initialize its pinned submodules. Do not copy the checkout token with it.
with:
persist-credentials: false
- name: Free up disk space
run: |
# Display initial space
echo "Initial disk space:"
df -h
# Remove large directories directly
sudo rm -rf /usr/share/dotnet
sudo rm -rf /usr/local/lib/android
sudo rm -rf /opt/ghc
sudo rm -rf /usr/local/share/boost
sudo rm -rf /usr/share/swift
sudo rm -rf /usr/local/lib/node_modules
sudo rm -rf /usr/local/share/powershell
sudo rm -rf /usr/share/rust
sudo rm -rf /usr/local/.ghcup
# Remove cached files
sudo rm -rf /var/lib/apt/lists/*
sudo rm -rf /var/cache/apt/archives/*
# Clean Docker
docker system prune -af --volumes
# Display available space after cleanup
echo "Disk space after cleanup:"
df -h
- name: Set up Docker Buildx
uses: docker/setup-buildx-action@v3
- name: Login to GitHub Container Registry
uses: docker/login-action@v3
with:
registry: ghcr.io
username: ${{ github.repository_owner }}
password: ${{ secrets.GITHUB_TOKEN }}
- name: Normalize image reference
id: image
env:
IMAGE: ghcr.io/${{ github.repository }}/${{ inputs.image_name }}
run: echo "name=${IMAGE,,}" >> "${GITHUB_OUTPUT}"
- name: Prepare tags
id: prepare-tags
run: |
SHORT_SHA=$(echo ${{ github.sha }} | cut -c1-7)
TAGS="type=raw,value=${{ inputs.tag_suffix }}-sha-${SHORT_SHA}"
if [[ "${{ inputs.include_latest_tags }}" == "true" ]]; then
TAGS="type=raw,value=${{ inputs.tag_suffix }}-latest\n${TAGS}"
fi
# Tag the designated default variant as the global `latest` image
if [[ "${{ inputs.include_latest_tags }}" == "true" && "${{ inputs.mark_as_latest }}" == "true" ]]; then
TAGS="${TAGS}\ntype=raw,value=latest"
fi
{
echo "tags<<EOF"
echo -e "$TAGS"
echo "EOF"
} >> $GITHUB_OUTPUT
- name: Extract metadata for Docker
id: meta
uses: docker/metadata-action@v5
with:
images: ${{ steps.image.outputs.name }}
tags: ${{ steps.prepare-tags.outputs.tags }}
- name: Build and push Docker image
if: ${{ !inputs.push_by_digest }}
id: build-push
uses: docker/build-push-action@v6
with:
context: .
file: ${{ inputs.dockerfile_path }}
platforms: linux/${{ inputs.architecture }}
push: true
tags: ${{ steps.meta.outputs.tags }}
labels: ${{ steps.meta.outputs.labels }}
build-args: ${{ inputs.build_args }}
cache-from: type=gha,scope=${{ inputs.image_name }}-${{ inputs.tag_suffix }}-${{ inputs.architecture }}
cache-to: type=gha,mode=max,scope=${{ inputs.image_name }}-${{ inputs.tag_suffix }}-${{ inputs.architecture }}
# Multi-architecture callers publish immutable manifests by digest here,
# then create the shared user-facing tags in a single downstream job.
# This prevents architecture jobs from racing to replace the same tag.
- name: Build and push Docker image by digest
if: ${{ inputs.push_by_digest }}
id: build-push-digest
uses: docker/build-push-action@v6
with:
context: .
file: ${{ inputs.dockerfile_path }}
platforms: linux/${{ inputs.architecture }}
labels: ${{ steps.meta.outputs.labels }}
build-args: ${{ inputs.build_args }}
outputs: type=image,name=${{ steps.image.outputs.name }},push-by-digest=true,name-canonical=true,push=true
cache-from: type=gha,scope=${{ inputs.image_name }}-${{ inputs.tag_suffix }}-${{ inputs.architecture }}
cache-to: type=gha,mode=max,scope=${{ inputs.image_name }}-${{ inputs.tag_suffix }}-${{ inputs.architecture }}
- name: Export digest
if: ${{ inputs.push_by_digest }}
env:
DIGEST: ${{ steps.build-push-digest.outputs.digest }}
run: |
if [[ -z "${{ inputs.digest_artifact_name }}" ]]; then
echo "digest_artifact_name is required when push_by_digest is true" >&2
exit 1
fi
mkdir -p /tmp/digests
touch "/tmp/digests/${DIGEST#sha256:}"
- name: Upload digest
if: ${{ inputs.push_by_digest }}
uses: actions/upload-artifact@v4
with:
name: ${{ inputs.digest_artifact_name }}
path: /tmp/digests/*
if-no-files-found: error
retention-days: 1
- name: Success message
if: ${{ !inputs.push_by_digest }}
run: |
echo "✅ Python ${{ inputs.python_version }} image successfully built and pushed to ${{ steps.image.outputs.name }}:${{ inputs.tag_suffix }}-sha-${GITHUB_SHA::7}"
echo "To run tests with this image, manually trigger the 'Run Tests' workflow."
- name: Digest success message
if: ${{ inputs.push_by_digest }}
env:
DIGEST: ${{ steps.build-push-digest.outputs.digest }}
run: |
echo "✅ Python ${{ inputs.python_version }} linux/${{ inputs.architecture }} image pushed as ${DIGEST}"
-80
View File
@@ -1,80 +0,0 @@
name: Aggregate Test Status
on:
status:
permissions:
statuses: write
jobs:
aggregate:
if: >-
github.event.context == 'direct-test-completed'
&& github.event.state == 'success'
runs-on: ubuntu-latest
steps:
- name: Check and update aggregate status
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7.0.1
with:
script: |
const sha = context.payload.sha;
const { data } = await github.rest.repos.getCombinedStatusForRef({
owner: context.repo.owner,
repo: context.repo.repo,
ref: sha,
per_page: 100,
});
const bkStatuses = data.statuses.filter(
s => s.context.startsWith('buildkite/ci/')
);
const FASTCHECK_PREFIX = 'buildkite/ci/microscope-';
const FULL_SUITE_PREFIXES = [
'buildkite/ci/test-tube-',
'buildkite/ci/bar-chart-',
];
const fastcheck = bkStatuses.filter(
s => s.context.startsWith(FASTCHECK_PREFIX)
);
const fullSuite = bkStatuses.filter(
s => FULL_SUITE_PREFIXES.some(p => s.context.startsWith(p))
);
if (
fastcheck.length > 0
&& fastcheck.every(s => s.state === 'success')
) {
core.info(
`All ${fastcheck.length} fastcheck tests passed — updating fastcheck-passed`
);
await github.rest.repos.createCommitStatus({
owner: context.repo.owner,
repo: context.repo.repo,
sha,
state: 'success',
context: 'fastcheck-passed',
description:
`All ${fastcheck.length} fastcheck tests passed`,
});
}
if (
fullSuite.length > 0
&& fullSuite.every(s => s.state === 'success')
) {
core.info(
`All ${fullSuite.length} full suite tests passed — updating full-suite-passed`
);
await github.rest.repos.createCommitStatus({
owner: context.repo.owner,
repo: context.repo.repo,
sha,
state: 'success',
context: 'full-suite-passed',
description:
`All ${fullSuite.length} full suite tests passed`,
});
}
-65
View File
@@ -1,65 +0,0 @@
name: pre-commit
on:
# pull_request_target instead of pull_request: the workflow definition and
# the hook config are always taken from the BASE branch, so fork /
# first-time-contributor PRs run immediately without a maintainer clicking
# "Approve and run". The PR head is checked out as data only.
pull_request_target:
branches: [main]
workflow_call:
inputs:
ref:
description: 'Git ref to checkout (defaults to github.ref)'
required: false
type: string
permissions:
contents: read
jobs:
pre-commit:
if: github.event.pull_request.draft != true
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v4
with:
ref: ${{ inputs.ref || '' }}
# For PR events, lint the PR head — but keep the hook definitions from
# the base branch so an untrusted PR cannot alter what gets executed.
# The gate scripts are saved too: the self-test step below executes them,
# so it must run the base-branch copies, not the PR head's.
- name: Save trusted hook config and gate scripts
if: github.event_name == 'pull_request_target'
run: |
cp .pre-commit-config.yaml "$RUNNER_TEMP/trusted-pre-commit-config.yaml"
cp -a .github/scripts "$RUNNER_TEMP/trusted-scripts"
echo "GATE_SCRIPTS_DIR=$RUNNER_TEMP/trusted-scripts" >> "$GITHUB_ENV"
# allow-unsafe-pr-checkout acknowledges checkout's pull_request_target
# guard: the head is data for the trusted hooks to lint; nothing from it
# is executed (config and gate scripts are pinned to the base branch
# above) and credentials are not persisted. SHA-pinned to v4.4.0 because
# actionlint's action schema does not know the new input yet.
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
if: github.event_name == 'pull_request_target'
with:
ref: ${{ github.event.pull_request.head.sha }}
persist-credentials: false
allow-unsafe-pr-checkout: true
- name: Restore trusted hook config
if: github.event_name == 'pull_request_target'
run: cp "$RUNNER_TEMP/trusted-pre-commit-config.yaml" .pre-commit-config.yaml
- uses: actions/setup-python@v5
with:
python-version: "3.12"
- run: echo "::add-matcher::.github/workflows/matchers/actionlint.json"
- run: echo "::add-matcher::.github/workflows/matchers/mypy.json"
- run: echo "::add-matcher::.github/workflows/matchers/ruff.json"
- uses: pre-commit/action@v3.0.1
with:
extra_args: --all-files --hook-stage manual
# After pre-commit so a self-test failure cannot mask lint failures.
# GATE_SCRIPTS_DIR points at the base-branch copy on fork PRs (set above);
# push / workflow_call runs use the checked-out tree directly.
- name: Full-suite gate self-test
run: bash "${GATE_SCRIPTS_DIR:-.github/scripts}/test_gate_full_suite.sh"
-280
View File
@@ -1,280 +0,0 @@
name: Slash Commands
on:
issue_comment:
types: [created]
permissions:
contents: read
pull-requests: write
statuses: write
jobs:
handle-merge:
if: >-
github.event.issue.pull_request != null
&& startsWith(github.event.comment.body, '/merge')
runs-on: ubuntu-latest
steps:
- name: Check write permission
id: perm
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7.0.1
with:
script: |
const { data: perm } = await github.rest.repos.getCollaboratorPermissionLevel({
owner: context.repo.owner,
repo: context.repo.repo,
username: context.payload.comment.user.login,
});
const hasWrite = ['admin', 'write'].includes(perm.permission);
if (!hasWrite) {
core.setFailed(`User ${context.payload.comment.user.login} lacks write permission (has: ${perm.permission}).`);
}
core.setOutput('has_write', String(hasWrite));
- name: Add ready label and react
id: label
if: steps.perm.outputs.has_write == 'true'
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7.0.1
with:
script: |
const owner = context.repo.owner;
const repo = context.repo.repo;
const prNumber = context.payload.issue.number;
try { await github.rest.issues.removeLabel({ owner, repo, issue_number: prNumber, name: 'ready' }); } catch {}
await github.rest.issues.addLabels({ owner, repo, issue_number: prNumber, labels: ['ready'] });
await github.rest.reactions.createForIssueComment({
owner, repo,
comment_id: context.payload.comment.id,
content: 'rocket',
});
const { data: pr } = await github.rest.pulls.get({ owner, repo, pull_number: prNumber });
core.setOutput('pr_sha', pr.head.sha);
core.setOutput('pr_branch', pr.head.ref);
core.setOutput('pr_number', String(prNumber));
core.setOutput('pr_title', pr.title);
- name: Trigger Full Suite
if: steps.perm.outputs.has_write == 'true'
env:
BUILDKITE_API_TOKEN: ${{ secrets.BUILDKITE_API_TOKEN }}
PR_SHA: ${{ steps.label.outputs.pr_sha }}
PR_BRANCH: ${{ steps.label.outputs.pr_branch }}
PR_NUMBER: ${{ steps.label.outputs.pr_number }}
PR_TITLE: ${{ steps.label.outputs.pr_title }}
BK_ORG: ${{ vars.BUILDKITE_ORG_SLUG }}
BK_PIPELINE: ${{ vars.BUILDKITE_PIPELINE_SLUG }}
run: |
curl -sS --fail-with-body -X POST \
"https://api.buildkite.com/v2/organizations/${BK_ORG}/pipelines/${BK_PIPELINE}/builds" \
-H "Authorization: Bearer $BUILDKITE_API_TOKEN" \
-H "Content-Type: application/json" \
--data-raw "$(jq -n \
--arg commit "$PR_SHA" \
--arg branch "$PR_BRANCH" \
--arg message "Full Suite for PR #${PR_NUMBER} (via /merge)" \
--arg pr_title "$PR_TITLE" \
--argjson pr_id "$PR_NUMBER" \
'{
commit: $commit,
branch: $branch,
message: $message,
ignore_pipeline_branch_filters: true,
pull_request_id: $pr_id,
pull_request_base_branch: "main",
env: {
TEST_SCOPE: "full",
FULL_SUITE: "true",
PR_NUMBER: ($pr_id | tostring),
PR_TITLE: $pr_title
}
}')"
parse-command:
if: >-
github.event.issue.pull_request != null
&& startsWith(github.event.comment.body, '/test')
runs-on: ubuntu-latest
outputs:
test_type: ${{ steps.parse.outputs.test_type }}
test_scope: ${{ steps.parse.outputs.test_scope }}
full_suite: ${{ steps.parse.outputs.full_suite }}
pr_sha: ${{ steps.pr.outputs.sha }}
pr_branch: ${{ steps.pr.outputs.branch }}
has_write: ${{ steps.perm.outputs.has_write }}
steps:
- name: Check write permission
id: perm
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7.0.1
with:
script: |
const { data: perm } = await github.rest.repos.getCollaboratorPermissionLevel({
owner: context.repo.owner,
repo: context.repo.repo,
username: context.payload.comment.user.login,
});
const hasWrite = ['admin', 'write'].includes(perm.permission);
core.setOutput('has_write', String(hasWrite));
if (!hasWrite) {
core.info(`User ${context.payload.comment.user.login} lacks write permission — ignoring.`);
}
- name: Parse /test command
id: parse
if: steps.perm.outputs.has_write == 'true'
shell: bash
env:
COMMENT: ${{ github.event.comment.body }}
run: |
set -euo pipefail
TEST_NAME=$(echo "$COMMENT" | grep -oP '(?<=/test\s)\S+' | head -1 || true)
VALID="encoder vae transformer kernel unit dreamverse ssim golden-gate training lora-inference lora-training lora-extraction distillation self-forcing vsa vmoba performance api train-framework eval full fastcheck pre-commit"
if [ -z "$TEST_NAME" ] || ! echo "$VALID" | grep -qw "$TEST_NAME"; then
echo "Unknown test: '$TEST_NAME'. Valid: $VALID"
exit 1
fi
declare -A MAP=(
[encoder]=encoder [vae]=vae [transformer]=transformer
[kernel]=kernel_tests [unit]=unit_test [dreamverse]=dreamverse_app
[ssim]=ssim [golden-gate]=golden_gate [training]=training
[lora-inference]=inference_lora [lora-training]=training_lora
[lora-extraction]=lora_extraction
[distillation]=distillation_dmd [self-forcing]=self_forcing
[vsa]=training_vsa [vmoba]=inference_vmoba
[performance]=performance [api]=api_server
[train-framework]=train_framework [eval]=eval
)
if [ "$TEST_NAME" = "full" ]; then
{
echo "test_type=all"
echo "test_scope=full"
echo "full_suite=true"
} >> "$GITHUB_OUTPUT"
elif [ "$TEST_NAME" = "fastcheck" ]; then
{
echo "test_type=fastcheck"
echo "test_scope=fastcheck"
echo "full_suite=false"
} >> "$GITHUB_OUTPUT"
elif [ "$TEST_NAME" = "pre-commit" ]; then
{
echo "test_type="
echo "test_scope=precommit"
echo "full_suite=false"
} >> "$GITHUB_OUTPUT"
else
{
echo "test_type=${MAP[$TEST_NAME]}"
echo "test_scope=direct"
echo "full_suite=false"
} >> "$GITHUB_OUTPUT"
fi
- name: Get PR details
id: pr
if: steps.perm.outputs.has_write == 'true'
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7.0.1
with:
script: |
const { data: pr } = await github.rest.pulls.get({
owner: context.repo.owner,
repo: context.repo.repo,
pull_number: context.payload.issue.number,
});
core.setOutput('sha', pr.head.sha);
core.setOutput('branch', pr.head.ref);
- name: React to comment
if: steps.perm.outputs.has_write == 'true'
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7.0.1
with:
script: |
await github.rest.reactions.createForIssueComment({
owner: context.repo.owner,
repo: context.repo.repo,
comment_id: context.payload.comment.id,
content: 'rocket',
});
pre-commit:
needs: parse-command
if: >-
needs.parse-command.outputs.has_write == 'true'
&& needs.parse-command.outputs.test_scope == 'precommit'
uses: ./.github/workflows/ci-precommit.yml
with:
ref: refs/pull/${{ github.event.issue.number }}/merge
post-precommit-status:
needs: [parse-command, pre-commit]
if: always() && needs.parse-command.outputs.test_scope == 'precommit'
runs-on: ubuntu-latest
steps:
- uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7.0.1
env:
PR_SHA: ${{ needs.parse-command.outputs.pr_sha }}
RESULT: ${{ needs.pre-commit.result }}
with:
script: |
const state = process.env.RESULT === 'success' ? 'success' : 'failure';
await github.rest.repos.createCommitStatus({
owner: context.repo.owner,
repo: context.repo.repo,
sha: process.env.PR_SHA,
state,
context: 'pre-commit',
description: `Triggered via /test pre-commit (${state})`,
});
trigger-buildkite:
needs: parse-command
if: >-
needs.parse-command.outputs.has_write == 'true'
&& needs.parse-command.outputs.test_type != ''
runs-on: ubuntu-latest
steps:
- name: Trigger Buildkite
env:
BUILDKITE_API_TOKEN: ${{ secrets.BUILDKITE_API_TOKEN }}
PR_SHA: ${{ needs.parse-command.outputs.pr_sha }}
PR_BRANCH: ${{ needs.parse-command.outputs.pr_branch }}
PR_NUMBER: ${{ github.event.issue.number }}
TEST_SCOPE: ${{ needs.parse-command.outputs.test_scope }}
FULL_SUITE: ${{ needs.parse-command.outputs.full_suite }}
TEST_TYPE: ${{ needs.parse-command.outputs.test_type }}
PR_TITLE: ${{ github.event.issue.title }}
BK_ORG: ${{ vars.BUILDKITE_ORG_SLUG }}
BK_PIPELINE: ${{ vars.BUILDKITE_PIPELINE_SLUG }}
run: |
curl -sS --fail-with-body -X POST \
"https://api.buildkite.com/v2/organizations/${BK_ORG}/pipelines/${BK_PIPELINE}/builds" \
-H "Authorization: Bearer $BUILDKITE_API_TOKEN" \
-H "Content-Type: application/json" \
--data-raw "$(jq -n \
--arg commit "$PR_SHA" \
--arg branch "$PR_BRANCH" \
--arg message "/test ${TEST_TYPE} on PR #${PR_NUMBER}" \
--argjson pr_id "$PR_NUMBER" \
--arg test_scope "$TEST_SCOPE" \
--arg full_suite "$FULL_SUITE" \
--arg test_type "$TEST_TYPE" \
--arg pr_number "$PR_NUMBER" \
--arg pr_title "$PR_TITLE" \
'{
commit: $commit,
branch: $branch,
message: $message,
ignore_pipeline_branch_filters: true,
pull_request_id: $pr_id,
pull_request_base_branch: "main",
env: {
TEST_SCOPE: $test_scope,
FULL_SUITE: $full_suite,
TEST_TYPE: $test_type,
PR_NUMBER: $pr_number,
PR_TITLE: $pr_title
}
}')"
-103
View File
@@ -1,103 +0,0 @@
name: Trigger Full Suite
on:
pull_request_target:
types: [labeled, synchronize]
permissions:
contents: read
pull-requests: read
actions: read
concurrency:
group: full-suite-${{ github.event.pull_request.number }}
cancel-in-progress: false
jobs:
trigger:
if: >-
(github.event.action == 'labeled' && github.event.label.name == 'ready')
|| github.event.action == 'synchronize'
runs-on: ubuntu-latest
# Gate below may wait for cheap checks (up to MAX_WAIT_SECS = 25 min).
timeout-minutes: 35
steps:
- name: Check ready label
id: check
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7.0.1
with:
script: |
const { data: pr } = await github.rest.pulls.get({
owner: context.repo.owner,
repo: context.repo.repo,
pull_number: context.payload.pull_request.number,
});
const hasReady = pr.labels.some(l => l.name === 'ready');
core.setOutput('has_ready', String(hasReady));
if (!hasReady) core.info('No ready label — skipping Full Suite trigger.');
- name: Cancel previous Buildkite builds
if: steps.check.outputs.has_ready == 'true'
env:
BUILDKITE_API_TOKEN: ${{ secrets.BUILDKITE_API_TOKEN }}
PR_BRANCH: ${{ github.event.pull_request.head.ref }}
run: |
# Find running builds for this branch with TEST_SCOPE=full and cancel them
builds=$(curl -sS -H "Authorization: Bearer $BUILDKITE_API_TOKEN" \
"https://api.buildkite.com/v2/organizations/${{ vars.BUILDKITE_ORG_SLUG }}/pipelines/${{ vars.BUILDKITE_PIPELINE_SLUG }}/builds?branch=${PR_BRANCH}&state=running,scheduled" \
| jq -r '.[] | select(try (.env.TEST_SCOPE == "full") catch false) | .number')
for build_num in $builds; do
echo "Cancelling Buildkite build #$build_num"
curl -sS -X PUT -H "Authorization: Bearer $BUILDKITE_API_TOKEN" \
"https://api.buildkite.com/v2/organizations/${{ vars.BUILDKITE_ORG_SLUG }}/pipelines/${{ vars.BUILDKITE_PIPELINE_SLUG }}/builds/${build_num}/cancel"
done
# Checks out the BASE branch (default for pull_request_target), so PR
# authors cannot tamper with the gate script.
- name: Checkout gate script
if: steps.check.outputs.has_ready == 'true'
uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683 # v4.2.2
- name: Wait for pre-commit and docs build
if: steps.check.outputs.has_ready == 'true'
env:
GH_TOKEN: ${{ github.token }}
PR_SHA: ${{ github.event.pull_request.head.sha }}
PR_NUMBER: ${{ github.event.pull_request.number }}
run: bash .github/scripts/gate_full_suite.sh
- name: Trigger Buildkite Full Suite
if: steps.check.outputs.has_ready == 'true'
env:
BUILDKITE_API_TOKEN: ${{ secrets.BUILDKITE_API_TOKEN }}
PR_SHA: ${{ github.event.pull_request.head.sha }}
PR_BRANCH: ${{ github.event.pull_request.head.ref }}
PR_NUMBER: ${{ github.event.pull_request.number }}
PR_TITLE: ${{ github.event.pull_request.title }}
BK_ORG: ${{ vars.BUILDKITE_ORG_SLUG }}
BK_PIPELINE: ${{ vars.BUILDKITE_PIPELINE_SLUG }}
run: |
curl -sS --fail-with-body -X POST \
"https://api.buildkite.com/v2/organizations/${BK_ORG}/pipelines/${BK_PIPELINE}/builds" \
-H "Authorization: Bearer $BUILDKITE_API_TOKEN" \
-H "Content-Type: application/json" \
--data-raw "$(jq -n \
--arg commit "$PR_SHA" \
--arg branch "$PR_BRANCH" \
--arg message "Full Suite for PR #${PR_NUMBER}" \
--arg pr_title "$PR_TITLE" \
--argjson pr_id "$PR_NUMBER" \
'{
commit: $commit,
branch: $branch,
message: $message,
ignore_pipeline_branch_filters: true,
pull_request_id: $pr_id,
pull_request_base_branch: "main",
env: {
TEST_SCOPE: "full",
FULL_SUITE: "true",
PR_NUMBER: ($pr_id | tostring),
PR_TITLE: $pr_title
}
}')"
@@ -1,65 +0,0 @@
name: Auto-Label Issues
on:
issues:
types: [opened, edited]
permissions:
issues: write
jobs:
label-issues:
if: github.repository == 'hao-ai-lab/FastVideo'
runs-on: ubuntu-latest
steps:
- name: Label by keywords
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7.0.1
with:
script: |
const title = context.payload.issue.title.toLowerCase();
const body = (context.payload.issue.body || '').toLowerCase();
const text = title + ' ' + body;
const labels = [];
const rules = [
// scope labels (shared with PR labeling via Mergify)
// Mapping: label → repo directories
// scope: training → fastvideo/train/, fastvideo/training/, fastvideo/distillation/
// scope: inference → fastvideo/pipelines/, fastvideo/entrypoints/, fastvideo/worker/
// scope: attention → fastvideo/attention/
// scope: kernel → fastvideo-kernel/, csrc/
// scope: model → fastvideo/models/, fastvideo/layers/, fastvideo/configs/models/
// scope: data → fastvideo/dataset/, fastvideo/pipelines/preprocess/
// scope: distributed → fastvideo/distributed/
// scope: docs → docs/
{ keywords: ['training', 'finetune', 'fine-tune', 'lora', 'fsdp', 'distill'], label: 'scope: training' },
{ keywords: ['inference', 'generate', 'pipeline', 'slow', 'latency'], label: 'scope: inference' },
{ keywords: ['attention', 'vsa', 'flash', 'sta', 'vmoba', 'sparse attn'], label: 'scope: attention' },
{ keywords: ['kernel', 'csrc', 'cuda kernel', 'thunderkittens'], label: 'scope: kernel' },
{ keywords: ['wan', 'hunyuan', 'mochi', 'ltx', 'cogvideo', 'flux', 'sd3', 'cosmos'], label: 'scope: model' },
{ keywords: ['dataset', 'dataloader', 'preprocessing', 'preprocess'], label: 'scope: data' },
{ keywords: ['distributed', 'sequence parallel', 'fsdp', 'tensor parallel', 'multi-node', 'multi-gpu'], label: 'scope: distributed' },
{ keywords: ['docs', 'documentation', 'tutorial', 'example'], label: 'scope: docs' },
// issue-only labels (cross-module, no single repo directory)
{ keywords: ['install', 'setup', 'pip', 'cuda', 'uv ', 'import error', 'modulenotfound'], label: 'installation' },
{ keywords: ['memory', 'oom', 'out of memory', 'gpu memory', 'vram'], label: 'performance' },
{ keywords: ['windows', 'macos', 'mac os', 'apple', 'mps', 'rocm', 'amd', 'npu'], label: 'platform' },
];
for (const rule of rules) {
if (rule.keywords.some(kw => text.includes(kw))) {
labels.push(rule.label);
}
}
if (labels.length > 0) {
await github.rest.issues.addLabels({
owner: context.repo.owner,
repo: context.repo.repo,
issue_number: context.payload.issue.number,
labels: labels,
});
console.log(`Added labels: ${labels.join(', ')}`);
} else {
console.log('No keyword matches found');
}
-51
View File
@@ -1,51 +0,0 @@
name: Close Stale Issues and PRs
on:
schedule:
# Daily at 1:30 AM UTC
- cron: '30 1 * * *'
jobs:
stale:
if: github.repository == 'hao-ai-lab/FastVideo'
permissions:
issues: write
pull-requests: write
actions: write
runs-on: ubuntu-latest
steps:
- uses: actions/stale@997185467fa4f803885201cee163a9f38240193d # v10.1.1
with:
operations-per-run: 500
exempt-draft-pr: true
exempt-issue-labels: 'keep-open,pinned,security,Bug,RFC'
exempt-pr-labels: 'keep-open,pinned'
labels-to-add-when-unstale: 'unstale'
labels-to-remove-when-stale: 'unstale'
days-before-issue-stale: 90
days-before-issue-close: 30
stale-issue-label: 'stale'
stale-issue-message: >
This issue has been automatically marked as stale because it has not
had any activity within 90 days. It will be automatically closed if
no further activity occurs within 30 days. Leave a comment if you
feel this issue should remain open. Thank you!
close-issue-message: >
This issue has been automatically closed due to inactivity. Please
feel free to reopen if you feel it is still relevant. Thank you!
days-before-pr-stale: 60
days-before-pr-close: 14
stale-pr-label: 'stale'
stale-pr-message: >
This pull request has been automatically marked as stale because it
has not had any activity within 60 days. It will be automatically
closed if no further activity occurs within 14 days. Leave a comment
if you feel this pull request should remain open. Thank you!
close-pr-message: >
This pull request has been automatically closed due to inactivity.
Please feel free to reopen if you intend to continue working on it.
Thank you!
-56
View File
@@ -1,56 +0,0 @@
name: Welcome First-Time Contributors
on:
issues:
types: [opened]
pull_request_target:
types: [opened]
permissions:
issues: write
pull-requests: write
jobs:
welcome:
if: github.repository == 'hao-ai-lab/FastVideo'
runs-on: ubuntu-latest
steps:
- uses: actions/first-interaction@34f15e814fe48ac9312ccf29db4e74fa767cbab7 # v1.3.0
with:
repo-token: ${{ secrets.GITHUB_TOKEN }}
issue-message: |
Welcome to FastVideo! Thanks for opening your first issue.
To help us investigate, please include:
- **FastVideo version**: `pip show fastvideo`
- **GPU**: `nvidia-smi` output (GPU model, driver, CUDA version)
- **Python version**: `python --version`
- **OS**: e.g., Ubuntu 22.04
If this is a bug, a minimal reproduction script helps us fix it faster.
Useful links:
- [Documentation](https://hao-ai-lab.github.io/FastVideo)
- [Contributing Guide](https://hao-ai-lab.github.io/FastVideo/contributing/overview/)
- [Slack](https://join.slack.com/t/fastvideo/shared_invite/zt-3f4lao1uq-u~Ipx6Lt4J27AlD2y~IdLQ)
pr-message: |
Welcome to FastVideo! Thanks for your first pull request.
**How our CI works:**
PRs run a two-tier CI system:
1. **Pre-commit** — formatting (yapf), linting (ruff), type checking (mypy). Runs immediately on every PR.
2. **Fastcheck** — core GPU tests (encoders, VAEs, transformers, kernels, unit tests). Runs automatically via Buildkite on relevant file changes (~10-15 min).
3. **Full Suite** — integration tests, training pipelines, SSIM regression. Runs only when a reviewer adds the `ready` label.
**Before your PR is reviewed:**
- [ ] `pre-commit run --all-files` passes locally
- [ ] You've added or updated tests for your changes
- [ ] The PR description explains what and why
If pre-commit fails, a bot comment will explain how to fix it. Fastcheck and Full Suite results appear in the Checks section below.
**Useful links:**
- [Contributing Guide](https://hao-ai-lab.github.io/FastVideo/contributing/overview/)
- [Development Roadmap](https://github.com/hao-ai-lab/FastVideo/issues/899)
- [Slack](https://join.slack.com/t/fastvideo/shared_invite/zt-3f4lao1uq-u~Ipx6Lt4J27AlD2y~IdLQ)
-230
View File
@@ -1,230 +0,0 @@
name: Build and Push Docker Images
on:
workflow_dispatch:
inputs:
build_cuda_matrix:
description: 'Build multi-arch CUDA images (12.6.3 default; 13.0.0 explicitly tagged)'
required: false
default: true
type: boolean
build_dreamverse_matrix:
description: 'Build the amd64 Dreamverse matrix (backend + UI x CUDA 12.6.3/13.0.0)'
required: false
default: false
type: boolean
# Auto-rebuild the CUDA images when a repository-controlled image input
# changes on main. This includes the trusted SM89 kernel artifact's source,
# metadata/key helper, ABI dependency metadata, and build orchestration.
# Dreamverse (apps/dreamverse/docker/Dockerfile) and the ROCm Dockerfile stay
# manual-dispatch only.
push:
branches: [main]
paths:
- '.dockerignore'
- '.github/workflows/_template-build-image.yml'
- '.github/workflows/infra-build-image.yml'
- '.gitmodules'
- 'docker/Dockerfile'
- 'docker/uv-excludes'
- 'fastvideo-kernel/**'
- 'fastvideo/tests/modal/kernel_build_cache.py'
- 'pyproject.toml'
permissions:
contents: read
packages: write
# One static group, no cancellation: every run of this workflow writes the same
# mutable registry tags (latest, py3.12-latest, ...), so runs must serialize —
# concurrent push/dispatch runs would race on those tags, and cancelling a run
# mid-publish can strand the cu126/cu130 tag families at different commits. An
# in-flight superseded build wastes its runner time, but its tags are then
# overwritten by the newer queued run. GitHub keeps a single pending run per
# group: the newest queued run replaces any older queued one.
concurrency:
group: infra-build-image
cancel-in-progress: false
jobs:
# CUDA matrix: Python 3.12 x {12.6.3, 13.0.0} x {amd64, arm64}. Each architecture
# builds natively and pushes only by digest; publish-cuda-manifests is the sole
# owner of the shared tags. CUDA 12.6 arm64 targets Hopper (sm_90a / GH200 class),
# because CUDA 12.6 cannot compile sm_121. CUDA 13 arm64 targets GB10 (sm_121).
# 12.6.3/cu126 owns the default `py3.12`/`latest` tags and keeps its versioned
# aliases; 13.0.0/cu130 is published under explicit versioned tags. Flash-attn
# 2.8.3 comes from the architecture-specific prebuilt releases.
build-cuda-images:
# Runs on a manual dispatch when build_cuda_matrix is set, or automatically
# on an in-scope main push (inputs are null on push). The
# repository guard keeps fork syncs from auto-building; manual dispatch
# still works in forks.
if: ${{ (github.event_name == 'push' && github.repository == 'hao-ai-lab/FastVideo') || github.event.inputs.build_cuda_matrix == 'true' }}
strategy:
fail-fast: false
matrix:
cuda:
- version: '12.6.3'
suffix: '-cuda12.6.3'
torch_backend: 'cu126'
fa_tag: 'cu126torch2.12'
cmake_build_parallel_level: '4'
torch_cuda_arch_list:
amd64: '9.0a'
arm64: '9.0a'
- version: '13.0.0'
suffix: '-cuda13.0.0'
torch_backend: 'cu130'
fa_tag: 'cu130torch2.12'
cmake_build_parallel_level: '1'
torch_cuda_arch_list:
amd64: '9.0a'
arm64: '12.1'
architecture:
- name: amd64
runner: ubuntu-latest
- name: arm64
runner: ubuntu-24.04-arm
uses: ./.github/workflows/_template-build-image.yml
with:
python_version: '3.12'
dockerfile_path: docker/Dockerfile
tag_suffix: py3.12${{ matrix.cuda.suffix }}
runner: ${{ matrix.architecture.runner }}
architecture: ${{ matrix.architecture.name }}
push_by_digest: true
digest_artifact_name: fastvideo-dev-cuda${{ matrix.cuda.version }}-${{ matrix.architecture.name }}
build_args: |
PYTHON_VERSION=3.12
CUDA_VERSION=${{ matrix.cuda.version }}
UV_TORCH_BACKEND=${{ matrix.cuda.torch_backend }}
TORCH_CUDA_ARCH_LIST=${{ matrix.cuda.torch_cuda_arch_list[matrix.architecture.name] }}
CMAKE_BUILD_PARALLEL_LEVEL=${{ matrix.cuda.cmake_build_parallel_level }}
FLASH_ATTN_WHEEL_TAG=${{ matrix.cuda.fa_tag }}
FLASH_ATTN_WHEEL_RELEASE=https://github.com/mjun0812/flash-attention-prebuild-wheels/releases/download/v0.9.17
FLASH_ATTN_WHEEL_RELEASE_ARM64=https://github.com/mjun0812/flash-attention-prebuild-wheels/releases/download/v0.9.22
secrets: inherit
publish-cuda-manifests:
# !cancelled(): publish lanes whose digests exist even if a sibling build
# leg failed (the digest-count check fails incomplete lanes); it also
# bypasses skipped-needs propagation, hence the explicit skipped check.
if: ${{ !cancelled() && needs.build-cuda-images.result != 'skipped' }}
needs: build-cuda-images
runs-on: ubuntu-latest
permissions:
contents: read
packages: write
strategy:
fail-fast: false
matrix:
cuda:
- version: '12.6.3'
suffix: '-cuda12.6.3'
is_default: true
- version: '13.0.0'
suffix: '-cuda13.0.0'
is_default: false
steps:
- name: Download architecture digests
uses: actions/download-artifact@v4
with:
path: /tmp/digests
pattern: fastvideo-dev-cuda${{ matrix.cuda.version }}-*
merge-multiple: true
- name: Set up Docker Buildx
uses: docker/setup-buildx-action@v3
- name: Login to GitHub Container Registry
uses: docker/login-action@v3
with:
registry: ghcr.io
username: ${{ github.repository_owner }}
password: ${{ secrets.GITHUB_TOKEN }}
- name: Normalize image reference
id: image
env:
IMAGE: ghcr.io/${{ github.repository }}/fastvideo-dev
run: echo "name=${IMAGE,,}" >> "${GITHUB_OUTPUT}"
- name: Prepare tags
id: prepare-tags
env:
VERSIONED_TAG_PREFIX: py3.12${{ matrix.cuda.suffix }}
PUBLISH_DEFAULT_TAGS: ${{ matrix.cuda.is_default }}
run: |
SHORT_SHA="${GITHUB_SHA::7}"
TAGS="type=raw,value=${VERSIONED_TAG_PREFIX}-latest\ntype=raw,value=${VERSIONED_TAG_PREFIX}-sha-${SHORT_SHA}"
if [[ "${PUBLISH_DEFAULT_TAGS}" == "true" ]]; then
TAGS="type=raw,value=latest\ntype=raw,value=py3.12-latest\ntype=raw,value=py3.12-sha-${SHORT_SHA}\n${TAGS}"
fi
{
echo "tags<<EOF"
echo -e "${TAGS}"
echo "EOF"
} >> "${GITHUB_OUTPUT}"
- name: Extract metadata for Docker
id: meta
uses: docker/metadata-action@v5
with:
images: ${{ steps.image.outputs.name }}
tags: ${{ steps.prepare-tags.outputs.tags }}
- name: Create multi-architecture manifests
env:
IMAGE: ${{ steps.image.outputs.name }}
run: |
mapfile -t DIGESTS < <(find /tmp/digests -maxdepth 1 -type f -printf '%f\n' | sort)
if [[ "${#DIGESTS[@]}" -ne 2 ]]; then
echo "Expected exactly two architecture digests, found ${#DIGESTS[@]}" >&2
exit 1
fi
mapfile -t TAGS < <(jq -r '.tags[]' <<< "${DOCKER_METADATA_OUTPUT_JSON}")
TAG_ARGS=()
for TAG in "${TAGS[@]}"; do
TAG_ARGS+=(--tag "${TAG}")
done
IMAGE_REFS=()
for DIGEST in "${DIGESTS[@]}"; do
IMAGE_REFS+=("${IMAGE}@sha256:${DIGEST}")
done
docker buildx imagetools create "${TAG_ARGS[@]}" "${IMAGE_REFS[@]}"
docker buildx imagetools inspect "${TAGS[0]}"
# Dreamverse matrix: {backend, UI} x {12.6.3, 13.0.0}, Python 3.12. Torch backend
# matches the base CUDA (cu126 / cu130). Keep these images amd64-only until the
# required FA4 dependency stack is available and validated on arm64.
build-dreamverse:
if: ${{ github.event.inputs.build_dreamverse_matrix == 'true' }}
strategy:
fail-fast: false
matrix:
cuda:
- version: '12.6.3'
torch_backend: 'cu126'
- version: '13.0.0'
torch_backend: 'cu130'
variant:
- name: backend
ui: '0'
- name: ui
ui: '1'
uses: ./.github/workflows/_template-build-image.yml
with:
python_version: '3.12'
dockerfile_path: apps/dreamverse/docker/Dockerfile
tag_suffix: dreamverse-${{ matrix.variant.name }}-cuda${{ matrix.cuda.version }}
image_name: dreamverse
build_args: |
CUDA_VERSION=${{ matrix.cuda.version }}
UV_TORCH_BACKEND=${{ matrix.cuda.torch_backend }}
BUILD_DREAMVERSE_UI=${{ matrix.variant.ui }}
include_latest_tags: false
secrets: inherit
-78
View File
@@ -1,78 +0,0 @@
name: Deploy Documentation
on:
push:
branches: [ main ]
paths:
- 'docs/**'
- 'examples/**'
- 'mkdocs.yml'
- 'requirements-mkdocs.in'
- 'requirements-mkdocs.txt'
- 'scripts/check_docs_links.py'
- '.github/workflows/infra-docs.yml'
pull_request:
branches: [ main ]
paths:
- 'docs/**'
- 'examples/**'
- 'mkdocs.yml'
- 'requirements-mkdocs.in'
- 'requirements-mkdocs.txt'
- 'scripts/check_docs_links.py'
- '.github/workflows/infra-docs.yml'
permissions:
contents: read
pages: write
id-token: write
concurrency:
group: "pages"
cancel-in-progress: false
jobs:
build:
runs-on: ubuntu-latest
steps:
- name: Checkout
uses: actions/checkout@v4
with:
fetch-depth: 0
- name: Setup Python
uses: actions/setup-python@v5
with:
python-version: '3.12'
- name: Install uv
uses: astral-sh/setup-uv@v3
- name: Install dependencies
run: uv pip install --system -r requirements-mkdocs.txt
- name: Setup Pages
uses: actions/configure-pages@v4
- name: Build documentation
run: mkdocs build
- name: Check docs links
run: python scripts/check_docs_links.py
- name: Upload artifact
uses: actions/upload-pages-artifact@v3
with:
path: ./site
deploy:
environment:
name: github-pages
url: ${{ steps.deployment.outputs.page_url }}
runs-on: ubuntu-latest
needs: build
if: github.ref == 'refs/heads/main'
steps:
- name: Deploy to GitHub Pages
id: deployment
uses: actions/deploy-pages@v4
@@ -1,17 +0,0 @@
{
"problemMatcher": [
{
"owner": "actionlint",
"pattern": [
{
"regexp": "^(?:\\x1b\\[\\d+m)?(.+?)(?:\\x1b\\[\\d+m)*:(?:\\x1b\\[\\d+m)*(\\d+)(?:\\x1b\\[\\d+m)*:(?:\\x1b\\[\\d+m)*(\\d+)(?:\\x1b\\[\\d+m)*: (?:\\x1b\\[\\d+m)*(.+?)(?:\\x1b\\[\\d+m)* \\[(.+?)\\]$",
"file": 1,
"line": 2,
"column": 3,
"message": 4,
"code": 5
}
]
}
]
}
-16
View File
@@ -1,16 +0,0 @@
{
"problemMatcher": [
{
"owner": "mypy",
"pattern": [
{
"regexp": "^(.+):(\\d+):\\s(error|warning):\\s(.+)$",
"file": 1,
"line": 2,
"severity": 3,
"message": 4
}
]
}
]
}
-17
View File
@@ -1,17 +0,0 @@
{
"problemMatcher": [
{
"owner": "ruff",
"pattern": [
{
"regexp": "^(.+):(\\d+):(\\d+): (\\w+) (.+)$",
"file": 1,
"line": 2,
"column": 3,
"code": 4,
"message": 5
}
]
}
]
}
-28
View File
@@ -1,28 +0,0 @@
name: Publish to Comfy registry
on:
workflow_dispatch:
push:
branches:
- main
- master
paths:
- "pyproject.toml"
permissions:
issues: write
jobs:
publish-node:
name: Publish Custom Node to registry
runs-on: ubuntu-latest
if: ${{ github.repository_owner == 'hao-ai-lab' }}
steps:
- name: Check out code
uses: actions/checkout@v4
with:
submodules: true
- name: Publish Custom Node
uses: Comfy-Org/publish-node-action@v1
with:
## Add your own personal access token to your Github Repository secrets and reference it here.
personal_access_token: ${{ secrets.REGISTRY_ACCESS_TOKEN }}
-72
View File
@@ -1,72 +0,0 @@
name: Publish FastVideo to PyPI on Version Change
on:
push:
branches:
- main
paths:
- 'pyproject.toml' # Trigger when pyproject.toml changes
workflow_dispatch:
jobs:
check-version-change:
runs-on: ubuntu-latest
outputs:
version-changed: ${{ steps.check-version.outputs.changed }}
new-version: ${{ steps.check-version.outputs.new-version }}
steps:
- name: Checkout code
uses: actions/checkout@v4
with:
fetch-depth: 2
- name: Check if version changed
id: check-version
run: |
# Get current commit's version
NEW_VERSION=$(grep -oP "version\\s*=\\s*\"\\K[^\"]+\"" pyproject.toml)
echo "New version: $NEW_VERSION"
# Get previous version from git history
OLD_VERSION=$(git show HEAD~1:./pyproject.toml | grep -oP "version\\s*=\\s*\"\\K[^\"]+\"" || echo "0.0.0")
echo "Old version: $OLD_VERSION"
if [ "$NEW_VERSION" != "$OLD_VERSION" ]; then
echo "Version changed from $OLD_VERSION to $NEW_VERSION"
echo "changed=true" >> $GITHUB_OUTPUT
echo "new-version=$NEW_VERSION" >> $GITHUB_OUTPUT
else
echo "Version did not change"
echo "changed=false" >> $GITHUB_OUTPUT
fi
build-publish-main:
needs: check-version-change
if: ${{ needs.check-version-change.outputs.version-changed == 'true' || github.event_name == 'workflow_dispatch' }}
runs-on: ubuntu-latest
permissions:
id-token: write # Needed for OIDC Trusted Publishing
steps:
- name: Checkout code
uses: actions/checkout@v4
- name: Set up Python
uses: actions/setup-python@v5
with:
python-version: '3.10'
- name: Install uv
uses: astral-sh/setup-uv@v3
- name: Install build dependencies
run: uv pip install --system build twine wheel
- name: Build package
run: |
python -m build
- name: Publish release distributions to PyPI
uses: pypa/gh-action-pypi-publish@release/v1
with:
packages-dir: dist/
-270
View File
@@ -1,270 +0,0 @@
name: Publish FastVideo Kernel to PyPI on Version Change
on:
push:
branches:
- main
paths:
- "fastvideo-kernel/pyproject.toml"
workflow_dispatch:
jobs:
check-version-change:
runs-on: ubuntu-latest
outputs:
version-changed: ${{ steps.check-version.outputs.changed }}
new-version: ${{ steps.check-version.outputs.new-version }}
steps:
- name: Checkout code
uses: actions/checkout@v4
with:
fetch-depth: 2
- name: Check if version changed
id: check-version
run: |
cd fastvideo-kernel
# Get current commit's version from pyproject.toml
# Use ^ to match start of line to avoid matching minimum-version
NEW_VERSION=$(grep -oP '^version\s*=\s*"\K[^"]+' pyproject.toml)
echo "New version: $NEW_VERSION"
# Get previous version from git history
# Note: git show expects path relative to repo root
OLD_VERSION=$(git show HEAD~1:fastvideo-kernel/pyproject.toml | grep -oP '^version\s*=\s*"\K[^"]+' || echo "0.0.0")
echo "Old version: $OLD_VERSION"
if [ "$NEW_VERSION" != "$OLD_VERSION" ]; then
echo "Version changed from $OLD_VERSION to $NEW_VERSION"
echo "changed=true" >> "$GITHUB_OUTPUT"
echo "new-version=$NEW_VERSION" >> "$GITHUB_OUTPUT"
else
echo "Version did not change"
echo "changed=false" >> "$GITHUB_OUTPUT"
fi
build_wheels:
name: Build Wheel
needs: check-version-change
if: ${{ needs.check-version-change.outputs.version-changed == 'true' || github.event_name == 'workflow_dispatch' }}
runs-on: ${{ matrix.platform.os }}
strategy:
fail-fast: false
matrix:
python-version: ['3.12']
torch-cuda:
# torch 2.12 dropped cu128; cu130 (CUDA 13) is the default wheel, cu126 covers older drivers.
- torch-version: '2.12.0'
cuda-version: '12.6.3'
torch-cuda-short: 'cu126'
- torch-version: '2.12.0'
cuda-version: '13.0.0'
torch-cuda-short: 'cu130'
platform:
# x86_64 builds the full cu126 + cu130 set (cu130 ships the consumer
# Blackwell sm_120a FP4 kernels).
- os: ubuntu-22.04
arch: x86_64
wheel-plat: manylinux_2_35_x86_64
# aarch64 is Blackwell (GB200 sm_100a + DGX Spark / consumer sm_120a), not
# Hopper, and Blackwell needs CUDA >= 12.8 — so only the cu130 leg applies.
# Added via include so x86 keeps cu126 + cu130 while aarch64 stays cu130-only.
include:
- python-version: '3.12'
torch-cuda:
torch-version: '2.12.0'
cuda-version: '13.0.0'
torch-cuda-short: 'cu130'
platform:
os: ubuntu-22.04-arm
arch: aarch64
wheel-plat: manylinux_2_35_aarch64
steps:
- name: Free up disk space
run: |
echo "Initial disk space:"
df -h
# Remove large directories
sudo rm -rf /usr/share/dotnet
sudo rm -rf /usr/local/lib/android
sudo rm -rf /opt/ghc
sudo rm -rf /usr/local/share/boost
sudo rm -rf /usr/share/swift
sudo rm -rf /usr/local/lib/node_modules
sudo rm -rf /usr/local/share/powershell
sudo rm -rf /usr/share/rust
sudo rm -rf /usr/local/.ghcup
# Remove cached files
sudo rm -rf /var/lib/apt/lists/*
sudo rm -rf /var/cache/apt/archives/*
echo "Disk space after cleanup:"
df -h
- name: Checkout
uses: actions/checkout@v4
- name: Set up Python
uses: actions/setup-python@v5
with:
python-version: ${{ matrix.python-version }}
- name: Install CUDA ${{ matrix.torch-cuda.cuda-version }}
uses: Jimver/cuda-toolkit@v0.2.35
id: cuda-toolkit
with:
cuda: ${{ matrix.torch-cuda.cuda-version }}
linux-local-args: '["--toolkit"]'
method: 'network'
- name: Install dependencies (GCC, Clang, CUDA Paths, Git)
run: |
sudo apt update
sudo apt install -y git patchelf gcc-11 g++-11 clang-11
sudo update-alternatives --install /usr/bin/gcc gcc /usr/bin/gcc-11 100 --slave /usr/bin/g++ g++ /usr/bin/g++-11
# Allow Git to Access Safe Directory
git config --global --add safe.directory /__w/FastVideo/FastVideo
# Set CUDA environment variables
export CUDA_HOME=/usr/local/cuda-${{ matrix.torch-cuda.cuda-version }}
export PATH=${CUDA_HOME}/bin:${PATH}
export LD_LIBRARY_PATH=${CUDA_HOME}/lib64:$LD_LIBRARY_PATH
# Verify installation
gcc --version
g++ --version
clang-11 --version
nvcc --version
- name: Install uv
uses: astral-sh/setup-uv@v3
- name: Install PyTorch ${{ matrix.torch-cuda.torch-version }}+cu${{ matrix.torch-cuda.cuda-version }}
run: |
uv pip install --system typing-extensions==4.12.2
uv pip install --system --no-cache-dir torch==${{ matrix.torch-cuda.torch-version }} --index-url https://download.pytorch.org/whl/${{matrix.torch-cuda.torch-cuda-short}}
nvcc --version
python --version
python -c "import torch; print('PyTorch:', torch.__version__)"
python -c "import torch; print('CUDA:', torch.version.cuda)"
python -c "from torch.utils import cpp_extension; print (cpp_extension.CUDA_HOME)"
- name: Build wheel
run: |
export PYTHONPATH=$GITHUB_WORKSPACE:$PYTHONPATH
uv pip install --system setuptools ninja packaging wheel triton scikit-build-core cmake build
cd fastvideo-kernel
git submodule update --init --recursive # Ensure ThunderKittens submodule is initialized
# Release builds run on GPU-less runners, so set kernels + arch explicitly:
# * aarch64 = Blackwell (GB200 sm_100a + DGX Spark/consumer sm_120a), NOT
# Hopper, so TK (sm_90a wgmma) is OFF. The C++ FP4 (attn_qat_infer, SM120)
# covers sm_120a; turbodiffusion covers sm_100a+sm_120a. The sm_100 FP4
# forward is the FA4 CuTe DSL path in the fastvideo package (PR #1221),
# JIT-compiled at runtime — not built into this wheel.
# * x86_64 cu130 = Hopper TK + consumer Blackwell sm_120a FP4.
# * x86_64 cu126 = Hopper TK only (older drivers; CUDA < 12.8 has no FP4).
# The per-arch split in CMakeLists pins the FP4 targets to sm_120a and builds
# the main extension for the full arch list. CMAKE_BUILD_PARALLEL_LEVEL caps
# Ninja so heavy CUTLASS/TK template TUs don't OOM the 16 GB runner (exit 143).
if [ "${{ matrix.platform.arch }}" = "aarch64" ]; then
export TORCH_CUDA_ARCH_LIST="10.0a;12.0a"
export CMAKE_ARGS="${CMAKE_ARGS:-} -DFASTVIDEO_KERNEL_BUILD_TK=OFF -DFASTVIDEO_KERNEL_BUILD_ATTN_QAT_INFER=ON"
export CMAKE_BUILD_PARALLEL_LEVEL=1
elif [ "${{ matrix.torch-cuda.torch-cuda-short }}" = "cu130" ]; then
export TORCH_CUDA_ARCH_LIST="9.0a;12.0a"
export CMAKE_ARGS="${CMAKE_ARGS:-} -DFASTVIDEO_KERNEL_BUILD_TK=ON -DFASTVIDEO_KERNEL_BUILD_ATTN_QAT_INFER=ON -DCMAKE_CUDA_ARCHITECTURES=90a"
# A single FP4 TU (attn_qat_infer) can use ~8-12 GB on its own, so serialize.
export CMAKE_BUILD_PARALLEL_LEVEL=1
else
export TORCH_CUDA_ARCH_LIST="9.0a"
export CMAKE_ARGS="${CMAKE_ARGS:-} -DFASTVIDEO_KERNEL_BUILD_TK=ON -DCMAKE_CUDA_ARCHITECTURES=90a"
# Hopper-only: the two TK TUs are the heavy ones (single-arch) and fit side
# by side; -j4 overlaps the light TUs without OOMing (~12-14 GB peak).
export CMAKE_BUILD_PARALLEL_LEVEL=4
fi
# Build standard wheel (no local version suffix) for PyPI
python -m build --wheel --outdir dist
# Fix the wheel to be manylinux compliant
uv pip install --system auditwheel
# Point auditwheel at torch libs, but do not vendor them into the wheel.
TORCH_LIB_DIR=$(python - <<'PY'
import os
import torch
print(os.path.join(os.path.dirname(torch.__file__), "lib"))
PY
)
export LD_LIBRARY_PATH="${TORCH_LIB_DIR}:${LD_LIBRARY_PATH}"
# Target manylinux_2_35 (Ubuntu 22.04 native), per-arch plat tag.
auditwheel repair dist/*.whl --plat ${{ matrix.platform.wheel-plat }} -w fixed_dist \
--exclude libtorch_cuda.so \
--exclude libtorch_cpu.so \
--exclude libtorch.so \
--exclude libc10.so \
--exclude libc10_cuda.so \
--exclude libtorch_python.so
# Move fixed wheels back to dist for upload consistency
rm dist/*.whl
mv fixed_dist/*.whl dist/
- name: Upload wheel artifact
# Upload every matrix leg as a distinct artifact (for inspection / GitHub releases).
# The publish job below selects which CUDA build is pushed to PyPI.
uses: actions/upload-artifact@v4
with:
name: fastvideo_kernel-py${{ matrix.python-version }}-${{ matrix.torch-cuda.torch-cuda-short }}-${{ matrix.platform.arch }}-torch${{ matrix.torch-cuda.torch-version }}
path: fastvideo-kernel/dist/*.whl
retention-days: 90
publish_package:
name: Publish package
needs: [build_wheels, check-version-change]
if: ${{ needs.check-version-change.outputs.version-changed == 'true' || github.event_name == 'workflow_dispatch' }}
runs-on: ubuntu-22.04
permissions:
id-token: write # Needed for OIDC Trusted Publishing
steps:
- uses: actions/checkout@v4
- uses: actions/setup-python@v5
with:
python-version: '3.12'
- name: Download PyPI wheels
# Publish the cu130 (CUDA 13) wheels to PyPI for both architectures:
# x86_64 — Hopper sm_90a TK + consumer Blackwell sm_120a FP4
# aarch64 — Blackwell: turbodiffusion (sm_100a/sm_120a) + C++ FP4 (sm_120a);
# no TK (Hopper). sm_100 FP4 forward is the FA4 CuTe DSL path in the
# fastvideo package (#1221), shipped/JIT separately.
# The x86_64 cu126 wheel stays available as a build artifact / GitHub-release asset.
uses: actions/download-artifact@v4
with:
path: fastvideo-kernel/dist/
pattern: 'fastvideo_kernel-py*-cu130-*'
merge-multiple: true
- name: Install uv
uses: astral-sh/setup-uv@v3
- name: Build source distribution
run: |
uv pip install --system build scikit-build-core cmake ninja
cd fastvideo-kernel
# We don't need full CUDA/Torch to just package the source (sdist)
python -m build --sdist --outdir dist
- name: Publish release distributions to PyPI
uses: pypa/gh-action-pypi-publish@release/v1
with:
packages-dir: fastvideo-kernel/dist/
+29 -113
View File
@@ -1,12 +1,17 @@
ucf101_stride4x4x4
__pycache__
*.mp4
.ipynb_checkpoints
*.pth
UCF-101/
results/
build/
fastvideo.egg-info/
wandb/
.idea
*.ipynb
*.jpg
!examples/datasets/lingbotworld2/image.jpg
*.mp3
*.safetensors
*.mp4
*.png
@@ -15,124 +20,35 @@ wandb/
*.pt
cache_dir/
wandb/
venv/
.venv/
test*
sample_video*
sample_image*
512*
720*
1024*
debug*
private*
caption*
*deepspeed*
revised*
129f*
all*
read*
YSH*
*pick*
*ysh*
hw*
257f*
513f*
taming*
221hw*
65x512x512
runs/
samples/
Miniconda3-latest-Linux-x86_64.sh
*validation/
data/
outputs/
outputs_video
checkpoints/
sbatch.sh
*.out
env
*.o
**/build/
**.pyc
**.txt
*.log
weights/
logs/
/Z-Image/
official_weights/
converted_weights/
# SSIM test outputs
fastvideo/tests/ssim/generated_videos/
**/.cache/**
# Distribution / packaging
build/
dist/
*.egg-info/
*.egg
eggs/
.eggs/
# MkDocs documentation
site/
docs/getting_started/examples/
docs/examples/
docs/inference/examples/
docs/training/examples/
docs/distillation/examples/
!requirements-mkdocs.txt
# VSCode
.vscode/
# DS Store
.DS_Store
# vim swap files
*.swo
*.swp
# Python pickle files
*.pkl
# Reference videos (negations must come after the catch-all on line below)
!fastvideo/tests/nightly/reference_video_*.mp4
# Static images
!docs/assets/images/**/*.png
!comfyui/assets/**/*.png
!comfyui/assets/**/*.gif
!assets/images/**/*.png
!assets/images/**/*.jpg
!assets/images/**/*.jpeg
!assets/images/**/*.gif
!assets/videos/**/*.mp4
dmd_t2v_output/
preprocess_output_text/
# SvelteKit / Node artifacts under apps/fastvideo_studio/: see apps/fastvideo_studio/.gitignore
# Next.js / Node artifacts under apps/dreamverse/web/
apps/dreamverse/web/node_modules/
apps/dreamverse/web/.next/
apps/dreamverse/web/out/
apps/dreamverse/web/coverage/
apps/dreamverse/web/test-results/
apps/dreamverse/web/playwright-report/
apps/dreamverse/web/.env.local
apps/dreamverse/web/.env.development.local
apps/dreamverse/web/.env.test.local
# Generated by apps/dreamverse/scripts/install_native_ffmpeg.sh — host-specific
apps/dreamverse/scripts/ffmpeg-env.sh
apps/dreamverse/web/.env.production.local
# Unignore migrated Dreamverse product assets — root .gitignore globally
# ignores *.png/*.jpg/*.mp4/*.gif, but apps/dreamverse/web/public/ MUST
# be tracked (logo, icons, k2.png, etc.).
!apps/dreamverse/web/public/**/*.png
!apps/dreamverse/web/public/**/*.jpg
!apps/dreamverse/web/public/**/*.jpeg
!apps/dreamverse/web/public/**/*.mp4
!apps/dreamverse/web/public/**/*.gif
!apps/dreamverse/web/prompts/**/*.png
!apps/dreamverse/web/prompts/**/*.jpg
!apps/dreamverse/web/prompts/**/*.jpeg
!apps/dreamverse/web/prompts/**/*.mp4
!apps/dreamverse/web/prompts/**/*.gif
!apps/dreamverse/gpu-pool.svg
!apps/dreamverse/gpu-pool.drawio
.claude/
.codex/
.agents/tmp/
.sisyphus/
openspec/
fastvideo/tests/ssim/reference_videos/**
!fastvideo/tests/ssim/reference_videos/**/*.mp4
!fastvideo/tests/ssim/reference_videos/**/*.png
# Editor logs and local Python version pins (accidentally committed)
*.nvimlog
.nvimlog
.python-version
-9
View File
@@ -1,9 +0,0 @@
[submodule "fastvideo-kernel/include/tk"]
path = fastvideo-kernel/include/tk
url = https://github.com/HazyResearch/ThunderKittens.git
[submodule "fastvideo-kernel/include/cutlass"]
path = fastvideo-kernel/include/cutlass
url = https://github.com/NVIDIA/cutlass.git
[submodule "fastvideo/third_party/eval/vbench"]
path = fastvideo/third_party/eval/vbench
url = https://github.com/Vchitect/VBench.git
-75
View File
@@ -1,75 +0,0 @@
default_stages:
- pre-commit # Run locally
- manual # Run in CI
exclude: |
(?x)(
fastvideo/third_party/.*|
fastvideo-kernel/.*|
assets/.*|
tests/.*|
scripts/.*|
fastvideo/dataset/.*|
fastvideo/models/.*|
^apps/dreamverse/web/.*|
examples/.*|
\.agents/.*|
.github/workflows/publish-fastvideo.yml|
.github/workflows/_template-build-image.yml
)
repos:
- repo: https://github.com/google/yapf
rev: v0.43.0
hooks:
- id: yapf
args: [--in-place, --verbose]
language_version: python3.12
additional_dependencies: [toml] # TODO: Remove when yapf is upgraded
- repo: https://github.com/astral-sh/ruff-pre-commit
rev: v0.11.12
hooks:
- id: ruff
args: [--output-format, github, --fix]
- repo: https://github.com/codespell-project/codespell
rev: v2.4.1
hooks:
- id: codespell
additional_dependencies: ['tomli']
args: ['--toml', 'pyproject.toml']
# - repo: https://github.com/PyCQA/isort
# rev: 6.0.1
# hooks:
# - id: isort
- repo: https://github.com/jackdewinter/pymarkdown
rev: v0.9.30
hooks:
- id: pymarkdown
args: [fix]
- repo: https://github.com/rhysd/actionlint
rev: v1.7.7
hooks:
- id: actionlint
- repo: https://github.com/pre-commit/mirrors-mypy
rev: v1.15.0
hooks:
- id: mypy
args: [--python-version, '3.10', --follow-imports, "skip", "--disable-error-code", "union-attr", "--disable-error-code", "override" ]
additional_dependencies: [types-aiofiles, types-cachetools, types-setuptools, types-PyYAML, types-requests]
- repo: local
hooks:
- id: check-filenames
name: Check for spaces in all filenames
entry: bash
args:
- -c
- 'git ls-files | grep -v "^\"*fastvideo/tests/ssim/" | grep -v "^\"*fastvideo/tests/inference/lora/L40S_reference_videos/" | grep " " && echo "Filenames should not contain spaces!" && exit 1 || exit 0'
language: system
always_run: true
pass_filenames: false
# Keep `suggestion` last
- id: suggestion
name: Suggestion
entry: bash -c 'echo "To bypass pre-commit hooks, add --no-verify to git commit."'
language: system
verbose: true
pass_filenames: false
# Insert new entries above the `suggestion` entry
-86
View File
@@ -1,86 +0,0 @@
# Repository Guidelines
## Project Structure & Module Organization
- Core Python package: `fastvideo/` (models, pipelines, training, distributed runtime, CLI entrypoints).
- CUDA/custom kernels: `fastvideo-kernel/` (separate build/test flow).
- Tests:
- `fastvideo/tests/` for package-level tests (dataset, encoders, inference, training, SSIM, workflow).
- `tests/local_tests/` for additional local/component checks.
- Docs and guides: `docs/` (MkDocs source), with contributor docs in `docs/contributing/`.
- Runnable examples and scripts: `examples/` and `scripts/`.
- Static assets: `assets/` (including `assets/images/`, `assets/videos/`, and `assets/prompts/`) and `comfyui/assets/`.
## Build, Test, and Development Commands
- `UV_TORCH_BACKEND=cu126 uv pip install -e ".[dev]"`: editable CUDA 12 install with lint/test extras (`cu130` on CUDA 13).
- `pre-commit install --hook-type pre-commit --hook-type commit-msg`: enable local hooks.
- `pre-commit run --all-files`: run formatter/lint/type/spelling checks.
- `pytest tests/`: run top-level test suite.
- `pytest fastvideo/tests/ -v`: run package tests.
- `pytest fastvideo/tests/ssim/ -vs`: run SSIM regression tests (GPU-heavy).
- `cd fastvideo-kernel && ./build.sh`: build kernel extensions.
## Coding Style & Naming Conventions
- Python 3.10+; 4-space indentation; keep code and imports readable and explicit.
- Style tools are configured in `pyproject.toml` and `.pre-commit-config.yaml`:
- `yapf` (format), `ruff` (lint, auto-fix), `mypy` (typing), `codespell`.
- Lint via `pre-commit run --files <changed paths>` (or `pre-commit run --all-files` for a full sweep) before committing. Do not shell out to `yapf`/`ruff`/`codespell`/`mypy` directly — pre-commit chains them with the project's config and respects the `.pre-commit-config.yaml` excludes (e.g. `fastvideo/tests/` is intentionally skipped). If pre-commit reports `(no files to check)` for your paths, that exclude is deliberate — don't bypass it.
- Target line length is 120 (configured in `pyproject.toml` for ruff, yapf, and isort).
- Naming: `snake_case` for functions/files, `PascalCase` for classes, `UPPER_SNAKE_CASE` for constants.
## Testing Guidelines
- Use `pytest` and place tests near relevant domains (e.g., `fastvideo/tests/encoders/`).
- Prefer descriptive names like `test_<feature>_<expected_behavior>.py`.
- For new pipelines/backends, include at least one regression-oriented test; add SSIM coverage when output quality must be preserved.
- Document GPU assumptions in tests that require specific hardware.
## Commit & Pull Request Guidelines
- Follow existing commit style: short subject with optional tag prefix, e.g. `[bugfix]: ...`, `[feat]: ...`, `[misc]: ...`, and include PR reference like `(#1234)` when applicable.
- Keep commits focused by concern (feature, refactor, fix).
- PRs should include:
- clear problem/solution summary,
- test evidence (`pytest`/SSIM outputs or rationale if skipped),
- linked issue/PR context,
- screenshots or sample outputs for UI/demo/docs changes.
## Agent Infrastructure
This repository is agent-friendly. Before doing any work:
1. Read the nearest in-scope `AGENTS.md` for every directory you may edit.
2. Read the relevant user-facing design or contributor guide.
3. Check `.agents/skills/*/SKILL.md` for a task-specific workflow.
4. Search `.agents/lessons/` for relevant pitfalls.
Architecture, commands, and operational guidance belong beside the code or
under `docs/`. Do not maintain static codebase maps, experiment journals,
branch-state snapshots, or duplicate user-facing documentation under
`.agents/`. Put a reusable procedure directly in a skill or contributor guide
and capture durable failures in `.agents/lessons/`.
## Per-Directory AGENTS.md
Local guidance lives next to the code. Read the in-scope file before editing:
| Directory | What it covers |
|-----------|----------------|
| `fastvideo/AGENTS.md` | Core package map, public API, registry-driven model dispatch |
| `fastvideo/configs/AGENTS.md` | Arch + pipeline config dataclasses, `param_names_mapping` |
| `fastvideo/models/AGENTS.md` | DiT / VAE / encoder / scheduler / loader layout (pre-commit excluded) |
| `fastvideo/layers/AGENTS.md` | Tensor-parallel linear/attention layer rules for ports |
| `fastvideo/attention/AGENTS.md` | Backend registry + env-var override |
| `fastvideo/pipelines/AGENTS.md` | Stage ABC, `basic/<model>/`, `preprocess/`, presets |
| `fastvideo/training/AGENTS.md` | Legacy monolithic pipelines (frozen for existing models) |
| `fastvideo/train/AGENTS.md` | New modular trainer (methods × models × callbacks, YAML) |
| `fastvideo/tests/AGENTS.md` | Test taxonomy, conftest, pre-commit-excluded path |
| `fastvideo/tests/ssim/AGENTS.md` | GPU SSIM regression authoring + reference video sync |
| `scripts/checkpoint_conversion/AGENTS.md` | Adding a converter for a new HF/official checkpoint |
## Critical: Two Training Stacks Coexist
- `fastvideo/training/` — legacy, monolithic per-model `*_training_pipeline.py` and
`*_distillation_pipeline.py`. Still authoritative for shipped models.
- `fastvideo/train/` — new modular framework (composable methods × models × callbacks
driven by YAML). Preferred for new training work.
Pick the matching stack before editing. Do not migrate a pipeline between them
without an explicit ask — the conventions and config surfaces differ.
-1
View File
@@ -1 +0,0 @@
@AGENTS.md
+17 -183
View File
@@ -1,187 +1,21 @@
Apache License
Version 2.0, January 2004
http://www.apache.org/licenses/
MIT License
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
Copyright (c) 2024 PKU-YUAN's Group (袁粒课题组-北大信工) and Rabbitpre AI
1. Definitions.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
"License" shall mean the terms and conditions for use, reproduction,
and distribution as defined by Sections 1 through 9 of this document.
The above copyright notice and this permission notice shall be included in all
copies or substantial portions of the Software.
"Licensor" shall mean the copyright owner or entity authorized by
the copyright owner that is granting the License.
"Legal Entity" shall mean the union of the acting entity and all
other entities that control, are controlled by, or are under common
control with that entity. For the purposes of this definition,
"control" means (i) the power, direct or indirect, to cause the
direction or management of such entity, whether by contract or
otherwise, or (ii) ownership of fifty percent (50%) or more of the
outstanding shares, or (iii) beneficial ownership of such entity.
"You" (or "Your") shall mean an individual or Legal Entity
exercising permissions granted by this License.
"Source" form shall mean the preferred form for making modifications,
including but not limited to software source code, documentation
source, and configuration files.
"Object" form shall mean any form resulting from mechanical
transformation or translation of a Source form, including but
not limited to compiled object code, generated documentation,
and conversions to other media types.
"Work" shall mean the work of authorship, whether in Source or
Object form, made available under the License, as indicated by a
copyright notice that is included in or attached to the work
(an example is provided in the Appendix below).
"Derivative Works" shall mean any work, whether in Source or Object
form, that is based on (or derived from) the Work and for which the
editorial revisions, annotations, elaborations, or other modifications
represent, as a whole, an original work of authorship. For the purposes
of this License, Derivative Works shall not include works that remain
separable from, or merely link (or bind by name) to the interfaces of,
the Work and Derivative Works thereof.
"Contribution" shall mean any work of authorship, including
the original version of the Work and any modifications or additions
to that Work or Derivative Works thereof, that is intentionally
submitted to Licensor for inclusion in the Work by the copyright owner
or by an individual or Legal Entity authorized to submit on behalf of
the copyright owner. For the purposes of this definition, "submitted"
means any form of electronic, verbal, or written communication sent
to the Licensor or its representatives, including but not limited to
communication on electronic mailing lists, source code control systems,
and issue tracking systems that are managed by, or on behalf of, the
Licensor for the purpose of discussing and improving the Work, but
excluding communication that is conspicuously marked or otherwise
designated in writing by the copyright owner as "Not a Contribution."
"Contributor" shall mean Licensor and any individual or Legal Entity
on behalf of whom a Contribution has been received by Licensor and
subsequently incorporated within the Work.
2. Grant of Copyright License. Subject to the terms and conditions of
this License, each Contributor hereby grants to You a perpetual,
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
copyright license to reproduce, prepare Derivative Works of,
publicly display, publicly perform, sublicense, and distribute the
Work and such Derivative Works in Source or Object form.
3. Grant of Patent License. Subject to the terms and conditions of
this License, each Contributor hereby grants to You a perpetual,
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
(except as stated in this section) patent license to make, have made,
use, offer to sell, sell, import, and otherwise transfer the Work,
where such license applies only to those patent claims licensable
by such Contributor that are necessarily infringed by their
Contribution(s) alone or by combination of their Contribution(s)
with the Work to which such Contribution(s) was submitted. If You
institute patent litigation against any entity (including a
cross-claim or counterclaim in a lawsuit) alleging that the Work
or a Contribution incorporated within the Work constitutes direct
or contributory patent infringement, then any patent licenses
granted to You under this License for that Work shall terminate
as of the date such litigation is filed.
4. Redistribution. You may reproduce and distribute copies of the
Work or Derivative Works thereof in any medium, with or without
modifications, and in Source or Object form, provided that You
meet the following conditions:
(a) You must give any other recipients of the Work or
Derivative Works a copy of this License; and
(b) You must cause any modified files to carry prominent notices
stating that You changed the files; and
(c) You must retain, in the Source form of any Derivative Works
that You distribute, all copyright, patent, trademark, and
attribution notices from the Source form of the Work,
excluding those notices that do not pertain to any part of
the Derivative Works; and
(d) If the Work includes a "NOTICE" text file as part of its
distribution, then any Derivative Works that You distribute must
include a readable copy of the attribution notices contained
within such NOTICE file, excluding those notices that do not
pertain to any part of the Derivative Works, in at least one
of the following places: within a NOTICE text file distributed
as part of the Derivative Works; within the Source form or
documentation, if provided along with the Derivative Works; or,
within a display generated by the Derivative Works, if and
wherever such third-party notices normally appear. The contents
of the NOTICE file are for informational purposes only and
do not modify the License. You may add Your own attribution
notices within Derivative Works that You distribute, alongside
or as an addendum to the NOTICE text from the Work, provided
that such additional attribution notices cannot be construed
as modifying the License.
You may add Your own copyright statement to Your modifications and
may provide additional or different license terms and conditions
for use, reproduction, or distribution of Your modifications, or
for any such Derivative Works as a whole, provided Your use,
reproduction, and distribution of the Work otherwise complies with
the conditions stated in this License.
5. Submission of Contributions. Unless You explicitly state otherwise,
any Contribution intentionally submitted for inclusion in the Work
by You to the Licensor shall be under the terms and conditions of
this License, without any additional terms or conditions.
Notwithstanding the above, nothing herein shall supersede or modify
the terms of any separate license agreement you may have executed
with Licensor regarding such Contributions.
6. Trademarks. This License does not grant permission to use the trade
names, trademarks, service marks, or product names of the Licensor,
except as required for reasonable and customary use in describing the
origin of the Work and reproducing the content of the NOTICE file.
7. Disclaimer of Warranty. Unless required by applicable law or
agreed to in writing, Licensor provides the Work (and each
Contributor provides its Contributions) on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
implied, including, without limitation, any warranties or conditions
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
PARTICULAR PURPOSE. You are solely responsible for determining the
appropriateness of using or redistributing the Work and assume any
risks associated with Your exercise of permissions under this License.
8. Limitation of Liability. In no event and under no legal theory,
whether in tort (including negligence), contract, or otherwise,
unless required by applicable law (such as deliberate and grossly
negligent acts) or agreed to in writing, shall any Contributor be
liable to You for damages, including any direct, indirect, special,
incidental, or consequential damages of any character arising as a
result of this License or out of the use or inability to use the
Work (including but not limited to damages for loss of goodwill,
work stoppage, computer failure or malfunction, or any and all
other commercial damages or losses), even if such Contributor
has been advised of the possibility of such damages.
9. Accepting Warranty or Additional Liability. While redistributing
the Work or Derivative Works thereof, You may choose to offer,
and charge a fee for, acceptance of support, warranty, indemnity,
or other liability obligations and/or rights consistent with this
License. However, in accepting such obligations, You may act only
on Your own behalf and on Your sole responsibility, not on behalf
of any other Contributor, and only if You agree to indemnify,
defend, and hold each Contributor harmless for any liability
incurred by, or claims asserted against, such Contributor by reason
of your accepting any such warranty or additional liability.
END OF TERMS AND CONDITIONS
APPENDIX: How to apply the Apache License to your work.
To apply the Apache License to your work, attach the following
boilerplate notice, with the fields enclosed by brackets "[]"
replaced with your own identifying information. (Don't include
the brackets!) The text should be enclosed in the appropriate
comment syntax for the file format. We also recommend that a
file or class name and description of purpose be included on the
same "printed page" as the copyright notice for easier
identification within third-party archives.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
SOFTWARE.
+86 -176
View File
@@ -1,198 +1,108 @@
<div align="center">
<img src=assets/logos/logo.svg width="30%"/>
</div>
# Fast Video
This is currently based on Open-Sora-1.2.0: https://github.com/PKU-YuanGroup/Open-Sora-Plan/tree/294993ca78bf65dec1c3b6fb25541432c545eda9
<p align="center">
| <a href="https://hao-ai-lab.github.io/FastVideo"><b>Documentation</b></a> | <a href="https://hao-ai-lab.github.io/FastVideo/inference/inference_quick_start/"><b> Quick Start</b></a> | <a href="https://github.com/hao-ai-lab/FastVideo/discussions/982" target="_blank"><b>Weekly Dev Meeting</b></a> | 🟣💬 <a href="https://join.slack.com/t/fastvideo/shared_invite/zt-3f4lao1uq-u~Ipx6Lt4J27AlD2y~IdLQ" target="_blank"> <b>Slack</b> </a> | 🟣💬 <a href="https://github.com/hao-ai-lab/FastVideo/discussions/1097" target="_blank"> <b> WeChat </b> </a> |
</p>
**FastVideo is a unified post-training and real-time inference framework for accelerated video generation.**
## NEWS
- `2026/06/23`: Release FastWan-QAD: 5s of Video generated in 1.8s E2E. See the [FastWan-QAD models](https://huggingface.co/FastVideo/FastWan-QAD-FP8-1.3B), [Attn-QAT training guide](https://haoailab.com/FastVideo/training/attn_qat/), and [blog](https://haoailab.com/blogs/fastwan-qad/).
- `2026/03/17`: Release demo: Into the Dreamverse: Vibe Directing in FastVideo, check out the [Blog](https://haoailab.com/blogs/dreamverse/).
- `2026/03/13`: Release demo: Create a 5s 1080p Video in 4.5s with FastVideo on a Single GPU, check out the [Blog](https://haoailab.com/blogs/fastvideo_realtime_1080p/).
- `2025/11/19`: Release [CausalWan2.2 I2V A14B Preview](https://huggingface.co/FastVideo/CausalWan2.2-I2V-A14B-Preview-Diffusers) models, [Blog](https://hao-ai-lab.github.io/blogs/fastvideo_causalwan_preview/) and [Inference Code!](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_self_forcing_causal_wan2_2_i2v.py).
- `2025/08/04`: Release [FastWan](https://hao-ai-lab.github.io/FastVideo/distillation/dmd) models and [Sparse-Distillation](https://hao-ai-lab.github.io/blogs/fastvideo_post_training/).
### More News
- `2025/06/14`: Release finetuning and inference code for [VSA](https://arxiv.org/pdf/2505.13389).
- `2025/04/24`: [FastVideo V1](https://hao-ai-lab.github.io/blogs/fastvideo/) is released!
- `2025/02/18`: Release the inference code for [Sliding Tile Attention](https://hao-ai-lab.github.io/blogs/sta/).
## Key Features
FastVideo has the following features:
- End-to-end post-training support for bidirectional and autoregressive models:
- Support full finetuning and LoRA finetuning for state-of-the-art open video DiTs
- Data preprocessing pipeline for video, image, and text data
- Distribution Matching Distillation (DMD2) stepwise distillation.
- Sparse attention with [Video Sparse Attention](https://arxiv.org/pdf/2505.13389)
- [Sparse distillation](https://hao-ai-lab.github.io/blogs/fastvideo_post_training/) to achieve >50x denoising speedup
- Scalable training with FSDP2, sequence parallelism, and selective activation checkpointing.
- Causal distillation through Self-Forcing
- See this [page](https://hao-ai-lab.github.io/FastVideo/training/overview/) for the supported training workflows, and the [support matrix](https://hao-ai-lab.github.io/FastVideo/inference/support_matrix/) for supported models.
- State-of-the-art performance optimizations for inference
- Sequence Parallelism for distributed inference
- Multiple state-of-the-art attention backends
- User-friendly CLI and Python API
- See this [page](https://hao-ai-lab.github.io/FastVideo/inference/optimizations/) for full list of supported optimizations.
- Diverse hardware and OS support
- Support H100, A100, 4090
- Support Linux, Windows, MacOS
- See this [page](https://hao-ai-lab.github.io/FastVideo/inference/support_matrix/) for full list of supported models, hardware assumptions, and optimization compatibility.
- Realtime video generation & editing
- [Dreamverse](apps/dreamverse/README.md): stream and "vibe direct" video in realtime ([live demo](https://dreamverse.fastvideo.org/)), deployable on local GPU, a self-hosted B200 server, Docker, or serverless Modal
## Getting Started
We recommend using [uv](https://docs.astral.sh/uv/) to create a clean environment. If you previously used Conda, switching to uv generally gives faster and more stable installs.
```bash
# Create and activate a new uv environment
uv venv --python 3.12 --seed
source .venv/bin/activate
# Install FastVideo on NVIDIA CUDA 12
UV_TORCH_BACKEND=cu126 uv pip install fastvideo
## Envrironment
Change the index-url cuda version according to your system.
```
conda create -n fastvideo python=3.10.12
conda activate fastvideo
pip3 install torch==2.5.0 torchvision --index-url https://download.pytorch.org/whl/cu121
pip install git+https://github.com/huggingface/diffusers.git@76b7d86a9a5c0c2186efa09c4a67b5f5666ac9e3
pip install packaging ninja && pip install flash-attn==2.7.0.post2 --no-build-isolation
```
Use `UV_TORCH_BACKEND=cu130` on CUDA 13. Apple silicon users should follow the
[MPS installation guide](https://hao-ai-lab.github.io/FastVideo/getting_started/installation/mps/).
Please see our [docs](https://hao-ai-lab.github.io/FastVideo/getting_started/installation/) for more detailed installation instructions.
> **On an NVIDIA DGX Spark (GB10 / ARM64 + CUDA 13)?** There's no prebuilt ARM wheel for the FastVideo CUDA kernel, so it's an editable from-source install (`UV_TORCH_BACKEND=cu130 uv pip install -e .`, which compiles that kernel for you) rather than `UV_TORCH_BACKEND=cu130 uv pip install fastvideo`. A compatible prebuilt ARM64 FlashAttention wheel is available separately. Follow the [DGX Spark install guide](https://hao-ai-lab.github.io/FastVideo/getting_started/installation/spark/).
### Install with an AI coding agent
FastVideo is a monorepo with rich agent guidance (see [`AGENTS.md`](AGENTS.md)). If you use Claude Code, Cursor, or another coding agent, paste the prompt below — it detects your platform and follows the matching guide:
```text
Install FastVideo (https://github.com/hao-ai-lab/FastVideo) into a fresh uv virtual environment.
1. Detect the platform: run `uname -m`, `nvidia-smi`, and `nvcc --version`.
2. Read and follow the matching install guide exactly (in this repo, or at
https://hao-ai-lab.github.io/FastVideo/getting_started/installation/):
- NVIDIA GPU, x86_64 -> docs/getting_started/installation/gpu.md
- NVIDIA DGX Spark / GB10, aarch64, CUDA 13 -> docs/getting_started/installation/spark.md
- Apple Silicon, macOS -> docs/getting_started/installation/mps.md
3. Use uv for every step. If a command fails, debug it and tell me what you changed.
4. Verify the result:
python -c "import fastvideo, torch; print('cuda', torch.cuda.is_available())"
fastvideo --help
5. Report which platform you detected and any deviations you had to make.
```
pip install -e . && pip install -e ".[train]"
sudo apt-get update && apt install screen && pip install watch gpustat
```
## Sparse Distillation
## Prepare Data & Models
We've prepared some debug data to facilitate development. To make sure the training pipeline is correct, train on the debug data and make sure the model overfit on it (feed it the same text prompt and see if the output video is the same as the training data)
For our sparse distillation techniques, please see our [distillation docs](https://hao-ai-lab.github.io/FastVideo/distillation/dmd/) and check out our [blog](https://hao-ai-lab.github.io/blogs/fastvideo_post_training/).
See below for recipes and datasets:
| Model | Sparse Distillation | Dataset |
| ------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------- |
| [FastWan2.1-T2V-1.3B](https://huggingface.co/FastVideo/FastWan2.1-T2V-1.3B-Diffusers) | [Recipe](https://github.com/hao-ai-lab/FastVideo/tree/main/examples/distill/Wan2.1-T2V/Wan-Syn-Data-480P) | [FastVideo Synthetic Wan2.1 480P](https://huggingface.co/datasets/FastVideo/Wan-Syn_77x448x832_600k) |
| [FastWan2.2-TI2V-5B](https://huggingface.co/FastVideo/FastWan2.2-TI2V-5B-Diffusers) | [Recipe](https://github.com/hao-ai-lab/FastVideo/tree/main/examples/distill/Wan2.2-TI2V-5B-Diffusers/Data-free) | [FastVideo Synthetic Wan2.2 720P](https://huggingface.co/datasets/FastVideo/Wan2.2-Syn-121x704x1280_32k) |
## Dreamverse — Realtime Video Generation & Editing
[Dreamverse](apps/dreamverse/README.md) is FastVideo's realtime video generation
and editing platform — "vibe directing" a video as it streams. It lives in the
monorepo under [`apps/dreamverse/`](apps/dreamverse/) and ships its own backend
(`dreamverse-server`) plus a web UI.
Try the [live demo](https://dreamverse.fastvideo.org/), read the
[blog](https://haoailab.com/blogs/dreamverse/), or run it yourself. Dreamverse
deploys on a local GPU, a self-hosted B200 server over SSH, Docker, or
serverless [Modal](apps/dreamverse/scripts/modal/README.md) — see the
[Dreamverse README](apps/dreamverse/README.md).
## Inference
### Generating Your First Video
Here's a minimal example to generate a video using the default settings. Make sure VSA kernels are [installed](https://hao-ai-lab.github.io/FastVideo/attention/vsa/#installation). Create a file called `example.py` with the following code:
```python
import os
from fastvideo import VideoGenerator
def main():
os.environ["FASTVIDEO_ATTENTION_BACKEND"] = "VIDEO_SPARSE_ATTN"
# Create a video generator with a pre-trained model
generator = VideoGenerator.from_pretrained(
"FastVideo/FastWan2.1-T2V-1.3B-Diffusers",
num_gpus=1, # Adjust based on your hardware
)
# Define a prompt for your video
prompt = "A curious raccoon peers through a vibrant field of yellow sunflowers, its eyes wide with interest."
# Generate the video
video = generator.generate_video(
prompt,
output_path="my_videos/", # Controls where videos are saved
save_video=True
)
if __name__ == '__main__':
main()
```
mkdir data && mkdir data/outputs/
python scripts/download_hf.py --repo_id=Stealths-Video/mochi_diffuser --local_dir=data/mochi --repo_type=model
python scripts/download_hf.py --repo_id=Stealths-Video/Merge-30k-Data --local_dir=data/Merge-30k-Data --repo_type=dataset
python scripts/download_hf.py --repo_id=Stealths-Video/validation_embeddings --local_dir=data/validation_embeddings --repo_type=dataset
cd data/Merge-30k-Data
cat Merged30K.tar.gz.part.* > Merged30K.tar.gz
rm Merged30K.tar.gz.part.*
tar --use-compress-program="pigz --processes 64" -xvf Merged30K.tar.gz
mv ephemeral/hao.zhang/codefolder/FastVideo-OSP/data/Merged-30K-Data/* .
rm -r ephemeral
rm Merged30K.tar.gz
cd ../..
```
Run the script with:
## Things Learned
1. shift8 clear but got structural artifacts
2. lq, 0.025 vague
3. adv not really helpful
4. shift8 euler steps 50 v.s. 100 very similar
5. 为啥image不会越distill越炸
6. EMA, 大batchsize, 1.5,2.5,3.5,4.5
7. Must have schedule
8. phase 1, 2 learning rate 5e-6不行
```bash
python example.py
```
## Experiments
Scripts are located at scripts/experiment_N.sh
For a more detailed guide, please see our [inference quick start](https://hao-ai-lab.github.io/FastVideo/inference/inference_quick_start/).
1. pcm_linear_quadratic, euler_steps 50, 0.025
2. pcm_linear_quadratic, euler_steps 50, 0.05
3. shift 8, euler_steps 100
4. shift 8, euler_steps 50
5. shift 8, euler_steps 100, adv
6. pcm_linear_quadratic, euler_steps 50, 0.025, adv
7. pcm_linear_quadratic, euler_steps 50, 0.05, multiphase 125
8. pcm_linear_quadratic, euler_steps 50, 0.1, linear_range 0.75
9. pcm_linear_quadratic, euler_steps 50, 0.05, range 0.75
10. pcm_linear_quadratic, euler_steps 50, 0.05, batchsize 32
11. pcm_linear_quadratic, euler_steps 50, learning rate,1e-7
12. shift1, euler_steps 50
## More Guides
13. pcm_linear_quadratic, euler_steps 50, 0.1, linear_range 0.75, phase 1
14. 4.5 cfg, validation no cfg, pcm_linear_quadratic, euler_steps 50, 0.1, linear_range 0.75
15. pcm_linear_quadratic, euler_steps 50, 0.15, linear_range 0.75
16. pcm_linear_quadratic, euler_steps 50, 0.1, linear_range 0.75 ema 0.95, decay 0.0
- [Design Overview](https://hao-ai-lab.github.io/FastVideo/design/overview/)
- [Distillation Guide](https://hao-ai-lab.github.io/FastVideo/distillation/dmd/)
- [Contribution Guide](https://hao-ai-lab.github.io/FastVideo/contributing/overview/)
## Awesome work using FastVideo or our research projects
17. no cfg, validation no cfg, pcm_linear_quadratic, euler_steps 50, 0.1, linear_range 0.75
18. shift16, euler_steps 50
- [SGLang](https://github.com/sgl-project/sglang/tree/main/python/sglang/multimodal_gen): SGLang's diffusion inference functionality is based on a fork of FastVideo on Sept. 24, 2025.
- [DanceGRPO](https://github.com/XueZeyue/DanceGRPO): A unified framework to adapt Group Relative Policy Optimization (GRPO) to visual generation paradigms. Code based on FastVideo.
- [SRPO](https://github.com/Tencent-Hunyuan/SRPO): A method to directly align the full diffusion trajectory with fine-grained human preference. Code based on FastVideo.
- [DCM](https://github.com/Vchitect/DCM): Dual-expert consistency model for efficient and high-quality video generation. Code based on FastVideo.
- [HY-WorldPlay](https://github.com/Tencent-Hunyuan/HY-WorldPlay): An action-conditioned world model model trained using FastVideo framework.
- [Hunyuan Video 1.5](https://github.com/Tencent-Hunyuan/HunyuanVideo-1.5): A leading lightweight video generation model, where they proposed SSTA based on Sliding Tile Attention.
- [Kandinsky-5.0](https://github.com/kandinskylab/kandinsky-5): A family of diffusion models for video & image generation, where their NABLA attention includes a Sliding Tile Attention branch.
- [LongCat Video](https://github.com/meituan-longcat/LongCat-Video): A foundational video generation model with 13.6B parameters with block-sparse attention similar to Video Sparse Attention.
19. 4step_infer_shift16_euler_50
20. 4step_infer_shift12_euler_50
21. 4step_infer_lq_euler_50_thresh0.1_lrg_0.75
22. pcm_linear_quadratic, euler_steps 50, 0.1, linear_range 0.75, phase 1, lr 1e-7
23. lq_euler_50_thres0.1_lrg_0.75_bs_64
24. lq_euler_50_thres0.1_lrg_0.75_lr5e-7
## 🤝 Contributing
We welcome all contributions. Please check out our guide [here](https://hao-ai-lab.github.io/FastVideo/contributing/overview/).
See details in [development roadmap](https://github.com/hao-ai-lab/FastVideo/issues/899).
## Acknowledgement
25. shift1_euler_50_0.75_phase1
26. kill
27. pcm_linear_quadratic, euler_steps 50, 0.1, linear_range 0.75, phase 1, ema 0.95, cfg 4.5
We learned the design and reused code from the following projects: [Wan-Video](https://github.com/Wan-Video), [ThunderKittens](https://github.com/HazyResearch/ThunderKittens), [DMD2](https://github.com/tianweiy/DMD2), [diffusers](https://github.com/huggingface/diffusers), [xDiT](https://github.com/xdit-project/xDiT), [vLLM](https://github.com/vllm-project/vllm), [SGLang](https://github.com/sgl-project/sglang). We thank [MBZUAI](https://ifm.mbzuai.ac.ae/), [Anyscale](https://www.anyscale.com/), and [GMI Cloud](https://www.gmicloud.ai/) for their support throughout this project.
28. lq_euler_50_thresh0.1_lrg_0.75_phase1_ema0.95
29. lq_euler_50_thres0.1_lrg_0.75_phase_ema0.95_cfg7
30. lq_euler_50_thresh0.1_lrg_0.75_phase1_ema0.98_cfg4.5
31. lq_euler_50_thresh0.1_lrg_0.75_phase1_lr_3e-7
32. lq_euler_50_thresh0.15_lrg_0.75_phase1_ema0.95_cfg4.5
33. lq_euler_50_thres0.1_linear_range_0.75_repro
34. lq_euler_50_thres0.1_lrg_0.75_reproduc
## Citation
35. pcm_linear_quadratic, euler_steps 50, 0.1, linear_range 0.75, phase 1, learning rate 5e-6
36. pcm_linear_quadratic, euler_steps 50, 0.1, linear_range 0.75, phase 2, learning rate 1e-6
37. pcm_linear_quadratic, euler_steps 50, 0.1, linear_range 0.75, phase 2, learning rate 5e-6
38. lq_euler_50_thres0.1_linear_range_0.75, learning rate 5e-6
39. lq_euler_50_thres0.1_linear_range_0.75, learning rate 1e-5
40. lq_euler_50_thres0.1_lrg_0.75_phase1_lr1e-6_repro
If you find FastVideo useful, please consider citing our research work:
```bibtex
@article{zhang2025vsa,
title={Vsa: Faster video diffusion with trainable sparse attention},
author={Zhang, Peiyuan and Chen, Yongqi and Huang, Haofeng and Lin, Will and Liu, Zhengzhong and Stoica, Ion and Xing, Eric and Zhang, Hao},
journal={arXiv preprint arXiv:2505.13389},
year={2025}
}
41. lq_euler_50_thres0.1_lrg_0.75_reproduce
42. pcm_linear_quadratic, euler_steps 50, 0.1, linear_range 0.75, phase 4, learning rate 1e-6
43. pcm_linear_quadratic, euler_steps 50, 0.1, linear_range 0.75, phase 1, learning rate 1e-6, cfg 6.0
44. lq_euler_50_thres0.1_lrg_0.75_phase1_lr_5e-6_test_norm
45. lq_euler_50_thres0.1_lrg_0.75_phase1_lr_5e-6_pred_decay_0.1_latent14
46-48. lq_euler_50_thres0.1_lrg_0.75_phase1_lr1e-6, l2 or l1, decay weight 0.1 to 0.001
@article{zhang2025fast,
title={Fast video generation with sliding tile attention},
author={Zhang, Peiyuan and Chen, Yongqi and Su, Runlong and Ding, Hangliang and Stoica, Ion and Liu, Zhengzhong and Zhang, Hao},
journal={arXiv preprint arXiv:2502.04507},
year={2025}
}
```
49.
-10
View File
@@ -1,10 +0,0 @@
try:
from .comfyui.video_generator.nodes import (NODE_CLASS_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS)
WEB_DIRECTORY = "./web"
__all__ = ['NODE_CLASS_MAPPINGS', 'NODE_DISPLAY_NAME_MAPPINGS', 'WEB_DIRECTORY']
except ImportError:
# ComfyUI environment not available, skip comfyui imports
NODE_CLASS_MAPPINGS = {}
NODE_DISPLAY_NAME_MAPPINGS = {}
WEB_DIRECTORY = "./web"
__all__ = ['NODE_CLASS_MAPPINGS', 'NODE_DISPLAY_NAME_MAPPINGS', 'WEB_DIRECTORY']
-215
View File
@@ -1,215 +0,0 @@
# Dreamverse Agent Notes
## Repo-local workflows
- Use `.agents/skills/dreamverse-deploy/` to redeploy or stop the local
backend/frontend stack on a chosen physical GPU.
- Use `apps/dreamverse/scripts/modal/README.md` for Modal deployments and
`apps/dreamverse/docker/README.md` for image builds. The local deploy skill
intentionally does not manage remote deployments.
## Repo layout
Current paths:
- `apps/dreamverse/web/`: Next.js frontend, client-side stores, websocket
event reduction, prompt-window editing, devtools UI.
- `apps/dreamverse/dreamverse/`: current Python FastAPI runtime, websocket
protocol, prompt enhancement, prompt rewrite orchestration, GPU worker
lifecycle.
- `apps/dreamverse/dreamverse/tests/`: backend unit and integration-oriented
tests.
- `apps/dreamverse/dreamverse/benchmarks/`: prompt-provider latency/token
benchmarking scripts.
Planned paths during the OSS reorg:
- `controller/`: local control plane for provider credentials, compute
lifecycle, and proxying.
- `runtime/`: eventual rename of `apps/dreamverse/dreamverse/` once the
controller/runtime split is stable.
- `providers/`: provider adapters for local, Runpod, and Modal.
Important rule:
- Until the split lands, treat `apps/dreamverse/dreamverse/` as the
authoritative runtime and keep provider orchestration out of
`apps/dreamverse/web`.
## System split
Dreamverse is moving toward a three-part local-first architecture.
- Frontend owns local UI state, drafts, inspection tools, and user-triggered
actions.
- Controller will own local-only credentials, compute provisioning, runtime
lifecycle, and HTTP/websocket proxying.
- Runtime owns generation sessions, prompt rewrite, prompt safety, and
websocket semantics.
The browser should only talk to the local Dreamverse process, never directly to
Modal or Runpod.
## Current runtime responsibilities
The runtime in `apps/dreamverse/dreamverse/` is responsible for:
- websocket session lifecycle on `/ws`
- queueing, GPU assignment, worker startup, and stream chunk emission
- seed prompt memory and the active prompt window used for generation
- prompt enhancement and prompt rewrite execution
- prompt safety checks
- persistence and reload of prompt system prompt files
- curated preset append/read routes in devtools mode
- health and readiness endpoints
Relevant files:
- `apps/dreamverse/dreamverse/main.py`: websocket protocol, session state
machine, REST routes
- `apps/dreamverse/dreamverse/gpu_pool.py`: FastVideo-backed generation
workers
- `apps/dreamverse/dreamverse/prompt_enhancer.py`: provider clients, prompt
enhancement, rewrite execution
- `apps/dreamverse/dreamverse/rewrite_prompt_payload.py`: canonical rewrite
request body format
- `apps/dreamverse/dreamverse/config.py`: prompt file paths, provider
configuration, runtime flags
## Frontend responsibilities
The frontend in `apps/dreamverse/web/` is responsible for:
- collecting user input and deciding whether to send raw prompts or rewrite
requests
- maintaining client-side stores for session, prompt-window, stream, rewrite,
and UI state
- rendering prompt history, playback state, devtools controls, and rewrite
inspection
- building the prompt-window snapshot sent with rewrite requests
- reducing websocket events into UI state
- showing compute status and controller-driven errors once the controller lands
Relevant files:
- `apps/dreamverse/web/src/app/page.tsx`: main orchestration, websocket
connect/send paths
- `apps/dreamverse/web/src/lib/ws/reducer.ts`: applies normalized websocket
events to stores
- `apps/dreamverse/web/src/stores/promptWindow.ts`: prompt window and
preset/editor state
- `apps/dreamverse/web/src/stores/rewrite.ts`: rewrite activity timeline and
flags
- `apps/dreamverse/web/src/lib/prompts/promptWindowSnapshot.ts`: rewrite snapshot
normalization and padding
## Planned controller responsibilities
The future local controller should own:
- local-only provider credential loading and storage
- provider selection
- runtime provisioning, reuse, shutdown, and health checks
- proxying frontend HTTP and websocket traffic to the active runtime
- surfacing provisioning, ready, failed, and idle states to the frontend
- durable local settings that should survive ephemeral remote runtimes
The controller should not own:
- prompt rewrite logic
- seed prompt memory
- generation queue semantics
- websocket event schemas
## Prompt rewrite contract
Prompt rewrite is a shared flow with a strict ownership split.
Frontend responsibilities:
- decide when a user action should trigger `rewrite_seed_prompts` instead of
`append_prompt`
- send `rewrite_instruction` and a snapshot of the current prompt window
- pad the rewrite snapshot to the runtime-expected segment count using
`buildRewritePromptWindowSnapshotFromPrompts(...)`
- show rewrite activity and raw LLM output in local inspection UI
Runtime responsibilities:
- validate and normalize `prompt_window_prompts`
- choose the rewrite system prompt and provider/model/temperature
- build the canonical LLM request body in
`apps/dreamverse/dreamverse/rewrite_prompt_payload.py`
- run the rewrite through `PromptEnhancer.rewrite_prompt_sequence(...)`
- apply safety filtering to rewritten prompts
- replace the authoritative seed prompt memory when rewrite succeeds
- emit `seed_prompts_updated` and `rewrite_seed_prompts_complete`
Controller responsibilities:
- proxy the request and response
- surface runtime availability and provider lifecycle failures
Important rule:
- The frontend may suggest the prompt window to rewrite, but the runtime owns
the actual rewritten rollout and the authoritative prompt window after
acceptance.
## Rewrite modes
There are two runtime rewrite modes:
- edit existing rollout: when `prompt_window_prompts` is non-empty, rewrite the
current rollout while preserving segment count and ordering
- new rollout: when the prompt window is empty but there is a
`rewrite_instruction`, generate a fresh rollout
The frontend should not emulate runtime rewrite behavior locally. It should
prepare the snapshot, send it, and display the result.
## Prompt window ownership
- Frontend owns editable drafts, selected preset UI, and prompt-window
inspection state.
- Runtime owns the active seed prompt memory used for actual generation.
- After any runtime event with reason `rewrite`, the frontend must replace its
prompt window from the server payload instead of keeping a locally-derived
version.
## Devtools and persistence ownership
Prompt config editing is runtime-owned persistence with frontend-owned forms
today.
- Frontend loads and edits drafts through `/prompt-system-config`.
- Runtime reads and writes prompt files and reloads runtime prompt config.
Curated presets follow the same pattern:
- frontend submits append requests and may update local UI optimistically from
the response
- runtime persists the JSON file and resolves overlay vs fallback file paths
During the controller reorg, avoid moving durable user settings into ephemeral
remote runtimes. Controller-owned local persistence is preferred for anything
that must survive provider restarts.
## Editing guidance
- Do not move rewrite logic into the frontend or controller.
- Do not make the frontend the source of truth for the generated prompt window
after rewrite.
- If you change websocket message types or payload fields in
`apps/dreamverse/dreamverse/main.py`, update the reducer in
`apps/dreamverse/web/src/lib/ws/reducer.ts` in the same change.
- If you change rewrite request shape, update both
`apps/dreamverse/web/src/lib/prompts/promptWindowSnapshot.ts` and
`apps/dreamverse/dreamverse/rewrite_prompt_payload.py`.
- If you add controller-managed status or error payloads, keep them separate
from runtime websocket events unless there is a strong reason to merge them.
- If you change prompt file paths or devtools persistence, update
`apps/dreamverse/dreamverse/`, the frontend devtools UI, and any
controller-owned local persistence logic together.
- Keep provider adapters focused on runtime lifecycle and reachability, not on
prompt or session semantics.
-294
View File
@@ -1,294 +0,0 @@
# Dreamverse
Dreamverse is the FastVideo realtime video generation & editing platform. It lives in this monorepo under `apps/dreamverse/`.
**Deploy on:** [local GPU](#quick-start-local-gpu) · [self-hosted B200 (SSH)](#server-b200-deployment-ssh) · [Docker](docker/README.md) · [Modal](scripts/modal/README.md)
## Install Dreamverse
You can install Dreamverse using one of the methods below.
### Method 1: With uv pip
```bash
pip install --upgrade pip
pip install uv
uv venv .venv --python 3.12
source .venv/bin/activate
UV_TORCH_BACKEND=cu126 uv pip install "fastvideo[dreamverse]"
```
### Method 2: From source
```bash
git clone https://github.com/hao-ai-lab/FastVideo.git
cd FastVideo
pip install --upgrade pip
pip install uv
uv venv .venv --python 3.12
source .venv/bin/activate
UV_TORCH_BACKEND=cu126 uv pip install -e ".[dreamverse]"
```
Use `UV_TORCH_BACKEND=cu130` instead on CUDA 13.
### Method 3: Using Docker
```bash
git clone https://github.com/hao-ai-lab/FastVideo.git
cd FastVideo
apps/dreamverse/docker/docker_build.sh
```
See `apps/dreamverse/docker/README.md` for Docker build and run option details.
## Optional: Building FFmpeg For Better Performance
For full streaming performance in a non-Docker install, build a custom FFmpeg
binary from a FastVideo source checkout. The command below is repo-relative,
so run it from the repository root:
```bash
bash apps/dreamverse/scripts/install_native_ffmpeg.sh
```
The installer supports Linux `x86_64` and `aarch64`. It prefers conda-forge
triplet compilers when those commands are on `PATH`, otherwise it falls back to
system `gcc`/`g++` (plain venv). On `x86_64`, x264's hand-tuned SIMD also
requires `nasm`; install via whichever path fits your host:
```bash
sudo apt install nasm # Debian/Ubuntu
conda install -c conda-forge nasm # inside an active conda env
```
No sudo and no conda? Build `nasm` from source (~30s, installs into `$HOME`):
```bash
(
mkdir -p "$HOME/src" "$HOME/opt" && cd "$HOME/src"
curl -fsSL -O https://www.nasm.us/pub/nasm/releasebuilds/2.16.03/nasm-2.16.03.tar.gz
tar -xf nasm-2.16.03.tar.gz && cd nasm-2.16.03
./configure --prefix="$HOME/opt/nasm" && make -j"$(nproc)" && make install
)
export PATH="$HOME/opt/nasm/bin:$PATH" # add to ~/.bashrc to persist
```
The installer writes to `~/opt/ffmpeg-native/` and emits
`apps/dreamverse/scripts/ffmpeg-env.sh`. Source it before starting the backend
so Dreamverse uses the custom FFmpeg binary:
```bash
source apps/dreamverse/scripts/ffmpeg-env.sh
dreamverse-server
```
Docker images already run this FFmpeg build during image creation and source the
generated environment file at container startup.
## Launch Dreamverse
Start the backend with the installed Dreamverse commands:
```bash
dreamverse-server --port 8009
dreamverse-mock-server --port 8009
```
> **Expect a slow first boot.** With `torch.compile` and startup warmup enabled
> (the default), the backend compiles the segment 1 and segment 2 inference
> paths before it reports ready — this can take **tens of minutes on a cold
> cache**, regardless of how you deploy (local, server, Docker, or Modal).
> `/healthz` responds as soon as the process is up; `/readyz` stays `503` until
> warmup finishes. For a faster, uncompiled startup while testing, set
> `FASTVIDEO_ENABLE_STARTUP_WARMUP=0` before starting the backend.
## Frontend Setup
Install the web dependencies once from the FastVideo checkout:
```bash
cd apps/dreamverse/web
npm ci
```
The frontend package uses `package-lock.json`; use npm for installs and scripts.
## Quick Start: Local GPU
### Start Backend
Export the API keys used for prompt rewrite and prompt enhancement:
```bash
export CEREBRAS_API_KEY=...
export GROQ_API_KEY=...
```
If you built the optional native FFmpeg binary above, source its environment
file in the same shell before starting the backend:
```bash
source apps/dreamverse/scripts/ffmpeg-env.sh
dreamverse-server --host 0.0.0.0 --port 8009
```
The Dreamverse backend defaults to `0.0.0.0:8009` and starts one GPU worker on
the first visible GPU by default.
### Check Readiness
In another shell, verify that the backend process is alive:
```bash
curl http://localhost:8009/healthz
```
Then wait for GPU workers and startup warmup to finish:
```bash
curl http://localhost:8009/readyz
```
You can also run the same readiness path with:
```bash
BACKEND_HOST=localhost BACKEND_PORT=8009 apps/dreamverse/scripts/smoke_local.sh
```
If a backend is already running and you only want the script to probe it:
```bash
DREAMVERSE_SMOKE_START_BACKEND=0 apps/dreamverse/scripts/smoke_local.sh
```
### Start Frontend
Start the frontend:
```bash
cd apps/dreamverse/web
BACKEND_HOST=localhost BACKEND_PORT=8009 npm run dev
```
Open `http://localhost:5299`.
## Server B200 deployment (SSH)
Deploying on a remote GPU host (for example a B200 box) is a local install run
over SSH, plus a few server-specific concerns. Two paths:
### Option A: Native (source install)
SSH in, then follow [Install → From source](#method-2-from-source) and
(recommended) [Building FFmpeg](#optional-building-ffmpeg-for-better-performance),
then start the backend as in [Quick Start: Local GPU](#quick-start-local-gpu).
For a remote host, a few things differ from localhost:
- Bind all interfaces: `dreamverse-server --host 0.0.0.0 --port 8009`.
- Point the frontend/client at the host: `BACKEND_HOST=<b200-host> BACKEND_PORT=8009 npm run dev`.
- Keep the backend alive across SSH sessions (`tmux` / `systemd` / `nohup`).
- Expose / firewall port `8009`, or front it with a reverse proxy + auth.
### Option B: Docker (on the server)
SSH in, then follow [Install → Using Docker](#method-3-using-docker) and the run
steps in [`docker/README.md`](docker/README.md):
```bash
CEREBRAS_API_KEY="<key>" GROQ_API_KEY="<key>" apps/dreamverse/docker/docker_run.sh
```
## Quick Start: Mock Backend (For UI development)
The mock server emulates the Dreamverse backend protocol and streams a
synthetic FFmpeg-generated fMP4 clip, so the frontend can run without a GPU.
```bash
dreamverse-mock-server --latency 200 --port 8009
```
## Tests
Run the focused backend tests that validate local startup wiring, config, GPU
selection, and mock-server behavior:
```bash
pytest apps/dreamverse/dreamverse/tests/test_config.py \
apps/dreamverse/dreamverse/tests/test_entrypoints.py \
apps/dreamverse/dreamverse/tests/test_gpu_pool.py \
apps/dreamverse/dreamverse/tests/test_mock_server.py -q
```
Run the broader Dreamverse backend suite:
```bash
pytest apps/dreamverse/dreamverse/tests -q
```
Run the frontend tests:
```bash
cd apps/dreamverse/web
npm test
```
Run the frontend e2e tests:
```bash
cd apps/dreamverse/web
npm run e2e
```
## Troubleshooting
`dreamverse-server` exits with an install hint
- install the Dreamverse extra with `UV_TORCH_BACKEND=cu126 uv pip install -e ".[dreamverse]"` from a
source checkout, or `UV_TORCH_BACKEND=cu126 uv pip install "fastvideo[dreamverse]"` from PyPI
(`cu130` on CUDA 13).
Prompt-provider environment variable errors
- set `CEREBRAS_API_KEY`
- set `GROQ_API_KEY`
- direct `dreamverse-server` launches do not source `~/.env`; export the keys
in the shell or use the bundled launch scripts, which source `~/.env`.
`/readyz` stays at `503`
- wait for model loading and startup warmup to finish
- confirm a compatible CUDA GPU is visible to the process
- check backend logs for worker startup or warmup failures
- for a startup/debug pass without warmup, set
`FASTVIDEO_ENABLE_STARTUP_WARMUP=0` before starting the backend
Only one GPU is used
- this is the default local behavior
- set `FASTVIDEO_GPU_COUNT=<N>` to start N GPU worker subprocesses inside one
backend instance
- set `FASTVIDEO_GPU_COUNT=all` to start one worker for every visible GPU
- use `CUDA_VISIBLE_DEVICES` first if you need to pin the visible GPU set
Frontend cannot connect to backend
- confirm the backend is running on `8009`; if not, point the frontend at it
with `BACKEND_HOST=<host> BACKEND_PORT=<port> npm run dev`
- confirm `http://localhost:8009/healthz` responds before starting the frontend
- confirm `http://localhost:8009/readyz` returns `200` before clicking Generate
- use `apps/dreamverse/scripts/smoke_local.sh` for a repeatable local startup
check
Mock backend fails during startup
- install FFmpeg or set `FASTVIDEO_FFMPEG_BIN` to an FFmpeg binary
- for non-mock local GPU streaming performance, use the native FFmpeg installer
described above
## Notes
Dreamverse owns its backend app under `apps/dreamverse/dreamverse/`. It expects
`dreamverse-server`, not `fastvideo serve`.
-354
View File
@@ -1,354 +0,0 @@
# Dreamverse Architecture
## Overview
Dreamverse currently has two main runtime pieces:
- `apps/dreamverse/web/`: Next.js frontend
- `apps/dreamverse/dreamverse/`: Python FastAPI runtime
Today, the browser talks directly to the Dreamverse runtime over HTTP and a
single websocket on `/ws`. The frontend owns UI state and interaction flow. The
server owns generation state, prompt rewrite, prompt safety, websocket session
semantics, and GPU-backed execution.
Near-term OSS note:
- `apps/dreamverse/dreamverse/` is the current runtime implementation.
- A future `controller/` layer is planned for local-only compute management and
provider orchestration, but it does not exist yet.
## Repo Map
### Frontend
- `apps/dreamverse/web/src/app/page.tsx`: main client orchestration,
websocket connect, init payloads, send paths, and top-level app behavior
- `apps/dreamverse/web/src/lib/ws/reducer.ts`: reduces normalized websocket
events into client stores
- `apps/dreamverse/web/src/stores/session.ts`: connection, mode, and top-level
session UI
- `apps/dreamverse/web/src/stores/promptWindow.ts`: editable prompt window and
seed prompt UI state
- `apps/dreamverse/web/src/stores/rewrite.ts`: rewrite activity timeline and
inspection state
- `apps/dreamverse/web/src/stores/stream.ts`: playback and stream-related
client state
- `apps/dreamverse/web/src/lib/prompts/promptWindowSnapshot.ts`: prompt-window snapshot
building for rewrite requests
### Server
- `apps/dreamverse/dreamverse/main.py`: websocket endpoint, request handling,
session state machine, rewrite orchestration, REST routes, and stream relay
- `apps/dreamverse/dreamverse/gpu_pool.py`: GPU worker processes, warmup, model
loading, and `generate_video()` calls through FastVideo
- `apps/dreamverse/dreamverse/prompt_enhancer.py`: prompt enhancement, rollout
rewrite execution, provider selection, and timeout/fallback behavior
- `apps/dreamverse/dreamverse/rewrite_prompt_payload.py`: canonical rewrite request payload
building
- `apps/dreamverse/dreamverse/config.py`: runtime flags, prompt file paths, provider settings, and
warmup config
- `apps/dreamverse/dreamverse/session_init_image.py`: validates and persists uploaded initial
images for segment 1
## Current Split Of Responsibility
### Frontend owns
- local UI state and client-side stores
- prompt drafts and prompt window editing
- websocket connection management
- deciding which user action to send:
- `session_init_v2`
- `project_init_v1`
- `append_prompt`
- `rewrite_seed_prompts`
- `simple_generate`
- showing rewrite progress, stream status, prompt history, and devtools views
### Server owns
- websocket session lifecycle and protocol
- GPU assignment and worker lifecycle
- the authoritative seed prompt memory used for generation
- prompt rewrite execution and prompt safety
- actual generation queue semantics
- stream chunk emission and segment lifecycle events
- prompt config and preset persistence routes
- health and readiness endpoints
Important rule:
- The frontend may propose prompt-window state for rewrite, but the server is
the source of truth for the rewritten rollout and the active prompt memory
used for generation.
## End-To-End Flow
1. The frontend opens `/ws`.
2. The frontend sends `session_init_v2` with the initial prompt-window state,
preset metadata, and current toggles.
3. The server validates init data, persists an optional initial image, acquires
a GPU slot, and emits session status such as `gpu_assigned`.
4. The server starts or resumes project generation and emits events like
`ltx2_stream_start`, `ltx2_segment_start`, media init/chunks, and completion
events.
5. The frontend reduces those websocket events into its stores and updates the
UI.
6. User actions such as appending prompts, rewriting seed prompts, or starting
a single custom clip go back to the server over the same websocket.
## Frontend Architecture
The frontend is store-driven.
- `page.tsx` wires together websocket setup, send helpers, reducer
application, and top-level interaction flows.
- Store modules separate concerns like session state, rewrite state, prompt
window state, and stream state.
- The websocket reducer is responsible for turning normalized runtime events
into store updates. If the server event schema changes, the reducer must
change with it.
The frontend is intentionally not responsible for:
- generating rewritten prompts locally
- deciding final prompt safety outcomes
- reconstructing server session state from scratch
- inventing its own generation semantics independent of the runtime
## Server Architecture
The current runtime is a FastAPI app with a single long-lived websocket per
session.
`apps/dreamverse/dreamverse/main.py` manages:
- websocket connect/init
- prompt queues
- project and segment state
- prompt enhancement/rewrite triggers
- stream relay from GPU workers to the browser
- session logging and REST endpoints
`apps/dreamverse/dreamverse/gpu_pool.py` manages:
- model loading through FastVideo
- one or more worker processes
- startup warmup
- user join/leave commands
- `USER_STEP` execution for each segment
- continuation state between segments
`apps/dreamverse/dreamverse/prompt_enhancer.py` manages:
- prompt enhancement for user-submitted prompts
- rollout rewrite requests for the prompt window
- provider selection and fallback across configured prompt providers
- response normalization and safety-aware failure handling
## FastAPI Surface
The server is a single FastAPI application created in
`apps/dreamverse/dreamverse/main.py`.
Current built-in FastAPI docs are enabled:
- `/docs`: Swagger UI
- `/redoc`: ReDoc
- `/openapi.json`: OpenAPI schema
The runtime also mounts the frontend static build at `/` when one of the
configured frontend static directories exists. It does not expose the backend
Python package as static content.
## HTTP API
The current HTTP API is small. Most realtime behavior still goes through the
websocket.
### Core health and status routes
- `GET /healthz`
- process liveness probe
- returns a small payload with `status`, `service`, and timestamp
- `GET /readyz`
- readiness probe
- returns `503` until prompt services are initialized and at least one GPU
worker is ready
- returns readiness and GPU pool summary fields such as ready workers, total
GPUs, warmup counts, and queue size
- `GET /status`
- returns the current GPU pool status payload from `gpu_pool`
- `GET /internal/monitor/sessions`
- internal monitoring payload for session dashboards
- includes pending session count, max available sessions, prompt provider
success counts, and timestamp
### Prompt config routes
- `GET /prompt-system-config`
- returns the editable prompt-system configuration currently loaded by
`PromptEnhancer`
- `POST /prompt-system-config`
- saves prompt-system configuration to disk and reloads prompt config in the
runtime
- current editable fields include:
- next-segment system prompt
- auto-extension system prompt
- rewrite-window system prompt
- rewrite-user system prompt
- rewrite model
- rewrite temperature
### Devtools-only preset routes
These exist only when `DEVTOOLS_ENABLED` is true in
`apps/dreamverse/dreamverse/config.py`.
- `GET /curated-presets`
- returns merged curated presets, applying the local overlay file on top of
the fallback file when both exist
- `POST /curated-presets/append`
- appends a new curated preset to the overlay presets file
- validates non-empty label, normalized id, and at least two non-empty
segment prompts
## Websocket API
`WS /ws` is the main runtime API.
The websocket owns:
- session init
- project init and reset
- prompt append
- prompt rewrite
- generation toggles
- segment lifecycle events
- media stream delivery
- runtime error delivery
The websocket is the authoritative API for realtime Dreamverse behavior. The
HTTP routes mainly support health checks, devtools persistence, and monitoring.
## Prompt Rewrite Architecture
Prompt rewrite is a shared flow with strict ownership boundaries.
### Frontend responsibilities
- collect the rewrite instruction
- build the prompt-window snapshot from current client state
- send `rewrite_seed_prompts`
- show rewrite progress, raw output, fallback state, and resulting prompt list
### Server responsibilities
- validate and normalize the prompt-window payload
- choose rewrite model, system prompt, timeout, and temperature
- build the canonical prompt payload in
`apps/dreamverse/dreamverse/rewrite_prompt_payload.py`
- execute rewrite through `PromptEnhancer`
- apply safety filtering to rewritten prompts
- replace the authoritative seed prompt memory when rewrite succeeds
- emit `seed_prompts_updated` and `rewrite_seed_prompts_complete`
Important rule:
- The frontend owns editable drafts.
- The server owns the accepted rollout.
After a successful rewrite, the frontend should replace its prompt-window view
from the server payload instead of preserving a locally-derived version.
## Prompt Modes
There are three related prompt paths in the current system:
### Initial rollout
- The frontend sends seed prompts during `session_init_v2`.
- The server uses those prompts as the initial seed prompt memory.
- If the rollout starts from an empty prompt window plus an initial rewrite
instruction, the server can pause generation until rewrite completes.
### Live append
- The frontend sends `append_prompt`.
- The server may enhance that prompt, safety-check it, enqueue it, and use it
as the next generated segment.
### Rewrite
- The frontend sends `rewrite_seed_prompts`.
- The server rewrites the entire seed prompt window or generates a new rollout,
depending on the payload and current state.
## Initial Image And Segment Handling
The frontend currently sends `initial_image` as part of session init or
`simple_generate`.
The server:
- validates and persists the image
- uses it only for segment 1 when present
- keeps continuation state for later segments in the GPU worker
This means the runtime, not the frontend, decides how segment 1 image
conditioning and later continuation conditioning are applied.
## Websocket Contract
The websocket is the main integration surface between UI and runtime.
Typical incoming messages from the frontend:
- `session_init_v2`
- `project_init_v1`
- `append_prompt`
- `rewrite_seed_prompts`
- `simple_generate`
- `set_enhancement`
- `set_auto_extension`
- `set_loop_generation`
Typical outgoing messages from the server:
- `gpu_assigned`
- `ltx2_stream_start`
- `ltx2_segment_start`
- `segment_prompt_source`
- `prompt_received`
- `prompt_ready`
- `prompt_enhancing`
- `seed_prompts_updated`
- `rewrite_seed_prompts_complete`
- `media_init`
- `media_segment_complete`
- `project_idle`
- `error`
Binary websocket frames carry media chunks for playback.
## Current And Planned Architecture
Current architecture:
- browser -> `apps/dreamverse/web`
- `apps/dreamverse/web` -> `apps/dreamverse/dreamverse/main.py`
- `apps/dreamverse/dreamverse/main.py` ->
`apps/dreamverse/dreamverse/gpu_pool.py`
- `gpu_pool.py` -> FastVideo runtime
Planned architecture:
- browser -> `apps/dreamverse/web`
- `apps/dreamverse/web` -> local `controller/`
- `controller/` -> local or remote Dreamverse runtime
- runtime -> FastVideo runtime
That future controller split should not move prompt rewrite, session state, or
generation semantics out of the runtime.
-600
View File
@@ -1,600 +0,0 @@
# Dreamverse OSS Design
## Overview
Dreamverse should ship as a local-first open source application.
- The browser talks only to a local Dreamverse control plane on the user's
machine.
- Provider credentials stay local to that machine.
- Dreamverse may provision compute on the user's behalf, but Dreamverse does
not host that control path as a service.
This keeps the UX simple without turning Dreamverse into a credential-holding
hosted platform.
## Goals
- Support three compute modes behind one product surface:
- local GPU
- managed remote GPU via Runpod
- managed remote GPU via Modal
- Keep prompt rewrite, websocket session state, and generation behavior
consistent across providers.
- Keep provider API keys out of browser state and out of any hosted service.
- Make the existing runtime reusable as the common serving contract.
- Minimize provider-specific code and isolate it behind a narrow interface.
## Non-goals
- Do not make the frontend call provider APIs directly.
- Do not unify providers at the level of SSH, VM, serverless, or pod
semantics.
- Do not move prompt rewrite logic into the frontend or controller.
- Do not require remote compute for the basic product path.
## Current State
Today the repo contains two major pieces:
- `apps/dreamverse/web/`: Next.js frontend
- `apps/dreamverse/dreamverse/`: FastAPI runtime that owns websocket state,
prompt rewrite, prompt safety, and GPU-backed generation
The current runtime already exposes useful health and streaming surfaces such
as `/healthz`, `/readyz`, `/status`, and `/ws`.
## Target Architecture
The target open source structure should be:
```text
Dreamverse/
├── apps/dreamverse/
│ ├── web/ # browser UI
│ ├── dreamverse/ # current FastAPI websocket/generation runtime
│ ├── controller/ # local control plane and provider lifecycle
│ ├── providers/ # provider adapters
│ ├── tests/
│ │ ├── contract/
│ │ ├── controller/
│ │ └── smoke/
│ └── design.md
└── ...
```
Near-term note:
- `apps/dreamverse/dreamverse/` is the current runtime implementation.
- We can keep the code there initially and rename it to `runtime/` only after
the controller lands.
## Trust Model
Dreamverse is local-only for control and secrets.
- The user launches Dreamverse on their own machine.
- Provider API keys are entered into the local app or local CLI.
- The controller uses those credentials to provision or connect to compute.
- The browser never talks to Modal or Runpod directly.
- Dreamverse-hosted infrastructure is not involved.
This is the key reason the provider-based path is acceptable for OSS.
## Responsibility Split
### `apps/dreamverse/web`
The frontend should own:
- UI state, drafts, and local interaction state
- websocket event reduction into client stores
- selection of compute mode and display of cost/health/status
- local forms for provider configuration
- sending prompt requests and rewrite requests to the local controller
The frontend should not own:
- provider credentials after submission
- provider API calls
- runtime lifecycle
- authoritative prompt window after rewrite
- prompt safety or generation policy
### `controller`
The local controller should own:
- provider credential loading and local-only storage
- compute mode selection
- provisioning, reuse, shutdown, and health monitoring of runtimes
- reverse proxying HTTP and websocket traffic from the frontend to the active
runtime
- user-visible status such as provisioning, ready, failed, and idle shutdown
- local persistence for user settings that must survive ephemeral runtimes
The controller should not own:
- prompt rewrite logic
- seed prompt memory semantics
- generation queue behavior
- provider-specific UI state
### `runtime`
The runtime should remain the authoritative owner of:
- `/ws` session state
- prompt rewrite execution
- prompt safety
- seed prompt memory and prompt-window state used for generation
- generation orchestration and GPU worker lifecycle
- websocket event schemas
This preserves the current model and avoids splitting state across layers.
## Runtime Contract
Provider abstraction should happen around a stable Dreamverse runtime contract,
not around infrastructure details.
Minimum runtime surface:
- `GET /healthz`
- `GET /readyz`
- `GET /status`
- `GET/POST /prompt-system-config` if devtools persists config through the
runtime
- curated preset routes if those remain runtime-backed
- `WS /ws`
Important rule:
- The controller only needs to know how to reach a healthy runtime.
- The runtime remains provider-agnostic.
## Provider Abstraction
Use a narrow provider interface:
```python
class ComputeProvider(Protocol):
async def ensure_runtime(self, spec: RuntimeSpec) -> RuntimeHandle: ...
async def wait_until_ready(self, handle: RuntimeHandle) -> None: ...
async def stop_runtime(self, handle: RuntimeHandle) -> None: ...
```
`RuntimeHandle` should include:
- `provider`
- `runtime_id`
- `base_url`
- `ws_url`
- runtime auth headers or tokens if needed
- lifecycle metadata
- cost or hardware metadata for UI display
The controller should work only with `RuntimeHandle`, never with raw SSH hosts
or provider-specific payloads after resolution.
## Provider Notes
### Local
Local mode should be the reference implementation.
- Start the runtime as a local subprocess or connect to an already-running
local runtime URL.
- Reuse the same runtime contract as remote providers.
- Make this the first supported path and the main smoke-test target.
### Runpod
Runpod should be treated as pod lifecycle plus runtime reachability.
- Prefer prepared images or templates that auto-start the Dreamverse runtime.
- Prefer exposed HTTP/TCP ports for steady-state traffic.
- Use SSH only for bootstrap fallback, diagnostics, or repair.
- Avoid a design where the controller shells into the pod for every action.
### Modal
Modal should be treated as deployment-based runtime hosting.
- Wrap the Dreamverse runtime in a thin Modal entrypoint if needed.
- Reuse the same runtime behavior behind that wrapper.
- Do not model Modal as a machine that Dreamverse logs into.
- Do not force the websocket runtime into a per-request serverless handler
shape.
## Config and Persistence
Remote compute may be ephemeral, so mutable user configuration should not live
only inside remote runtimes.
Keep durable state local to the user's machine unless there is a strong reason
otherwise:
- provider selection
- provider credentials or credential references
- default hardware preferences
- editable prompt presets
- prompt system prompt overrides
- idle shutdown policy
Runtime-local state should be treated as disposable unless explicitly synced.
## Prompt Rewrite Ownership
Prompt rewrite remains runtime-owned even after the controller is added.
Frontend responsibilities:
- collect the rewrite instruction
- build the prompt-window snapshot
- display rewrite activity and results
Runtime responsibilities:
- validate and normalize the prompt window
- choose the rewrite system prompt and model settings
- execute rewrite
- apply safety filtering
- replace authoritative seed prompt memory
- emit the canonical completion events
Controller responsibilities:
- proxy the request and response
- surface runtime availability and failure state
This boundary should not move.
## Recommended Rollout
1. Finish the path reorg so docs and code agree on `apps/dreamverse/web`.
2. Introduce `controller/` as a local-only API/proxy process.
3. Keep `apps/dreamverse/dreamverse/` as the runtime and adapt it behind the
controller.
4. Add `local` provider first.
5. Add "bring your own runtime URL" as an escape hatch.
6. Add automated Runpod provisioning.
7. Add Modal deployment support.
8. Rename `apps/dreamverse/dreamverse/` to `runtime/` once the split is stable.
## Implementation Plan
The implementation should start with the smallest milestone that gives users a
working local GPU setup without forcing the controller/provider architecture
into the first patch series.
### Milestone 0: Make local GPU the official baseline
Goal:
- A user with a working `fastvideo` install can run the Dreamverse backend on a
local GPU and connect to it from `apps/dreamverse/web`.
Non-goals for this milestone:
- no controller process yet
- no provider abstraction yet
- no Runpod or Modal support yet
- no secret-management UI yet
Reasoning:
- `apps/dreamverse/dreamverse/` already is the real local GPU runtime.
- `apps/dreamverse/web` already knows how to talk to a backend over `/ws` and
REST rewrites.
- The shortest path is to make the existing local path explicit, reliable, and
tested before adding another layer.
### Milestone 0 work items
#### 0.1 Fix repo path assumptions after the frontend move
Current issue:
- Some paths still assume `prod-ui/`, but the frontend now lives at
`apps/dreamverse/web/`.
Required changes:
- update prompt/preset path resolution in
`apps/dreamverse/dreamverse/config.py`
- update docs that still mention `prod-ui`
- audit any frontend build settings that assume the old repo root
This is prerequisite cleanup. Local GPU mode should not depend on stale
monorepo paths.
#### 0.2 Make local runtime startup the primary supported entrypoint
Required outcome:
- one documented backend command
- one documented frontend command
- one clear env contract for local development
Expected shape:
```bash
uv pip install -e ".[dreamverse]"
dreamverse-server --host 0.0.0.0 --port 8009
cd apps/dreamverse/web
npm ci
BACKEND_HOST=localhost BACKEND_PORT=8009 npm run dev
```
Optional but useful:
- add a small root helper script or Make target for local startup
- add a `dreamverse-doctor` or lightweight startup check later
#### 0.3 Define the minimum local runtime contract
For Milestone 0, the frontend should rely only on the current runtime surface:
- `/ws`
- `/status`
- `/healthz`
- `/readyz`
- existing prompt/devtools routes
Do not add a second local API layer yet unless the current runtime surface is
proven insufficient.
#### 0.4 Make failure states explicit in the UI
Local GPU mode fails in a few predictable ways:
- backend not reachable
- backend reachable but not ready
- `fastvideo` or model runtime missing
- no compatible GPU available
Minimum implementation:
- show a clear connection error when `/ws` or `/status` fails
- surface readiness failures in a human-readable way
- avoid silent retry loops that hide backend startup failures
This is a small UI pass, not a controller project.
#### 0.5 Add a minimal local smoke test path
At this milestone, local GPU support is "done" only if there is a repeatable
test path for the local runtime contract.
Minimum test additions:
- backend tests for `/healthz`, `/readyz`, and `/status`
- a frontend integration test that assumes a reachable backend URL and verifies
connection lifecycle behavior
- one local smoke script that starts the backend and verifies readiness before
the frontend is launched
### Milestone 1: Introduce a thin local controller
Goal:
- Preserve the same local GPU behavior, but place a stable local control-plane
API in front of the runtime.
This should happen only after Milestone 0 is stable.
Scope:
- add `controller/`
- proxy `/ws` and the needed REST routes to `apps/dreamverse/dreamverse/`
- expose controller-owned status for "backend starting", "runtime ready", and
"runtime failed"
- optionally spawn the local runtime as a subprocess
Non-goal:
- do not add remote provider logic yet
Reasoning:
- the controller earns its complexity only once it stabilizes the local
contract that future providers will share
### Milestone 2: Provider abstraction on top of the controller
Goal:
- Keep the same frontend contract while allowing the controller to resolve a
runtime via `local`, then later `runpod` and `modal`.
At this point:
- define `ComputeProvider`
- implement `providers/local.py`
- move local-runtime subprocess management behind the provider interface
The first provider should be `local`, because it is cheapest to debug and
matches the runtime most closely.
## Minimal Code Change Order
If we want the shortest path to a working local GPU milestone, the change order
should be:
1. Fix `apps/dreamverse/dreamverse/config.py` and any remaining path
assumptions from `prod-ui` to `apps/dreamverse/web`.
2. Update `README.md` to document the real local GPU startup flow.
3. Confirm `apps/dreamverse/web` connects cleanly to the local wrapper-backed
backend.
4. Improve frontend error handling for backend-not-ready and backend-missing
cases.
5. Add a local smoke test and keep existing backend/frontend tests green.
6. Only then introduce `controller/`.
## Test Plan for the Local GPU Milestone
### Backend
Keep the current Python test suite as the base:
- `apps/dreamverse/dreamverse/tests/test_health_endpoints.py`
- `apps/dreamverse/dreamverse/tests/test_mock_server.py`
- `apps/dreamverse/dreamverse/tests/test_prompt_enhancer.py`
- `apps/dreamverse/dreamverse/tests/test_rewrite_prompt_payload.py`
- related config and logging tests
Add or tighten tests for:
- path resolution in `apps/dreamverse/dreamverse/config.py`
- readiness behavior when GPU pool initialization fails
- startup error messaging when `fastvideo` is unavailable
### Frontend
Keep the current Vitest suite as the base:
- websocket reducer tests
- prompt-window snapshot tests
- integration tests under `apps/dreamverse/web/src/app/`
Add or tighten tests for:
- connection failure UX when backend is down
- readiness failure UX when backend returns non-ready status
- backend routing configuration through `BACKEND_HOST` and `BACKEND_PORT`
### Manual smoke path
The first manual smoke checklist should be:
1. start `dreamverse-server`
2. confirm `GET /healthz` returns 200
3. confirm `GET /readyz` returns 200 after warmup
4. start `apps/dreamverse/web`
5. confirm the UI opens and the websocket connects
6. submit a prompt and verify the first generation starts
This checklist should be written down in the README once the milestone is
implemented.
## Test Strategy
The test suite should preserve one rule: provider changes must not be able to
break prompt rewrite, websocket semantics, or runtime behavior silently.
### 1. Runtime unit and integration tests
Keep and expand the current `pytest` coverage in
`apps/dreamverse/dreamverse/tests/test_*.py`.
Focus areas:
- config loading
- prompt rewrite payload normalization
- prompt enhancement and prompt safety
- health/readiness endpoints
- websocket session behavior
- session logging
- mock runtime behavior
These tests should remain provider-agnostic.
### 2. Controller unit tests
Add a Python test suite for the controller state machine.
Key cases:
- provider selection and validation
- credential loading from local config or env
- runtime lifecycle transitions:
- idle
- provisioning
- ready
- failed
- stopping
- idle timeout and cleanup behavior
- retry and backoff behavior
- HTTP and websocket proxy routing
These tests should use fake providers and fake runtimes by default.
### 3. Provider contract tests
Each provider should pass the same contract tests.
Examples:
- `ensure_runtime()` returns a usable `RuntimeHandle`
- `wait_until_ready()` surfaces timeout vs readiness correctly
- `stop_runtime()` is safe to call twice
- provider errors are mapped into stable controller error types
Use recorded fixtures or fakes wherever possible to avoid spend in CI.
### 4. Web contract tests
The frontend already has useful Vitest coverage under
`apps/dreamverse/web/src`.
Preserve that and expand around the controller split.
Priority areas:
- websocket event reduction
- prompt-window snapshot construction
- rewrite request shaping
- compute-status UI
- failure and reconnect UX
The frontend should mock the local controller API, not provider APIs.
### 5. Cross-layer protocol tests
Add contract fixtures that validate shared payloads across layers.
Important fixtures:
- websocket event payloads
- rewrite request payloads
- runtime status payloads
- controller status payloads
These can be simple JSON fixtures validated by both Python and TypeScript
tests. They will catch drift earlier than end-to-end tests.
### 6. Smoke tests
Add a small number of high-signal smoke tests:
- local provider + mock runtime
- local provider + real runtime when GPU is available
- controller startup + frontend health path
These should be cheap enough for routine local use.
### 7. Provider-backed manual or nightly tests
Real Modal and Runpod tests should be opt-in.
- Do not run them in default CI.
- Gate them behind explicit credentials and flags.
- Keep them focused on provisioning and reachability, not full product
regression.
This avoids flaky and expensive CI while still validating real provider flows.
## Testing Recommendations for the Next Step
The next practical additions should be:
1. A controller test suite in Python using fake providers.
2. Shared contract fixtures for websocket and rewrite payloads.
3. A smoke test that starts the local controller against the existing mock
runtime.
I would not add Playwright yet. The current web stack already has Vitest and
integration-style component tests, which are cheaper and better aligned with
the immediate reorg. Add browser automation only after the controller path is
stable.
-103
View File
@@ -1,103 +0,0 @@
# syntax=docker/dockerfile:1.7
# CUDA base. CUDA_VERSION/UBUNTU_VERSION feed the default tag (matches the
# unified docker/Dockerfile); override BUILD_BASE_IMAGE wholesale for a mirror.
ARG CUDA_VERSION=13.0.0
ARG UBUNTU_VERSION=22.04
ARG BUILD_BASE_IMAGE=nvidia/cuda:${CUDA_VERSION}-cudnn-devel-ubuntu${UBUNTU_VERSION}
FROM ${BUILD_BASE_IMAGE}
ARG BUILD_FASTVIDEO_KERNEL_FROM_SOURCE=0
ARG BUILD_DREAMVERSE_UI=0
ENV DEBIAN_FRONTEND=noninteractive \
PYTHONUNBUFFERED=1 \
UV_LINK_MODE=copy \
UV_CACHE_DIR=/opt/uv/cache
# pyproject no longer pins a PyTorch index; GPU-less build host -> pin explicitly
# (auto would fall back to CPU). Matches the base CUDA: cu130 (13.0) / cu126 (12.6).
ARG UV_TORCH_BACKEND=cu130
ENV UV_TORCH_BACKEND=${UV_TORCH_BACKEND}
SHELL ["/bin/bash", "-c"]
RUN apt-get update && apt-get install -y --no-install-recommends \
gcc-11 g++-11 clang-11 \
make cmake ninja-build pkg-config nasm \
git curl wget ca-certificates \
libssl-dev zlib1g-dev \
&& rm -rf /var/lib/apt/lists/* \
&& update-alternatives --install /usr/bin/gcc gcc /usr/bin/gcc-11 100 \
--slave /usr/bin/g++ g++ /usr/bin/g++-11
# Version-agnostic /usr/local/cuda symlink so this works for any CUDA_VERSION.
ENV CUDA_HOME=/usr/local/cuda
ENV PATH=/root/.local/bin:/opt/venv/bin:${CUDA_HOME}/bin:${PATH}
ENV LD_LIBRARY_PATH=${CUDA_HOME}/lib64:${LD_LIBRARY_PATH}
ENV VIRTUAL_ENV=/opt/venv
RUN curl -LsSf https://astral.sh/uv/install.sh | sh
RUN uv venv --python 3.12 --seed /opt/venv \
&& echo 'source /opt/venv/bin/activate' >> /root/.bashrc
WORKDIR /opt/FastVideo
COPY . /opt/FastVideo
RUN --mount=type=cache,target=/opt/uv/cache \
source /opt/venv/bin/activate \
&& uv pip install "/opt/FastVideo[dreamverse]"
# Standard docker build does not expose GPUs, while fastvideo-kernel/build.sh
# detects the CUDA architecture with torch at build time. The FastVideo package
# install above brings in the pinned fastvideo-kernel package; rebuild from the
# copied source only on hosts configured for build-time GPU access.
RUN --mount=type=cache,target=/opt/uv/cache \
if [[ "${BUILD_FASTVIDEO_KERNEL_FROM_SOURCE}" == "1" ]]; then \
source /opt/venv/bin/activate \
&& cd /opt/FastVideo/fastvideo-kernel \
&& ./build.sh; \
else \
echo "Skipping source fastvideo-kernel build; using installed fastvideo-kernel package."; \
fi
RUN if [[ "${BUILD_DREAMVERSE_UI}" == "1" ]]; then \
curl -fsSL https://deb.nodesource.com/setup_22.x | bash - \
&& apt-get install -y --no-install-recommends nodejs \
&& rm -rf /var/lib/apt/lists/* \
&& cd /opt/FastVideo/apps/dreamverse/web \
&& npm ci --ignore-scripts \
&& NEXT_OUTPUT_EXPORT=1 npm run build \
&& rm -rf node_modules .next; \
else \
echo "Skipping Dreamverse UI build; Building backend-only image."; \
fi
# The monorepo ffmpeg installer force-selects conda compiler triplets for
# local dev shells. Inside this image we explicitly opt into the system
# gcc/g++ toolchain.
RUN source /opt/venv/bin/activate \
&& INSTALL_PREFIX=/opt/ffmpeg-native \
SOURCE_DIR=/tmp/ffmpeg-native-src \
FFMPEG_NATIVE_CC=/usr/bin/gcc \
FFMPEG_NATIVE_CXX=/usr/bin/g++ \
bash /opt/FastVideo/apps/dreamverse/scripts/install_native_ffmpeg.sh
# FASTVIDEO_FA4: FA4 (flash_attn.cute) is opt-in; this image installs it via
# the dreamverse extra and is validated with it, so enable it here.
ENV FASTVIDEO_DREAMVERSE_HOME=/var/lib/dreamverse \
STREAM_MODE=av_fmp4 \
FASTVIDEO_ENABLE_PROMPT_SAFETY=0 \
FASTVIDEO_FA4=1 \
HF_HOME=/root/.cache/huggingface
RUN mkdir -p /var/lib/dreamverse
EXPOSE 8009
HEALTHCHECK --interval=30s --timeout=5s --start-period=2700s --retries=3 \
CMD curl -fsS http://127.0.0.1:8009/healthz || exit 1
ENTRYPOINT ["/opt/FastVideo/apps/dreamverse/docker/docker_entrypoint.sh"]
CMD ["dreamverse-server", "--host", "0.0.0.0", "--port", "8009"]
@@ -1,46 +0,0 @@
.git/
**/.git/
.venv/
**/.venv/
.*dreamverse*/
**/__pycache__/
**/*.pyc
**/*.pyo
**/*.egg-info/
.pytest_cache/
**/.pytest_cache/
.mypy_cache/
.ruff_cache/
.cache/
apps/dreamverse/web/node_modules/
apps/dreamverse/web/.next/
apps/dreamverse/web/out/
apps/dreamverse/web/dist/
apps/dreamverse/web/test-results/
apps/dreamverse/web/playwright-report/
apps/dreamverse/outputs/
apps/dreamverse/dreamverse/outputs/
apps/dreamverse/dreamverse/prompts.local/
apps/dreamverse/logs/
outputs/
outputs_video/
quality_check_outputs/
data/
logs/
slurm-logs/
wandb/
.env
.env.*
**/prompts.local/
.codex/
.agents/
.opencode/
.agent_tmp/
.vscode/
.idea/
*.log
*.tmp
*.pdf
-97
View File
@@ -1,97 +0,0 @@
# Dreamverse Docker Image
This folder contains the Docker image for Dreamverse inside the FastVideo
monorepo. Build commands use the FastVideo repository root as the Docker
context, so run the helper scripts from this folder or from any path in the
checkout.
## Build
```bash
apps/dreamverse/docker/docker_build.sh
```
The image defaults to `dreamverse:dev`. Override it with:
```bash
DREAMVERSE_IMAGE=dreamverse:local apps/dreamverse/docker/docker_build.sh
```
Backend-only remains the default image. To include the static Dreamverse UI
served by the backend, set `BUILD_DREAMVERSE_UI=1` and choose a specific image
tag:
```bash
BUILD_DREAMVERSE_UI=1 DREAMVERSE_IMAGE=<image-tag> apps/dreamverse/docker/docker_build.sh
```
Prefer SHA-specific tags for deployable images; avoid `latest`.
The Dockerfile defaults to CUDA 13.0.0 with the cu130 PyTorch backend. CI also
builds a CUDA 12.6.3 / cu126 image. Select that local build explicitly with:
```bash
CUDA_VERSION=12.6.3 apps/dreamverse/docker/docker_build.sh
```
The helper derives `UV_TORCH_BACKEND=cu126` for CUDA 12.x and `cu130` for CUDA
13.x. Set `UV_TORCH_BACKEND` explicitly for a custom CUDA version. The legacy
complete-image-tag override remains supported and selects the same matching
backend:
```bash
CUDA_TAG=12.6.3-cudnn-devel-ubuntu22.04 apps/dreamverse/docker/docker_build.sh
```
Do not set `CUDA_TAG` and `CUDA_VERSION` together. The image installs FastVideo
from this checkout with the `dreamverse` extra, including FA4 flash-attention
and FlashInfer for NVFP4 quantization, and builds native FFmpeg.
FastVideo's pinned `fastvideo-kernel==0.3.2` package is installed by default.
To rebuild `fastvideo-kernel` from this checkout during the image build, set:
```bash
BUILD_FASTVIDEO_KERNEL_FROM_SOURCE=1 apps/dreamverse/docker/docker_build.sh
```
That source build detects the GPU architecture with torch during `docker
build`. On hosts where Docker does not expose GPUs during build, leave the
default package install path enabled.
## Run
```bash
CEREBRAS_API_KEY="<your-key>" \
GROQ_API_KEY="<your-key>" \
apps/dreamverse/docker/docker_run.sh
```
The container serves Dreamverse on host port `8009` by default and mounts:
```text
$HOME/.cache/huggingface -> /root/.cache/huggingface
apps/dreamverse/outputs -> /var/lib/dreamverse/outputs
```
Override the host port and output directory with `BACKEND_PORT` and
`DREAMVERSE_OUTPUTS_DIR`.
To pin the container to a specific host GPU, pass Docker's GPU request syntax:
```bash
DREAMVERSE_DOCKER_GPUS=device=4 FASTVIDEO_GPU_COUNT=1 \
CEREBRAS_API_KEY="<your-key>" \
GROQ_API_KEY="<your-key>" \
apps/dreamverse/docker/docker_run.sh
```
## Smoke
```bash
CEREBRAS_API_KEY=placeholder \
GROQ_API_KEY=placeholder \
apps/dreamverse/docker/docker_smoke.sh
```
The smoke script starts the container, polls `/healthz`, then polls `/readyz`.
It removes the container on exit unless `DREAMVERSE_KEEP_CONTAINER=1` is set.
-88
View File
@@ -1,88 +0,0 @@
#!/usr/bin/env bash
set -euo pipefail
SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" && pwd)"
REPO_ROOT="$(cd -- "${SCRIPT_DIR}/../../.." && pwd)"
IMAGE="${DREAMVERSE_IMAGE:-dreamverse:dev}"
ROOT_DOCKERIGNORE="${REPO_ROOT}/.dockerignore"
DOCKERFILE_DOCKERIGNORE="${SCRIPT_DIR}/Dockerfile.dockerignore"
CREATED_ROOT_DOCKERIGNORE_SYMLINK=0
cleanup_root_dockerignore_symlink() {
if [[ "${CREATED_ROOT_DOCKERIGNORE_SYMLINK}" == "1" && -L "${ROOT_DOCKERIGNORE}" ]] && \
[[ "$(readlink "${ROOT_DOCKERIGNORE}")" == "${DOCKERFILE_DOCKERIGNORE}" ]]; then
rm -- "${ROOT_DOCKERIGNORE}"
fi
}
trap cleanup_root_dockerignore_symlink EXIT INT TERM
if [[ -z "${DOCKER_BUILDKIT:-}" ]] && docker buildx version >/dev/null 2>&1; then
export DOCKER_BUILDKIT=1
fi
build_args=()
cuda_version="${CUDA_VERSION:-}"
torch_backend="${UV_TORCH_BACKEND:-}"
# CUDA_TAG was the Dockerfile's original override and contains the complete
# nvidia/cuda tag (for example, 12.6.3-cudnn-devel-ubuntu22.04). Keep accepting
# it while translating it to the parameterized Dockerfile inputs.
if [[ -n "${CUDA_TAG:-}" ]]; then
if [[ -n "${CUDA_VERSION:-}" ]]; then
printf 'CUDA_TAG and CUDA_VERSION cannot both be set. Use CUDA_VERSION for new builds.\n' >&2
exit 2
fi
cuda_version="${CUDA_TAG%%-*}"
if [[ ! "${cuda_version}" =~ ^[0-9]+\.[0-9]+(\.[0-9]+)?$ ]]; then
printf 'Cannot infer CUDA_VERSION from CUDA_TAG=%s. Use CUDA_VERSION and UV_TORCH_BACKEND instead.\n' \
"${CUDA_TAG}" >&2
exit 2
fi
build_args+=(--build-arg "BUILD_BASE_IMAGE=nvidia/cuda:${CUDA_TAG}")
fi
if [[ -n "${cuda_version}" ]]; then
build_args+=(--build-arg "CUDA_VERSION=${cuda_version}")
fi
if [[ -z "${torch_backend}" && -n "${cuda_version}" ]]; then
case "${cuda_version}" in
12.*) torch_backend=cu126 ;;
13.*) torch_backend=cu130 ;;
*)
printf 'No default UV_TORCH_BACKEND for CUDA_VERSION=%s. Set UV_TORCH_BACKEND explicitly.\n' \
"${cuda_version}" >&2
exit 2
;;
esac
fi
if [[ -n "${torch_backend}" ]]; then
build_args+=(--build-arg "UV_TORCH_BACKEND=${torch_backend}")
fi
[[ -n "${BUILD_FASTVIDEO_KERNEL_FROM_SOURCE:-}" ]] && \
build_args+=(--build-arg "BUILD_FASTVIDEO_KERNEL_FROM_SOURCE=${BUILD_FASTVIDEO_KERNEL_FROM_SOURCE}")
build_args+=(--build-arg "BUILD_DREAMVERSE_UI=${BUILD_DREAMVERSE_UI:-0}")
if [[ "${DOCKER_BUILDKIT:-}" == "1" ]]; then
if [[ -L "${ROOT_DOCKERIGNORE}" ]] && \
[[ "$(readlink "${ROOT_DOCKERIGNORE}")" == "${DOCKERFILE_DOCKERIGNORE}" ]]; then
rm -- "${ROOT_DOCKERIGNORE}"
printf 'Removed stale temporary root .dockerignore symlink: %s\n' "${ROOT_DOCKERIGNORE}"
fi
elif [[ -e "${ROOT_DOCKERIGNORE}" || -L "${ROOT_DOCKERIGNORE}" ]]; then
printf 'Using existing root .dockerignore: %s\n' "${ROOT_DOCKERIGNORE}"
else
ln -s -- "${DOCKERFILE_DOCKERIGNORE}" "${ROOT_DOCKERIGNORE}"
CREATED_ROOT_DOCKERIGNORE_SYMLINK=1
printf 'Created temporary root .dockerignore symlink for legacy Docker builder: %s -> %s\n' \
"${ROOT_DOCKERIGNORE}" "${DOCKERFILE_DOCKERIGNORE}"
fi
docker build \
-f "${SCRIPT_DIR}/Dockerfile" \
-t "${IMAGE}" \
"${build_args[@]}" \
"${REPO_ROOT}"
@@ -1,13 +0,0 @@
#!/usr/bin/env bash
set -euo pipefail
source /opt/venv/bin/activate
if [[ -f /opt/FastVideo/apps/dreamverse/scripts/ffmpeg-env.sh ]]; then
source /opt/FastVideo/apps/dreamverse/scripts/ffmpeg-env.sh
fi
: "${CEREBRAS_API_KEY:?CEREBRAS_API_KEY must be set (pass with -e CEREBRAS_API_KEY=...)}"
: "${GROQ_API_KEY:?GROQ_API_KEY must be set (pass with -e GROQ_API_KEY=...)}"
exec "$@"
-30
View File
@@ -1,30 +0,0 @@
#!/usr/bin/env bash
set -euo pipefail
: "${CEREBRAS_API_KEY:?CEREBRAS_API_KEY not set on host}"
: "${GROQ_API_KEY:?GROQ_API_KEY not set on host}"
SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" && pwd)"
DREAMVERSE_ROOT="$(cd -- "${SCRIPT_DIR}/.." && pwd)"
IMAGE="${DREAMVERSE_IMAGE:-dreamverse:dev}"
PORT="${BACKEND_PORT:-8009}"
HF_CACHE="${HF_HOME:-$HOME/.cache/huggingface}"
OUTPUTS_DIR="${DREAMVERSE_OUTPUTS_DIR:-${DREAMVERSE_ROOT}/outputs}"
GPU_REQUEST="${DREAMVERSE_DOCKER_GPUS:-all}"
mkdir -p "${HF_CACHE}" "${OUTPUTS_DIR}"
env_args=(
-e "CEREBRAS_API_KEY=${CEREBRAS_API_KEY}"
-e "GROQ_API_KEY=${GROQ_API_KEY}"
)
[[ -n "${ENABLE_TORCH_COMPILE:-}" ]] && env_args+=(-e "ENABLE_TORCH_COMPILE=${ENABLE_TORCH_COMPILE}")
[[ -n "${FASTVIDEO_GPU_COUNT:-}" ]] && env_args+=(-e "FASTVIDEO_GPU_COUNT=${FASTVIDEO_GPU_COUNT}")
exec docker run --rm --gpus "${GPU_REQUEST}" --init \
-p "${PORT}:8009" \
"${env_args[@]}" \
-v "${HF_CACHE}:/root/.cache/huggingface" \
-v "${OUTPUTS_DIR}:/var/lib/dreamverse/outputs" \
"${IMAGE}"
-75
View File
@@ -1,75 +0,0 @@
#!/usr/bin/env bash
set -euo pipefail
: "${CEREBRAS_API_KEY:?CEREBRAS_API_KEY not set on host}"
: "${GROQ_API_KEY:?GROQ_API_KEY not set on host}"
SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" && pwd)"
DREAMVERSE_ROOT="$(cd -- "${SCRIPT_DIR}/.." && pwd)"
IMAGE="${DREAMVERSE_IMAGE:-dreamverse:dev}"
PORT="${BACKEND_PORT:-8009}"
HF_CACHE="${HF_HOME:-$HOME/.cache/huggingface}"
OUTPUTS_DIR="${DREAMVERSE_OUTPUTS_DIR:-${DREAMVERSE_ROOT}/outputs}"
NAME="${DREAMVERSE_NAME:-dreamverse}"
TIMEOUT_SECONDS="${DREAMVERSE_SMOKE_TIMEOUT_SECONDS:-1200}"
POLL_SECONDS="${DREAMVERSE_SMOKE_POLL_SECONDS:-5}"
GPU_REQUEST="${DREAMVERSE_DOCKER_GPUS:-all}"
mkdir -p "${HF_CACHE}" "${OUTPUTS_DIR}"
docker rm -f "${NAME}" >/dev/null 2>&1 || true
env_args=(
-e "CEREBRAS_API_KEY=${CEREBRAS_API_KEY}"
-e "GROQ_API_KEY=${GROQ_API_KEY}"
-e "ENABLE_TORCH_COMPILE=${ENABLE_TORCH_COMPILE:-0}"
)
[[ -n "${FASTVIDEO_GPU_COUNT:-}" ]] && env_args+=(-e "FASTVIDEO_GPU_COUNT=${FASTVIDEO_GPU_COUNT}")
container_id="$(
docker run -d --rm --gpus "${GPU_REQUEST}" --init \
-p "${PORT}:8009" \
"${env_args[@]}" \
-v "${HF_CACHE}:/root/.cache/huggingface" \
-v "${OUTPUTS_DIR}:/var/lib/dreamverse/outputs" \
--name "${NAME}" \
"${IMAGE}"
)"
cleanup() {
if [[ "${DREAMVERSE_KEEP_CONTAINER:-0}" != "1" ]]; then
docker rm -f "${NAME}" >/dev/null 2>&1 || true
fi
}
trap cleanup EXIT
wait_for_endpoint() {
local path="$1"
local label="$2"
local deadline=$((SECONDS + TIMEOUT_SECONDS))
local url="http://127.0.0.1:${PORT}${path}"
echo "Waiting for ${label} at ${url}"
while (( SECONDS < deadline )); do
if curl -fsS "${url}" >/dev/null 2>&1; then
echo "${label} ok"
return 0
fi
if ! docker ps --format '{{.Names}}' | grep -qx "${NAME}"; then
echo "Container exited before ${label} became healthy." >&2
docker logs "${container_id}" >&2 || true
return 1
fi
sleep "${POLL_SECONDS}"
done
echo "Timed out waiting for ${label}." >&2
docker logs "${container_id}" >&2 || true
return 1
}
wait_for_endpoint "/healthz" "healthz"
wait_for_endpoint "/readyz" "readyz"
echo "Dreamverse Docker smoke passed for ${IMAGE} on host port ${PORT}."
-15
View File
@@ -1,15 +0,0 @@
from __future__ import annotations
DREAMVERSE_RUNTIME_DEPS_MESSAGE = (
"Dreamverse runtime deps missing — install with pip install 'fastvideo[dreamverse]'.")
def require_dreamverse_runtime_deps() -> None:
try:
import cerebras.cloud.sdk # noqa: F401
import openai # noqa: F401
except ModuleNotFoundError as exc:
missing_root = (exc.name or "").split(".", 1)[0]
if missing_root in {"cerebras", "openai"}:
raise SystemExit(DREAMVERSE_RUNTIME_DEPS_MESSAGE) from exc
raise
-445
View File
@@ -1,445 +0,0 @@
# pyright: reportMissingTypeArgument=false, reportArgumentType=false, reportOptionalSubscript=false, reportOptionalMemberAccess=false, reportConstantRedefinition=false, reportCallIssue=false
# ruff: noqa: UP007, SIM108, SIM105
# mypy: ignore-errors
"""ffmpeg fMP4 muxing with chunk-level event emission.
Self-contained: spawns ffmpeg as a subprocess, pipes raw frames into
its stdin, reads fragmented-MP4 chunks from stdout, and publishes each
chunk as a ``StreamEvent`` via the caller-supplied ``publish``
callback. Knows nothing about multiprocessing queues, the GPU pool,
or individual users — the caller decides what "publish" means.
"""
import fcntl
import os
import shutil
import subprocess
import tempfile
import threading
import time
import uuid
import wave
from dataclasses import dataclass
from typing import Union
from collections.abc import Callable
import numpy as np
import torch
FFMPEG_BIN = shutil.which(os.getenv("FASTVIDEO_FFMPEG_BIN", "ffmpeg"))
AV_MEDIA_MIME = os.getenv(
"STREAM_MIME_TYPE",
'video/mp4; codecs="avc1.42E01E,mp4a.40.2"',
)
AV_CHUNK_SIZE_BYTES = 1048576
TARGET_FPS = 24
AV_FRAGMENT_DURATION_US = int(os.getenv("FASTVIDEO_FRAG_US", "250000"))
X264_GOP_FRAMES = int(os.getenv("FASTVIDEO_X264_GOP", "12"))
X264_PROFILE = os.getenv("FASTVIDEO_X264_PROFILE", "baseline").strip().lower()
if X264_PROFILE not in {"baseline", "main", "high", "main10", "high10"}:
print(f"[WARN] Unsupported FASTVIDEO_X264_PROFILE={X264_PROFILE}; using baseline")
X264_PROFILE = "baseline"
USE_SHARED_STREAM_BUFFER = (os.getenv("FASTVIDEO_USE_SHARED_STREAM_BUFFER", "1").strip().lower()
not in {"0", "false", "no"})
SHARED_STREAM_BUFFER_BYTES = int(os.getenv("FASTVIDEO_SHARED_STREAM_BUFFER_BYTES", str(256 * 1024 * 1024)))
@dataclass
class StreamInit:
"""First event emitted — tells the consumer the stream is starting."""
stream_id: str
mime: str
uses_shared_buffer: bool
@dataclass
class StreamChunk:
"""One fMP4 chunk. Either ``chunk`` (raw bytes) or
``chunk_offset``+``chunk_length`` (read from the shared buffer)
will be populated, never both."""
stream_id: str
chunk: bytes | None = None
chunk_offset: int | None = None
chunk_length: int | None = None
uses_shared_buffer: bool = False
@dataclass
class StreamComplete:
"""Final event emitted — muxing finished successfully."""
stream_id: str
chunks: int
StreamEvent = Union[StreamInit, StreamChunk, StreamComplete]
def generate_stream_id(segment_idx: int) -> str:
"""Convenience: build a stream id of the form ``seg007-abcd1234``."""
return f"seg{segment_idx:03d}-{uuid.uuid4().hex[:8]}"
def _normalize_audio_tensor(audio: object) -> tuple[np.ndarray, int] | None:
"""Convert audio tensor/array into int16 ndarray [samples, channels]."""
if audio is None:
return None
if torch.is_tensor(audio):
audio_np = audio.detach().cpu().float().numpy()
else:
audio_np = np.asarray(audio, dtype=np.float32)
if audio_np.ndim == 1:
audio_np = audio_np[:, None]
elif audio_np.ndim == 2:
if audio_np.shape[0] <= 8 and audio_np.shape[1] > audio_np.shape[0]:
audio_np = audio_np.T
else:
return None
audio_np = np.clip(audio_np, -1.0, 1.0)
audio_int16 = (audio_np * 32767.0).astype(np.int16)
num_channels = audio_int16.shape[1]
return audio_int16, num_channels
def _write_audio_wav(
audio_int16: np.ndarray,
num_channels: int,
sample_rate: int,
) -> str:
"""Write normalized int16 audio to a temporary WAV file."""
with tempfile.NamedTemporaryFile(suffix=".wav", delete=False) as f:
wav_path = f.name
try:
with wave.open(wav_path, "wb") as wav_file:
wav_file.setnchannels(num_channels)
wav_file.setsampwidth(2)
wav_file.setframerate(sample_rate)
wav_file.writeframes(audio_int16.tobytes())
except Exception:
try:
os.unlink(wav_path)
except FileNotFoundError:
pass
raise
return wav_path
def stream_fmp4(
*,
frames: list[np.ndarray],
audio: object,
audio_sample_rate: int | None,
stream_id: str,
timings: dict,
head_trim_frames: int = 0,
head_trim_audio_frames: int | None = None,
shared_buffer=None,
shared_buffer_bytes: int = 0,
publish: Callable[[StreamEvent], None],
log_prefix: str = "",
) -> tuple[bool, str | None]:
"""Encode frames+audio with ffmpeg, publish each fMP4 chunk as an event.
Args:
frames: RGB24 video frames as HxWx3 uint8 arrays.
audio: 1D/2D tensor or ndarray, float values in [-1, 1].
audio_sample_rate: sample rate of ``audio``.
stream_id: caller-supplied identifier carried on every event.
timings: dict mutated in place with ffmpeg/stream timing metrics.
head_trim_frames: video frames to drop from the start
(conditioning overlap).
head_trim_audio_frames: video-frame-equivalent audio to drop.
Defaults to ``head_trim_frames``.
shared_buffer: optional ``mp.RawArray``-compatible object; when
provided, chunks are written into it and emitted by offset
rather than by bytes (avoids IPC copies).
shared_buffer_bytes: size of ``shared_buffer`` in bytes.
publish: callback invoked once per stream event.
log_prefix: prepended to warning prints (e.g. ``"[GPU 0]"``).
Returns:
``(True, None)`` on success, ``(False, error_message)`` on
failure. On mid-stream failure, a ``StreamInit`` may have
already been published — the caller is responsible for
handling that.
"""
if not frames:
return False, "no frames returned"
if audio is None:
return False, "audio is None"
if audio_sample_rate is None:
return False, "audio_sample_rate is None"
if FFMPEG_BIN is None:
return False, "ffmpeg not found"
if head_trim_audio_frames is None:
head_trim_audio_frames = head_trim_frames
normalized_audio = _normalize_audio_tensor(audio)
if normalized_audio is None:
shape_hint = getattr(audio, "shape", None)
return False, f"unsupported audio shape={shape_hint}"
audio_int16, num_channels = normalized_audio
if head_trim_frames < 0:
return False, (f"head_trim_frames must be >= 0, "
f"got {head_trim_frames}")
if head_trim_frames >= len(frames):
return False, (f"head_trim_frames={head_trim_frames} removes "
f"all {len(frames)} frames in segment")
out_frames = (frames[head_trim_frames:] if head_trim_frames > 0 else frames)
sample_rate = int(audio_sample_rate)
if head_trim_audio_frames > 0:
trim_start_samples = int(round((head_trim_audio_frames / float(TARGET_FPS)) * sample_rate))
if trim_start_samples >= audio_int16.shape[0]:
return False, ("audio too short after overlap trim: "
f"trim_start_samples={trim_start_samples}"
f", audio_samples={audio_int16.shape[0]}")
keep_samples = int(round((len(out_frames) / float(TARGET_FPS)) * sample_rate))
trim_end_samples = min(
audio_int16.shape[0],
trim_start_samples + keep_samples,
)
if trim_end_samples <= trim_start_samples:
return False, ("invalid audio trim range: "
f"start={trim_start_samples}, "
f"end={trim_end_samples}")
audio_int16 = audio_int16[trim_start_samples:trim_end_samples]
height = int(out_frames[0].shape[0])
width = int(out_frames[0].shape[1])
codec = os.getenv("FASTVIDEO_VIDEO_CODEC", "libx264")
t_wav_start = time.perf_counter()
wav_path = _write_audio_wav(audio_int16, num_channels, sample_rate)
wav_write_ms = (time.perf_counter() - t_wav_start) * 1000
cmd = [
FFMPEG_BIN,
"-hide_banner",
"-loglevel",
"error",
"-y",
"-f",
"rawvideo",
"-pix_fmt",
"rgb24",
"-s:v",
f"{width}x{height}",
"-r",
str(TARGET_FPS),
"-i",
"pipe:0",
"-i",
wav_path,
"-c:v",
codec,
]
if codec.endswith("_nvenc"):
cmd += [
"-preset",
os.getenv("FASTVIDEO_NVENC_PRESET", "p1"),
"-tune",
os.getenv("FASTVIDEO_NVENC_TUNE", "ull"),
"-rc",
os.getenv("FASTVIDEO_NVENC_RC", "constqp"),
"-qp",
os.getenv("FASTVIDEO_NVENC_QP", "28"),
"-bf",
os.getenv("FASTVIDEO_NVENC_BF", "0"),
]
else:
cmd += [
"-preset",
os.getenv("FASTVIDEO_X264_PRESET", "ultrafast"),
"-tune",
"zerolatency",
"-profile:v",
X264_PROFILE,
# Emit frequent keyframes so fragments are independently playable.
"-g",
str(X264_GOP_FRAMES),
"-keyint_min",
str(X264_GOP_FRAMES),
"-x264-params",
"scenecut=0",
]
cmd += [
"-c:a",
"aac",
"-pix_fmt",
os.getenv("FASTVIDEO_OUTPUT_PIX_FMT", "yuv420p"),
"-shortest",
"-movflags",
"+frag_keyframe+empty_moov+default_base_moof",
"-frag_duration",
str(AV_FRAGMENT_DURATION_US),
"-flush_packets",
"1",
"-muxdelay",
"0",
"-muxpreload",
"0",
"-f",
"mp4",
"pipe:1",
]
proc: subprocess.Popen | None = None
stderr_chunks: list[bytes] = []
writer_error: list[Exception | None] = [None]
t_stream_start = time.perf_counter()
use_shared_buffer = (USE_SHARED_STREAM_BUFFER and shared_buffer is not None and shared_buffer_bytes > 0)
shared_write_offset = 0
shared_buffer_fallback = False
shared_np = (np.frombuffer(
shared_buffer,
dtype=np.uint8,
count=shared_buffer_bytes,
) if use_shared_buffer else None)
chunk_intervals_ms: list[float] = []
chunk_publish_ms: list[float] = []
chunk_read_ms: list[float] = []
try:
t_proc_spawn_start = time.perf_counter()
proc = subprocess.Popen(
cmd,
stdin=subprocess.PIPE,
stdout=subprocess.PIPE,
stderr=subprocess.PIPE,
bufsize=0,
)
assert proc.stdin is not None
assert proc.stdout is not None
assert proc.stderr is not None
if hasattr(fcntl, "F_SETPIPE_SZ"):
fcntl.fcntl(proc.stdin.fileno(), fcntl.F_SETPIPE_SZ, 1048576)
fcntl.fcntl(proc.stdout.fileno(), fcntl.F_SETPIPE_SZ, 1048576)
ffmpeg_spawn_ms = (time.perf_counter() - t_proc_spawn_start) * 1000
def _write_frames():
try:
for frame in out_frames:
proc.stdin.write(np.ascontiguousarray(frame).tobytes())
proc.stdin.close()
except Exception as exc:
writer_error[0] = exc
try:
proc.stdin.close()
except Exception:
pass
def _read_stderr():
try:
while True:
data = proc.stderr.read(4096)
if not data:
break
stderr_chunks.append(data)
except Exception:
pass
writer_thread = threading.Thread(target=_write_frames, daemon=True)
stderr_thread = threading.Thread(target=_read_stderr, daemon=True)
writer_thread.start()
stderr_thread.start()
publish(StreamInit(
stream_id=stream_id,
mime=AV_MEDIA_MIME,
uses_shared_buffer=use_shared_buffer,
))
chunk_count = 0
total_bytes = 0
first_chunk_ms: float | None = None
last_chunk_emit_t = time.perf_counter()
while True:
t_read_start = time.perf_counter()
chunk = proc.stdout.read(AV_CHUNK_SIZE_BYTES)
t_read_end = time.perf_counter()
if not chunk:
break
chunk_read_ms.append((t_read_end - t_read_start) * 1000)
chunk_count += 1
total_bytes += len(chunk)
if first_chunk_ms is None:
first_chunk_ms = (t_read_end - t_stream_start) * 1000
chunk_intervals_ms.append((t_read_end - last_chunk_emit_t) * 1000)
t_publish_start = time.perf_counter()
if use_shared_buffer and not shared_buffer_fallback:
chunk_len = len(chunk)
write_end = shared_write_offset + chunk_len
if write_end <= shared_buffer_bytes:
shared_np[shared_write_offset:write_end] = np.frombuffer(chunk, dtype=np.uint8)
publish(
StreamChunk(
stream_id=stream_id,
chunk_offset=shared_write_offset,
chunk_length=chunk_len,
uses_shared_buffer=True,
))
shared_write_offset = write_end
chunk_publish_ms.append((time.perf_counter() - t_publish_start) * 1000)
last_chunk_emit_t = time.perf_counter()
continue
shared_buffer_fallback = True
print(f"{log_prefix} Shared stream buffer exhausted at "
f"{shared_write_offset / (1024 * 1024):.1f}MB; "
"falling back to queue chunk bytes")
publish(StreamChunk(
stream_id=stream_id,
chunk=chunk,
))
chunk_publish_ms.append((time.perf_counter() - t_publish_start) * 1000)
last_chunk_emit_t = time.perf_counter()
writer_thread.join(timeout=5.0)
rc = proc.wait()
stderr_thread.join(timeout=1.0)
if rc != 0:
stderr_tail = b"".join(stderr_chunks).decode(errors="ignore")[-1200:]
return False, f"ffmpeg av_fmp4 stream failed (rc={rc}): {stderr_tail}"
if writer_error[0] is not None:
return False, f"ffmpeg frame writer failed: {writer_error[0]}"
timings["av_encode_stream_ms"] = (time.perf_counter() - t_stream_start) * 1000
timings["av_stream_bytes"] = total_bytes
timings["av_trim_head_frames"] = head_trim_frames
timings["av_trim_head_audio_frames"] = head_trim_audio_frames
timings["av_frames_encoded"] = len(out_frames)
timings["av_shared_buffer_used"] = (bool(use_shared_buffer and not shared_buffer_fallback))
timings["av_wav_write_ms"] = wav_write_ms
timings["av_ffmpeg_spawn_ms"] = ffmpeg_spawn_ms
timings["av_first_chunk_ms"] = first_chunk_ms or 0.0
if chunk_intervals_ms:
timings["av_chunk_interval_ms_min"] = min(chunk_intervals_ms)
timings["av_chunk_interval_ms_median"] = float(np.median(chunk_intervals_ms))
timings["av_chunk_interval_ms_p95"] = (float(np.percentile(chunk_intervals_ms, 95)))
timings["av_chunk_interval_ms_max"] = max(chunk_intervals_ms)
if chunk_publish_ms:
timings["av_chunk_publish_ms_median"] = float(np.median(chunk_publish_ms))
timings["av_chunk_publish_ms_p95"] = float(np.percentile(chunk_publish_ms, 95))
if chunk_read_ms:
timings["av_chunk_read_ms_median"] = float(np.median(chunk_read_ms))
timings["av_chunk_read_ms_p95"] = float(np.percentile(chunk_read_ms, 95))
publish(StreamComplete(
stream_id=stream_id,
chunks=chunk_count,
))
return True, None
except Exception as exc:
return False, str(exc)
finally:
if proc is not None and proc.poll() is None:
try:
proc.kill()
except Exception:
pass
try:
os.remove(wav_path)
except OSError:
pass
@@ -1,288 +0,0 @@
"""Benchmark the AV streaming hot-path used by dreamverse-server.
Measures wall-time, encoded byte volume, and chunk count for
``av_streaming.stream_fmp4`` over synthetic frames + audio at
production resolution. Sweeps codecs (default: ``libx264`` and
``h264_nvenc`` if the active ffmpeg supports it) and ffmpeg presets so
the deploy can pick a configuration that achieves a >=1.0 realtime
ratio (5.04s of generated video produced in <=5.04s wall-time).
Usage::
python -m dreamverse.benchmarks.benchmark_av_streaming
python -m dreamverse.benchmarks.benchmark_av_streaming \\
--frames 121 --width 1920 --height 1088 --runs 3 \\
--codecs libx264 h264_nvenc --x264-preset ultrafast \\
--nvenc-preset p1
FASTVIDEO_FFMPEG_BIN=$HOME/opt/ffmpeg-native/bin/ffmpeg \\
python -m dreamverse.benchmarks.benchmark_av_streaming
Skips ``h264_nvenc`` automatically if the binary lacks the encoder. This is
the regression guard for software-encoding overhead that can drain the
inter-segment playback buffer.
"""
from __future__ import annotations
import argparse
import os
import shutil
import statistics
import subprocess
import sys
import time
from collections.abc import Iterable
from dataclasses import dataclass
import numpy as np
import torch
_HERE = os.path.dirname(os.path.abspath(__file__))
_SERVER_ROOT = os.path.dirname(_HERE)
if _SERVER_ROOT not in sys.path:
sys.path.insert(0, _SERVER_ROOT)
from dreamverse.av_streaming import stream_fmp4 # noqa: E402
@dataclass
class BenchResult:
codec: str
preset: str
runs: int
wall_ms_min: float
wall_ms_median: float
wall_ms_p95: float
wall_ms_max: float
bytes_median: int
chunks_median: float
realtime_ratio_median: float
error: str | None = None
def _make_synthetic_frames(num: int, width: int, height: int, seed: int) -> list[np.ndarray]:
rng = np.random.default_rng(seed)
base = rng.integers(0, 255, size=(height, width, 3), dtype=np.uint8)
out: list[np.ndarray] = []
for i in range(num):
f = base.copy()
f[:, :, 0] = (f[:, :, 0].astype(np.int32) + i * 2) % 256
out.append(f)
return out
def _make_synthetic_audio(num_frames: int, fps: int, sample_rate: int, seed: int) -> torch.Tensor:
duration_s = num_frames / fps
samples = int(round(duration_s * sample_rate))
rng = np.random.default_rng(seed + 1)
audio = (rng.uniform(-0.1, 0.1, (2, samples))).astype(np.float32)
return torch.from_numpy(audio)
def _ffmpeg_supports(codec: str, ffmpeg_bin: str) -> bool:
try:
out = subprocess.run(
[ffmpeg_bin, "-hide_banner", "-encoders"],
capture_output=True,
text=True,
check=False,
timeout=10,
).stdout
except Exception:
return False
needle = f" {codec} "
return any(needle in line for line in out.splitlines())
def _run_one(frames: list[np.ndarray], audio: torch.Tensor, sample_rate: int, codec: str,
preset: str) -> tuple[float, int, int, str | None]:
timings: dict = {}
chunks: list = []
def _publish(event):
chunks.append(event)
prev_codec = os.environ.get("FASTVIDEO_VIDEO_CODEC")
prev_preset = os.environ.get("FASTVIDEO_X264_PRESET")
prev_nvenc_preset = os.environ.get("FASTVIDEO_NVENC_PRESET")
os.environ["FASTVIDEO_VIDEO_CODEC"] = codec
if codec.endswith("_nvenc"):
os.environ["FASTVIDEO_NVENC_PRESET"] = preset
else:
os.environ["FASTVIDEO_X264_PRESET"] = preset
t0 = time.perf_counter()
try:
ok, err = stream_fmp4(
frames=frames,
audio=audio,
audio_sample_rate=sample_rate,
stream_id="bench",
timings=timings,
head_trim_frames=0,
head_trim_audio_frames=0,
shared_buffer=None,
shared_buffer_bytes=0,
publish=_publish,
log_prefix="[bench]",
)
finally:
if prev_codec is None:
os.environ.pop("FASTVIDEO_VIDEO_CODEC", None)
else:
os.environ["FASTVIDEO_VIDEO_CODEC"] = prev_codec
if prev_preset is None:
os.environ.pop("FASTVIDEO_X264_PRESET", None)
else:
os.environ["FASTVIDEO_X264_PRESET"] = prev_preset
if prev_nvenc_preset is None:
os.environ.pop("FASTVIDEO_NVENC_PRESET", None)
else:
os.environ["FASTVIDEO_NVENC_PRESET"] = prev_nvenc_preset
wall_ms = (time.perf_counter() - t0) * 1000.0
if not ok:
return wall_ms, 0, 0, err or "stream_fmp4 returned False"
total_bytes = int(timings.get("av_stream_bytes", 0))
return wall_ms, total_bytes, len(chunks), None
def benchmark(codecs: Iterable[str], runs: int, frames_n: int, width: int, height: int, fps: int, sample_rate: int,
x264_preset: str, nvenc_preset: str, seed: int, ffmpeg_bin: str) -> list[BenchResult]:
print(f"[bench] ffmpeg_bin={ffmpeg_bin}")
print(f"[bench] frames={frames_n} {width}x{height} fps={fps} "
f"audio_sr={sample_rate} runs/codec={runs}")
frames = _make_synthetic_frames(frames_n, width, height, seed)
audio = _make_synthetic_audio(frames_n, fps, sample_rate, seed)
playable_s = frames_n / fps
print(f"[bench] playable={playable_s:.3f}s "
f"(realtime_ratio = playable / wall_time; >= 1.0 means no "
f"buffer drain)")
results: list[BenchResult] = []
for codec in codecs:
preset = nvenc_preset if codec.endswith("_nvenc") else x264_preset
if not _ffmpeg_supports(codec, ffmpeg_bin):
results.append(
BenchResult(codec=codec,
preset=preset,
runs=0,
wall_ms_min=0,
wall_ms_median=0,
wall_ms_p95=0,
wall_ms_max=0,
bytes_median=0,
chunks_median=0,
realtime_ratio_median=0,
error=f"{codec} not in ffmpeg"))
continue
walls: list[float] = []
sizes: list[int] = []
chunkcounts: list[int] = []
last_err: str | None = None
for run in range(runs):
wall_ms, total_bytes, chunk_count, err = _run_one(frames, audio, sample_rate, codec, preset)
print(f"[bench] codec={codec:12s} preset={preset:9s} "
f"run={run + 1}/{runs} wall={wall_ms:7.1f}ms "
f"bytes={total_bytes:>9d} chunks={chunk_count:>3d} "
f"realtime={playable_s / (wall_ms / 1000.0):5.2f}x"
f"{' ERR=' + err if err else ''}")
if err is not None:
last_err = err
continue
walls.append(wall_ms)
sizes.append(total_bytes)
chunkcounts.append(chunk_count)
if not walls:
results.append(
BenchResult(codec=codec,
preset=preset,
runs=0,
wall_ms_min=0,
wall_ms_median=0,
wall_ms_p95=0,
wall_ms_max=0,
bytes_median=0,
chunks_median=0,
realtime_ratio_median=0,
error=last_err or "all runs failed"))
continue
walls_sorted = sorted(walls)
p95_idx = max(0, int(round(0.95 * (len(walls_sorted) - 1))))
wall_med = statistics.median(walls)
results.append(
BenchResult(
codec=codec,
preset=preset,
runs=len(walls),
wall_ms_min=min(walls),
wall_ms_median=wall_med,
wall_ms_p95=walls_sorted[p95_idx],
wall_ms_max=max(walls),
bytes_median=int(statistics.median(sizes)),
chunks_median=statistics.median(chunkcounts),
realtime_ratio_median=playable_s / (wall_med / 1000.0),
))
return results
def _print_summary(results: list[BenchResult]) -> None:
print()
print("=== summary ===")
header = (f"{'codec':14s} {'preset':10s} {'runs':>4s} "
f"{'wall_med_ms':>11s} {'wall_p95_ms':>11s} "
f"{'bytes_med':>10s} {'realtime':>8s} notes")
print(header)
print("-" * len(header))
for r in results:
if r.error is not None:
print(f"{r.codec:14s} {r.preset:10s} {r.runs:>4d} "
f"{'-':>11s} {'-':>11s} {'-':>10s} {'-':>8s} "
f"ERR: {r.error}")
continue
print(f"{r.codec:14s} {r.preset:10s} {r.runs:>4d} "
f"{r.wall_ms_median:>11.1f} {r.wall_ms_p95:>11.1f} "
f"{r.bytes_median:>10d} {r.realtime_ratio_median:>7.2f}x "
f"{'OK' if r.realtime_ratio_median >= 1.0 else 'BUFFER DRAINS'}")
def main() -> int:
p = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
p.add_argument("--frames", type=int, default=121, help="num frames (default: 121, matches NUM_FRAMES)")
p.add_argument("--width", type=int, default=1920)
p.add_argument("--height", type=int, default=1088)
p.add_argument("--fps", type=int, default=24)
p.add_argument("--sample-rate", type=int, default=24000)
p.add_argument("--runs", type=int, default=3, help="runs per codec (default: 3 — 1 warmup + 2 timed in median)")
p.add_argument("--codecs",
nargs="+",
default=["libx264", "h264_nvenc"],
help="codecs to benchmark; missing ones are skipped")
p.add_argument("--x264-preset", default="ultrafast")
p.add_argument("--nvenc-preset", default="p1")
p.add_argument("--seed", type=int, default=0)
args = p.parse_args()
ffmpeg_bin = shutil.which(os.getenv("FASTVIDEO_FFMPEG_BIN", "ffmpeg"))
if ffmpeg_bin is None:
print("ffmpeg not found", file=sys.stderr)
return 2
results = benchmark(
codecs=args.codecs,
runs=args.runs,
frames_n=args.frames,
width=args.width,
height=args.height,
fps=args.fps,
sample_rate=args.sample_rate,
x264_preset=args.x264_preset,
nvenc_preset=args.nvenc_preset,
seed=args.seed,
ffmpeg_bin=ffmpeg_bin,
)
_print_summary(results)
any_below = any(r.error is None and r.realtime_ratio_median < 1.0 for r in results)
return 1 if any_below else 0
if __name__ == "__main__":
raise SystemExit(main())

Some files were not shown because too many files have changed in this diff Show More