Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
3ae184bc53 | ||
|
|
a0a58e02e2 | ||
|
|
c17d33bf33 | ||
|
|
2aaeee2ab8 | ||
|
|
eb3a394224 | ||
|
|
f673423b51 | ||
|
|
eb0a41528a | ||
|
|
140bd1a6cf | ||
|
|
11f5a8e582 | ||
|
|
71b3cb8c34 | ||
|
|
c85f6a477f | ||
|
|
40d4930d73 | ||
|
|
f9be085243 | ||
|
|
36b53ff350 | ||
|
|
9801037c3d | ||
|
|
74d09b0efd | ||
|
|
38dc8820ac | ||
|
|
c77a76c6af | ||
|
|
d14d5aadea | ||
|
|
4c915b7742 | ||
|
|
9a8bbe18fa | ||
|
|
ea25441ef0 | ||
|
|
48957fcde1 | ||
|
|
7b872cc41e | ||
|
|
37418946c8 | ||
|
|
95fd29e0cb | ||
|
|
e17cd2633c | ||
|
|
e0dc5f2b0c | ||
|
|
70ee5d230c | ||
|
|
24ced500f5 | ||
|
|
4ddcdf541f | ||
|
|
0e3529869c | ||
|
|
e1e0d91c00 | ||
|
|
145a3f166b | ||
|
|
88a5a933ab | ||
|
|
c591d6d2a6 | ||
|
|
65dff806a8 | ||
|
|
b85f0f4c2a | ||
|
|
76c62d7a00 | ||
|
|
f6e65ff668 | ||
|
|
c220aa8000 | ||
|
|
4713fc17ed | ||
|
|
5789955bbe | ||
|
|
2ad84a3b78 | ||
|
|
12d699cd78 | ||
|
|
34f14ded21 | ||
|
|
71d1ab411f | ||
|
|
805e487773 | ||
|
|
8803b4547e | ||
|
|
3b3806b3f6 | ||
|
|
38d962e89d | ||
|
|
3966a365d0 | ||
|
|
d73fd14af0 | ||
|
|
a87cc89916 | ||
|
|
81fd80c8ee | ||
|
|
ff22439f28 | ||
|
|
de0de04212 | ||
|
|
7f2c3e1f64 | ||
|
|
ab55e57c22 | ||
|
|
46f6b43a53 | ||
|
|
833a33b663 | ||
|
|
9ea1307cd4 | ||
|
|
be35003cb1 | ||
|
|
26bd4db253 | ||
|
|
e294ca011c | ||
|
|
2085a4fc4a | ||
|
|
c0c8e39c04 | ||
|
|
f72618dafb | ||
|
|
b3edfacdd8 | ||
|
|
30129a3350 | ||
|
|
4d49f7b0aa | ||
|
|
71bfc13d75 | ||
|
|
74db6e18d1 | ||
|
|
7d263c6a36 | ||
|
|
454c32d1d1 | ||
|
|
d1240b9238 | ||
|
|
4105094fa5 | ||
|
|
f036469d3d | ||
|
|
14261bc98c | ||
|
|
d92858659d | ||
|
|
1a383f3f66 | ||
|
|
bc27a032c5 | ||
|
|
2b13e117f0 | ||
|
|
99c166c381 | ||
|
|
95066245db | ||
|
|
6dcaac768b | ||
|
|
02c1c49b75 | ||
|
|
cd1b7cf139 | ||
|
|
e63b7d8ac4 | ||
|
|
5190c1bb1e | ||
|
|
2cb3bba658 | ||
|
|
f9e1c46c3c | ||
|
|
e1eda47589 | ||
|
|
d902967208 | ||
|
|
fea556269b | ||
|
|
69dd3c68f6 | ||
|
|
5433f6e80b | ||
|
|
e315657066 | ||
|
|
f8d9a0c57f | ||
|
|
fa6d276925 | ||
|
|
fc80d95d7e | ||
|
|
37cab18780 | ||
|
|
128d0b7fc5 | ||
|
|
8092f02e6d | ||
|
|
03d9ce2edb | ||
|
|
10fc92dba5 | ||
|
|
6736dc06a5 | ||
|
|
8c002c62af | ||
|
|
7061313d04 | ||
|
|
76d3ba69e0 | ||
|
|
8e39ce38c9 | ||
|
|
d4bd8bf2c0 | ||
|
|
e83d7bc50c | ||
|
|
ff3d5aff75 | ||
|
|
959dbcc8a2 | ||
|
|
36bf37e9ba | ||
|
|
7a83e0e6fc | ||
|
|
8be1313b86 | ||
|
|
d925ad05f3 | ||
|
|
7f795600c8 | ||
|
|
ec16b6b01d | ||
|
|
31c0f1b341 | ||
|
|
4bee0fa199 | ||
|
|
530e6b8363 | ||
|
|
9ab2725db1 | ||
|
|
0aff68f51d | ||
|
|
ad58f802f3 | ||
|
|
04fa356ee3 | ||
|
|
f9c076fe2b | ||
|
|
f76efe798e | ||
|
|
09f455233e | ||
|
|
aea300f690 | ||
|
|
b92219f6a6 | ||
|
|
a321b95a8a | ||
|
|
98308db7e0 | ||
|
|
c1e18f6722 | ||
|
|
75e193a2c9 | ||
|
|
d6e0a7d0dd | ||
|
|
7fc5f241da | ||
|
|
aae48a7e90 | ||
|
|
88f38eb0f4 | ||
|
|
d750b463dc | ||
|
|
e10b26a3d8 | ||
|
|
74636ba246 | ||
|
|
7e2f3f14e7 | ||
|
|
caa1c402ba | ||
|
|
38a6bd93d3 | ||
|
|
b867ef7e7c | ||
|
|
3ae58c277a | ||
|
|
0c6862ca55 | ||
|
|
e8c854bcf1 | ||
|
|
06860e96fe | ||
|
|
1b503554d1 | ||
|
|
351ceb7c59 | ||
|
|
10875e0d7b | ||
|
|
1eaae8a10b | ||
|
|
59e00f6164 | ||
|
|
745cc05b10 | ||
|
|
c5dc244871 | ||
|
|
dbf3917bf4 | ||
|
|
050f189c95 | ||
|
|
029216029f | ||
|
|
31f44110b5 | ||
|
|
21f3ce6577 | ||
|
|
785d123e36 | ||
|
|
d58c551c11 | ||
|
|
560628709c | ||
|
|
0f53b51e6c | ||
|
|
06093a9c4e | ||
|
|
dbddfab6d2 | ||
|
|
7188170277 | ||
|
|
b7f69c2c1d | ||
|
|
23a4531491 | ||
|
|
7d52ad0118 | ||
|
|
4d7bf35fa3 | ||
|
|
a6a9c9ca07 | ||
|
|
f4704847c2 | ||
|
|
d9c996310b | ||
|
|
d6651afd2e | ||
|
|
cf67618cad | ||
|
|
2f0a2b3c57 | ||
|
|
e7748d9952 | ||
|
|
8eb3140b2f | ||
|
|
d6ddcea682 | ||
|
|
3559ba2377 | ||
|
|
61e63ea0d7 | ||
|
|
4ce4ac4734 | ||
|
|
e7f6db9bd1 | ||
|
|
d83f45a6a0 | ||
|
|
dd91542cd1 | ||
|
|
581e8115fe | ||
|
|
dea69cf651 | ||
|
|
60ac6537df | ||
|
|
5285116e73 | ||
|
|
40ce2d72f5 | ||
|
|
704bc9aaf9 | ||
|
|
de264fcc99 | ||
|
|
7b952e4673 | ||
|
|
551b2d2048 | ||
|
|
7bfaf82fd7 | ||
|
|
9cd6a86b95 | ||
|
|
16e9552778 | ||
|
|
87f8a2782d | ||
|
|
cbbb09d7b8 | ||
|
|
2f6230abcf | ||
|
|
f8bfc76015 | ||
|
|
8f1e6c3336 | ||
|
|
8e7d2e7879 | ||
|
|
6ab2870942 | ||
|
|
e0ad145152 | ||
|
|
da04d08426 | ||
|
|
1f70032af5 | ||
|
|
7f71994653 | ||
|
|
2bb3349da1 | ||
|
|
8fe1689968 | ||
|
|
e53730f324 | ||
|
|
7a4fe9086a | ||
|
|
d277361aae | ||
|
|
734a54e7a9 | ||
|
|
91364982df | ||
|
|
50145e4fcb | ||
|
|
4112507e99 | ||
|
|
424fc2b4ae | ||
|
|
e6066223e6 | ||
|
|
b6fa3d24d8 | ||
|
|
55c2e7cd76 | ||
|
|
5a549af823 | ||
|
|
92fb660c2e | ||
|
|
3ff640b2e6 | ||
|
|
c722429ab5 | ||
|
|
e04a192de6 | ||
|
|
c9ca6d1298 | ||
|
|
754292c419 | ||
|
|
0082bc66fc | ||
|
|
8b1937422e | ||
|
|
fb6cbf23e6 | ||
|
|
c8fdd5ed7b | ||
|
|
1c19a6a00c | ||
|
|
d44409c704 | ||
|
|
5d1c7852b7 | ||
|
|
77a211d006 | ||
|
|
bef8169bb1 | ||
|
|
681f1583f9 | ||
|
|
e3b4564d5a | ||
|
|
c0d03fc43d | ||
|
|
404ee8538e | ||
|
|
e57ac59462 | ||
|
|
8c55fdaf7e | ||
|
|
c30779184f | ||
|
|
9d188c0b6c | ||
|
|
9dd7c54221 | ||
|
|
62b95d8287 | ||
|
|
fdf21702f5 | ||
|
|
2972fc9449 | ||
|
|
8f5712629f | ||
|
|
436c701b9f | ||
|
|
543fea88e3 | ||
|
|
bdec816b31 | ||
|
|
2cd2e57d2e | ||
|
|
9370234294 | ||
|
|
50da62e722 | ||
|
|
4f3e8751db | ||
|
|
f4c58894d9 | ||
|
|
01c94ef385 | ||
|
|
2415226d25 | ||
|
|
404314d00f | ||
|
|
87489f0872 | ||
|
|
9ce7c8039e | ||
|
|
e1e25e95f9 | ||
|
|
490bde90e1 | ||
|
|
dc7596b973 | ||
|
|
335afa4457 | ||
|
|
3f77a6805a | ||
|
|
13d0aae706 | ||
|
|
404cbf4f3c | ||
|
|
958ffec844 | ||
|
|
31f000d1cc | ||
|
|
cd32b3e02f | ||
|
|
bf27908095 | ||
|
|
c5f9ea53b2 | ||
|
|
d32a7184da | ||
|
|
2930abe456 | ||
|
|
b93ef4289d | ||
|
|
401bdbd316 | ||
|
|
1048d79cf8 | ||
|
|
1e8406162d | ||
|
|
03edd35c83 | ||
|
|
ac11127397 | ||
|
|
e028dcc7c0 | ||
|
|
076f45c1ee | ||
|
|
85eb7265db | ||
|
|
d3ceb67e66 | ||
|
|
7ac153a5ca | ||
|
|
d1e7aa0abd | ||
|
|
2d846c55a1 | ||
|
|
b318063c0a | ||
|
|
4aa307be55 | ||
|
|
055e52e5ea | ||
|
|
7d2069596b | ||
|
|
c45009c9a4 | ||
|
|
b91020b407 | ||
|
|
2dcc5ea4f6 | ||
|
|
359151d9a0 | ||
|
|
ce67cd3729 | ||
|
|
7c554e5da8 | ||
|
|
663ea33ff1 | ||
|
|
3ef04f1654 | ||
|
|
0eced76a41 | ||
|
|
3ab6470d1a | ||
|
|
989a03532c | ||
|
|
fa15369a02 | ||
|
|
78a9cb88d8 | ||
|
|
a0bff12746 | ||
|
|
98f2af94e5 | ||
|
|
46f7b6d574 | ||
|
|
911a6a6a35 | ||
|
|
38c7949d5c | ||
|
|
7e7a0dba9d | ||
|
|
f62e210ae6 | ||
|
|
6ceb4942a0 | ||
|
|
8cae5e4708 | ||
|
|
2a773fa34e | ||
|
|
5357f63327 | ||
|
|
60f61c8101 | ||
|
|
3d75ba8251 | ||
|
|
6c6bcd914d | ||
|
|
f79b08de81 | ||
|
|
f2bc037fff | ||
|
|
86604a684b | ||
|
|
47bd1e0178 | ||
|
|
c41305ad18 | ||
|
|
98ce9034f0 | ||
|
|
0ceff110da | ||
|
|
1d018acb3e | ||
|
|
7d8cf38dbe | ||
|
|
8d483fe4aa | ||
|
|
c1191250bf | ||
|
|
4b7266349a | ||
|
|
22f9b7681f | ||
|
|
589d32cc39 | ||
|
|
89199837db | ||
|
|
d6ebaf1b49 | ||
|
|
fac927777c | ||
|
|
ecbd697dae | ||
|
|
7d4acef64d | ||
|
|
9f0ce517cf | ||
|
|
c718e56b0d | ||
|
|
b65f0316d1 | ||
|
|
8d8bcb76b0 | ||
|
|
5f42748ed1 | ||
|
|
c9005045dc | ||
|
|
6c81befc87 | ||
|
|
dfe0b288e1 | ||
|
|
31200fbb83 | ||
|
|
9185978c55 | ||
|
|
fcba463553 | ||
|
|
2c53d3eecf | ||
|
|
516ecd374a | ||
|
|
3b1b54a74d | ||
|
|
6914e7c904 | ||
|
|
5452369749 | ||
|
|
a113311e77 | ||
|
|
44da97da92 | ||
|
|
f759980a58 | ||
|
|
37e0f8c236 | ||
|
|
51711d5906 | ||
|
|
6375223b16 | ||
|
|
4cb046768d | ||
|
|
3322542444 | ||
|
|
65f707354b | ||
|
|
109e2e7e9d | ||
|
|
cbc3a6bb9d | ||
|
|
2fa8d4ae6d | ||
|
|
7b6c8aee99 | ||
|
|
6284eaa363 | ||
|
|
636524e87f | ||
|
|
202b2f3972 | ||
|
|
247fe273d8 | ||
|
|
cb320dfa3a | ||
|
|
d8bb5abc46 | ||
|
|
cc703eca51 | ||
|
|
81c9df629c | ||
|
|
d3c0c52208 | ||
|
|
744e0555c0 | ||
|
|
3a38f7dfdc | ||
|
|
f572319bd9 | ||
|
|
48528f468c | ||
|
|
4264a80ca9 | ||
|
|
8573d4f05e | ||
|
|
210a733515 | ||
|
|
0aef0e6f63 | ||
|
|
dd022ad9be | ||
|
|
832ad61e5b | ||
|
|
9419c04ee3 | ||
|
|
a37b39d83c | ||
|
|
bb8c769c8e | ||
|
|
576c214f28 | ||
|
|
b79d1fc15b | ||
|
|
eb66e1c18d | ||
|
|
616d43c1cf | ||
|
|
7244a4b27f | ||
|
|
7e5ebb4582 | ||
|
|
65ed588570 | ||
|
|
14adfe2edc | ||
|
|
6198c6a640 | ||
|
|
e6b71b531b | ||
|
|
ae1d112c6a | ||
|
|
bf4de1f38f | ||
|
|
66fdcc8e76 | ||
|
|
ed1e8d6bad | ||
|
|
b9423ca3f8 | ||
|
|
ad16289871 | ||
|
|
2a41da1e6b | ||
|
|
32133171da | ||
|
|
19674c6f29 | ||
|
|
508afb7002 | ||
|
|
288ea88105 | ||
|
|
eb0f1318f3 | ||
|
|
ce9b5910cc | ||
|
|
d0e5a6214a | ||
|
|
834562b2db | ||
|
|
060cc7b9ba | ||
|
|
6c58a5ba62 | ||
|
|
48d9f61f86 | ||
|
|
5f938b5844 | ||
|
|
74da2a7370 | ||
|
|
580d6dfe1f | ||
|
|
344e43006a | ||
|
|
c5155b256e | ||
|
|
e005c7f3ac | ||
|
|
ff5a79ef60 | ||
|
|
ab01dc4ba5 | ||
|
|
285a950c1b | ||
|
|
46a0a85d85 | ||
|
|
4aeabbc629 | ||
|
|
949bb5c835 | ||
|
|
aab74c1271 | ||
|
|
f89d86944f | ||
|
|
8741d204a5 | ||
|
|
cdc85f58a8 | ||
|
|
0262d2f089 | ||
|
|
62c0343465 | ||
|
|
1e1a023fb0 | ||
|
|
1d2517ad8e | ||
|
|
d41186cb4a | ||
|
|
78e0c7eec9 | ||
|
|
1c41a94b62 | ||
|
|
2e66aafe20 | ||
|
|
55074bda76 | ||
|
|
de65bec2b7 | ||
|
|
7664dd0de3 | ||
|
|
019a88ced4 | ||
|
|
72de11abcc | ||
|
|
d71a4ebffc | ||
|
|
1089ab43bf | ||
|
|
97d4b984c9 | ||
|
|
2a8953d74d | ||
|
|
8801b10da7 | ||
|
|
6b413f2ec4 | ||
|
|
28b72694aa | ||
|
|
4afb0cfe4f | ||
|
|
3eec1281cf | ||
|
|
0660489e38 | ||
|
|
dd871a17bf | ||
|
|
dc11529862 | ||
|
|
ffabf85e31 | ||
|
|
c0026ca5ba | ||
|
|
0f2bbe71ac | ||
|
|
2a46902ecb | ||
|
|
66012d3a4c | ||
|
|
f666b9de41 | ||
|
|
7e3c073b55 | ||
|
|
a6aa21bd07 | ||
|
|
6519b57aab | ||
|
|
675aea6ece | ||
|
|
46e7a15e0d | ||
|
|
e4f702d7ec | ||
|
|
bb68fcc809 | ||
|
|
b392e6a874 | ||
|
|
0991003905 | ||
|
|
8f8ce6d9e1 | ||
|
|
e3d0cbe185 | ||
|
|
d5ec468d43 | ||
|
|
e55fa6e5dc | ||
|
|
61b6ddeee1 | ||
|
|
a9a000f45d | ||
|
|
66b8b8561e | ||
|
|
6684872616 | ||
|
|
7f654e3332 | ||
|
|
8631c1b806 | ||
|
|
5357e12b5a | ||
|
|
bdfdf1dfee | ||
|
|
d156461785 | ||
|
|
6edf113838 | ||
|
|
dcf7738cbc | ||
|
|
b2ebaaf865 | ||
|
|
7768bb80f6 | ||
|
|
a335811869 | ||
|
|
357b0533fe | ||
|
|
2ec3732758 | ||
|
|
a004408a93 | ||
|
|
007e237e69 | ||
|
|
8e18dc9f71 | ||
|
|
7ab32539af | ||
|
|
6ef8fcb61d | ||
|
|
016e24da63 | ||
|
|
85b8717545 | ||
|
|
657fd745e1 | ||
|
|
12647457a7 | ||
|
|
298f74f956 | ||
|
|
ee8babb298 | ||
|
|
60295cc03f | ||
|
|
1572e13b6e | ||
|
|
a157275b4c | ||
|
|
c4dbe7dac3 | ||
|
|
d39591108e | ||
|
|
ace6e971e5 | ||
|
|
b4f6758253 | ||
|
|
535d29b392 | ||
|
|
b4255517e0 | ||
|
|
6eeb60613f | ||
|
|
53d2c7791f | ||
|
|
53cb693dca | ||
|
|
6f72d24876 | ||
|
|
d1459e9976 | ||
|
|
59ab481eb1 | ||
|
|
0cf001986a | ||
|
|
51956369a5 | ||
|
|
4b0970cbbf | ||
|
|
94bf47a572 | ||
|
|
1a3ac9074b | ||
|
|
fb0581d5b0 | ||
|
|
6c74ab4132 | ||
|
|
9a91021c56 | ||
|
|
51c94d6a73 | ||
|
|
dba38dbc03 | ||
|
|
2034cc3c4f | ||
|
|
c69afce2f6 | ||
|
|
b08e758eb3 | ||
|
|
3f3462d7ce | ||
|
|
c9c47dd89c | ||
|
|
9c4ef7c2f1 | ||
|
|
f25eb4b905 | ||
|
|
048d55ccbb | ||
|
|
a271c55fe4 | ||
|
|
f663ae0d8a | ||
|
|
c0911aa3dd | ||
|
|
5f59687ae7 | ||
|
|
5adbc81cdc | ||
|
|
f26d5c37c1 | ||
|
|
6a4ef42378 |
@@ -0,0 +1,94 @@
|
||||
# Agent Infrastructure — Status Dashboard
|
||||
|
||||
Developer-maintained overview of all agent components and their maturity.
|
||||
Use this to understand what exists, how complete it is, and how much to trust it.
|
||||
|
||||
_Last synced: 2026-03-02_
|
||||
|
||||
> To resync this dashboard, use the workflow: `.agents/workflows/sync-dashboard.md`
|
||||
|
||||
---
|
||||
|
||||
## Summary
|
||||
|
||||
| Category | Total | ✅ Ready | 🟡 Draft | 🔴 Stub | Trust |
|
||||
|----------|-------|---------|---------|---------|-------|
|
||||
| Skills | 8 | 0 | 8 | 0 | Low — newly created, untested |
|
||||
| Workflows (SOPs) | 4 | 0 | 4 | 0 | Low — newly created, untested |
|
||||
| Memory files | 4 | 1 | 3 | 0 | Medium — codebase_map is solid |
|
||||
| Lessons | 0 | — | — | — | N/A — empty |
|
||||
| Exploration logs | 0 | — | — | — | N/A — empty |
|
||||
|
||||
---
|
||||
|
||||
## Skills (`.agents/skills/`)
|
||||
|
||||
| Skill | File | Status | Trust | Tested | Notes |
|
||||
|-------|------|--------|-------|--------|-------|
|
||||
| Launch Experiment | `launch-experiment.md` | 🟡 Draft | Low | ❌ | Needs dry-run validation |
|
||||
| Monitor Experiment | `monitor-experiment.md` | 🟡 Draft | Low | ❌ | Requires W&B API access to test |
|
||||
| Summarize Run | `summarize-run.md` | 🟡 Draft | Low | ❌ | Pattern from existing test infra |
|
||||
| Log Experiment | `log-experiment.md` | 🟡 Draft | Low | ❌ | Journal formatting only |
|
||||
| Evaluate Video Quality | `evaluate-video-quality.md` | 🟡 Draft | Low | ❌ | SSIM section most mature |
|
||||
| Index Related Work | `index-related-work.md` | 🟡 Draft | Low | ❌ | Schema defined, no entries yet |
|
||||
| Search Related Work | `search-related-work.md` | 🟡 Draft | Low | ❌ | Depends on indexed entries |
|
||||
| Skill Template | `SKILL_TEMPLATE.md` | ✅ Ready | High | ✅ | Meta-template, stable |
|
||||
|
||||
### Trust Level Definitions
|
||||
- **High**: Tested in production, validated against real experiments
|
||||
- **Medium**: Logic is sound, partially tested or based on existing patterns
|
||||
- **Low**: Newly written, not yet validated
|
||||
- **None**: Placeholder only
|
||||
|
||||
---
|
||||
|
||||
## Workflows / SOPs (`.agents/workflows/`)
|
||||
|
||||
| Workflow | File | Status | Trust | Tested | Notes |
|
||||
|----------|------|--------|-------|--------|-------|
|
||||
| Experiment Lifecycle | `experiment-lifecycle.md` | 🟡 Draft | Low | ❌ | End-to-end flow, untested |
|
||||
| Evaluation Development | `evaluation-development.md` | 🟡 Draft | Low | ❌ | Metric dev process |
|
||||
| Experiment Journaling | `experiment-journaling.md` | 🟡 Draft | Low | ❌ | Journaling cadence |
|
||||
| Lesson Capture | `lesson-capture.md` | 🟡 Draft | Low | ❌ | Post-experiment reflection |
|
||||
| Sync Dashboard | `sync-dashboard.md` | 🟡 Draft | Low | ❌ | This dashboard's updater |
|
||||
|
||||
---
|
||||
|
||||
## Memory (`.agents/memory/`)
|
||||
|
||||
| File | Status | Trust | Notes |
|
||||
|------|--------|-------|-------|
|
||||
| `codebase_map.md` | ✅ Ready | High | Synthesized from full repo research |
|
||||
| `experiment_journal.md` | 🟡 Draft | Medium | Schema defined, no entries yet |
|
||||
| `evaluation_registry.md` | 🟡 Draft | Medium | SSIM/loss metrics documented |
|
||||
| `related_work/README.md` | 🟡 Draft | Medium | Schema defined, no entries yet |
|
||||
|
||||
---
|
||||
|
||||
## Lessons (`.agents/lessons/`)
|
||||
|
||||
| File | Category | Severity | Notes |
|
||||
|------|----------|----------|-------|
|
||||
|
||||
_No lessons captured yet._
|
||||
|
||||
---
|
||||
|
||||
## Exploration Logs (`.agents/exploration/`)
|
||||
|
||||
| File | Status | Topic | Notes |
|
||||
|------|--------|-------|-------|
|
||||
|
||||
_No exploration logs yet._
|
||||
|
||||
---
|
||||
|
||||
## What to Do Next
|
||||
|
||||
1. **Validate skills**: Run a minimal training experiment using the
|
||||
`experiment-lifecycle` SOP to test `launch-experiment` → `monitor-experiment`
|
||||
→ `summarize-run` end-to-end.
|
||||
2. **Index first related work**: Use `index-related-work` to add at least one
|
||||
paper (e.g., the Self-Forcing paper used in the codebase).
|
||||
3. **Capture first lesson**: After the validation run, capture any findings.
|
||||
4. **Promote to Ready**: As each skill/SOP is tested, update its status here.
|
||||
@@ -0,0 +1,46 @@
|
||||
# Exploration Logs
|
||||
|
||||
This directory holds draft procedures and investigation notes for tasks that
|
||||
don't yet have a standardized skill or SOP. Each exploration should follow this
|
||||
template.
|
||||
|
||||
## When to Create an Exploration Log
|
||||
|
||||
- You are working on a task with no existing skill or workflow.
|
||||
- You are experimenting with a new metric, training technique, or tool.
|
||||
- You want to document findings before they are promoted to a standard.
|
||||
|
||||
## File Naming
|
||||
|
||||
`<topic-slug>.md` — e.g., `fvd-metric-investigation.md`
|
||||
|
||||
## Template
|
||||
|
||||
```markdown
|
||||
# Exploration Log: <Topic>
|
||||
|
||||
## Status: draft | under_review | promoted | abandoned
|
||||
|
||||
## Context
|
||||
<Why this exploration is needed — link to experiment or task if applicable.>
|
||||
|
||||
## Progress
|
||||
- [ ] Step 1: ...
|
||||
- [ ] Step 2: ...
|
||||
|
||||
## Findings
|
||||
<What you have learned so far.>
|
||||
|
||||
## Mistakes / Dead Ends
|
||||
<What didn't work and why — these become lessons.>
|
||||
|
||||
## Proposed Standardization
|
||||
<If this works, describe the skill/SOP/workflow to create.>
|
||||
```
|
||||
|
||||
## Lifecycle
|
||||
|
||||
1. **Create** during exploration mode.
|
||||
2. **Update** as you make progress.
|
||||
3. **Promote**: If findings are solid, create a skill in `.agents/skills/` or an SOP in `.agents/workflows/`.
|
||||
4. **Archive mistakes**: Move failures into `.agents/lessons/`.
|
||||
@@ -0,0 +1,48 @@
|
||||
# Lessons Learned Database
|
||||
|
||||
This directory stores documented mistakes, unexpected behaviors, and their fixes.
|
||||
Each lesson is a permanent record that helps agents and humans avoid repeating
|
||||
past errors.
|
||||
|
||||
## When to Create a Lesson
|
||||
|
||||
- An experiment failed for a non-obvious reason.
|
||||
- A configuration or hyperparameter choice led to wasted compute.
|
||||
- A porting, data, or infrastructure issue was discovered and resolved.
|
||||
- A workaround was needed for a known framework/library bug.
|
||||
|
||||
## File Naming
|
||||
|
||||
`<YYYY-MM-DD>_<short-slug>.md` — e.g., `2026-03-02_lr-too-high-for-lora.md`
|
||||
|
||||
## Template
|
||||
|
||||
```markdown
|
||||
---
|
||||
date: <ISO-8601>
|
||||
experiment: <reference to experiment_journal.md entry, if applicable>
|
||||
category: hyperparameter | data | infrastructure | evaluation | porting | other
|
||||
severity: critical | important | minor
|
||||
---
|
||||
|
||||
# <Short Descriptive Title>
|
||||
|
||||
## What Happened
|
||||
<Description of the problem and its symptoms.>
|
||||
|
||||
## Root Cause
|
||||
<Analysis of why it happened.>
|
||||
|
||||
## Fix / Workaround
|
||||
<What resolved the issue.>
|
||||
|
||||
## Prevention
|
||||
<How to avoid this in the future — updated skills, SOPs, or checks.>
|
||||
```
|
||||
|
||||
## Usage
|
||||
|
||||
- Before starting a task, **search this directory** for relevant lessons.
|
||||
- After completing or failing a task, **check if a new lesson should be created**.
|
||||
- Periodically review lessons for **patterns** — recurring themes may warrant
|
||||
a new skill, SOP, or codebase fix.
|
||||
@@ -0,0 +1,129 @@
|
||||
# FastVideo-WorldModel — Codebase Map
|
||||
|
||||
High-level structural index for agent orientation. Updated 2026-03-08.
|
||||
|
||||
## Repository Layout
|
||||
|
||||
```
|
||||
FastVideo-WorldModel/
|
||||
├── fastvideo/ # Core Python package
|
||||
│ ├── models/ # Model implementations
|
||||
│ │ ├── dits/ # DiT transformers (wanvideo, ltx2, ...)
|
||||
│ │ ├── vaes/ # VAE models
|
||||
│ │ ├── encoders/ # Text/image encoders (T5, CLIP)
|
||||
│ │ ├── schedulers/ # Noise schedulers
|
||||
│ │ ├── upsamplers/ # Super-resolution models
|
||||
│ │ ├── audio/ # Audio models
|
||||
│ │ └── loader/ # Component loaders for HF repos
|
||||
│ ├── configs/ # Configuration system
|
||||
│ │ ├── models/ # Arch configs + param_names_mapping
|
||||
│ │ ├── pipelines/ # Pipeline wiring
|
||||
│ │ └── sample/ # Default sampling parameters
|
||||
│ ├── pipelines/ # End-to-end pipelines
|
||||
│ │ ├── basic/ # Per-model pipelines (wan/, ltx2/, ...)
|
||||
│ │ └── stages/ # Reusable pipeline stages
|
||||
│ ├── train/ # Refactored training framework (YAML-driven, preferred)
|
||||
│ │ ├── trainer.py # Main training loop coordinator
|
||||
│ │ ├── entrypoint/ # Training entrypoint (train.py) + checkpoint conversion
|
||||
│ │ ├── methods/ # Training algorithms (FineTune, DFSFT, DMD2, SelfForcing)
|
||||
│ │ │ ├── base.py # TrainingMethod ABC
|
||||
│ │ │ ├── fine_tuning/ # FineTuneMethod, DiffusionForcingSFTMethod
|
||||
│ │ │ └── distribution_matching/ # DMD2Method, SelfForcingMethod
|
||||
│ │ ├── models/ # Per-role model wrappers (ModelBase, CausalModelBase)
|
||||
│ │ │ └── wan/ # WanModel, WanCausalModel
|
||||
│ │ ├── callbacks/ # Composable hooks (grad_clip, ema, validation)
|
||||
│ │ └── utils/ # Config, builder, checkpoint, optimizer, tracking
|
||||
│ ├── training/ # Legacy training infrastructure (being phased out)
|
||||
│ │ ├── trackers.py # W&B tracker (BaseTracker → WandbTracker)
|
||||
│ │ ├── training_utils.py # Checkpointing, grad clipping, state dicts
|
||||
│ │ ├── training_pipeline.py # Base training pipeline
|
||||
│ │ ├── wan_training_pipeline.py # Wan T2V training
|
||||
│ │ ├── wan_i2v_training_pipeline.py # Wan I2V training
|
||||
│ │ ├── distillation_pipeline.py # Distillation base
|
||||
│ │ ├── wan_distillation_pipeline.py # Wan distillation
|
||||
│ │ ├── self_forcing_distillation_pipeline.py # Self-forcing distill
|
||||
│ │ ├── ltx2_training_pipeline.py # LTX-2 training
|
||||
│ │ └── matrixgame_training_pipeline.py # MatrixGame training
|
||||
│ ├── attention/ # Attention backends
|
||||
│ ├── distributed/ # Sequence/tensor parallel utilities
|
||||
│ ├── layers/ # Tensor-parallel layers
|
||||
│ ├── tests/ # Package-level tests
|
||||
│ │ ├── training/ # Training regression tests (W&B summary comparison)
|
||||
│ │ ├── ssim/ # SSIM visual regression tests
|
||||
│ │ ├── encoders/ # Encoder parity tests
|
||||
│ │ └── modal/ # Modal CI test runner
|
||||
│ └── registry.py # Unified config registry
|
||||
├── fastvideo-kernel/ # CUDA/custom kernels (separate build: ./build.sh)
|
||||
├── scripts/ # Utility scripts
|
||||
│ ├── distill/ # Distillation launch scripts
|
||||
│ ├── inference/ # Inference scripts
|
||||
│ ├── checkpoint_conversion/ # Weight conversion tools
|
||||
│ ├── finetune/ # Finetune scripts
|
||||
│ └── preprocess/ # Data preprocessing
|
||||
├── examples/ # Ready-to-run examples
|
||||
│ ├── training/ # Training examples (finetune/, consistency_finetune/)
|
||||
│ ├── distill/ # Distillation examples
|
||||
│ ├── inference/ # Inference examples
|
||||
│ └── dataset/ # Dataset examples
|
||||
├── docs/ # MkDocs documentation source
|
||||
│ ├── design/overview.md # Architecture overview
|
||||
│ ├── training/ # Training guides
|
||||
│ └── contributing/ # Contributor guides + coding_agents.md
|
||||
├── tests/ # Top-level tests (local_tests/)
|
||||
├── AGENTS.md # Agent coding guidelines
|
||||
└── .agents/ # Agent infrastructure (you are here)
|
||||
```
|
||||
|
||||
## Key Training Entrypoints
|
||||
|
||||
### New framework (`fastvideo/train/`) — preferred
|
||||
|
||||
| Method | Config Example | Launch Pattern |
|
||||
|--------|---------------|----------------|
|
||||
| FineTune (Wan) | `examples/train/finetune_wan2.1_t2v_1.3B_vsa_*.yaml` | `torchrun -m fastvideo.train.entrypoint.train --config <yaml>` |
|
||||
| DFSFT (Wan causal) | `examples/train/dfsft_wan_causal_t2v_1.3B.yaml` | `torchrun -m fastvideo.train.entrypoint.train --config <yaml>` |
|
||||
| DMD2 distillation | `examples/train/distill_wan2.1_t2v_1.3B_dmd2.yaml` | `torchrun -m fastvideo.train.entrypoint.train --config <yaml>` |
|
||||
| Self-Forcing | `examples/train/self_forcing_wan_causal_t2v_1.3B.yaml` | `torchrun -m fastvideo.train.entrypoint.train --config <yaml>` |
|
||||
|
||||
### Legacy pipelines (`fastvideo/training/`) — being phased out
|
||||
|
||||
| Pipeline | Entrypoint | Launch Pattern |
|
||||
|----------|-----------|----------------|
|
||||
| Wan T2V finetune | `fastvideo/training/wan_training_pipeline.py` | `torchrun --nproc_per_node N` |
|
||||
| Wan I2V finetune | `fastvideo/training/wan_i2v_training_pipeline.py` | `torchrun --nproc_per_node N` |
|
||||
| Wan distillation (DMD) | `fastvideo/training/wan_distillation_pipeline.py` | `torchrun --nproc_per_node N` |
|
||||
| Self-forcing distill | `fastvideo/training/wan_self_forcing_distillation_pipeline.py` | `torchrun --nproc_per_node N` |
|
||||
| LTX-2 finetune | `fastvideo/training/ltx2_training_pipeline.py` | `torchrun --nproc_per_node N` |
|
||||
| MatrixGame | `fastvideo/training/matrixgame_training_pipeline.py` | `torchrun --nproc_per_node N` |
|
||||
|
||||
## W&B Integration
|
||||
|
||||
- **Tracker classes**: `fastvideo/training/trackers.py`
|
||||
- `WandbTracker` — logs metrics, videos, timing
|
||||
- `SequentialTracker` — fan-out to multiple trackers
|
||||
- `DummyTracker` — no-op for offline/test
|
||||
- **Run summary location**: `<output_dir>/tracker/wandb/latest-run/files/wandb-summary.json`
|
||||
- **Reference summaries**: `fastvideo/tests/training/*/` (e.g., `a40_reference_wandb_summary.json`)
|
||||
- **Environment**: `WANDB_API_KEY`, `WANDB_BASE_URL`, `WANDB_MODE`
|
||||
|
||||
## Critical Environment Variables
|
||||
|
||||
| Variable | Purpose |
|
||||
|----------|---------|
|
||||
| `WANDB_API_KEY` | W&B authentication |
|
||||
| `WANDB_MODE` | `online` / `offline` |
|
||||
| `FASTVIDEO_ATTENTION_BACKEND` | `FLASH_ATTN` / `TORCH_SDPA` |
|
||||
| `TOKENIZERS_PARALLELISM` | Set `false` to avoid fork warnings |
|
||||
| `HF_HOME` | HuggingFace cache directory |
|
||||
|
||||
## Build & Test Commands
|
||||
|
||||
```bash
|
||||
uv pip install -e ".[dev]" # Editable install
|
||||
pre-commit run --all-files # Lint/format/spell
|
||||
pytest tests/ # Top-level tests
|
||||
pytest fastvideo/tests/ -v # Package tests
|
||||
pytest fastvideo/tests/training/Vanilla -srP # Training loss regression
|
||||
pytest fastvideo/tests/ssim/ -vs # SSIM visual regression
|
||||
cd fastvideo-kernel && ./build.sh # Build kernels
|
||||
```
|
||||
@@ -0,0 +1,327 @@
|
||||
# Evaluation Metrics Registry
|
||||
|
||||
Living catalog of all evaluation metrics for FastVideo-WorldModel video quality
|
||||
assessment. Each metric includes a detailed explanation, implementation status,
|
||||
usage instructions, and interpretation guide.
|
||||
|
||||
_Last updated: 2026-03-02_
|
||||
|
||||
---
|
||||
|
||||
## Metric Summary
|
||||
|
||||
| Metric | Category | Status | Location | Trust |
|
||||
|--------|----------|--------|----------|-------|
|
||||
| **FVD** | Distribution | ✅ Implemented | `benchmarks/fvd/` | High |
|
||||
| **SSIM** | Reference | ✅ Implemented | `fastvideo/tests/ssim/` | High |
|
||||
| **LPIPS** | Perceptual | ✅ Implemented | `scripts/lora_extraction/` | Medium |
|
||||
| **Loss trajectory** | Training signal | ✅ Implemented | W&B `train_loss` | Medium |
|
||||
| **Grad norm stability** | Training signal | ✅ Implemented | W&B `grad_norm` | Medium |
|
||||
| **GameWorld Score** | Multi-dim benchmark | 🟡 External | Matrix-Game repo | Low |
|
||||
| **Human preference** | Gold standard | 🔴 Manual | N/A | Highest |
|
||||
|
||||
---
|
||||
|
||||
## Implemented Metrics
|
||||
|
||||
### FVD — Fréchet Video Distance
|
||||
|
||||
**Category**: Distribution-level quality metric
|
||||
**Status**: ✅ Fully implemented in `benchmarks/fvd/`
|
||||
**Trust**: High — standard protocol, I3D feature extractor
|
||||
|
||||
#### What It Measures
|
||||
FVD measures the distance between the **distribution** of generated videos and
|
||||
a distribution of real/reference videos. It works by:
|
||||
1. Extracting spatiotemporal features from both real and generated video sets
|
||||
using a pretrained **I3D** (Inflated 3D ConvNet) model.
|
||||
2. Modeling each set of features as a multivariate Gaussian (mean + covariance).
|
||||
3. Computing the **Fréchet distance** between the two Gaussians.
|
||||
|
||||
Lower FVD = generated videos are more statistically similar to real videos.
|
||||
|
||||
#### Why It Matters
|
||||
- FVD is the **de facto standard** for benchmarking video generation models.
|
||||
- It captures both **visual quality** (are individual frames realistic?) and
|
||||
**temporal coherence** (do frames flow naturally?).
|
||||
- Matrix-Game 2.0, Open-Sora, and most video generation papers report FVD.
|
||||
|
||||
#### Limitations
|
||||
- Requires a **large sample set** (standard protocol uses 2048 videos) to
|
||||
produce stable statistics. Small sample sizes yield noisy results.
|
||||
- Measures **distributional similarity**, not per-video quality. A model could
|
||||
have low FVD by generating a diverse set of "roughly okay" videos.
|
||||
- The I3D model was trained on Kinetics-400 (human actions). It may be less
|
||||
sensitive to domain-specific artifacts in non-human-action videos (e.g.,
|
||||
driving, game environments).
|
||||
- Does not directly measure text-video alignment or action controllability.
|
||||
|
||||
#### How to Use
|
||||
|
||||
```python
|
||||
# Programmatic
|
||||
from benchmarks.fvd import compute_fvd_with_config, FVDConfig
|
||||
|
||||
config = FVDConfig.fvd2048_16f() # Standard: 2048 videos, 16 frames
|
||||
results = compute_fvd_with_config('data/real/', 'outputs/gen/', config)
|
||||
print(f"FVD: {results['fvd']:.2f}")
|
||||
```
|
||||
|
||||
```bash
|
||||
# CLI
|
||||
python -m benchmarks.fvd.cli \
|
||||
--real-path data/real/ \
|
||||
--gen-path outputs/gen/ \
|
||||
--protocol fvd2048_16f
|
||||
```
|
||||
|
||||
**Preset protocols**:
|
||||
| Protocol | Videos | Frames | Use Case |
|
||||
|----------|--------|--------|----------|
|
||||
| `fvd2048_16f` | 2048 | 16 | Standard benchmark (papers) |
|
||||
| `fvd2048_128f` | 2048 | 128 | Long video evaluation |
|
||||
| `quick_test` | 100 | 16 | Fast dev iteration |
|
||||
|
||||
**Feature extractors**: `i3d` (default, standard), `clip`, `videomae`
|
||||
|
||||
#### Interpretation
|
||||
| FVD Range | Interpretation |
|
||||
|-----------|---------------|
|
||||
| < 100 | Excellent — near-real quality |
|
||||
| 100–300 | Good — competitive with SOTA |
|
||||
| 300–600 | Fair — noticeable gap from real |
|
||||
| > 600 | Poor — significant quality issues |
|
||||
|
||||
> FVD values are dataset-dependent. Always compare against baselines evaluated
|
||||
> on the same real video distribution.
|
||||
|
||||
---
|
||||
|
||||
### SSIM — Structural Similarity Index
|
||||
|
||||
**Category**: Per-frame reference comparison
|
||||
**Status**: ✅ Implemented in `fastvideo/tests/ssim/`
|
||||
**Trust**: High — used in CI regression tests
|
||||
|
||||
#### What It Measures
|
||||
SSIM compares two images (or video frames) based on three components:
|
||||
1. **Luminance**: brightness similarity
|
||||
2. **Contrast**: dynamic range similarity
|
||||
3. **Structure**: spatial pattern similarity
|
||||
|
||||
The final score is a value in [0, 1] where 1.0 = identical.
|
||||
|
||||
#### Why It Matters
|
||||
- Used as a **regression guard** in CI: ensures model updates don't degrade
|
||||
visual output below a threshold.
|
||||
- More perceptually meaningful than raw pixel MSE.
|
||||
- Fast to compute — suitable for automated testing.
|
||||
|
||||
#### Limitations
|
||||
- Requires a **pixel-aligned reference** video. Cannot compare videos with
|
||||
different seeds, prompts, or angles.
|
||||
- Operates **per-frame** — does not capture temporal coherence.
|
||||
- Insensitive to some perceptual artifacts (color shifts, high-frequency noise).
|
||||
|
||||
#### How to Use
|
||||
|
||||
```bash
|
||||
pytest fastvideo/tests/ssim/ -vs
|
||||
```
|
||||
|
||||
#### Interpretation
|
||||
| SSIM Range | Quality |
|
||||
|------------|---------|
|
||||
| > 0.90 | Excellent — very close to reference |
|
||||
| 0.80–0.90 | Good — acceptable for most uses |
|
||||
| 0.70–0.80 | Fair — noticeable differences |
|
||||
| < 0.70 | Poor — significant divergence |
|
||||
|
||||
---
|
||||
|
||||
### LPIPS — Learned Perceptual Image Patch Similarity
|
||||
|
||||
**Category**: Per-frame perceptual distance
|
||||
**Status**: ✅ Implemented in `scripts/lora_extraction/lora_inference_comparison.py`
|
||||
**Trust**: Medium — available but only used for LoRA comparison currently
|
||||
|
||||
#### What It Measures
|
||||
LPIPS uses a pretrained neural network (AlexNet by default) to extract
|
||||
deep features from two images and computes the distance between them in
|
||||
feature space. Unlike SSIM, LPIPS correlates much more strongly with
|
||||
**human perceptual judgments**.
|
||||
|
||||
Lower LPIPS = more perceptually similar.
|
||||
|
||||
#### Why It Matters
|
||||
- Best available automated proxy for **human visual judgments** at the frame
|
||||
level.
|
||||
- Captures semantic and structural differences that SSIM misses (e.g., texture
|
||||
changes, minor recoloring).
|
||||
- Used for validating LoRA merge quality.
|
||||
|
||||
#### Limitations
|
||||
- Per-frame metric — no temporal awareness.
|
||||
- Requires reference video (paired comparison only).
|
||||
- Slightly slower than SSIM due to neural network forward pass.
|
||||
|
||||
#### How to Use
|
||||
|
||||
```bash
|
||||
python scripts/lora_extraction/lora_inference_comparison.py \
|
||||
--base merged_model \
|
||||
--ft path/to/finetuned \
|
||||
--adapter NONE \
|
||||
--output-dir results \
|
||||
--prompt "A cat" \
|
||||
--compute-lpips
|
||||
```
|
||||
|
||||
#### Interpretation
|
||||
| LPIPS Range | Quality |
|
||||
|-------------|---------|
|
||||
| < 0.10 | Excellent — nearly indistinguishable |
|
||||
| 0.10–0.20 | Good — minor perceptual differences |
|
||||
| 0.20–0.40 | Fair — noticeable differences |
|
||||
| > 0.40 | Poor — clearly different |
|
||||
|
||||
---
|
||||
|
||||
### Loss Trajectory
|
||||
|
||||
**Category**: Training signal proxy
|
||||
**Status**: ✅ Active (from W&B `train_loss`)
|
||||
**Trust**: Medium — proxy, not direct quality measure
|
||||
|
||||
#### What It Measures
|
||||
Tracks the training loss over time. A healthy training run shows:
|
||||
- **Decreasing loss** over the first hundreds of steps.
|
||||
- **Stable gradient norms** (no wild spikes).
|
||||
- **Consistent step times** (no infrastructure issues).
|
||||
|
||||
#### Why It Matters
|
||||
- Cheapest evaluation signal — available in real-time from W&B.
|
||||
- Critical for the **30-minute quality check** workflow.
|
||||
- At later training stages (when loss becomes meaningful), trajectory shape
|
||||
can predict final model quality.
|
||||
|
||||
#### Context: How This Evolves
|
||||
The team's experience shows evaluation signals change during a project:
|
||||
- **Early stage**: Loss may be flat or meaningless → focus on SSIM & visual
|
||||
inspection instead.
|
||||
- **Mid stage**: Loss starts decreasing → trajectory shape becomes useful.
|
||||
- **Late stage**: Loss is meaningful → can compare trajectories across runs.
|
||||
|
||||
This dynamic is a key insight from the team's workflow: don't over-rely on
|
||||
loss early; don't ignore it late.
|
||||
|
||||
---
|
||||
|
||||
### Grad Norm Stability
|
||||
|
||||
**Category**: Training health diagnostic
|
||||
**Status**: ✅ Active (from W&B `grad_norm`)
|
||||
**Trust**: Medium — diagnostic, not quality metric
|
||||
|
||||
#### What It Measures
|
||||
The magnitude of gradients during training. Stable grad norms indicate
|
||||
healthy optimization. Spikes or NaN values indicate training instability.
|
||||
|
||||
#### Alert Thresholds
|
||||
| Condition | Meaning |
|
||||
|-----------|---------|
|
||||
| Stable ~0.3–0.5 | Normal training |
|
||||
| Single spike > 3× average | Possible bad batch, monitor |
|
||||
| NaN or Inf | 🔴 Training has diverged — stop run |
|
||||
| Increasing trend | Learning rate may be too high |
|
||||
|
||||
---
|
||||
|
||||
## External Benchmarks
|
||||
|
||||
### GameWorld Score Benchmark (Matrix-Game)
|
||||
|
||||
**Category**: Multi-dimensional evaluation framework for interactive world models
|
||||
**Status**: 🟡 External — not implemented in-repo
|
||||
**Source**: [Matrix-Game 1.0 benchmark](https://github.com/SkyworkAI/Matrix-Game), used in [Matrix-Game 2.0 paper](https://arxiv.org/abs/2508.13009)
|
||||
|
||||
#### What It Measures
|
||||
A comprehensive benchmark examining **four critical capabilities**:
|
||||
|
||||
| Dimension | What It Evaluates | Example Signals |
|
||||
|-----------|-------------------|-----------------|
|
||||
| **Visual quality** | Frame-level realism, absence of artifacts | Color fidelity, sharpness, coherence |
|
||||
| **Temporal quality** | Smoothness across frames, motion consistency | Jitter, flickering, temporal aliasing |
|
||||
| **Action controllability** | Response to input actions (keyboard/mouse) | Action delay, correctness, smoothness |
|
||||
| **Physical rule understanding** | Adherence to physics (gravity, collision) | Object persistence, plausible motion |
|
||||
|
||||
#### Context from Matrix-Game 2.0
|
||||
- Evaluation uses **597-frame composite action sequences** over 32 Minecraft
|
||||
scenes and 16 wild scenes.
|
||||
- Action controllability assessment is **Minecraft-specific** — cannot be
|
||||
directly applied to wild/general scenes.
|
||||
- The paper notes that models that "collapse" to static frames can
|
||||
paradoxically score higher on consistency metrics — beware of this confound.
|
||||
|
||||
#### Relevance to FastVideo
|
||||
- Matrix-Game 2.0 is built on SkyReels-V2/Wan2.1 architecture — **same model
|
||||
family as FastVideo**.
|
||||
- Their distillation uses DMD-based Self-Forcing — **same technique** as our
|
||||
`self_forcing_distillation_pipeline.py`.
|
||||
- GameWorld Score dimensions are a useful framework for thinking about world
|
||||
model quality even outside gaming contexts.
|
||||
|
||||
---
|
||||
|
||||
## Human Preference Evaluation
|
||||
|
||||
**Category**: Gold-standard quality assessment
|
||||
**Status**: 🔴 Manual process — no automated implementation
|
||||
**Priority**: **Highest** — this is the most important evaluation signal
|
||||
**Trust**: Highest — but expensive
|
||||
|
||||
### What It Measures
|
||||
Human evaluators compare generated videos and rate them on dimensions like:
|
||||
- Overall quality and realism
|
||||
- Temporal coherence and smoothness
|
||||
- Prompt adherence / action correctness
|
||||
- Absence of artifacts
|
||||
|
||||
#### Why It's the Most Important Metric
|
||||
All automated metrics are **proxies** for human judgment. They can be gamed
|
||||
or may miss artifacts that humans easily notice. Human preference is the
|
||||
ultimate ground truth for video generation quality.
|
||||
|
||||
#### Cost & Practicality
|
||||
| Approach | Cost | Scale | When to Use |
|
||||
|----------|------|-------|-------------|
|
||||
| Internal team review | Low | ~10–50 videos | Every major checkpoint |
|
||||
| Crowdsource (MTurk, Scale) | Medium | 100+ videos | Pre-release validation |
|
||||
| A/B preference test | Medium | Pairs | Comparing two model versions |
|
||||
|
||||
#### Recommended Protocol
|
||||
1. Sample 10–20 videos from the model at a checkpoint.
|
||||
2. Include diverse prompts (easy + hard, short + long).
|
||||
3. Have 2–3 evaluators score each video 1–5 on: quality, coherence, fidelity.
|
||||
4. Record scores in the experiment journal.
|
||||
|
||||
---
|
||||
|
||||
## Metrics NOT Used
|
||||
|
||||
| Metric | Reason |
|
||||
|--------|--------|
|
||||
| ~~CLIP-Score~~ | Not used by the team. Measures text-image alignment using CLIP embeddings, but not well-suited for video temporal quality. |
|
||||
| Inception Score (IS) | Less informative than FVD for video; primarily an image metric. |
|
||||
| PSNR | Pixel-level metric; less perceptually meaningful than SSIM/LPIPS. |
|
||||
|
||||
---
|
||||
|
||||
## Adding a New Metric
|
||||
|
||||
Follow the SOP: `.agents/workflows/evaluation-development.md`
|
||||
|
||||
1. Prototype in `.agents/exploration/`
|
||||
2. Validate on known-good and known-bad samples
|
||||
3. Add to this registry
|
||||
4. Update the `evaluate-video-quality` skill
|
||||
@@ -0,0 +1,21 @@
|
||||
# Experiment Journal
|
||||
|
||||
Living log of all experiments. Each entry captures what was tried, the result,
|
||||
and any insights. Newest entries go at the top.
|
||||
|
||||
_No experiments logged yet. Use the `log-experiment` skill to add entries._
|
||||
|
||||
<!-- TEMPLATE — copy and fill for each new experiment:
|
||||
|
||||
## [YYYY-MM-DD] Experiment: <name>
|
||||
- **Hypothesis**: <what you expected to learn>
|
||||
- **Config**: model=..., lr=..., sp_size=..., gpus=..., script=...
|
||||
- **W&B run**: <run_id or URL>
|
||||
- **Duration**: <total wall time>
|
||||
- **Key metrics**: loss=..., step_time=..., grad_norm=...
|
||||
- **Checkpoint**: <path>
|
||||
- **Insight**: <what was learned>
|
||||
- **Status**: running | completed | failed | abandoned
|
||||
- **Related lessons**: `.agents/lessons/<filename>.md`
|
||||
|
||||
-->
|
||||
@@ -0,0 +1,4 @@
|
||||
{"name": "codebase-map", "description": "High-level structural index of the FastVideo-WorldModel repository", "path": "codebase-map/README.md", "status": "ready", "trust": "high"}
|
||||
{"name": "evaluation-registry", "description": "Catalog of all evaluation metrics with detailed explanations, implementation status, and usage guides", "path": "evaluation-registry/README.md", "status": "draft", "trust": "medium"}
|
||||
{"name": "experiment-journal", "description": "Living log of all experiments with hypotheses, configs, metrics, and insights", "path": "experiment-journal/README.md", "status": "draft", "trust": "medium"}
|
||||
{"name": "related-work", "description": "Index of related papers, repos, and blog posts with structured comparisons to FastVideo", "path": "related-work/README.md", "status": "draft", "trust": "low"}
|
||||
@@ -0,0 +1,34 @@
|
||||
# Related Work Index
|
||||
|
||||
Each file in this directory is a structured summary of a related paper, repo,
|
||||
or blog post relevant to FastVideo-WorldModel training.
|
||||
|
||||
## File Format
|
||||
|
||||
Each file is named `<slug>.md` and follows this structure:
|
||||
|
||||
```markdown
|
||||
---
|
||||
title: <paper/repo title>
|
||||
source: <URL or citation>
|
||||
type: paper | repo | blog
|
||||
date_indexed: <ISO-8601>
|
||||
tags: [world-model, distillation, evaluation, reward-shaping, ...]
|
||||
---
|
||||
|
||||
## Summary
|
||||
<1-2 paragraph summary of the work.>
|
||||
|
||||
## Key Differences from FastVideo
|
||||
- <Bullet points comparing their approach to ours.>
|
||||
|
||||
## Actionable Insights
|
||||
- <What we could adopt or adapt.>
|
||||
```
|
||||
|
||||
## How to Add New Entries
|
||||
|
||||
Use the `index-related-work` skill, or manually create a file following the
|
||||
template above.
|
||||
|
||||
_No related work indexed yet._
|
||||
@@ -0,0 +1,76 @@
|
||||
# Agent Onboarding — FastVideo-WorldModel
|
||||
|
||||
Welcome, agent. This is the **master onboarding** guide. Follow the steps below,
|
||||
then check if a **domain-specific onboarding** exists for your task.
|
||||
|
||||
## Domain-Specific Onboarding
|
||||
|
||||
If your task falls into one of these areas, read the specialized guide **after**
|
||||
completing the general steps below:
|
||||
|
||||
| Domain | Guide | When to Use |
|
||||
|--------|-------|-------------|
|
||||
| **WorldModel Training** | `worldmodel-training/README.md` | Training, finetuning, distillation, experiment management |
|
||||
|
||||
---
|
||||
|
||||
## Step 1: Understand the Codebase
|
||||
|
||||
Read these files to build your context:
|
||||
|
||||
| Priority | File | What you learn |
|
||||
|----------|------|----------------|
|
||||
| 1 | `AGENTS.md` | Coding guidelines, build/test commands, PR conventions |
|
||||
| 2 | `docs/design/overview.md` | Architecture: models, pipelines, configs, registry |
|
||||
| 3 | `fastvideo/train/` | Refactored training framework (YAML-driven, modular methods/models/callbacks) |
|
||||
| 4 | `docs/training/overview.md` | Training data flow and preprocessing |
|
||||
| 5 | `docs/training/finetune.md` | Training arguments, parallelism, LoRA, validation |
|
||||
| 6 | `docs/contributing/coding_agents.md` | How to add model pipelines with agent assistance |
|
||||
|
||||
## Step 2: Discover Available Resources
|
||||
|
||||
Read these two index files to see what skills and memory modules exist:
|
||||
|
||||
- **`.agents/skills/index.jsonl`** — catalog of all agent skills (name + description)
|
||||
- **`.agents/memory/index.jsonl`** — catalog of all memory modules (name + description)
|
||||
|
||||
Each entry has a `path` field pointing to the full content. Only load the
|
||||
full README.md for modules relevant to your current task.
|
||||
|
||||
## Step 3: Check for Existing Skills & SOPs
|
||||
|
||||
Before writing new code or procedures:
|
||||
|
||||
1. **Skills**: Read `.agents/skills/index.jsonl` — find a matching skill by description.
|
||||
2. **Workflows/SOPs**: Browse `.agents/workflows/` — step-by-step procedures for common tasks.
|
||||
3. **Lessons**: Browse `.agents/lessons/` — known pitfalls and their fixes.
|
||||
|
||||
If a skill or SOP exists for your task, **use it**. If not, you are in **exploration mode** — see Step 4.
|
||||
|
||||
## Step 4: Exploration Mode
|
||||
|
||||
If no existing skill/SOP covers your task:
|
||||
|
||||
1. Document your progress in `.agents/exploration/<topic>.md` using the template in `.agents/exploration/README.md`.
|
||||
2. At the end of your session, reflect:
|
||||
- **What worked** → propose a new skill or SOP in the exploration log.
|
||||
- **What failed** → create a lesson in `.agents/lessons/`.
|
||||
3. Flag the exploration log for human review.
|
||||
|
||||
## Quick Reference
|
||||
|
||||
```
|
||||
.agents/
|
||||
├── ONBOARDING.md ← you are here
|
||||
├── STATUS.md ← dashboard: completeness & trust of all components
|
||||
├── skills/ ← reusable agent skills
|
||||
├── workflows/ ← SOPs and procedures
|
||||
├── memory/ ← persistent context (folder per topic + index.jsonl)
|
||||
│ ├── index.jsonl
|
||||
│ ├── codebase-map/
|
||||
│ ├── experiment-journal/
|
||||
│ ├── evaluation-registry/
|
||||
│ └── related-work/
|
||||
├── lessons/ ← mistakes and fixes
|
||||
└── exploration/ ← draft procedures
|
||||
```
|
||||
@@ -0,0 +1,302 @@
|
||||
# WorldModel Training — Agent Onboarding
|
||||
|
||||
Specialized onboarding for agents working on FastVideo-WorldModel training,
|
||||
distillation, and evaluation. Read the master onboarding (`.agents/onboarding/README.md`)
|
||||
first, then come here.
|
||||
|
||||
---
|
||||
|
||||
## Domain Context
|
||||
|
||||
FastVideo-WorldModel trains **interactive world models** — video generation systems
|
||||
that respond to user actions (keyboard/mouse) in real-time. The architecture is
|
||||
based on **Wan2.1** (SkyReels-V2) DiT models with causal attention for
|
||||
auto-regressive streaming generation.
|
||||
|
||||
**Key techniques you will work with:**
|
||||
- Full finetuning and LoRA on Wan / LTX-2 / MatrixGame models
|
||||
- DMD-based distillation (few-step generation)
|
||||
- Self-Forcing distillation (causal streaming)
|
||||
- Diffusion-Forcing SFT (DFSFT) for causal models
|
||||
- VSA (Variable Sparsity Acceleration) for efficient training
|
||||
|
||||
---
|
||||
|
||||
## Training Code: Two Generations
|
||||
|
||||
### New modular framework: `fastvideo/train/` (preferred)
|
||||
|
||||
The refactored training code uses a **YAML-only config-driven** architecture
|
||||
with composable methods, per-role models, and a callback system. All new
|
||||
training work should use this framework.
|
||||
|
||||
### Legacy pipelines: `fastvideo/training/` (deprecated)
|
||||
|
||||
The old monolithic pipeline classes (`WanTrainingPipeline`,
|
||||
`DistillationPipeline`, etc.) still exist but are being phased out. The new
|
||||
framework imports select utilities from `fastvideo/training/` for backward
|
||||
compatibility (EMA, gradient clipping, checkpoint wrappers).
|
||||
|
||||
---
|
||||
|
||||
## Essential Reading (Training-Specific)
|
||||
|
||||
Read these **in order** before touching any training code:
|
||||
|
||||
| # | File | What You Learn |
|
||||
|---|------|----------------|
|
||||
| 1 | `docs/training/overview.md` | Training data flow: raw video → text embeddings + video latents → training |
|
||||
| 2 | `docs/training/finetune.md` | Training arguments, parallelism (SP/TP), LoRA, validation settings |
|
||||
| 3 | `docs/training/data_preprocess.md` | How to preprocess datasets into the expected format |
|
||||
| 4 | `docs/design/overview.md` | Architecture: models, pipelines, configs, registry |
|
||||
|
||||
---
|
||||
|
||||
## New Training Framework (`fastvideo/train/`)
|
||||
|
||||
### Architecture Overview
|
||||
|
||||
```
|
||||
fastvideo/train/
|
||||
├── __init__.py → exports Trainer
|
||||
├── trainer.py → main training loop coordinator
|
||||
├── entrypoint/
|
||||
│ ├── train.py → YAML-only training entrypoint
|
||||
│ └── dcp_to_diffusers.py → checkpoint conversion utility
|
||||
├── methods/ → training algorithms (TrainingMethod ABC)
|
||||
│ ├── base.py → TrainingMethod base class
|
||||
│ ├── fine_tuning/
|
||||
│ │ ├── finetune.py → FineTuneMethod (supervised finetuning)
|
||||
│ │ └── dfsft.py → DiffusionForcingSFTMethod (causal)
|
||||
│ ├── distribution_matching/
|
||||
│ │ ├── dmd2.py → DMD2Method (distribution matching distill)
|
||||
│ │ └── self_forcing.py → SelfForcingMethod (causal streaming)
|
||||
│ ├── knowledge_distillation/ → (stub, not yet implemented)
|
||||
│ └── consistency_model/ → (stub, not yet implemented)
|
||||
├── models/ → per-role model instances
|
||||
│ ├── base.py → ModelBase & CausalModelBase (ABC)
|
||||
│ └── wan/
|
||||
│ ├── wan.py → WanModel (non-causal)
|
||||
│ └── wan_causal.py → WanCausalModel (causal streaming)
|
||||
├── callbacks/ → training hooks & monitoring
|
||||
│ ├── callback.py → Callback base class + CallbackDict
|
||||
│ ├── grad_clip.py → GradNormClipCallback
|
||||
│ ├── ema.py → EMACallback (shadow weights)
|
||||
│ └── validation.py → ValidationCallback (sampling + eval)
|
||||
└── utils/ → configuration, building, checkpointing
|
||||
├── builder.py → build_from_config() (config → runtime)
|
||||
├── checkpoint.py → CheckpointManager (DCP-based)
|
||||
├── config.py → load_run_config() (YAML → RunConfig)
|
||||
├── training_config.py → TypedConfig dataclasses
|
||||
├── optimizer.py → build_optimizer_and_scheduler()
|
||||
├── instantiate.py → resolve_target() + instantiate()
|
||||
├── tracking.py → build_tracker() (W&B, etc.)
|
||||
├── dataloader.py → dataloader utilities
|
||||
├── module_state.py → apply_trainable()
|
||||
└── moduleloader.py → load_module_from_path()
|
||||
```
|
||||
|
||||
### Key Concepts
|
||||
|
||||
**TrainingMethod** (`methods/base.py`): Abstract base class for all training
|
||||
algorithms. Owns role models (student, teacher, critic), manages checkpoint
|
||||
state, and defines the training step interface.
|
||||
|
||||
**ModelBase** (`models/base.py`): Per-role model wrapper. Each role (student,
|
||||
teacher, critic) gets its own `ModelBase` instance owning a `transformer` and
|
||||
`noise_scheduler`. `CausalModelBase` extends this for streaming models.
|
||||
|
||||
**Callback system** (`callbacks/`): Composable hooks for gradient clipping,
|
||||
EMA, validation, etc. Configured via YAML, dispatched by `CallbackDict`.
|
||||
|
||||
**Config system** (`utils/config.py`, `utils/training_config.py`): YAML files
|
||||
are parsed into typed `RunConfig` dataclass trees. Models and methods use
|
||||
`_target_` fields for instantiation (similar to Hydra).
|
||||
|
||||
### Training Flow
|
||||
|
||||
```
|
||||
run_training_from_config(config_path)
|
||||
→ load_run_config() # YAML → RunConfig
|
||||
→ init_distributed() # TP/SP setup
|
||||
→ build_from_config() # instantiate models, method, dataloader
|
||||
→ Trainer.run() # main loop:
|
||||
├─ callbacks.on_train_start()
|
||||
├─ checkpoint_manager.maybe_resume()
|
||||
├─ for step in range(max_steps):
|
||||
│ ├─ method.single_train_step(batch)
|
||||
│ ├─ method.backward()
|
||||
│ ├─ callbacks.on_before_optimizer_step()
|
||||
│ ├─ method.optimizers_schedulers_step()
|
||||
│ ├─ tracker.log(metrics, step)
|
||||
│ ├─ callbacks.on_training_step_end()
|
||||
│ └─ checkpoint_manager.maybe_save(step)
|
||||
├─ callbacks.on_train_end()
|
||||
└─ checkpoint_manager.save_final()
|
||||
```
|
||||
|
||||
### Training Methods
|
||||
|
||||
| Method | Class | Use Case |
|
||||
|--------|-------|----------|
|
||||
| **FineTune** | `FineTuneMethod` | Single-role supervised finetuning |
|
||||
| **DFSFT** | `DiffusionForcingSFTMethod` | Diffusion-forcing SFT with inhomogeneous timesteps |
|
||||
| **DMD2** | `DMD2Method` | Multi-role distribution matching distillation (student + teacher + critic) |
|
||||
| **Self-Forcing** | `SelfForcingMethod` | Extends DMD2 for causal student rollouts |
|
||||
|
||||
### Launching Training (New Framework)
|
||||
|
||||
Training is launched via `torchrun` with a single YAML config:
|
||||
|
||||
```bash
|
||||
torchrun --nproc_per_node <N_GPUS> \
|
||||
-m fastvideo.train.entrypoint.train \
|
||||
--config examples/train/<config>.yaml
|
||||
```
|
||||
|
||||
### Example YAML Configs
|
||||
|
||||
| Config | Method | Description |
|
||||
|--------|--------|-------------|
|
||||
| `examples/train/finetune_wan2.1_t2v_1.3B_vsa_phase3.4_0.9sparsity.yaml` | FineTune | Wan 1.3B finetuning with VSA sparsity |
|
||||
| `examples/train/distill_wan2.1_t2v_1.3B_dmd2.yaml` | DMD2 | Wan 1.3B distillation (student + teacher + critic) |
|
||||
| `examples/train/dfsft_wan_causal_t2v_1.3B.yaml` | DFSFT | Causal Wan 1.3B diffusion-forcing SFT |
|
||||
| `examples/train/self_forcing_wan_causal_t2v_1.3B.yaml` | Self-Forcing | Causal streaming distillation |
|
||||
|
||||
### Checkpointing (New Framework)
|
||||
|
||||
**CheckpointManager** (`utils/checkpoint.py`) saves via `torch.distributed.checkpoint`:
|
||||
|
||||
```
|
||||
output_dir/
|
||||
└─ checkpoint-{step}/
|
||||
├─ dcp/ # DCP state dict
|
||||
├─ config.json # resolved training config
|
||||
└─ .fastvideo_metadata.json
|
||||
```
|
||||
|
||||
Checkpoint state includes: role model weights, per-role optimizers/schedulers,
|
||||
CUDA RNG state, and callback state (e.g., EMA shadow weights).
|
||||
|
||||
### Config Structure
|
||||
|
||||
A YAML config defines the full training pipeline:
|
||||
|
||||
```yaml
|
||||
models:
|
||||
student:
|
||||
_target_: fastvideo.train.models.wan.WanModel
|
||||
model_path: ...
|
||||
trainable: true
|
||||
teacher: # optional, for distillation
|
||||
_target_: fastvideo.train.models.wan.WanModel
|
||||
model_path: ...
|
||||
trainable: false
|
||||
|
||||
method:
|
||||
_target_: fastvideo.train.methods.fine_tuning.FineTuneMethod
|
||||
# method-specific params...
|
||||
|
||||
training:
|
||||
distributed: { num_gpus: 8, tp_size: 1, sp_size: 8 }
|
||||
data: { data_path: ..., batch_size: 1 }
|
||||
optimizer: { lr: 1e-5, lr_scheduler: constant_with_warmup }
|
||||
loop: { max_train_steps: 1000 }
|
||||
checkpoint: { output_dir: ./outputs }
|
||||
tracker: { trackers: [wandb], project_name: ... }
|
||||
|
||||
callbacks:
|
||||
grad_clip:
|
||||
_target_: fastvideo.train.callbacks.GradNormClipCallback
|
||||
max_grad_norm: 1.0
|
||||
validation:
|
||||
_target_: fastvideo.train.callbacks.ValidationCallback
|
||||
validation_steps: 100
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Legacy Training Pipelines (`fastvideo/training/`)
|
||||
|
||||
> **Note:** Use the new `fastvideo/train/` framework for new work. This section
|
||||
> is retained for reference on existing pipelines not yet migrated.
|
||||
|
||||
| Pipeline | Entrypoint | Use Case |
|
||||
|----------|-----------|----------|
|
||||
| Wan T2V finetune | `fastvideo/training/wan_training_pipeline.py` | Standard text-to-video finetune / LoRA |
|
||||
| Wan I2V finetune | `fastvideo/training/wan_i2v_training_pipeline.py` | Image-to-video (first frame conditioned) |
|
||||
| MatrixGame finetune | `fastvideo/training/matrixgame_training_pipeline.py` | Action-conditioned world model |
|
||||
| LTX-2 finetune | `fastvideo/training/ltx2_training_pipeline.py` | LTX-2 architecture finetuning |
|
||||
| Wan DMD distillation | `fastvideo/training/wan_distillation_pipeline.py` | Few-step distillation via DMD |
|
||||
| Self-Forcing distill | `fastvideo/training/wan_self_forcing_distillation_pipeline.py` | Causal streaming distillation |
|
||||
|
||||
---
|
||||
|
||||
## Key Infrastructure
|
||||
|
||||
### W&B Integration
|
||||
- **Tracker**: `fastvideo/training/trackers.py` — `WandbTracker` class
|
||||
- **New framework tracker**: `fastvideo/train/utils/tracking.py` — `build_tracker()`
|
||||
- **Env vars**: `WANDB_API_KEY`, `WANDB_BASE_URL`, `WANDB_MODE`
|
||||
|
||||
### Parallelism
|
||||
- **SP** (Sequence Parallel): splits video frames across GPUs — `sp_size: N`
|
||||
- **TP** (Tensor Parallel): splits model layers across GPUs — `tp_size: N`
|
||||
- Typical configs: SP=2–8, TP=1–2
|
||||
|
||||
---
|
||||
|
||||
## Evaluation (for training runs)
|
||||
|
||||
Read `.agents/memory/evaluation-registry/README.md` for the full metric catalog.
|
||||
|
||||
**Quick summary for training agents:**
|
||||
| Metric | When to Use | Trust |
|
||||
|--------|-------------|-------|
|
||||
| **Loss trajectory** | Every run, real-time from W&B | Medium |
|
||||
| **SSIM** | When comparing against reference outputs | High |
|
||||
| **FVD** | For benchmarking model quality (`benchmarks/fvd/`) | High |
|
||||
| **LPIPS** | LoRA merge validation | Medium |
|
||||
| **Human preference** | Major checkpoints | Highest |
|
||||
|
||||
---
|
||||
|
||||
## Common Workflows
|
||||
|
||||
| Task | Skill / SOP |
|
||||
|------|-------------|
|
||||
| Launch a training run | `.agents/skills/launch-experiment/SKILL.md` |
|
||||
| Monitor a running experiment | `.agents/skills/monitor-experiment/SKILL.md` |
|
||||
| Summarize final results | `.agents/skills/summarize-run/SKILL.md` |
|
||||
| Full experiment lifecycle | `.agents/workflows/experiment-lifecycle.md` |
|
||||
| Capture lessons from failures | `.agents/workflows/lesson-capture.md` |
|
||||
|
||||
---
|
||||
|
||||
## World Model–Specific Concepts
|
||||
|
||||
### Action Injection (MatrixGame)
|
||||
The MatrixGame pipeline adds **action modules** to each DiT block, enabling
|
||||
frame-level mouse/keyboard input conditioning. The action sequence is injected
|
||||
per-frame alongside the latent video tokens.
|
||||
|
||||
### Causal Architecture
|
||||
For streaming generation, the model uses **causal attention** (each frame only
|
||||
attends to previous frames). This enables auto-regressive chunk-by-chunk
|
||||
generation — critical for real-time interactive world models.
|
||||
|
||||
### Self-Forcing Distillation
|
||||
A **data-free** distillation method where the student model is trained to
|
||||
generate coherent video sequences by being forced to use its own previous
|
||||
outputs (rather than ground-truth) as context. This produces models robust to
|
||||
their own error accumulation during long auto-regressive generation.
|
||||
|
||||
### DMD Distillation (Distribution Matching Distillation)
|
||||
Reduces inference steps from ~50 to 3–4 by training a student model to match
|
||||
the output distribution of the teacher model. Uses a critic network to estimate
|
||||
distribution divergence.
|
||||
|
||||
### Diffusion-Forcing SFT (DFSFT)
|
||||
Supervised finetuning with **inhomogeneous timesteps** across chunks — each
|
||||
chunk in a causal sequence can have a different noise level, training the model
|
||||
to handle mixed-fidelity contexts.
|
||||
@@ -0,0 +1,96 @@
|
||||
#!/usr/bin/env bash
|
||||
# Sync .agents/skills/ into .claude/skills/ via per-skill symlinks.
|
||||
#
|
||||
# Why: Claude Code only scans .claude/skills/ and ~/.claude/skills/ for
|
||||
# user-invocable skills (no skillsPath config exists — see
|
||||
# https://code.claude.com/docs/en/skills.md). This repo's skills live
|
||||
# in .agents/skills/ so they travel with the repo and stay under git.
|
||||
# Run this once after cloning (or after adding/removing a skill) to
|
||||
# expose them to Claude Code without maintaining a parallel tree.
|
||||
#
|
||||
# Usage:
|
||||
# .agents/scripts/sync-skills.sh
|
||||
#
|
||||
# Idempotent and safe to re-run. Prunes stale symlinks whose source
|
||||
# has been removed from .agents/skills/. Leaves hand-written
|
||||
# .claude/skills/<name>/ directories untouched (only symlinks are
|
||||
# managed).
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
REPO_ROOT="$(git -C "$(dirname "$0")" rev-parse --show-toplevel)"
|
||||
SRC_DIR="$REPO_ROOT/.agents/skills"
|
||||
DST_DIR="$REPO_ROOT/.claude/skills"
|
||||
|
||||
if [[ ! -d "$SRC_DIR" ]]; then
|
||||
echo "Error: $SRC_DIR does not exist." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
mkdir -p "$DST_DIR"
|
||||
|
||||
linked=0
|
||||
unchanged=0
|
||||
skipped=0
|
||||
pruned=0
|
||||
|
||||
link_skill() {
|
||||
local name="$1"
|
||||
local src="$SRC_DIR/$name"
|
||||
local dst="$DST_DIR/$name"
|
||||
# Relative target keeps symlinks portable across clones.
|
||||
local rel="../../.agents/skills/$name"
|
||||
|
||||
if [[ -L "$dst" ]]; then
|
||||
if [[ "$(readlink "$dst")" == "$rel" ]]; then
|
||||
unchanged=$((unchanged + 1))
|
||||
return
|
||||
fi
|
||||
rm "$dst"
|
||||
elif [[ -e "$dst" ]]; then
|
||||
echo "Skipped (not a symlink): .claude/skills/$name" >&2
|
||||
skipped=$((skipped + 1))
|
||||
return
|
||||
fi
|
||||
|
||||
ln -s "$rel" "$dst"
|
||||
echo "Linked: .claude/skills/$name -> $rel"
|
||||
linked=$((linked + 1))
|
||||
}
|
||||
|
||||
prune_stale() {
|
||||
local link="$1"
|
||||
local target
|
||||
target="$(readlink "$link")"
|
||||
case "$target" in
|
||||
../../.agents/skills/*) ;;
|
||||
*) return ;;
|
||||
esac
|
||||
local name="${target##*/}"
|
||||
if [[ ! -d "$SRC_DIR/$name" ]]; then
|
||||
rm "$link"
|
||||
echo "Pruned stale: .claude/skills/$(basename "$link")"
|
||||
pruned=$((pruned + 1))
|
||||
fi
|
||||
}
|
||||
|
||||
for src in "$SRC_DIR"/*/; do
|
||||
[[ -d "$src" ]] || continue
|
||||
name="$(basename "$src")"
|
||||
# Only treat directories that actually contain a SKILL.md as skills.
|
||||
[[ -f "$src/SKILL.md" ]] || continue
|
||||
link_skill "$name"
|
||||
done
|
||||
|
||||
shopt -s nullglob
|
||||
for link in "$DST_DIR"/*; do
|
||||
[[ -L "$link" ]] || continue
|
||||
prune_stale "$link"
|
||||
done
|
||||
shopt -u nullglob
|
||||
|
||||
printf "\nSummary: %d linked, %d unchanged, %d pruned" "$linked" "$unchanged" "$pruned"
|
||||
if [[ "$skipped" -gt 0 ]]; then
|
||||
printf ", %d skipped (non-symlink collision)" "$skipped"
|
||||
fi
|
||||
printf "\n"
|
||||
@@ -0,0 +1,57 @@
|
||||
---
|
||||
name: <skill-name>
|
||||
description: <one-line description — Codex uses this for implicit invocation matching>
|
||||
---
|
||||
|
||||
# <Skill Name>
|
||||
|
||||
## Purpose
|
||||
<Why this skill exists and when to use it.>
|
||||
|
||||
## Prerequisites
|
||||
- <What must be true before using this skill>
|
||||
|
||||
## Inputs
|
||||
| Parameter | Required | Description |
|
||||
|-----------|----------|-------------|
|
||||
| `param1` | Yes | ... |
|
||||
|
||||
## Steps
|
||||
|
||||
1. **Step 1 title**
|
||||
- Detail...
|
||||
|
||||
2. **Step 2 title**
|
||||
- Detail...
|
||||
|
||||
## Outputs
|
||||
- <What this skill produces>
|
||||
|
||||
## Example Usage
|
||||
|
||||
```
|
||||
<Example invocation or prompt snippet>
|
||||
```
|
||||
|
||||
## References
|
||||
- <Links to relevant files in the codebase>
|
||||
|
||||
---
|
||||
|
||||
## Folder Structure
|
||||
|
||||
Each skill lives in its own directory under `.agents/skills/`:
|
||||
|
||||
```
|
||||
.agents/skills/<skill-name>/
|
||||
├── SKILL.md # Required: instructions + metadata (this file)
|
||||
├── scripts/ # Optional: executable helper scripts
|
||||
├── references/ # Optional: documentation, papers
|
||||
└── assets/ # Optional: templates, resources
|
||||
```
|
||||
|
||||
After creating a new skill, add an entry to `.agents/skills/index.jsonl`:
|
||||
|
||||
```json
|
||||
{"name": "<skill-name>", "description": "<description>", "path": "<skill-name>/SKILL.md", "status": "draft", "trust": "low"}
|
||||
```
|
||||
@@ -0,0 +1,128 @@
|
||||
---
|
||||
name: evaluate-video-quality
|
||||
description: Evaluate generated video quality using available metrics (SSIM, loss trajectory, caption consistency)
|
||||
---
|
||||
|
||||
# Evaluate Video Quality
|
||||
|
||||
## Purpose
|
||||
Assess the quality of videos generated by a training run. Combines multiple
|
||||
signals to give a holistic quality assessment. This skill is **evolving** —
|
||||
new metrics will be added as they are developed.
|
||||
|
||||
## Prerequisites
|
||||
- Generated videos available locally or via W&B artifacts.
|
||||
- For SSIM: reference videos from official implementations.
|
||||
- For caption consistency: LLM access (optional, stub for now).
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Required | Description |
|
||||
|-----------|----------|-------------|
|
||||
| `video_paths` | Yes | List of paths to generated videos |
|
||||
| `reference_paths` | No | Paths to reference videos (for SSIM) |
|
||||
| `prompts` | No | Prompts used to generate videos (for caption check) |
|
||||
| `loss_summary` | No | Path to W&B summary JSON (for loss trajectory) |
|
||||
| `metrics` | No | Which metrics to run (default: all available) |
|
||||
|
||||
## Available Metrics
|
||||
|
||||
Check `.agents/memory/evaluation-registry/README.md` for the current catalog.
|
||||
|
||||
### SSIM (Active)
|
||||
|
||||
Leverages the existing infrastructure in `fastvideo/tests/ssim/`.
|
||||
|
||||
```bash
|
||||
pytest fastvideo/tests/ssim/ -vs --video-path <generated> --reference-path <reference>
|
||||
```
|
||||
|
||||
Or use the SSIM utility directly:
|
||||
|
||||
```python
|
||||
from fastvideo.tests.ssim.ssim_utils import compute_ssim
|
||||
score = compute_ssim(generated_video, reference_video)
|
||||
# score > 0.85 is typically "acceptable"
|
||||
```
|
||||
|
||||
**Interpretation**:
|
||||
| SSIM Range | Quality |
|
||||
|------------|---------|
|
||||
| > 0.90 | Excellent — very close to reference |
|
||||
| 0.80–0.90 | Good — acceptable for most uses |
|
||||
| 0.70–0.80 | Fair — noticeable differences |
|
||||
| < 0.70 | Poor — significant quality issues |
|
||||
|
||||
### Loss Trajectory (Active)
|
||||
|
||||
Analyze the loss curve shape from W&B summary:
|
||||
|
||||
```python
|
||||
import json
|
||||
with open(loss_summary_path) as f:
|
||||
summary = json.load(f)
|
||||
|
||||
final_loss = summary["train_loss"]
|
||||
runtime = summary["_runtime"]
|
||||
steps = summary["_step"]
|
||||
```
|
||||
|
||||
**Early-stage heuristics** (first 500 steps):
|
||||
- Loss should be decreasing (even slightly).
|
||||
- Grad norm should be stable (no wild oscillations).
|
||||
- If loss is flat or increasing, flag for review.
|
||||
|
||||
### Caption Consistency (Draft — Not Yet Calibrated)
|
||||
|
||||
Use an LLM to evaluate whether the video content matches the input prompt.
|
||||
|
||||
```
|
||||
Prompt: "A golden retriever playing in the snow"
|
||||
Video: <path>
|
||||
|
||||
Score the video on:
|
||||
1. Object presence (is there a golden retriever?)
|
||||
2. Action accuracy (is it playing?)
|
||||
3. Environment match (is there snow?)
|
||||
4. Overall coherence (does it look natural?)
|
||||
|
||||
Each 1-5, total /20.
|
||||
```
|
||||
|
||||
> ⚠️ This metric is in **draft** status. Results should not be treated as
|
||||
> ground truth until calibrated against human judgments.
|
||||
|
||||
## Steps
|
||||
|
||||
1. **Identify available metrics** — Check `.agents/memory/evaluation-registry/README.md`.
|
||||
2. **Run each metric** — Collect scores.
|
||||
3. **Aggregate** — Produce a combined quality report.
|
||||
4. **Log** — Update the experiment journal with quality results.
|
||||
|
||||
## Outputs
|
||||
|
||||
```markdown
|
||||
## Video Quality Report: <experiment_name>
|
||||
|
||||
| Metric | Score | Threshold | Status |
|
||||
|--------|-------|-----------|--------|
|
||||
| SSIM (avg) | 0.87 | > 0.80 | ✅ Pass |
|
||||
| Loss trajectory | decreasing | decreasing | ✅ Pass |
|
||||
| Caption consistency | 16/20 | > 14/20 | ✅ Pass |
|
||||
|
||||
### Per-Video Scores
|
||||
| Video | SSIM | Caption |
|
||||
|-------|------|---------|
|
||||
| video_001.mp4 | 0.89 | 17/20 |
|
||||
| video_002.mp4 | 0.85 | 15/20 |
|
||||
```
|
||||
|
||||
## References
|
||||
- `fastvideo/tests/ssim/` — SSIM test infrastructure
|
||||
- `fastvideo/tests/training/Vanilla/test_training_loss.py` — loss comparison
|
||||
- `.agents/memory/evaluation-registry/README.md` — metric catalog
|
||||
|
||||
## Changelog
|
||||
| Date | Change |
|
||||
|------|--------|
|
||||
| 2026-03-02 | Initial version with SSIM, loss trajectory, caption consistency stub |
|
||||
@@ -0,0 +1,94 @@
|
||||
---
|
||||
name: index-related-work
|
||||
description: Ingest a paper or repository into the related work index
|
||||
---
|
||||
|
||||
# Index Related Work
|
||||
|
||||
## Purpose
|
||||
Create a structured summary of a related paper, repository, or blog post and
|
||||
add it to `.agents/memory/related-work/` for future reference. This builds the
|
||||
agent's knowledge base for making informed decisions about training, evaluation,
|
||||
and architecture choices.
|
||||
|
||||
## Prerequisites
|
||||
- Access to the paper/repo (URL, PDF, or local clone).
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Required | Description |
|
||||
|-----------|----------|-------------|
|
||||
| `source` | Yes | URL, citation, or local path |
|
||||
| `type` | Yes | `paper`, `repo`, or `blog` |
|
||||
| `tags` | No | List of tags (default: inferred from content) |
|
||||
|
||||
## Steps
|
||||
|
||||
### 1. Extract key information
|
||||
|
||||
For **papers**: Read abstract, method section, experimental setup, and results.
|
||||
For **repos**: Read README, key source files, and training scripts.
|
||||
For **blogs**: Read the full post.
|
||||
|
||||
Focus on:
|
||||
- What problem does it solve?
|
||||
- What architecture/technique is used?
|
||||
- How does it relate to FastVideo's approach?
|
||||
|
||||
### 2. Create the index entry
|
||||
|
||||
Write to `.agents/memory/related-work/<slug>.md`:
|
||||
|
||||
```markdown
|
||||
---
|
||||
title: <title>
|
||||
source: <URL or citation>
|
||||
type: paper | repo | blog
|
||||
date_indexed: <ISO-8601>
|
||||
tags: [world-model, distillation, evaluation, ...]
|
||||
---
|
||||
|
||||
## Summary
|
||||
<1-2 paragraph summary.>
|
||||
|
||||
## Key Differences from FastVideo
|
||||
- <comparison points>
|
||||
|
||||
## Actionable Insights
|
||||
- <what we could adopt or adapt>
|
||||
```
|
||||
|
||||
### 3. Update the catalog
|
||||
|
||||
If `.agents/memory/related-work/_catalog.md` exists, append the new entry.
|
||||
If not, create it:
|
||||
|
||||
```markdown
|
||||
# Related Work Catalog
|
||||
|
||||
| Slug | Title | Type | Tags | Date |
|
||||
|------|-------|------|------|------|
|
||||
| <slug> | <title> | <type> | <tags> | <date> |
|
||||
```
|
||||
|
||||
## Outputs
|
||||
- New file in `.agents/memory/related-work/<slug>.md`.
|
||||
- Updated catalog.
|
||||
|
||||
## Example Usage
|
||||
|
||||
```
|
||||
Index the Self-Forcing paper:
|
||||
|
||||
source: https://arxiv.org/abs/2406.xxxxx
|
||||
type: paper
|
||||
tags: [world-model, self-forcing, distillation]
|
||||
```
|
||||
|
||||
## References
|
||||
- `.agents/memory/related-work/README.md` — schema documentation
|
||||
|
||||
## Changelog
|
||||
| Date | Change |
|
||||
|------|--------|
|
||||
| 2026-03-02 | Initial version |
|
||||
@@ -0,0 +1,10 @@
|
||||
{"name": "launch-experiment", "description": "Generate and execute a training launch command for FastVideo models", "path": "launch-experiment/SKILL.md", "status": "draft", "trust": "low"}
|
||||
{"name": "monitor-experiment", "description": "Poll a running W&B training run for progress and emit structured alerts", "path": "monitor-experiment/SKILL.md", "status": "draft", "trust": "low"}
|
||||
{"name": "summarize-run", "description": "Extract a W&B run summary into a structured experiment report", "path": "summarize-run/SKILL.md", "status": "draft", "trust": "low"}
|
||||
{"name": "log-experiment", "description": "Append or update an experiment entry in the experiment journal", "path": "log-experiment/SKILL.md", "status": "draft", "trust": "low"}
|
||||
{"name": "evaluate-video-quality", "description": "Evaluate generated video quality using available metrics (SSIM, loss trajectory, caption consistency)", "path": "evaluate-video-quality/SKILL.md", "status": "draft", "trust": "low"}
|
||||
{"name": "index-related-work", "description": "Ingest a paper or repository into the related work index", "path": "index-related-work/SKILL.md", "status": "draft", "trust": "low"}
|
||||
{"name": "search-related-work", "description": "Query the related work index for relevant papers, repos, or comparisons", "path": "search-related-work/SKILL.md", "status": "draft", "trust": "low"}
|
||||
{"name": "seed-ssim-references", "description": "Run a new or updated fastvideo/tests/ssim/ test on Modal, pull generated videos, and upload them to FastVideo/ssim-reference-videos so the test has a regression baseline", "path": "seed-ssim-references/SKILL.md", "status": "draft", "trust": "low"}
|
||||
{"name": "reseed-ssim-references", "description": "Re-seed (overwrite) HF reference videos for an existing fastvideo/tests/ssim/ test and a single model id on Modal L40S. Always backs up current refs first, regenerates on Modal, pauses for the user to eyeball before-vs-after, then uploads with --force scoped to --model-id. Sister skill to seed-ssim-references; use when intentional code change has invalidated existing refs", "path": "reseed-ssim-references/SKILL.md", "status": "draft", "trust": "low"}
|
||||
{"name": "reseed-performance-baseline", "description": "Re-seed the HF performance-tracking baseline for an intentional runtime, dependency, or environment-caused benchmark shift. Use when performance CI fails because metrics such as latency, throughput, component time, or peak memory changed for an accepted reason and the rolling median baseline must be advanced by replicating one reviewed shifted source result into three success=true records, or five records when explicitly requested", "path": "reseed-performance-baseline/SKILL.md", "status": "draft", "trust": "low"}
|
||||
@@ -0,0 +1,127 @@
|
||||
---
|
||||
name: launch-experiment
|
||||
description: Generate and execute a training launch command for FastVideo models
|
||||
---
|
||||
|
||||
# Launch Experiment
|
||||
|
||||
## Purpose
|
||||
Construct a fully-specified `torchrun` training command for a FastVideo model
|
||||
given a target pipeline, dataset, and hyperparameter overrides. This skill
|
||||
automates the boilerplate of setting environment variables, picking the right
|
||||
entrypoint, and applying defaults from the closest example script.
|
||||
|
||||
## Prerequisites
|
||||
- The repo is cloned and `fastvideo` is installed (`uv pip install -e ".[dev]"`).
|
||||
- Dataset is preprocessed (see `docs/training/data_preprocess.md`).
|
||||
- `WANDB_API_KEY` is set in the environment (or `WANDB_MODE=offline` for local).
|
||||
- GPU resources are available (multi-GPU requires NCCL).
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Required | Description |
|
||||
|-----------|----------|-------------|
|
||||
| `pipeline` | Yes | Training pipeline type: `finetune`, `distill-dmd`, `self-forcing`, `lora`, `consistency` |
|
||||
| `model` | Yes | Model family: `wan-t2v-1.3B`, `wan-i2v-14B`, `ltx2`, `matrixgame` |
|
||||
| `data_path` | Yes | Path to preprocessed dataset (parquet) |
|
||||
| `num_gpus` | Yes | Number of GPUs |
|
||||
| `overrides` | No | Dict of hyperparameter overrides (any CLI arg) |
|
||||
| `output_dir` | No | Output directory (default: `outputs/<model>_<pipeline>`) |
|
||||
| `run_name` | No | W&B run name (default: auto-generated) |
|
||||
|
||||
## Steps
|
||||
|
||||
### 1. Identify the training entrypoint
|
||||
|
||||
| Pipeline | Entrypoint |
|
||||
|----------|-----------|
|
||||
| `finetune` (Wan T2V) | `fastvideo/training/wan_training_pipeline.py` |
|
||||
| `finetune` (Wan I2V) | `fastvideo/training/wan_i2v_training_pipeline.py` |
|
||||
| `finetune` (LTX-2) | `fastvideo/training/ltx2_training_pipeline.py` |
|
||||
| `finetune` (MatrixGame) | `fastvideo/training/matrixgame_training_pipeline.py` |
|
||||
| `distill-dmd` | `fastvideo/training/wan_distillation_pipeline.py` |
|
||||
| `self-forcing` | `fastvideo/training/wan_self_forcing_distillation_pipeline.py` |
|
||||
|
||||
### 2. Resolve default hyperparameters
|
||||
|
||||
Find the closest example script in `examples/training/` for the model:
|
||||
|
||||
| Model | Example Script Directory |
|
||||
|-------|-------------------------|
|
||||
| `wan-t2v-1.3B` | `examples/training/finetune/wan_t2v_1.3B/crush_smol/` |
|
||||
| `wan-i2v-14B` | `examples/training/finetune/wan_i2v_14B_480p/crush_smol/` |
|
||||
| `ltx2` | `examples/training/finetune/ltx2/` |
|
||||
| `matrixgame` | `examples/training/finetune/MatrixGame2.0/` |
|
||||
| `distill-dmd` | `scripts/distill/v1_distill_dmd_wan.sh` |
|
||||
|
||||
Read the script to extract default values for:
|
||||
- `--learning_rate`, `--train_batch_size`, `--sp_size`, `--tp_size`
|
||||
- `--num_latent_t`, `--num_height`, `--num_width`, `--num_frames`
|
||||
- `--gradient_accumulation_steps`, `--max_train_steps`
|
||||
- `--mixed_precision`, `--weight_decay`, `--max_grad_norm`
|
||||
- `--validation_steps`, `--validation_sampling_steps`
|
||||
|
||||
### 3. Set environment variables
|
||||
|
||||
```bash
|
||||
export WANDB_API_KEY="${WANDB_API_KEY}"
|
||||
export WANDB_BASE_URL="https://api.wandb.ai"
|
||||
export FASTVIDEO_ATTENTION_BACKEND=FLASH_ATTN
|
||||
export TOKENIZERS_PARALLELISM=false
|
||||
export TRITON_CACHE_DIR=/tmp/triton_cache
|
||||
```
|
||||
|
||||
### 4. Construct the torchrun command
|
||||
|
||||
```bash
|
||||
torchrun --nnodes 1 --nproc_per_node <num_gpus> \
|
||||
<entrypoint> \
|
||||
--pretrained_model_name_or_path <model_hf_id> \
|
||||
--data_path "<data_path>" \
|
||||
--output_dir "<output_dir>" \
|
||||
--wandb_run_name "<run_name>" \
|
||||
--tracker_project_name "<project_name>" \
|
||||
--log_validation \
|
||||
<...all hyperparameters...>
|
||||
```
|
||||
|
||||
### 5. Log to experiment journal
|
||||
|
||||
After launching, append an entry to `.agents/memory/experiment-journal/README.md`:
|
||||
|
||||
```markdown
|
||||
## [YYYY-MM-DD] Experiment: <run_name>
|
||||
- **Hypothesis**: <user-provided or auto-generated>
|
||||
- **Config**: model=<model>, lr=<lr>, sp_size=<sp>, gpus=<n>, script=<entrypoint>
|
||||
- **W&B run**: <pending — will be updated by monitor skill>
|
||||
- **Status**: running
|
||||
```
|
||||
|
||||
## Outputs
|
||||
- A ready-to-execute shell command.
|
||||
- An experiment journal entry.
|
||||
|
||||
## Example Usage
|
||||
|
||||
```
|
||||
Launch a Wan T2V 1.3B finetune on 4 GPUs with lr=5e-5 and max_train_steps=1000:
|
||||
|
||||
pipeline: finetune
|
||||
model: wan-t2v-1.3B
|
||||
data_path: data/crush_smol_preprocessed/
|
||||
num_gpus: 4
|
||||
overrides:
|
||||
learning_rate: 5e-5
|
||||
max_train_steps: 1000
|
||||
```
|
||||
|
||||
## References
|
||||
- `examples/training/finetune/wan_t2v_1.3B/crush_smol/finetune_t2v.sh`
|
||||
- `scripts/distill/v1_distill_dmd_wan.sh`
|
||||
- `docs/training/finetune.md` (training arguments table)
|
||||
- `fastvideo/training/trackers.py` (tracker initialization)
|
||||
|
||||
## Changelog
|
||||
| Date | Change |
|
||||
|------|--------|
|
||||
| 2026-03-02 | Initial version |
|
||||
@@ -0,0 +1,87 @@
|
||||
---
|
||||
name: log-experiment
|
||||
description: Append or update an experiment entry in the experiment journal
|
||||
---
|
||||
|
||||
# Log Experiment
|
||||
|
||||
## Purpose
|
||||
Create or update an entry in `.agents/memory/experiment-journal/README.md` to maintain
|
||||
a living record of all experiments and their outcomes.
|
||||
|
||||
## Prerequisites
|
||||
- `.agents/memory/experiment-journal/README.md` exists.
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Required | Description |
|
||||
|-----------|----------|-------------|
|
||||
| `name` | Yes | Experiment name / identifier |
|
||||
| `hypothesis` | No | What you expected to learn |
|
||||
| `config` | Yes | Key config: model, lr, sp_size, gpus, script |
|
||||
| `wandb_run` | No | W&B run ID or URL |
|
||||
| `duration` | No | Total wall time |
|
||||
| `metrics` | No | Key metrics dict (loss, step_time, grad_norm) |
|
||||
| `checkpoint` | No | Path to checkpoint |
|
||||
| `insight` | No | What was learned |
|
||||
| `status` | Yes | `running`, `completed`, `failed`, `abandoned` |
|
||||
| `lessons` | No | Paths to related lesson files |
|
||||
|
||||
## Steps
|
||||
|
||||
### 1. Check for existing entry
|
||||
|
||||
Search `.agents/memory/experiment-journal/README.md` for an entry with the same name.
|
||||
If found, update it instead of creating a duplicate.
|
||||
|
||||
### 2. Format the entry
|
||||
|
||||
```markdown
|
||||
## [YYYY-MM-DD] Experiment: <name>
|
||||
- **Hypothesis**: <hypothesis or "N/A">
|
||||
- **Config**: model=<model>, lr=<lr>, sp_size=<sp>, gpus=<n>, script=<script>
|
||||
- **W&B run**: <wandb_run or "pending">
|
||||
- **Duration**: <duration or "in progress">
|
||||
- **Key metrics**: loss=<loss>, step_time=<step_time>, grad_norm=<grad_norm>
|
||||
- **Checkpoint**: <checkpoint or "N/A">
|
||||
- **Insight**: <insight or "pending">
|
||||
- **Status**: <status>
|
||||
- **Related lessons**: <lessons or "none">
|
||||
```
|
||||
|
||||
### 3. Insert at the top of the journal
|
||||
|
||||
New entries go at the top of the file (after the header), so the most recent
|
||||
experiments are always visible first.
|
||||
|
||||
### 4. Warn on duplicates
|
||||
|
||||
If a similar experiment name exists with `status: completed`, warn that this
|
||||
may be a repeat. If it's `status: running`, assume this is an update.
|
||||
|
||||
## Outputs
|
||||
- Updated `.agents/memory/experiment-journal/README.md`.
|
||||
|
||||
## Example Usage
|
||||
|
||||
```
|
||||
Log a completed experiment:
|
||||
|
||||
name: wan-t2v-finetune-lr5e5-sp4
|
||||
config: model=wan-t2v-1.3B, lr=5e-5, sp_size=4, gpus=4
|
||||
wandb_run: fastvideo/training/run_abc123
|
||||
duration: 2h 15m
|
||||
metrics: {loss: 0.065, step_time: 2.3, grad_norm: 0.35}
|
||||
checkpoint: outputs/wan_finetune/checkpoint-1000
|
||||
insight: LR 5e-5 converges 30% faster than 1e-5 with no quality loss
|
||||
status: completed
|
||||
```
|
||||
|
||||
## References
|
||||
- `.agents/memory/experiment-journal/README.md` — journal file
|
||||
- `.agents/workflows/experiment-lifecycle.md` — when to log
|
||||
|
||||
## Changelog
|
||||
| Date | Change |
|
||||
|------|--------|
|
||||
| 2026-03-02 | Initial version |
|
||||
@@ -0,0 +1,134 @@
|
||||
---
|
||||
name: monitor-experiment
|
||||
description: Poll a running W&B training run for progress and emit structured alerts
|
||||
---
|
||||
|
||||
# Monitor Experiment
|
||||
|
||||
## Purpose
|
||||
Continuously (or on-demand) check a running experiment's W&B metrics and emit
|
||||
alerts for anomalies. Supports the "30-minute quality check" paradigm: after
|
||||
the first 30 minutes of a long training run, produce a checkpoint quality
|
||||
report before committing more resources.
|
||||
|
||||
## Prerequisites
|
||||
- `WANDB_API_KEY` is set in the environment.
|
||||
- The experiment is actively logging to W&B (not in `WANDB_MODE=offline`).
|
||||
- For offline mode: read from local `wandb-summary.json` instead.
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Required | Description |
|
||||
|-----------|----------|-------------|
|
||||
| `run_id` | Yes* | W&B run ID (e.g., `entity/project/run_id`) |
|
||||
| `output_dir` | Yes* | Local output directory (for offline mode fallback) |
|
||||
| `poll_interval` | No | Seconds between polls (default: 60) |
|
||||
| `alert_on` | No | List of alert conditions to enable (default: all) |
|
||||
|
||||
\* One of `run_id` or `output_dir` is required.
|
||||
|
||||
## Steps
|
||||
|
||||
### 1. Connect to the run
|
||||
|
||||
**Online mode** (preferred):
|
||||
|
||||
```python
|
||||
import wandb
|
||||
api = wandb.Api()
|
||||
run = api.run("<run_id>")
|
||||
```
|
||||
|
||||
**Offline fallback**:
|
||||
|
||||
```python
|
||||
import json
|
||||
summary_path = f"{output_dir}/tracker/wandb/latest-run/files/wandb-summary.json"
|
||||
with open(summary_path) as f:
|
||||
summary = json.load(f)
|
||||
```
|
||||
|
||||
### 2. Track key metrics
|
||||
|
||||
| Metric | W&B Key | Description |
|
||||
|--------|---------|-------------|
|
||||
| Training loss | `train_loss` | Primary training loss |
|
||||
| Gradient norm | `grad_norm` | Gradient magnitude |
|
||||
| Step time | `step_time` | Wall-clock seconds per step |
|
||||
| Learning rate | `learning_rate` | Current LR |
|
||||
| Avg step time | `avg_step_time` | Running average step time |
|
||||
| Validation videos | `validation_videos_*` | Generated validation samples |
|
||||
|
||||
### 3. Evaluate alert conditions
|
||||
|
||||
| Alert | Condition | Severity |
|
||||
|-------|-----------|----------|
|
||||
| **Loss spike** | `current_loss > 3 × rolling_avg_loss` | 🔴 Critical |
|
||||
| **NaN/Inf gradient** | `grad_norm` is NaN or Inf | 🔴 Critical |
|
||||
| **Step time regression** | `step_time > 2 × baseline_step_time` | 🟡 Warning |
|
||||
| **No progress** | No new W&B logs for > 10 minutes | 🟡 Warning |
|
||||
| **Loss plateau** | Loss change < 1% over last 100 steps | 🟢 Info |
|
||||
|
||||
### 4. Emit structured status
|
||||
|
||||
Output format (agent-consumable):
|
||||
|
||||
```json
|
||||
{
|
||||
"run_id": "...",
|
||||
"step": 500,
|
||||
"metrics": {
|
||||
"train_loss": 0.078,
|
||||
"grad_norm": 0.41,
|
||||
"step_time": 2.5,
|
||||
"learning_rate": 1e-6
|
||||
},
|
||||
"alerts": [
|
||||
{"type": "loss_spike", "severity": "critical", "message": "Loss jumped to 0.45 (avg: 0.08)"}
|
||||
],
|
||||
"status": "running"
|
||||
}
|
||||
```
|
||||
|
||||
### 5. 30-Minute Quality Check
|
||||
|
||||
After the first 30 minutes of wall-clock time:
|
||||
1. Summarize the loss curve shape (decreasing? at what rate?).
|
||||
2. Check if validation videos have been generated.
|
||||
3. Report step count, loss at start vs. current, and estimated time to completion.
|
||||
4. Produce a go/no-go recommendation.
|
||||
|
||||
```markdown
|
||||
## 30-Minute Check: <run_name>
|
||||
- **Steps completed**: 150
|
||||
- **Loss**: 0.12 → 0.08 (↓ 33%)
|
||||
- **Grad norm**: stable at ~0.4
|
||||
- **Step time**: 2.5s/step (consistent)
|
||||
- **Validation videos**: 5 generated at step 100
|
||||
- **Recommendation**: ✅ Continue — loss is decreasing normally
|
||||
```
|
||||
|
||||
## Outputs
|
||||
- Structured JSON status updates.
|
||||
- Alert messages for anomalous conditions.
|
||||
- 30-minute checkpoint quality report.
|
||||
|
||||
## Example Usage
|
||||
|
||||
```
|
||||
Monitor W&B run "fastvideo/Wan_distillation/abc123":
|
||||
|
||||
run_id: fastvideo/Wan_distillation/abc123
|
||||
poll_interval: 120
|
||||
alert_on: [loss_spike, nan_gradient, step_time_regression]
|
||||
```
|
||||
|
||||
## References
|
||||
- `fastvideo/training/trackers.py` — `WandbTracker` implementation
|
||||
- `fastvideo/tests/training/Vanilla/test_training_loss.py` — how summaries are compared
|
||||
- `fastvideo/tests/training/Vanilla/a40_reference_wandb_summary.json` — reference summary format
|
||||
|
||||
## Changelog
|
||||
| Date | Change |
|
||||
|------|--------|
|
||||
| 2026-03-02 | Initial version |
|
||||
@@ -0,0 +1,426 @@
|
||||
---
|
||||
name: reseed-performance-baseline
|
||||
description: Re-seed the HF performance-tracking baseline for an intentional runtime, dependency, or environment-caused benchmark shift. Use when performance CI fails because metrics such as latency, throughput, component time, or peak memory changed for an accepted reason and the rolling median baseline in FastVideo/performance-tracking must be advanced by replicating one reviewed shifted source result into three success=true records, or five records when explicitly requested.
|
||||
---
|
||||
|
||||
# Re-seed Performance Baseline
|
||||
|
||||
## Purpose
|
||||
|
||||
Replace or advance the rolling performance baseline for a single
|
||||
`(model_id, gpu_type)` pair in the HF dataset
|
||||
`FastVideo/performance-tracking`.
|
||||
|
||||
Performance comparison uses the median of up to the last 5 successful records
|
||||
for the same model and GPU. Failed records are useful audit history, but they
|
||||
do not move the future baseline because `compare_baseline.py` loads records
|
||||
with `successful_only=True`.
|
||||
|
||||
For a 5-record median, one shifted record is not enough to move the median if
|
||||
the other four records are from the old runtime. This skill therefore creates
|
||||
3 reviewed `success=true` records from one accepted shifted source result by
|
||||
default. If the user explicitly asks for a full reset, create 5 records.
|
||||
|
||||
These replicated records are an intentional operator-approved baseline reset,
|
||||
not independent measurements. Mark them clearly with provenance fields so the
|
||||
HF history remains auditable.
|
||||
|
||||
Use this skill when a performance test fails for an intentional and reviewed
|
||||
reason, such as a torch/runtime/container upgrade that legitimately increases
|
||||
peak memory or changes timings. This is the performance equivalent of
|
||||
`reseed-ssim-references`: backup first, scope tightly, require explicit human
|
||||
approval, then upload reviewed accepted baseline records.
|
||||
|
||||
## When to use
|
||||
|
||||
- A PR or main run failed the rolling performance comparison by more than the
|
||||
allowed regression threshold, and maintainers agree the shift is caused by
|
||||
an intentional runtime, dependency, hardware image, or benchmark environment
|
||||
change rather than a FastVideo logic regression.
|
||||
- One shifted source result has been reviewed and accepted, and the operator
|
||||
wants to replicate it into 3 successful records so the rolling median moves
|
||||
immediately. Use 5 records only when the user explicitly asks to fully reset
|
||||
the last-5 window.
|
||||
|
||||
## When not to use
|
||||
|
||||
- The benchmark failure might be a real code regression. Fix or investigate
|
||||
the code path first.
|
||||
- The fixed benchmark thresholds in
|
||||
`.buildkite/performance-benchmarks/tests/*.json` are too low. Those are a
|
||||
separate gate from the rolling HF baseline and may need a code review change.
|
||||
- There is no clear source run, commit, and rationale. Baseline history is a
|
||||
production signal; do not edit it without provenance.
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Required | Description |
|
||||
|-----------|----------|-------------|
|
||||
| `model_id` | Yes | Benchmark id, e.g. `wan-t2v-1.3b-2gpu`. This maps to the HF subdirectory after `sanitize(model_id)`. |
|
||||
| `gpu_type` | Yes | Exact GPU device string from the performance record, e.g. the L40S device name emitted by CI. Baselines are GPU-specific. |
|
||||
| `source_result` | Yes | Path or Buildkite artifact URL for one accepted shifted performance JSON. Prefer the normalized `normalized_perf_*.json` artifact emitted by `compare_baseline.py`. |
|
||||
| `replica_count` | No | Number of success records to create from `source_result`. Default: `3`. Only use `5` if the user explicitly asks for a full reset. |
|
||||
| `intent_rationale` | Yes | One-line explanation for why the baseline shift is legitimate. This is written into provenance and should be reused in the PR. |
|
||||
|
||||
Hardcoded defaults:
|
||||
|
||||
- HF repo: `FastVideo/performance-tracking` (`HF_REPO_ID` override is
|
||||
supported by the code, but use the default unless the user explicitly asks).
|
||||
- Local sync root: `/tmp/perf-tracking` or a timestamped local backup under
|
||||
`performance_reseed_backup/`.
|
||||
- Baseline window: last 5 `success=true` records for the same
|
||||
`(model_id, gpu_type)`.
|
||||
- Default reseed count: 3 replicated `success=true` records from one reviewed
|
||||
source result. Explicit full-reset count: 5.
|
||||
|
||||
## Steps
|
||||
|
||||
### 1. Validate the target and source result
|
||||
|
||||
If `source_result` is a Buildkite artifact URL, download it first into a
|
||||
local scratch directory such as `performance_reseed_source/` and use that
|
||||
downloaded JSON path for the rest of the workflow. If the agent cannot access
|
||||
the artifact because Buildkite authentication is missing, ask the user to
|
||||
download the artifact manually and provide the local path.
|
||||
|
||||
Prefer the normalized Buildkite artifact emitted by `compare_baseline.py`:
|
||||
|
||||
```text
|
||||
perf_reports/results/normalized_perf_*.json
|
||||
```
|
||||
|
||||
That file is already in the HF tracking schema. Load it directly and confirm
|
||||
it has the expected baseline fields:
|
||||
|
||||
```python
|
||||
import json
|
||||
|
||||
with open(source_result, encoding="utf-8") as f:
|
||||
record = json.load(f)
|
||||
```
|
||||
|
||||
If only the older raw `fastvideo/tests/performance/results/perf_*.json`
|
||||
artifact is available, normalize it with `compare_baseline.py`'s shared helper
|
||||
before continuing. Run this from the repository root with
|
||||
`PYTHONPATH=fastvideo/tests/performance` so the script-local `hf_store` import
|
||||
resolves the same way it does in CI:
|
||||
|
||||
```python
|
||||
import json
|
||||
from compare_baseline import normalize_performance_result
|
||||
|
||||
with open(source_result, encoding="utf-8") as f:
|
||||
record = normalize_performance_result(json.load(f))
|
||||
```
|
||||
|
||||
The raw-to-normalized helper maps:
|
||||
|
||||
- `model_id` comes from `benchmark_id`.
|
||||
- `gpu_type` comes from `device`.
|
||||
- `memory` comes from `max_peak_memory_mb`.
|
||||
- `latency` comes from `avg_generation_time_s`.
|
||||
- `throughput` comes from `throughput_fps`.
|
||||
- component timings come from the raw `text_encoder_time_s`, `dit_time_s`,
|
||||
and `vae_decode_time_s` fields when present. If an older raw artifact lacks
|
||||
those keys, they normalize to `None`; that source can still reseed latency,
|
||||
throughput, and memory, but it cannot move component-time baselines.
|
||||
|
||||
Stop if the normalized record's `model_id` or `gpu_type` does not match the
|
||||
requested `model_id` and `gpu_type`.
|
||||
|
||||
The source record may have `success: false` when it came from a failed rolling
|
||||
baseline comparison. That is expected; only the reviewed reseed replicas become
|
||||
new `success: true` baseline records after explicit approval.
|
||||
|
||||
Set `replica_count` to `3` by default. Set it to `5` only when the user
|
||||
explicitly asks to upload the same shifted source result 5 times for a full
|
||||
last-5 reset. Reject other counts unless the user gives a concrete reason.
|
||||
|
||||
Check that `HF_API_KEY` is exported. The sync path may be public, but the
|
||||
upload path requires write access.
|
||||
|
||||
### 1a. How to obtain `source_result` from CI
|
||||
|
||||
The performance CI exports normalized source results for failed rolling
|
||||
baseline comparisons when `compare_baseline.py` ran. The preferred artifact
|
||||
comes from:
|
||||
|
||||
```text
|
||||
perf_reports/results/normalized_perf_*.json
|
||||
```
|
||||
|
||||
and is uploaded by Buildkite with the performance reports. The normal operator
|
||||
flow is:
|
||||
|
||||
1. Open the failed Buildkite performance job.
|
||||
2. Download the `normalized_perf_*.json` artifact for the failed benchmark.
|
||||
3. Pass the local path or artifact URL as `source_result`.
|
||||
|
||||
Do not scrape the Markdown performance summary to reconstruct the JSON. The
|
||||
normalized JSON artifact is the source of truth for reseed metrics and
|
||||
provenance. If only a raw `fastvideo/tests/performance/results/perf_*.json`
|
||||
artifact is present, normalize it with `normalize_performance_result()` before
|
||||
continuing. If no JSON artifact is present, the benchmark likely failed before
|
||||
writing results, so that run is not a valid source for baseline reseeding.
|
||||
|
||||
### 2. Sync and back up existing HF records
|
||||
|
||||
Use `fastvideo/tests/performance/hf_store.py` helpers directly. Do **not** use
|
||||
`compare_baseline.py` as a sync shortcut; on full main runs it can persist
|
||||
records, while this step must only fetch and back up existing history.
|
||||
|
||||
The sync command pattern is:
|
||||
|
||||
```bash
|
||||
export PERFORMANCE_TRACKING_ROOT="${PERFORMANCE_TRACKING_ROOT:-/tmp/perf-tracking}"
|
||||
export HF_REPO_ID="${HF_REPO_ID:-FastVideo/performance-tracking}"
|
||||
PYTHONPATH=fastvideo/tests/performance python -c 'from hf_store import sync_from_hf; import os; sync_from_hf(os.environ["PERFORMANCE_TRACKING_ROOT"], strict=True)'
|
||||
```
|
||||
|
||||
Then back up only the sanitized model directory:
|
||||
|
||||
```bash
|
||||
SHORT_COMMIT=$(git rev-parse --short=12 HEAD)
|
||||
TIMESTAMP=$(date -u +%Y%m%d_%H%M%S)
|
||||
MODEL_SAFE=$(python - <<'PY'
|
||||
from fastvideo.tests.performance.hf_store import sanitize
|
||||
print(sanitize("<model_id>"))
|
||||
PY
|
||||
)
|
||||
BACKUP_DIR="performance_reseed_backup/${TIMESTAMP}_${SHORT_COMMIT}_${MODEL_SAFE}"
|
||||
mkdir -p "$BACKUP_DIR"
|
||||
cp -R "${PERFORMANCE_TRACKING_ROOT}/${MODEL_SAFE}" "$BACKUP_DIR/" 2>/dev/null || true
|
||||
```
|
||||
|
||||
Write provenance next to the backup:
|
||||
|
||||
```bash
|
||||
cat > "$BACKUP_DIR/PROVENANCE.txt" <<EOF
|
||||
model_id: <model_id>
|
||||
gpu_type: <gpu_type>
|
||||
source_result: <source_result>
|
||||
replica_count: <3_or_5>
|
||||
head_commit: $(git rev-parse HEAD)
|
||||
timestamp_utc: $(date -u +%FT%TZ)
|
||||
reason: <intent_rationale>
|
||||
EOF
|
||||
```
|
||||
|
||||
If the backup has no prior records, this is not a destructive reseed; it is a
|
||||
first baseline seed. Continue, but report that baseline history was empty.
|
||||
|
||||
### 3. Compute old baseline and candidate shift
|
||||
|
||||
Load the last 5 successful records for the target:
|
||||
|
||||
```python
|
||||
from fastvideo.tests.performance.hf_store import load_records_for_model
|
||||
|
||||
records = load_records_for_model(
|
||||
"/tmp/perf-tracking",
|
||||
"<model_id>",
|
||||
"<gpu_type>",
|
||||
last_n=5,
|
||||
successful_only=True,
|
||||
)
|
||||
```
|
||||
|
||||
Print a small table showing the source result metrics, the replicated
|
||||
candidate median, and the old medians for:
|
||||
|
||||
- `latency`
|
||||
- `throughput`
|
||||
- `memory`
|
||||
- `text_encoder_time_s`
|
||||
- `dit_time_s`
|
||||
- `vae_decode_time_s`
|
||||
|
||||
Also print how many successful old records exist. Make clear:
|
||||
|
||||
- 1 shifted record only seeds audit history and usually does not move the
|
||||
median.
|
||||
- 3 replicated shifted records in a 5-record window move the median
|
||||
immediately.
|
||||
- 5 replicated shifted records fully reset the rolling window to the source
|
||||
result's runtime profile.
|
||||
- Replicated records are not independent measurements; they are an intentional
|
||||
approved baseline reset and must be labeled that way.
|
||||
|
||||
### 4. Confirm intent
|
||||
|
||||
Require an explicit confirmation phrase before preparing the upload:
|
||||
|
||||
> About to RE-SEED performance baseline for `<model_id>` on `<gpu_type>`.
|
||||
> This will upload `<N>` new `success=true` records to
|
||||
> `FastVideo/performance-tracking/<sanitize(model_id)>/`.
|
||||
>
|
||||
> Reason: `<intent_rationale>`
|
||||
> Source result: `<source_result>`
|
||||
> Replica count: `<replica_count>`
|
||||
> Note: these records replicate one reviewed measurement to force the rolling
|
||||
> median to the accepted runtime profile.
|
||||
> HEAD: `<git rev-parse --short=12 HEAD>`
|
||||
> Backup: `<BACKUP_DIR>`
|
||||
>
|
||||
> Reply `confirm performance reseed` to proceed, anything else to abort.
|
||||
|
||||
Do not continue unless the user types exactly `confirm performance reseed`.
|
||||
|
||||
### 5. Create the accepted seed records
|
||||
|
||||
Create `replica_count` normalized records from the single source result. Use
|
||||
an explicit allowlist; do not copy the raw result JSON wholesale.
|
||||
|
||||
Each record must include only these baseline fields plus the reseed provenance
|
||||
fields below:
|
||||
|
||||
- `model_id`
|
||||
- `timestamp`
|
||||
- `commit_sha`
|
||||
- `gpu_type`
|
||||
- `latency`
|
||||
- `throughput`
|
||||
- `memory`
|
||||
- `text_encoder_time_s`
|
||||
- `dit_time_s`
|
||||
- `vae_decode_time_s`
|
||||
- `success: true`
|
||||
|
||||
For normalized `normalized_perf_*.json` sources, these fields already exist.
|
||||
For older raw `perf_*.json` sources, map the raw fields exactly as
|
||||
`normalize_performance_result()` in `compare_baseline.py` does:
|
||||
|
||||
| Normalized field | Raw source field |
|
||||
|------------------|------------------|
|
||||
| `model_id` | `benchmark_id` |
|
||||
| `gpu_type` | `device` |
|
||||
| `latency` | `avg_generation_time_s` |
|
||||
| `throughput` | `throughput_fps` |
|
||||
| `memory` | `max_peak_memory_mb` |
|
||||
| `text_encoder_time_s` | `text_encoder_time_s` |
|
||||
| `dit_time_s` | `dit_time_s` |
|
||||
| `vae_decode_time_s` | `vae_decode_time_s` |
|
||||
| `commit_sha` | `commit` |
|
||||
|
||||
Do not upload raw-only fields such as `model_short_name`, `num_gpus`,
|
||||
`num_warmup_runs`, `num_measurement_runs`, `individual_times_s`,
|
||||
`individual_peak_memories_mb`, `thresholds`, or `pr_number`.
|
||||
|
||||
Optional provenance fields are allowed and useful:
|
||||
|
||||
- `baseline_reseed: true`
|
||||
- `baseline_reseed_reason`
|
||||
- `baseline_reseed_source_result`
|
||||
- `baseline_reseed_source_timestamp`
|
||||
- `baseline_reseed_replicated_source: true`
|
||||
- `baseline_reseed_batch_size`
|
||||
- `baseline_reseed_batch_index`
|
||||
- `baseline_reseed_operator`
|
||||
|
||||
Use a fresh reseed timestamp for each replicated record, not the original
|
||||
source result timestamp. This is required because
|
||||
`load_records_for_model(..., last_n=5)` keeps the last records after loading
|
||||
the model directory; stale filenames/timestamps may not enter the last-5
|
||||
window and therefore may not move the median. Preserve the original source
|
||||
timestamp in `baseline_reseed_source_timestamp`.
|
||||
|
||||
Use the existing filename convention from `_write_tracking_record()`:
|
||||
`<sanitize(timestamp)>_<sanitize(commit_sha)>.json` under the sanitized model
|
||||
directory, but include a deterministic suffix such as `_reseed_01`,
|
||||
`_reseed_02`, and `_reseed_03` before `.json` so the replicated files do not
|
||||
overwrite each other. For a 5-record full reset, continue through
|
||||
`_reseed_05`.
|
||||
|
||||
If the source record already exists on HF with `success=false`, do not edit it
|
||||
in place unless the user explicitly asked for an audit-preserving correction.
|
||||
Prefer uploading new accepted seed records so failed history remains visible.
|
||||
|
||||
### 6. Pause before upload
|
||||
|
||||
Print:
|
||||
|
||||
- Backup directory path.
|
||||
- HF paths that will receive the new records.
|
||||
- Old rolling medians.
|
||||
- Source metrics, replica count, and candidate median.
|
||||
- Rationale.
|
||||
|
||||
Ask the user to reply exactly `upload`. Anything else aborts and leaves the
|
||||
prepared records plus backup on disk.
|
||||
|
||||
### 7. Upload only the scoped records
|
||||
|
||||
Use the shared storage helper so the path and repo type match CI:
|
||||
|
||||
```python
|
||||
from fastvideo.tests.performance.hf_store import upload_record
|
||||
|
||||
upload_record("<local_record_path>", record, strict=True)
|
||||
```
|
||||
|
||||
Run it once per prepared record. Each upload goes to:
|
||||
|
||||
```text
|
||||
FastVideo/performance-tracking/<sanitize(model_id)>/<record_filename>.json
|
||||
```
|
||||
|
||||
Never bulk upload the whole tracking root. Never modify another model's
|
||||
directory in the same operation.
|
||||
|
||||
### 8. Report outcome
|
||||
|
||||
Report:
|
||||
|
||||
- Uploaded HF paths.
|
||||
- Backup directory.
|
||||
- Old baseline window count and medians.
|
||||
- Source metrics, replica count, and candidate median.
|
||||
- Expected effect: 3 replicated shifted records move the 5-record median; 5
|
||||
replicated shifted records fully reset the window to the accepted source
|
||||
result.
|
||||
- Any separate threshold changes still needed in
|
||||
`.buildkite/performance-benchmarks/tests/*.json`.
|
||||
|
||||
Include the `intent_rationale` in the PR or follow-up comment so reviewers can
|
||||
distinguish an accepted baseline shift from a hidden regression.
|
||||
|
||||
## Failure modes and handling
|
||||
|
||||
- **`HF_API_KEY` unset.** Stop before upload. Do not create an untracked
|
||||
process that appears to have reseeded but never reached HF.
|
||||
- **Source result does not match target.** Stop. The wrong benchmark or GPU
|
||||
would poison a separate baseline.
|
||||
- **`replica_count` is 5 but the user did not explicitly ask for a full
|
||||
reset.** Stop and use the default count of 3.
|
||||
- **The source result is noisy or suspicious.** Stop. Replicating one result
|
||||
amplifies that measurement into the baseline, so it must be reviewed first.
|
||||
- **HF sync fails.** Stop for destructive reseeds. A stale or empty sync can
|
||||
make the old baseline look missing.
|
||||
- **Candidate still violates fixed thresholds.** Report that this skill only
|
||||
handles the rolling HF baseline; update benchmark JSON thresholds in code
|
||||
review if maintainers accept the new absolute limit.
|
||||
- **The user aborts at either confirmation.** Leave the backup and prepared
|
||||
records on disk. Nothing should be uploaded.
|
||||
- **A bad seed was uploaded.** Use the backup and HF history to identify the
|
||||
uploaded file, then remove or supersede it with an explicitly reviewed
|
||||
corrective record. Do not silently rewrite unrelated history.
|
||||
|
||||
## References
|
||||
|
||||
- `.agents/skills/reseed-ssim-references/SKILL.md` — safety pattern for
|
||||
intentional baseline replacement.
|
||||
- `fastvideo/tests/performance/compare_baseline.py` — normalization, rolling
|
||||
median comparison, and persistence rules.
|
||||
- `fastvideo/tests/performance/hf_store.py` — HF sync, record loading,
|
||||
`sanitize()`, and `upload_record()`.
|
||||
- `fastvideo/tests/performance/test_inference_performance.py` — source result
|
||||
JSON schema.
|
||||
- `.buildkite/performance-benchmarks/tests/*.json` — fixed absolute benchmark
|
||||
thresholds, separate from rolling baseline comparisons.
|
||||
|
||||
## Changelog
|
||||
|
||||
| Date | Change |
|
||||
|------|--------|
|
||||
| 2026-05-03 | Initial version. Sister workflow to `reseed-ssim-references`, scoped to one performance `(model_id, gpu_type)` baseline seed with backup, confirmation, provenance, and `success=true` upload. |
|
||||
| 2026-05-03 | Current policy: replicate one approved shifted source result into 3 success records by default, or 5 only when explicitly requested. Add provenance marker for replicated-source reseeds. |
|
||||
@@ -0,0 +1,343 @@
|
||||
---
|
||||
name: reseed-ssim-references
|
||||
description: Re-seed HF reference videos for a single existing SSIM test on Modal L40S. Always backs up current refs locally first, regenerates on Modal, pauses for the user to eyeball before-vs-after quality, then overwrites the targeted `<model_id>` subtree on `FastVideo/ssim-reference-videos` with `--force`. Use when an intentional code change (model port fix, attention backend swap, kernel upgrade, hyperparameter change) has invalidated existing refs and they need to be regenerated. Pairs with `seed-ssim-references`, which is for first-time seeding only.
|
||||
---
|
||||
|
||||
# Re-seed SSIM Reference Videos
|
||||
|
||||
## Purpose
|
||||
|
||||
Replace the existing SSIM reference videos for a single `(test_file, model_id)`
|
||||
pair on the HF dataset (`FastVideo/ssim-reference-videos`). This is **destructive**
|
||||
on HF — the old refs are overwritten — so the skill always:
|
||||
|
||||
1. Confirms intent with a one-liner the user has to type.
|
||||
2. Downloads the existing refs as a local, timestamped backup.
|
||||
3. Regenerates on Modal L40S (same code path that CI uses).
|
||||
4. Pauses for a side-by-side eyeball of backup vs new mp4s.
|
||||
5. Uploads with `--force`, scoped to the single `--model-id`.
|
||||
6. Reminds the user to keep the backup until the PR lands.
|
||||
|
||||
Pairs with `seed-ssim-references`, which is the inverse (first-time seeding
|
||||
only, refuses to overwrite). Re-seeding is intentionally a separate, more
|
||||
ceremonial operation because mistakenly clobbering production refs is much
|
||||
harder to recover from than failing closed.
|
||||
|
||||
## When to use
|
||||
|
||||
- An intentional code change (model port fix, kernel upgrade, attention
|
||||
backend swap, hyperparameter change in the test itself) has shifted the
|
||||
expected SSIM output and the existing refs no longer represent the new
|
||||
ground truth.
|
||||
- A test is failing in CI **for the right reason** (the new code is correct,
|
||||
the old refs are stale).
|
||||
|
||||
## When not to use
|
||||
|
||||
- A test is failing for the **wrong** reason (the port is buggy, not the
|
||||
refs). Fix the port; re-seeding hides the bug.
|
||||
- A brand-new test that has no refs on HF yet. Use `seed-ssim-references`.
|
||||
- "Just to clean up drift" without a concrete code change to point at. The
|
||||
PR description has to justify *why* refs changed; without a concrete
|
||||
change, there's nothing to write.
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Required | Description |
|
||||
|-----------|----------|-------------|
|
||||
| `test_file` | Yes | Path to the SSIM test, e.g. `fastvideo/tests/ssim/test_matrixgame_similarity.py`. Validated against `fastvideo/tests/ssim/test_*_similarity.py`. |
|
||||
| `model_id` | Yes | Single model id from the test's `*_MODEL_TO_PARAMS`, e.g. `Matrix-Game-2.0-Diffusers-Base`. Re-seed runs are **per model**. For multi-model tests, invoke the skill once per model. |
|
||||
| `intent_rationale` | Yes | One-line explanation of *why* refs are being regenerated (e.g. "Relax FA-2 head_size whitelist to include 80 — matrix_game now uses FLASH_ATTN instead of TORCH_SDPA"). Recorded in the backup directory and reused in the PR description. |
|
||||
|
||||
Hardcoded:
|
||||
|
||||
- Modal GPU: **L40S** (matches CI; re-seeding from another SKU produces refs
|
||||
that L40S CI cannot match).
|
||||
- Quality tier: **`default`**. `full_quality` is a separate, deliberate
|
||||
operation.
|
||||
- HF repo: `FastVideo/ssim-reference-videos` (override via
|
||||
`FASTVIDEO_SSIM_REFERENCE_HF_REPO`).
|
||||
- Device folder: `L40S_reference_videos`.
|
||||
|
||||
## Prerequisites
|
||||
|
||||
The user has confirmed:
|
||||
|
||||
- `modal` CLI authenticated.
|
||||
- `hf` CLI authenticated, **and** `HF_API_KEY` (or `HUGGINGFACE_HUB_TOKEN` /
|
||||
`HF_TOKEN`) exported with **write** access to
|
||||
`FastVideo/ssim-reference-videos`.
|
||||
- The current branch's code is the change that motivated the re-seed (i.e.
|
||||
`git rev-parse HEAD` is the commit that intentionally invalidated refs).
|
||||
|
||||
Fail fast if any of these are missing.
|
||||
|
||||
## Steps
|
||||
|
||||
### 1. Validate inputs and confirm intent
|
||||
|
||||
- Verify `test_file` exists and matches `fastvideo/tests/ssim/test_*_similarity.py`.
|
||||
- Grep the file for `*_MODEL_TO_PARAMS` and assert `model_id` is one of its
|
||||
keys. If the file has only a single hardcoded model, accept that model id
|
||||
as the only valid value.
|
||||
- Print the rationale and ask the user to type **`confirm reseed`** (not just
|
||||
`y` — make it deliberate):
|
||||
|
||||
> About to RE-SEED references for model `<model_id>` from test `<test_file>`.
|
||||
> This will OVERWRITE existing refs on
|
||||
> `FastVideo/ssim-reference-videos/reference_videos/default/L40S_reference_videos/<model_id>/`
|
||||
> after backup + Modal regen + eyeball.
|
||||
>
|
||||
> Reason: `<intent_rationale>`
|
||||
> HEAD: `<git rev-parse --short=12 HEAD>`
|
||||
>
|
||||
> Reply `confirm reseed` to proceed, anything else to abort.
|
||||
|
||||
Stop until the user types exactly `confirm reseed`. Anything else aborts
|
||||
with no side effects.
|
||||
|
||||
### 2. Back up existing refs
|
||||
|
||||
Always required. The backup is the only graceful path back if anything goes
|
||||
wrong later.
|
||||
|
||||
```bash
|
||||
SHORT_COMMIT=$(git rev-parse --short=12 HEAD)
|
||||
TIMESTAMP=$(date -u +%Y%m%d_%H%M%S)
|
||||
MODEL_SAFE=$(echo "<model_id>" | tr '/' '_')
|
||||
BACKUP_DIR="ssim_reseed_backup/${TIMESTAMP}_${SHORT_COMMIT}_${MODEL_SAFE}"
|
||||
mkdir -p "$BACKUP_DIR"
|
||||
|
||||
hf download \
|
||||
--repo-type dataset FastVideo/ssim-reference-videos \
|
||||
--include "reference_videos/default/L40S_reference_videos/<model_id>/**" \
|
||||
--local-dir "$BACKUP_DIR"
|
||||
|
||||
mp4_count=$(find "$BACKUP_DIR" -name "*.mp4" | wc -l)
|
||||
echo "Backup mp4 count: $mp4_count"
|
||||
[ "$mp4_count" -gt 0 ] || {
|
||||
echo "ERROR: backup is empty for <model_id>. Either the model id is wrong"
|
||||
echo "or there are no existing refs (use seed-ssim-references instead)."
|
||||
exit 1
|
||||
}
|
||||
|
||||
# Provenance — used in the PR description
|
||||
cat > "$BACKUP_DIR/PROVENANCE.txt" <<EOF
|
||||
test_file: <test_file>
|
||||
model_id: <model_id>
|
||||
head_commit: $(git rev-parse HEAD)
|
||||
timestamp_utc: $(date -u +%FT%TZ)
|
||||
reason: <intent_rationale>
|
||||
EOF
|
||||
```
|
||||
|
||||
If the `hf download` produces zero mp4s, abort — the user has either picked a
|
||||
non-existent `model_id` or there are no refs yet (in which case
|
||||
`seed-ssim-references` is the right tool).
|
||||
|
||||
### 3. Regenerate on Modal L40S
|
||||
|
||||
Mirror CI's exact env recipe so the regenerated refs are byte-comparable to
|
||||
what CI will produce on the same commit. Two differences from CI:
|
||||
|
||||
1. **Pass the same env prefix CI uses** (`IMAGE_VERSION`, `BUILDKITE_*`) — see
|
||||
`.buildkite/pipeline.yml:1-3` and `.buildkite/scripts/pr_test.sh:62-83`.
|
||||
Without this, `ssim_test.py:17-18` resolves a different GHCR image tag
|
||||
(default is `latest`, CI is `py3.12-latest`), and `ssim_test.py:38-46`
|
||||
bakes different values into the image's frozen env block. **Mismatched
|
||||
image or env is the most common source of SSIM drift between reseed and
|
||||
CI runs.**
|
||||
2. **Do not pass `--skip-reference-download`**. Letting the test fetch the
|
||||
existing refs and run the full SSIM compare gives "before" SSIM numbers
|
||||
for the PR description, and the test still produces the new mp4s
|
||||
regardless of whether the comparison passes or fails.
|
||||
|
||||
```bash
|
||||
SUBDIR="${TIMESTAMP}_${SHORT_COMMIT}"
|
||||
|
||||
IMAGE_VERSION="py3.12-latest" \
|
||||
BUILDKITE_REPO="$(git config --get remote.origin.url)" \
|
||||
BUILDKITE_COMMIT="$(git rev-parse HEAD)" \
|
||||
BUILDKITE_PULL_REQUEST="${BUILDKITE_PULL_REQUEST:-false}" \
|
||||
modal run fastvideo/tests/modal/ssim_test.py \
|
||||
--git-repo="$(git config --get remote.origin.url)" \
|
||||
--git-commit="$(git rev-parse HEAD)" \
|
||||
--hf-api-key="$HF_API_KEY" \
|
||||
--test-files="<test_file>" \
|
||||
--sync-generated-to-volume \
|
||||
--generated-volume-subdir="$SUBDIR" \
|
||||
--no-fail-fast
|
||||
```
|
||||
|
||||
Capture the printed `modal volume get ...` hint — its `<SUBDIR>` matches
|
||||
`$SUBDIR` and is needed for step 4. Capture the SSIM numbers from the test
|
||||
output (or from the JSON next to the generated mp4) for the PR description.
|
||||
|
||||
### 4. Download generated videos
|
||||
|
||||
```bash
|
||||
modal volume get --force hf-model-weights \
|
||||
ssim_generated_videos/default/"$SUBDIR"/generated_videos \
|
||||
./generated_videos_modal/default
|
||||
```
|
||||
|
||||
After this, the new mp4s live at:
|
||||
|
||||
```
|
||||
./generated_videos_modal/default/generated_videos/L40S_reference_videos/<model_id>/<backend>/<prompt>.mp4
|
||||
```
|
||||
|
||||
`--force` is required when `./generated_videos_modal/default` already exists
|
||||
from a prior run; safe on the first run too.
|
||||
|
||||
### 5. PAUSE — user reviews quality side-by-side
|
||||
|
||||
Print the diff and the comparison:
|
||||
|
||||
```bash
|
||||
echo "=== File list diff (backup vs new) ==="
|
||||
diff -u \
|
||||
<(find "$BACKUP_DIR/reference_videos/default/L40S_reference_videos/<model_id>" -name "*.mp4" \
|
||||
| sed "s|$BACKUP_DIR/reference_videos/default/L40S_reference_videos/||" | sort) \
|
||||
<(find ./generated_videos_modal/default/generated_videos/L40S_reference_videos/<model_id> -name "*.mp4" \
|
||||
| sed "s|./generated_videos_modal/default/generated_videos/L40S_reference_videos/||" | sort) \
|
||||
|| true
|
||||
|
||||
echo
|
||||
echo "=== SSIM numbers from this run (paste into PR) ==="
|
||||
find ./generated_videos_modal/default/generated_videos/L40S_reference_videos/<model_id> -name "*_ssim.json" -exec cat {} \;
|
||||
```
|
||||
|
||||
Then stop and tell the user:
|
||||
|
||||
> Old refs backed up to `$BACKUP_DIR`.
|
||||
> New videos in `./generated_videos_modal/default/generated_videos/L40S_reference_videos/<model_id>/`.
|
||||
>
|
||||
> Open both in a video player. Confirm the new videos:
|
||||
> 1. Look correct (no obvious artifacts, no black/static frames).
|
||||
> 2. Are *intentionally* different from the backup in the way described
|
||||
> in `<intent_rationale>` (e.g. slight numerical drift only, not a
|
||||
> different scene / different motion / corrupted output).
|
||||
>
|
||||
> Reply **`upload`** to overwrite HF, anything else to abort.
|
||||
> Aborting leaves the backup and new videos on disk for inspection — nothing
|
||||
> on HF changes.
|
||||
|
||||
Do not proceed until the user types exactly `upload`. If they abort, leave
|
||||
everything on disk and stop here.
|
||||
|
||||
### 6. Copy into the local reference layout
|
||||
|
||||
Same as `seed-ssim-references` step 5:
|
||||
|
||||
```bash
|
||||
python fastvideo/tests/ssim/reference_videos_cli.py copy-local \
|
||||
--quality-tier default \
|
||||
--device-folder L40S_reference_videos \
|
||||
--generated-dir ./generated_videos_modal/default/generated_videos/L40S_reference_videos
|
||||
```
|
||||
|
||||
Result: `fastvideo/tests/ssim/reference_videos/default/L40S_reference_videos/<model_id>/<backend>/<prompt>.mp4`.
|
||||
|
||||
### 7. Upload with `--force`, scoped to `--model-id`
|
||||
|
||||
The `--force` flag is what makes this skill different from `seed-ssim-references`.
|
||||
Always pair it with `--model-id` so a typo cannot accidentally overwrite a
|
||||
neighboring model's refs.
|
||||
|
||||
```bash
|
||||
python fastvideo/tests/ssim/reference_videos_cli.py upload \
|
||||
--quality-tier default \
|
||||
--device-folder L40S_reference_videos \
|
||||
--model-id "<model_id>" \
|
||||
--force
|
||||
```
|
||||
|
||||
The CLI's overwrite guard refuses without `--force`; with `--force` it
|
||||
overwrites only files under
|
||||
`reference_videos/default/L40S_reference_videos/<model_id>/`.
|
||||
|
||||
### 8. Report success and retention guidance
|
||||
|
||||
Print:
|
||||
|
||||
- The HF path that was overwritten (`<repo>/reference_videos/default/L40S_reference_videos/<model_id>/`).
|
||||
- The local backup directory path.
|
||||
- The new SSIM numbers from step 5.
|
||||
- This restore command, in case the PR review surfaces a problem after
|
||||
upload:
|
||||
|
||||
```bash
|
||||
python fastvideo/tests/ssim/reference_videos_cli.py upload \
|
||||
--quality-tier default \
|
||||
--device-folder L40S_reference_videos \
|
||||
--model-id "<model_id>" \
|
||||
--reference-dir "$BACKUP_DIR/reference_videos/default/L40S_reference_videos" \
|
||||
--force
|
||||
```
|
||||
|
||||
- This PR-description checklist (see `fastvideo/tests/ssim/AGENTS.md` →
|
||||
*Updating Reference Videos*):
|
||||
1. Source commit that produced the new refs (HEAD at re-seed time).
|
||||
2. Test command and GPU SKU (`L40S`).
|
||||
3. Before/after SSIM numbers.
|
||||
4. The `<intent_rationale>` from step 1.
|
||||
5. A note that the backup lives at `$BACKUP_DIR` and should be retained
|
||||
until CI on the PR is green.
|
||||
|
||||
Do **not** auto-rerun the SSIM test — the user does that as part of the PR.
|
||||
|
||||
## Failure modes and how to handle them
|
||||
|
||||
- **`HF_API_KEY` unset.** Stop before step 2.
|
||||
- **Backup is empty (zero mp4s).** Stop before step 3 — the model id is
|
||||
wrong or the refs don't exist yet (use `seed-ssim-references`).
|
||||
- **Modal run fails before generation.** No mp4s on the volume. Don't
|
||||
upload. Investigate the failure (test crash, OOM, partition exhaustion),
|
||||
fix, then retry from step 3. Backup is still intact.
|
||||
- **Quality regressed (visual or metric).** User aborts at step 5. Backup
|
||||
retained. New videos retained on disk for inspection. Nothing on HF
|
||||
changed. Either fix the underlying code change or abandon the re-seed.
|
||||
- **User confirmed `upload` but later realized the new refs are wrong.**
|
||||
Run the restore command from step 8 with the backup `--reference-dir`.
|
||||
This is exactly why the backup exists.
|
||||
- **Multi-model test, only one model is being re-seeded.** Run the skill
|
||||
once per model id. The `--model-id` scope on upload guarantees the others
|
||||
are untouched.
|
||||
|
||||
## Design notes (for future skill maintainers)
|
||||
|
||||
- Per-`model_id` scope is mandatory. The dataset houses many model subtrees;
|
||||
re-seeding the wrong one is hard to undo without backup.
|
||||
- `default` tier only; `full_quality` is a separate, deliberate operation
|
||||
with different params and ~doubled runtime, and isn't what CI gates on.
|
||||
- The skill deliberately does **not** pass `--skip-reference-download` to
|
||||
Modal so we get pre-reseed SSIM numbers for the PR. The `seed`-skill
|
||||
passes it because no refs exist yet; for re-seed, refs do exist and
|
||||
exposing the comparison is informative.
|
||||
- The two-token confirm (`confirm reseed`, then `upload`) is intentional.
|
||||
Re-seeding is high-blast-radius and should not be one-keystroke.
|
||||
- The backup directory is plain mp4s + `PROVENANCE.txt`. No HF metadata is
|
||||
preserved; the restore path uses `reference_videos_cli.py upload
|
||||
--reference-dir` which doesn't need it.
|
||||
|
||||
## References
|
||||
|
||||
- `.agents/skills/seed-ssim-references/SKILL.md` — the first-time seed
|
||||
skill this one parallels. Read it for the Modal flag rationale shared
|
||||
between the two flows.
|
||||
- `fastvideo/tests/ssim/AGENTS.md` — directory rules, including the PR
|
||||
expectations for any reference-video change (rationale, before/after
|
||||
SSIM, source commit/model/backend).
|
||||
- `fastvideo/tests/ssim/reference_videos_cli.py` — `copy-local`, `upload`
|
||||
(with `--model-id`, `--force`), `download`. The overwrite guard at
|
||||
`upload_reference_videos` is the safety net this skill leans on.
|
||||
- `fastvideo/tests/modal/ssim_test.py` — Modal orchestrator;
|
||||
`--sync-generated-to-volume`, `--generated-volume-subdir`,
|
||||
`--skip-reference-download`, `--no-fail-fast`.
|
||||
|
||||
## Changelog
|
||||
|
||||
| Date | Change |
|
||||
|------|--------|
|
||||
| 2026-05-02 | Initial version. Sister skill to `seed-ssim-references`, scoped to single `(test_file, model_id)` re-seeds, with mandatory backup and two-token confirm. |
|
||||
@@ -0,0 +1,82 @@
|
||||
---
|
||||
name: search-related-work
|
||||
description: Query the related work index for relevant papers, repos, or comparisons
|
||||
---
|
||||
|
||||
# Search Related Work
|
||||
|
||||
## Purpose
|
||||
Search through `.agents/memory/related-work/` to find indexed papers, repos,
|
||||
or blog posts relevant to a query. Use this when you need to understand how
|
||||
other work compares to FastVideo's approach, or when looking for techniques
|
||||
to adopt.
|
||||
|
||||
## Prerequisites
|
||||
- The related work index has entries (`.agents/memory/related-work/*.md`).
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Required | Description |
|
||||
|-----------|----------|-------------|
|
||||
| `query` | Yes | Natural language query |
|
||||
| `tags` | No | Filter by tags (e.g., `[distillation, evaluation]`) |
|
||||
| `type` | No | Filter by type (`paper`, `repo`, `blog`) |
|
||||
|
||||
## Steps
|
||||
|
||||
### 1. Search the index
|
||||
|
||||
Use grep-based search through `.agents/memory/related-work/`:
|
||||
|
||||
```bash
|
||||
# Search by content
|
||||
grep -rl "<query>" .agents/memory/related-work/
|
||||
|
||||
# Search by tags (in frontmatter)
|
||||
grep -l "tags:.*<tag>" .agents/memory/related-work/*.md
|
||||
```
|
||||
|
||||
### 2. Rank results
|
||||
|
||||
For each matching file:
|
||||
1. Read the file.
|
||||
2. Score relevance to the query based on:
|
||||
- Title match
|
||||
- Tag match
|
||||
- Content match (summary, differences, insights)
|
||||
3. Return top results.
|
||||
|
||||
### 3. Format output
|
||||
|
||||
```markdown
|
||||
## Related Work Search: "<query>"
|
||||
|
||||
### 1. <Title> (relevance: high)
|
||||
- **Source**: <URL>
|
||||
- **Tags**: <tags>
|
||||
- **Key insight**: <most relevant excerpt>
|
||||
- **File**: `.agents/memory/related-work/<slug>.md`
|
||||
|
||||
### 2. <Title> (relevance: medium)
|
||||
...
|
||||
```
|
||||
|
||||
## Outputs
|
||||
- Ranked list of relevant related work entries with excerpts.
|
||||
|
||||
## Example Usage
|
||||
|
||||
```
|
||||
Search for work related to video quality evaluation metrics:
|
||||
|
||||
query: "video generation quality evaluation metrics"
|
||||
tags: [evaluation]
|
||||
```
|
||||
|
||||
## References
|
||||
- `.agents/memory/related-work/README.md` — index schema
|
||||
|
||||
## Changelog
|
||||
| Date | Change |
|
||||
|------|--------|
|
||||
| 2026-03-02 | Initial version |
|
||||
@@ -0,0 +1,376 @@
|
||||
---
|
||||
name: seed-ssim-references
|
||||
description: Seed HF reference artefacts for a single newly-added SSIM test (pixel `.mp4` for `run_text_to_video_similarity_test`-style tests, or latent `.pt` for `run_text_to_latent_similarity_test`-style tests). Runs the test on Modal L40S, downloads the generated artefacts via `modal volume get`, pauses for the user to verify (visual eyeball for mp4, numerics dump for pt), then uploads only that test's files to `FastVideo/ssim-reference-videos`. Use when a new `fastvideo/tests/ssim/test_*_similarity.py` has just been added and has no references on HF yet.
|
||||
---
|
||||
|
||||
# Seed SSIM Reference Artefacts (mp4 or pt)
|
||||
|
||||
## Purpose
|
||||
|
||||
A brand-new SSIM test in `fastvideo/tests/ssim/` fails forever until its
|
||||
reference artefacts exist on the HF dataset
|
||||
(`FastVideo/ssim-reference-videos`). The dataset hosts two kinds of artefacts
|
||||
side-by-side per `(model_id, backend, prompt)`:
|
||||
|
||||
- **`.mp4`** — pixel ground-truth for tests that call
|
||||
`run_text_to_video_similarity_test` / `run_image_to_video_similarity_test`
|
||||
in `inference_similarity_utils.py`. Compared via SSIM.
|
||||
- **`.pt`** — pre-VAE latent bundle (fp16 full latent + fp32 slice +
|
||||
metadata + `slice_spec` + `format_version`) for tests that call
|
||||
`run_text_to_latent_similarity_test` in `latent_similarity_utils.py`.
|
||||
Compared via cosine distance on the slice and the full tensor.
|
||||
|
||||
This skill:
|
||||
|
||||
1. Detects which artefact type the test produces (pixel vs latent).
|
||||
2. Runs the test on Modal's L40S pool to generate the artefacts.
|
||||
3. Downloads them to the local repo via `modal volume get`.
|
||||
4. Pauses so the user can verify quality:
|
||||
- **mp4**: visual eyeball in a video player.
|
||||
- **pt**: numerics dump (shape, slice stats, NaN/Inf check, metadata).
|
||||
5. Uploads only the new test's files to HF, with a guard that refuses to
|
||||
overwrite anything already present.
|
||||
|
||||
The skill is run **manually**, once per new test. Before invoking it, the user
|
||||
has already sanity-tested the new test locally — it launches `VideoGenerator`
|
||||
and writes an artefact without crashing (the missing-reference assertion at
|
||||
the end is expected). The skill does not re-test locally; it goes straight
|
||||
to Modal L40S (which is what CI uses).
|
||||
|
||||
## When to use
|
||||
|
||||
- A new `test_*_similarity.py` file has been added in `fastvideo/tests/ssim/`
|
||||
and the HF dataset has no `reference_videos/default/L40S_reference_videos/<model_id>/`
|
||||
subtree for it yet.
|
||||
|
||||
## When not to use
|
||||
|
||||
- Regular CI runs — once refs exist, `pytest fastvideo/tests/ssim/` downloads
|
||||
them automatically.
|
||||
- Re-seeding an existing test. That requires `--force` on the upload step, and
|
||||
is out of scope here; treat as a separate, deliberate operation.
|
||||
|
||||
## Inputs
|
||||
|
||||
The skill has **one required input**: the path to the new SSIM test file.
|
||||
Prompt the user for it if they didn't supply it.
|
||||
|
||||
| Parameter | Required | Description |
|
||||
|-----------|----------|-------------|
|
||||
| `test_file` | Yes | e.g. `fastvideo/tests/ssim/test_ltx2_similarity.py`. The skill's first action is to ask for this if missing. |
|
||||
|
||||
Everything else is fixed:
|
||||
|
||||
- Modal runner GPU: **L40S** (hardcoded in `fastvideo/tests/modal/ssim_test.py`).
|
||||
- Device folder: `L40S_reference_videos`.
|
||||
- Quality tier: `default` (the tier CI runs). The `full_quality` tier is not
|
||||
seeded by this skill.
|
||||
- HF repo: `FastVideo/ssim-reference-videos` (dataset).
|
||||
- Multi-model test files: all model ids in `*_MODEL_TO_PARAMS` are seeded
|
||||
together; the Modal run produces one mp4 per (model, prompt, backend) and
|
||||
the upload scopes by `--model-id`, looping if there is more than one.
|
||||
|
||||
## Prerequisites
|
||||
|
||||
The user has confirmed:
|
||||
|
||||
- `modal` CLI authenticated.
|
||||
- `HF_API_KEY` (or `HUGGINGFACE_HUB_TOKEN` / `HF_TOKEN`) exported with write
|
||||
access to `FastVideo/ssim-reference-videos`.
|
||||
- The test file runs locally end-to-end (generates an mp4; SSIM assertion
|
||||
failure due to missing reference is expected and fine).
|
||||
|
||||
Fail fast if the token env var is missing.
|
||||
|
||||
## Steps
|
||||
|
||||
### 1. Ask for the test file, then detect artefact type
|
||||
|
||||
If the user didn't name one, ask: *"Which SSIM test file do you want to seed
|
||||
references for? (e.g. `fastvideo/tests/ssim/test_ltx2_similarity.py`)"*.
|
||||
|
||||
Validate:
|
||||
|
||||
- Path exists and matches `fastvideo/tests/ssim/test_*_similarity.py`.
|
||||
- File defines a `*_MODEL_TO_PARAMS` dict — grep it to extract the set of
|
||||
model ids. Those ids drive step 5.
|
||||
|
||||
Detect artefact type by inspecting the file's imports / helper call:
|
||||
|
||||
- **latent** (`.pt`) — file imports `run_text_to_latent_similarity_test`
|
||||
from `fastvideo.tests.ssim.latent_similarity_utils` (or any other helper
|
||||
that ends with `_latent_similarity_test`).
|
||||
- **pixel** (`.mp4`) — file imports
|
||||
`run_text_to_video_similarity_test` / `run_image_to_video_similarity_test`
|
||||
from `fastvideo.tests.ssim.inference_similarity_utils`, OR uses the
|
||||
legacy custom-inline helper pattern (see `test_gamecraft`,
|
||||
`test_longcat`, etc.). Default to pixel when both heuristics fail.
|
||||
|
||||
Record `ARTEFACT_TYPE ∈ {pixel, latent}` for use in step 4. Steps 2, 3, 5,
|
||||
and 6 are artefact-type-agnostic — `_iter_reference_files`,
|
||||
`copy_generated_to_reference`, and `upload_reference_videos` already walk
|
||||
both `.mp4` and `.pt` (see `reference_videos_cli.py`).
|
||||
|
||||
If either check fails, stop and tell the user what's wrong.
|
||||
|
||||
### 2. Run the test on Modal L40S
|
||||
|
||||
Pick a subdir name so repeated runs don't collide:
|
||||
|
||||
```bash
|
||||
SHORT_COMMIT=$(git rev-parse --short=12 HEAD)
|
||||
TIMESTAMP=$(date -u +%Y%m%d_%H%M%S)
|
||||
SUBDIR="${TIMESTAMP}_${SHORT_COMMIT}"
|
||||
```
|
||||
|
||||
Then launch the Modal run. The `IMAGE_VERSION` and `BUILDKITE_*` env-prefix
|
||||
**must** match what CI exports in `.buildkite/scripts/pr_test.sh`, otherwise
|
||||
`fastvideo/tests/modal/ssim_test.py` resolves a different GHCR image tag
|
||||
(default is `latest`, CI is `py3.12-latest`) and bakes different values into
|
||||
the image's frozen env block (`ssim_test.py:17-18, 38-46`). Mismatched image
|
||||
or env produces SSIM drift that doesn't show up until the same commit runs
|
||||
in CI.
|
||||
|
||||
```bash
|
||||
IMAGE_VERSION="py3.12-latest" \
|
||||
BUILDKITE_REPO="$(git config --get remote.origin.url)" \
|
||||
BUILDKITE_COMMIT="$(git rev-parse HEAD)" \
|
||||
BUILDKITE_PULL_REQUEST="${BUILDKITE_PULL_REQUEST:-false}" \
|
||||
modal run fastvideo/tests/modal/ssim_test.py \
|
||||
--git-repo="$(git config --get remote.origin.url)" \
|
||||
--git-commit="$(git rev-parse HEAD)" \
|
||||
--hf-api-key="$HF_API_KEY" \
|
||||
--test-files="<test_file>" \
|
||||
--sync-generated-to-volume \
|
||||
--generated-volume-subdir="$SUBDIR" \
|
||||
--skip-reference-download \
|
||||
--no-fail-fast
|
||||
```
|
||||
|
||||
Env prefix rationale (parity with CI; see `.buildkite/pipeline.yml:1-3` and
|
||||
`.buildkite/scripts/pr_test.sh:62-83`):
|
||||
- `IMAGE_VERSION=py3.12-latest`: pins the Modal image tag to the same one CI
|
||||
uses. Without this, `ssim_test.py:17` falls back to `latest`, which on
|
||||
GHCR is built from `Dockerfile.python3.10` — different Python, torch, and
|
||||
flash-attn wheel than CI's `py3.12-latest` (`infra-build-image.yml:51-67`,
|
||||
`_template-build-image.yml:65-101`).
|
||||
- `BUILDKITE_REPO`/`BUILDKITE_COMMIT`/`BUILDKITE_PULL_REQUEST`: mirror what
|
||||
Buildkite exports. `ssim_test.py:38-46` bakes these into the image's
|
||||
`.env(...)` block; mismatched values can perturb in-container code paths
|
||||
that branch on PR-vs-non-PR. `false` for `BUILDKITE_PULL_REQUEST` matches
|
||||
Buildkite's "non-PR build" sentinel.
|
||||
|
||||
Flag rationale:
|
||||
- `--skip-reference-download`: no refs exist yet, so conftest must not try to
|
||||
pull them.
|
||||
- `--no-fail-fast`: lets the test finish generation before `_assert_similarity`
|
||||
raises `FileNotFoundError: Reference video folder does not exist`. The
|
||||
expected failure is what we want — the mp4 has already been written.
|
||||
- `--sync-generated-to-volume` + `--generated-volume-subdir`: copies the
|
||||
generated mp4s to the `hf-model-weights` Modal volume under
|
||||
`ssim_generated_videos/default/<SUBDIR>/generated_videos/` so we can pull
|
||||
them locally.
|
||||
|
||||
The Modal run will end with a nonzero exit (expected) and print a
|
||||
`modal volume get hf-model-weights ssim_generated_videos/default/<SUBDIR>/generated_videos ./generated_videos_modal/default`
|
||||
command. Capture that `<SUBDIR>` — you need it for step 3.
|
||||
|
||||
### 3. Download generated videos locally
|
||||
|
||||
```bash
|
||||
modal volume get --force hf-model-weights \
|
||||
ssim_generated_videos/default/"$SUBDIR"/generated_videos \
|
||||
./generated_videos_modal/default
|
||||
```
|
||||
|
||||
`--force` is required when the parent `./generated_videos_modal/default`
|
||||
already exists; without it, `modal volume get` errors with `[Errno 21] Is a
|
||||
directory`. Safe to pass on the first run too.
|
||||
|
||||
After this, the mp4s live at
|
||||
`./generated_videos_modal/default/generated_videos/L40S_reference_videos/<model_id>/<backend>/<prompt>.mp4`.
|
||||
The extra `generated_videos/` level comes from the volume layout in
|
||||
`_sync_generated_videos_to_volume` (`ssim_test.py`) — the command copies
|
||||
`<repo>/fastvideo/tests/ssim/generated_videos/<tier>` to
|
||||
`ssim_generated_videos/<tier>/<SUBDIR>/generated_videos/`, and `modal volume
|
||||
get` preserves that trailing `generated_videos/` segment.
|
||||
|
||||
### 4. PAUSE — user reviews quality
|
||||
|
||||
Type-aware verification.
|
||||
|
||||
**For `ARTEFACT_TYPE = pixel`** — list the downloaded mp4s and ask the user to
|
||||
open them in a video player:
|
||||
|
||||
> "Generated videos downloaded to `./generated_videos_modal/default/generated_videos/L40S_reference_videos/`. Please open them and confirm the quality looks correct. Reply **`upload`** to continue, or anything else to abort."
|
||||
|
||||
**For `ARTEFACT_TYPE = latent`** — `.pt` files are not human-watchable. Print
|
||||
a numerics dump for each `.pt` so the user can sanity-check shape, distribution,
|
||||
and metadata:
|
||||
|
||||
```python
|
||||
import torch
|
||||
from pathlib import Path
|
||||
ROOT = Path("./generated_videos_modal/default/generated_videos/L40S_reference_videos")
|
||||
for p in sorted(ROOT.rglob("*.pt")):
|
||||
d = torch.load(p, map_location="cpu", weights_only=False)
|
||||
s = d["expected_slice"]
|
||||
L = d["latent"].float()
|
||||
print(f"=== {p.relative_to(ROOT)} ===")
|
||||
print(f" format_version: {d['format_version']}")
|
||||
print(f" shape: {d['shape']}")
|
||||
print(f" dtype_original: {d['dtype_original']}")
|
||||
print(f" slice_spec: {d['slice_spec']}")
|
||||
print(f" slice shape={tuple(s.shape)} mean={s.mean():+.4f} std={s.std():.4f} min={s.min():+.4f} max={s.max():+.4f}")
|
||||
print(f" latent shape={tuple(L.shape)} mean={L.mean():+.4f} std={L.std():.4f} min={L.min():+.4f} max={L.max():+.4f}")
|
||||
print(f" finite: latent NaN={torch.isnan(L).any().item()} Inf={torch.isinf(L).any().item()}; "
|
||||
f"slice NaN={torch.isnan(s).any().item()} Inf={torch.isinf(s).any().item()}")
|
||||
print(f" metadata: {d['metadata']}\n")
|
||||
```
|
||||
|
||||
Sanity criteria:
|
||||
- `format_version == 1` (matches `LATENT_REFERENCE_FORMAT_VERSION`).
|
||||
- `shape` matches what the model produces (e.g. LTX-2 distilled =
|
||||
`[1, 128, T_lat, H_lat, W_lat]`; Stable Audio Open 1.0 = `[1, 64, 1024]`).
|
||||
- `slice_spec.kind` matches a registered kind (`corner_3x3_first_frame`
|
||||
for video, `audio_first_8_timesteps` for audio).
|
||||
- No `NaN`/`Inf`. `mean ≈ 0`, `std ≈ 1` (denoised latents stay close to
|
||||
the initial Gaussian distribution; very wide deviations suggest
|
||||
numerical drift).
|
||||
- `metadata.prompt` matches the test's prompt.
|
||||
|
||||
Then ask:
|
||||
|
||||
> "Numerics look right? Reply **`upload`** to continue, or anything else to abort."
|
||||
|
||||
Do not proceed until the user explicitly says `upload`. If they abort, leave
|
||||
everything on disk so they can inspect further — no cleanup.
|
||||
|
||||
### 5. Copy into the local reference layout
|
||||
|
||||
Scoped copy — only the new test's artefacts. Single command works for both
|
||||
artefact types because `_iter_reference_files` walks `.mp4` and `.pt`:
|
||||
|
||||
```bash
|
||||
python fastvideo/tests/ssim/reference_videos_cli.py copy-local \
|
||||
--quality-tier default \
|
||||
--device-folder L40S_reference_videos \
|
||||
--generated-dir ./generated_videos_modal/default/generated_videos/L40S_reference_videos
|
||||
```
|
||||
|
||||
(The `--generated-dir` points at the device-folder root inside the
|
||||
downloaded tree; `copy-local` walks all `<model>/<backend>/*.{mp4,pt}`
|
||||
underneath it. Since the Modal run was scoped to a single test file via
|
||||
`--test-files`, only that test's model(s) are present — so the copy is
|
||||
implicitly per-test.)
|
||||
|
||||
Result for pixel: `fastvideo/tests/ssim/reference_videos/default/L40S_reference_videos/<model_id>/<backend>/<prompt>.mp4`.
|
||||
Result for latent: same path with `.pt` extension.
|
||||
|
||||
### 6. Upload to HF — scoped per model_id, with overwrite guard
|
||||
|
||||
For each `<model_id>`:
|
||||
|
||||
```bash
|
||||
python fastvideo/tests/ssim/reference_videos_cli.py upload \
|
||||
--quality-tier default \
|
||||
--device-folder L40S_reference_videos \
|
||||
--model-id "<model_id>"
|
||||
```
|
||||
|
||||
The upload command:
|
||||
|
||||
- Uploads **only** `reference_videos/default/L40S_reference_videos/<model_id>/`.
|
||||
- **Refuses** if any file already exists at that path on HF (this is the
|
||||
guard — seeding a new test should never clobber existing refs). To override,
|
||||
the user must re-run with `--force`. If the guard fires, stop and report
|
||||
exactly which files exist; do not silently `--force`.
|
||||
|
||||
Reads the HF token from `HF_API_KEY` / `HUGGINGFACE_HUB_TOKEN` / `HF_TOKEN`.
|
||||
|
||||
### 7. Report success
|
||||
|
||||
List what was uploaded (paths in repo) and remind the user to push any
|
||||
related code changes. Do **not** auto-verify by re-running Modal — the user
|
||||
can run `pytest fastvideo/tests/ssim/<test_file>` later to confirm end-to-end;
|
||||
it will auto-download the refs they just uploaded.
|
||||
|
||||
## Failure modes and how to handle them
|
||||
|
||||
- **`HF_API_KEY` unset.** Stop before step 2. The Modal run needs it (passed
|
||||
via `--hf-api-key`), and step 6 needs it for upload. If the user
|
||||
ran `hf auth login` instead of exporting an env var, read the cached
|
||||
token via `huggingface_hub.get_token()` and forward it to Modal as
|
||||
`--hf-api-key="$CACHED_TOKEN"`.
|
||||
- **Modal run fails before generation.** No artefacts on the volume — nothing
|
||||
to download. Fix the test locally (`pytest fastvideo/tests/ssim/<test_file>`)
|
||||
and retry from step 2.
|
||||
- **`./generated_videos_modal/default/L40S_reference_videos/` missing after
|
||||
`modal volume get`.** The run didn't produce artefacts (most likely the
|
||||
test crashed before writing, or `REQUIRED_GPUS` exceeded the partition
|
||||
capacity — see Modal logs).
|
||||
- **Latent test crashed with FSDP / inference_mode error
|
||||
(`RuntimeError: Inference tensors do not track version counter`).** The
|
||||
test must pass `init_kwargs_override={"use_fsdp_inference": False}` when
|
||||
`sp_size == 1` — see `test_stable_audio_similarity.py` for the pattern.
|
||||
Fix in the test, push, retry.
|
||||
- **Upload guard fires (files already exist).** The test name / model id
|
||||
collides with something already on HF. Verify the user actually wants to
|
||||
replace existing refs; if so, re-run the upload with `--force`. If not,
|
||||
rename the model id in `*_MODEL_TO_PARAMS` and re-seed.
|
||||
- **Quality looks wrong in step 4.** Abort. The artefacts stay on disk for
|
||||
inspection. The fix is usually in the test's params (resolution, steps,
|
||||
seed) — edit the test, then re-run the skill.
|
||||
- For latent: also check `slice_spec.kind` matches the latent rank
|
||||
(`corner_3x3_first_frame` requires 5-D, `audio_first_8_timesteps`
|
||||
requires 3-D); a rank/kind mismatch raises in `_extract_expected_slice`.
|
||||
|
||||
## Design notes (for future skill maintainers)
|
||||
|
||||
- The skill deliberately runs on Modal, **not** locally, because the CI
|
||||
runner is L40S. Seeding from a different GPU SKU produces refs that CI's
|
||||
L40S runs can't match (pixel SSIM drifts across SKUs; latent cosine has
|
||||
tighter cross-SKU bf16 drift but the configured tolerances assume
|
||||
same-SKU seed → same-SKU verify).
|
||||
- The skill is default-tier only. `full_quality` refs are seeded by a
|
||||
separate, deliberate operation — they double runtime and aren't what CI
|
||||
gates on.
|
||||
- The overwrite guard in `reference_videos_cli.py upload` is default-on
|
||||
specifically because this skill exists. Re-seeding is a distinct operation
|
||||
that requires explicit `--force`.
|
||||
- Both artefact types share the same Modal flow: the orchestrator sets
|
||||
`--skip-reference-download` + `--no-fail-fast`, runs pytest, the test's
|
||||
helper writes the artefact (`.mp4` via `imageio` for pixel,
|
||||
`save_latent_reference` → `torch.save` for latent) BEFORE the
|
||||
missing-reference assertion raises. `_sync_generated_videos_to_volume` in
|
||||
`ssim_test.py` does a `shutil.copytree` of the whole `generated_videos/`
|
||||
tree, picking up `.mp4`, `.pt`, and the `*_ssim.json` / `*_latent.json`
|
||||
metric files alongside.
|
||||
|
||||
## References
|
||||
|
||||
- `fastvideo/tests/modal/ssim_test.py` — Modal orchestrator; see
|
||||
`--sync-generated-to-volume`, `--generated-volume-subdir`,
|
||||
`--skip-reference-download`, `--no-fail-fast`.
|
||||
- `fastvideo/tests/ssim/reference_videos_cli.py` — `copy-local`, `upload`
|
||||
(with `--model-id`, `--force`), `download`, `ensure` subcommands.
|
||||
Extension allowlist is `REFERENCE_EXTENSIONS = VIDEO_EXTENSIONS +
|
||||
LATENT_EXTENSIONS` (`.pt`).
|
||||
- `fastvideo/tests/ssim/README.md` — reference layout, HF repo conventions.
|
||||
- `fastvideo/tests/ssim/inference_similarity_utils.py` — pixel helpers
|
||||
(`run_text_to_video_similarity_test`,
|
||||
`run_image_to_video_similarity_test`, `build_init_kwargs`).
|
||||
- `fastvideo/tests/ssim/latent_similarity_utils.py` — latent helper
|
||||
(`run_text_to_latent_similarity_test`), slice spec dispatch
|
||||
(`_extract_expected_slice`), reference schema
|
||||
(`save_latent_reference` / `load_latent_reference`),
|
||||
`LATENT_REFERENCE_FORMAT_VERSION`.
|
||||
|
||||
## Changelog
|
||||
|
||||
| Date | Change |
|
||||
|------|--------|
|
||||
| 2026-04-17 | Initial version (Modal sync-to-volume flow). |
|
||||
| 2026-04-21 | Rewrite: single-test scope, explicit user-review pause, per-`model_id` upload, HF overwrite guard. Dropped `scripts/seed_ssim.sh`. |
|
||||
| 2026-04-21 | Post-first-run fixes: `modal volume get` needs `--force` when parent exists; download tree has an extra `generated_videos/` level so `--generated-dir` must reflect it. |
|
||||
| 2026-05-01 | Latent (`*.pt`) artefact support: artefact-type detection in step 1, type-aware verification (visual eyeball for mp4, numerics dump for pt) in step 4, FSDP+inference_mode failure-mode added, design notes for the unified Modal flow. Triggered by PR #1253 (LTX-2 latent migration + Stable Audio latent test). |
|
||||
@@ -0,0 +1,137 @@
|
||||
---
|
||||
name: summarize-run
|
||||
description: Extract a W&B run summary into a structured experiment report
|
||||
---
|
||||
|
||||
# Summarize Run
|
||||
|
||||
## Purpose
|
||||
After a training run completes (or at any checkpoint), extract key metrics from
|
||||
the W&B run summary and produce a structured markdown report. Supports both
|
||||
online (W&B API) and offline (local `wandb-summary.json`) modes.
|
||||
|
||||
## Prerequisites
|
||||
- Run has completed or reached a checkpoint with a saved summary.
|
||||
- For online: `WANDB_API_KEY` set in environment.
|
||||
- For offline: access to `<output_dir>/tracker/wandb/latest-run/files/wandb-summary.json`.
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Required | Description |
|
||||
|-----------|----------|-------------|
|
||||
| `run_id` | Yes* | W&B run ID for online access |
|
||||
| `output_dir` | Yes* | Local output dir for offline access |
|
||||
| `reference_run` | No | Path to reference `wandb-summary.json` for comparison |
|
||||
| `experiment_name` | No | Name for the journal entry (default: from W&B) |
|
||||
|
||||
\* One of `run_id` or `output_dir` is required.
|
||||
|
||||
## Steps
|
||||
|
||||
### 1. Load run summary
|
||||
|
||||
**Online**:
|
||||
|
||||
```python
|
||||
import wandb
|
||||
api = wandb.Api()
|
||||
run = api.run("<run_id>")
|
||||
summary = dict(run.summary)
|
||||
config = dict(run.config)
|
||||
```
|
||||
|
||||
**Offline** (existing codebase pattern from `fastvideo/tests/training/`):
|
||||
|
||||
```python
|
||||
import json
|
||||
summary_path = f"{output_dir}/tracker/wandb/latest-run/files/wandb-summary.json"
|
||||
with open(summary_path) as f:
|
||||
summary = json.load(f)
|
||||
```
|
||||
|
||||
### 2. Extract key fields
|
||||
|
||||
| Field | Source | Description |
|
||||
|-------|--------|-------------|
|
||||
| `train_loss` | `summary["train_loss"]` | Final training loss |
|
||||
| `avg_step_time` | `summary["avg_step_time"]` | Average seconds per step |
|
||||
| `step_time` | `summary["step_time"]` | Last step time |
|
||||
| `grad_norm` | `summary["grad_norm"]` | Final gradient norm |
|
||||
| `learning_rate` | `summary["learning_rate"]` | Final LR |
|
||||
| `_step` | `summary["_step"]` | Total steps completed |
|
||||
| `_runtime` | `summary["_runtime"]` | Total wall-clock seconds |
|
||||
| `validation_videos_*` | `summary[key]` | Validation video artifacts |
|
||||
|
||||
### 3. Compare against reference (optional)
|
||||
|
||||
Follow the pattern in `fastvideo/tests/training/Vanilla/test_training_loss.py`:
|
||||
|
||||
```python
|
||||
# Fields to compare
|
||||
compare_fields = ["train_loss", "grad_norm", "avg_step_time"]
|
||||
tolerance = 0.05 # 5% relative tolerance
|
||||
|
||||
for field in compare_fields:
|
||||
ref_val = reference_summary[field]
|
||||
cur_val = summary[field]
|
||||
diff_pct = abs(cur_val - ref_val) / abs(ref_val) * 100
|
||||
status = "✅" if diff_pct < tolerance * 100 else "⚠️"
|
||||
print(f"{status} {field}: {cur_val:.4f} (ref: {ref_val:.4f}, diff: {diff_pct:.1f}%)")
|
||||
```
|
||||
|
||||
### 4. Generate report
|
||||
|
||||
```markdown
|
||||
# Run Summary: <experiment_name>
|
||||
|
||||
| Metric | Value | Reference | Diff |
|
||||
|--------|-------|-----------|------|
|
||||
| Train Loss | 0.0788 | 0.0800 | -1.5% ✅ |
|
||||
| Avg Step Time | 2.81s | 2.80s | +0.4% ✅ |
|
||||
| Grad Norm | 0.408 | 0.410 | -0.5% ✅ |
|
||||
| Total Steps | 500 | — | — |
|
||||
| Wall Time | 23m 30s | — | — |
|
||||
|
||||
## Configuration
|
||||
- Model: Wan-AI/Wan2.1-T2V-1.3B-Diffusers
|
||||
- Learning Rate: 1e-6
|
||||
- Batch Size: 1
|
||||
- GPUs: 8 × (SP=1, TP=1)
|
||||
- Mixed Precision: bf16
|
||||
|
||||
## Validation Videos
|
||||
<list of validation video paths if available>
|
||||
|
||||
## Notes
|
||||
<any observations or anomalies>
|
||||
```
|
||||
|
||||
### 5. Update experiment journal
|
||||
|
||||
Append or update the experiment's entry in `.agents/memory/experiment-journal/README.md`
|
||||
with the final metrics and status.
|
||||
|
||||
## Outputs
|
||||
- Structured markdown report.
|
||||
- Updated experiment journal entry.
|
||||
|
||||
## Example Usage
|
||||
|
||||
```
|
||||
Summarize the run in output directory "outputs/wan_finetune":
|
||||
|
||||
output_dir: outputs/wan_finetune
|
||||
reference_run: fastvideo/tests/training/Vanilla/a40_reference_wandb_summary.json
|
||||
experiment_name: wan-t2v-finetune-lr1e6
|
||||
```
|
||||
|
||||
## References
|
||||
- `fastvideo/tests/training/Vanilla/test_training_loss.py` — reference comparison pattern
|
||||
- `fastvideo/tests/training/Vanilla/a40_reference_wandb_summary.json` — example summary
|
||||
- `fastvideo/tests/training/lora/test_lora_training.py` — LoRA summary comparison
|
||||
- `fastvideo/training/trackers.py` — tracker summary generation
|
||||
|
||||
## Changelog
|
||||
| Date | Change |
|
||||
|------|--------|
|
||||
| 2026-03-02 | Initial version |
|
||||
@@ -0,0 +1,54 @@
|
||||
---
|
||||
description: How to develop, validate, and register a new evaluation metric
|
||||
---
|
||||
|
||||
# Evaluation Development SOP
|
||||
|
||||
Standard procedure for adding new video quality evaluation metrics to the
|
||||
FastVideo agent toolkit.
|
||||
|
||||
## When to Use
|
||||
|
||||
- You need a metric that doesn't exist in `.agents/memory/evaluation-registry/README.md`.
|
||||
- An existing metric needs significant changes to its methodology.
|
||||
- You're exploring a new evaluation approach.
|
||||
|
||||
## Steps
|
||||
|
||||
### 1. Research
|
||||
|
||||
- Search `.agents/memory/related-work/` for existing evaluation approaches.
|
||||
- Check the `evaluation_registry.md` for current metrics and their limitations.
|
||||
- Review literature: FVD, CLIP-Score, human preference, etc.
|
||||
|
||||
### 2. Prototype
|
||||
|
||||
- Write a standalone script in `.agents/exploration/<metric-name>.md`.
|
||||
- Keep it simple: one script, minimal dependencies.
|
||||
- Test on a few known-good and known-bad video samples.
|
||||
|
||||
### 3. Validate
|
||||
|
||||
- **Known-good test**: Metric should score high on reference-quality videos.
|
||||
- **Known-bad test**: Metric should score low on degraded/unrelated videos.
|
||||
- **Sensitivity test**: Small quality differences should produce meaningful
|
||||
score differences.
|
||||
- Document thresholds and their justification.
|
||||
|
||||
### 4. Register
|
||||
|
||||
Update `.agents/memory/evaluation-registry/README.md`:
|
||||
- Add the metric with status `Active`.
|
||||
- Document location, thresholds, and trust level.
|
||||
|
||||
### 5. Integrate
|
||||
|
||||
Update `.agents/skills/evaluate-video-quality.md`:
|
||||
- Add the new metric as a section.
|
||||
- Include code examples and interpretation guide.
|
||||
|
||||
### 6. Document
|
||||
|
||||
- Move the exploration log content into the skill.
|
||||
- Clean up the exploration file or mark it as `promoted`.
|
||||
- If anything went wrong during development, create a lesson.
|
||||
@@ -0,0 +1,47 @@
|
||||
---
|
||||
description: When and how to log experiments in the experiment journal
|
||||
---
|
||||
|
||||
# Experiment Journaling SOP
|
||||
|
||||
Ensures every experiment is properly recorded with context and outcomes.
|
||||
|
||||
## When to Log
|
||||
|
||||
**Always.** Every experiment — even quick tests — should be journaled.
|
||||
|
||||
## Steps
|
||||
|
||||
### 1. Before Launch — Create Draft Entry
|
||||
|
||||
Use the `log-experiment` skill with `status: running`:
|
||||
- Include hypothesis and config.
|
||||
- Leave metrics, duration, and insight blank.
|
||||
|
||||
### 2. After 30-Minute Check — Update with Initial Metrics
|
||||
|
||||
Update the entry with:
|
||||
- Current loss and its trajectory direction.
|
||||
- Step time.
|
||||
- Number of validation videos generated.
|
||||
- Preliminary go/no-go assessment.
|
||||
|
||||
### 3. On Completion — Fill Final Entry
|
||||
|
||||
Update the entry with `status: completed`:
|
||||
- Final loss, grad norm, avg step time.
|
||||
- Total duration and steps.
|
||||
- Checkpoint path.
|
||||
- Key insight.
|
||||
|
||||
### 4. On Failure — Document Failure Mode
|
||||
|
||||
Update the entry with `status: failed`:
|
||||
- What went wrong (OOM, NaN, crash, etc.).
|
||||
- At what step the failure occurred.
|
||||
- Create a lesson in `.agents/lessons/` for non-trivial failures.
|
||||
|
||||
### 5. Cross-Reference
|
||||
|
||||
- Link related lessons: `**Related lessons**: .agents/lessons/<filename>.md`
|
||||
- Link related experiments: if this is a follow-up, reference the prior entry.
|
||||
@@ -0,0 +1,87 @@
|
||||
---
|
||||
description: End-to-end experiment lifecycle from hypothesis to lessons learned
|
||||
---
|
||||
|
||||
# Experiment Lifecycle SOP
|
||||
|
||||
Standard operating procedure for running ML training experiments on
|
||||
FastVideo-WorldModel. Every experiment should follow this flow.
|
||||
|
||||
## Overview
|
||||
|
||||
```
|
||||
Plan → Launch → Monitor → Summarize → Journal → Reflect
|
||||
```
|
||||
|
||||
## Steps
|
||||
|
||||
### 1. Plan the Experiment
|
||||
|
||||
Before launching:
|
||||
- [ ] Define a clear **hypothesis** (what you expect to learn).
|
||||
- [ ] Select the **model** and **pipeline** type (finetune, distill, lora, etc.).
|
||||
- [ ] Prepare the **dataset** (preprocessed into parquet format).
|
||||
- [ ] Review existing experiments in `.agents/memory/experiment-journal/README.md` for related work.
|
||||
- [ ] Check `.agents/lessons/` for known pitfalls with this configuration.
|
||||
- [ ] Document the plan in the experiment journal as a draft entry.
|
||||
|
||||
### 2. Launch the Experiment
|
||||
|
||||
Use the `launch-experiment` skill:
|
||||
- Provide: pipeline, model, data_path, num_gpus, and any hyperparameter overrides.
|
||||
- The skill generates the `torchrun` command and creates a journal entry.
|
||||
- Verify the command looks correct before executing.
|
||||
|
||||
Reference: `.agents/skills/launch-experiment.md`
|
||||
|
||||
### 3. Monitor the Experiment
|
||||
|
||||
Use the `monitor-experiment` skill:
|
||||
- Provide the W&B run ID (or output_dir for offline).
|
||||
- Monitor alerts: loss spikes, NaN gradients, step time regressions.
|
||||
- At the **30-minute mark**: perform the quality check.
|
||||
- Is loss decreasing?
|
||||
- Are validation videos reasonable?
|
||||
- Is step time consistent?
|
||||
- **Decision point**: Continue or abort based on the 30-min check.
|
||||
|
||||
Reference: `.agents/skills/monitor-experiment.md`
|
||||
|
||||
### 4. Summarize the Run
|
||||
|
||||
After completion (or at any checkpoint), use the `summarize-run` skill:
|
||||
- Extract final metrics from W&B summary.
|
||||
- Compare against reference runs if available.
|
||||
- Generate a structured report.
|
||||
|
||||
Reference: `.agents/skills/summarize-run.md`
|
||||
|
||||
### 5. Update the Experiment Journal
|
||||
|
||||
Use the `log-experiment` skill to update the journal entry:
|
||||
- Fill in final metrics, duration, checkpoint paths.
|
||||
- Record the key insight learned.
|
||||
- Set status to `completed`, `failed`, or `abandoned`.
|
||||
|
||||
Reference: `.agents/skills/log-experiment.md`
|
||||
|
||||
### 6. Reflect and Capture Lessons
|
||||
|
||||
After every experiment:
|
||||
- **What went right?** → Note in the journal insight field.
|
||||
- **What went wrong?** → Create a lesson in `.agents/lessons/`:
|
||||
- Use the template in `.agents/lessons/README.md`.
|
||||
- Cross-reference the experiment journal entry.
|
||||
- **What was surprising?** → Consider creating an exploration log if this
|
||||
warrants further investigation.
|
||||
|
||||
Reference: `.agents/workflows/lesson-capture.md`
|
||||
|
||||
## Validation Criteria
|
||||
|
||||
This SOP is validated when an agent can:
|
||||
1. Follow steps 1–6 end-to-end for a minimal training run
|
||||
(e.g., `examples/training/finetune/wan_t2v_1.3B/crush_smol/finetune_t2v.sh`
|
||||
with `--max_train_steps 5`).
|
||||
2. Produce a complete experiment journal entry.
|
||||
3. Generate a run summary report.
|
||||
@@ -0,0 +1,71 @@
|
||||
---
|
||||
description: Post-experiment reflection to capture lessons learned
|
||||
---
|
||||
|
||||
# Lesson Capture SOP
|
||||
|
||||
Systematic procedure for turning experiment outcomes into persistent knowledge.
|
||||
|
||||
## When to Use
|
||||
|
||||
After **every** completed or failed experiment. Even successful experiments
|
||||
can yield lessons (e.g., "LR 5e-5 works better than 1e-5 for LoRA").
|
||||
|
||||
## Steps
|
||||
|
||||
### 1. Review the Experiment
|
||||
|
||||
Read the experiment journal entry. Ask:
|
||||
- Did anything go wrong?
|
||||
- Was anything surprising?
|
||||
- Did anything take longer than expected?
|
||||
- Was a workaround needed?
|
||||
|
||||
### 2. Decide: Lesson or Not?
|
||||
|
||||
| Situation | Action |
|
||||
|-----------|--------|
|
||||
| Something broke | Create a lesson (category: `infrastructure` or `data`) |
|
||||
| Hyperparameter choice mattered | Create a lesson (category: `hyperparameter`) |
|
||||
| Porting issue found | Create a lesson (category: `porting`) |
|
||||
| Evaluation metric was misleading | Create a lesson (category: `evaluation`) |
|
||||
| Everything went smoothly | No lesson needed, but note in the journal insight |
|
||||
|
||||
### 3. Create the Lesson File
|
||||
|
||||
In `.agents/lessons/`, create `<YYYY-MM-DD>_<short-slug>.md`:
|
||||
|
||||
```markdown
|
||||
---
|
||||
date: <ISO-8601>
|
||||
experiment: <journal entry reference>
|
||||
category: hyperparameter | data | infrastructure | evaluation | porting
|
||||
severity: critical | important | minor
|
||||
---
|
||||
|
||||
# <Short Descriptive Title>
|
||||
|
||||
## What Happened
|
||||
<description>
|
||||
|
||||
## Root Cause
|
||||
<analysis>
|
||||
|
||||
## Fix / Workaround
|
||||
<resolution>
|
||||
|
||||
## Prevention
|
||||
<how to avoid in future>
|
||||
```
|
||||
|
||||
### 4. Cross-Reference
|
||||
|
||||
- Update the experiment journal entry with a link to the lesson file.
|
||||
- If a similar lesson already exists, add a reference or update it.
|
||||
|
||||
### 5. Periodic Pattern Review
|
||||
|
||||
Every ~10 lessons, scan for patterns:
|
||||
- Multiple lessons in the same category → consider a new skill or SOP.
|
||||
- Repeated mistakes → strengthen the relevant SOP with a checklist item.
|
||||
- Infrastructure issues → propose a codebase fix.
|
||||
@@ -0,0 +1,67 @@
|
||||
---
|
||||
description: Synchronize the STATUS.md dashboard by scanning .agents/ directories
|
||||
---
|
||||
|
||||
# Sync Dashboard
|
||||
|
||||
Updates `.agents/STATUS.md` by scanning the skills, workflows, memory, lessons,
|
||||
and exploration directories to reflect what actually exists on disk.
|
||||
|
||||
## When to Use
|
||||
|
||||
- After adding, removing, or renaming any file in `.agents/`.
|
||||
- Periodically (e.g., at end of each conversation session).
|
||||
- When the dashboard feels out of date.
|
||||
|
||||
## Steps
|
||||
|
||||
### 1. Scan directories
|
||||
|
||||
List all files in each directory:
|
||||
|
||||
```bash
|
||||
echo "=== Skills ==="
|
||||
ls -1 .agents/skills/*.md 2>/dev/null | grep -v SKILL_TEMPLATE
|
||||
|
||||
echo "=== Workflows ==="
|
||||
ls -1 .agents/workflows/*.md 2>/dev/null
|
||||
|
||||
echo "=== Memory ==="
|
||||
ls -1 .agents/memory/*.md 2>/dev/null
|
||||
ls -1 .agents/memory/related-work/*.md 2>/dev/null | grep -v README
|
||||
|
||||
echo "=== Lessons ==="
|
||||
ls -1 .agents/lessons/*.md 2>/dev/null | grep -v README
|
||||
|
||||
echo "=== Exploration ==="
|
||||
ls -1 .agents/exploration/*.md 2>/dev/null | grep -v README
|
||||
```
|
||||
|
||||
### 2. Compare with STATUS.md
|
||||
|
||||
For each file found:
|
||||
- If it's in STATUS.md → leave it (preserve status/trust/tested fields).
|
||||
- If it's NOT in STATUS.md → add it with status `🔴 Stub`, trust `None`, tested `❌`.
|
||||
|
||||
For each entry in STATUS.md:
|
||||
- If the file no longer exists → mark it as `❌ Removed` or delete the row.
|
||||
|
||||
### 3. Update counts
|
||||
|
||||
Recalculate the summary table at the top:
|
||||
- Count files per category.
|
||||
- Count by status (Ready, Draft, Stub).
|
||||
|
||||
### 4. Update timestamp
|
||||
|
||||
Set `_Last synced: <current date>_` at the top of STATUS.md.
|
||||
|
||||
### 5. Review
|
||||
|
||||
Read through the updated STATUS.md for accuracy. Flag anything that looks wrong.
|
||||
|
||||
## Notes
|
||||
|
||||
- Do NOT change trust levels during sync — those are set manually after testing.
|
||||
- Do NOT change status during sync — status changes require actual validation.
|
||||
- This workflow only handles structural sync (file existence), not content review.
|
||||
@@ -0,0 +1,46 @@
|
||||
{
|
||||
"benchmark_id": "wan-t2v-1.3b-2gpu",
|
||||
"description": "Wan2.1 T2V 1.3B inference performance",
|
||||
"model": {
|
||||
"model_path": "Wan-AI/Wan2.1-T2V-1.3B-Diffusers",
|
||||
"model_short_name": "Wan2.1-T2V-1.3B"
|
||||
},
|
||||
"init_kwargs": {
|
||||
"num_gpus": 2,
|
||||
"flow_shift": 7.0,
|
||||
"sp_size": 2,
|
||||
"tp_size": 1,
|
||||
"vae_sp": true,
|
||||
"vae_tiling": true,
|
||||
"text_encoder_precisions": ["fp32"]
|
||||
},
|
||||
"generation_kwargs": {
|
||||
"height": 480,
|
||||
"width": 832,
|
||||
"num_frames": 45,
|
||||
"num_inference_steps": 4,
|
||||
"guidance_scale": 3,
|
||||
"embedded_cfg_scale": 6,
|
||||
"seed": 1024,
|
||||
"fps": 24,
|
||||
"neg_prompt": "Bright tones, overexposed, static, blurred details, subtitles, style, works, paintings, images, static, overall gray, worst quality, low quality, JPEG compression residue, ugly, incomplete, extra fingers, poorly drawn hands, poorly drawn faces, deformed, disfigured, misshapen limbs, fused fingers, still picture, messy background, three legs, many people in the background, walking backwards"
|
||||
},
|
||||
"test_prompts": [
|
||||
"Will Smith casually eats noodles, his relaxed demeanor contrasting with the energetic background of a bustling street food market. The scene captures a mix of humor and authenticity. Mid-shot framing, vibrant lighting."
|
||||
],
|
||||
"run_config": {
|
||||
"num_warmup_runs": 2,
|
||||
"num_measurement_runs": 5,
|
||||
"required_gpus": 2
|
||||
},
|
||||
"thresholds": {
|
||||
"L40S": {
|
||||
"max_generation_time_s": 34.0,
|
||||
"max_peak_memory_mb": 11000.0
|
||||
},
|
||||
"default": {
|
||||
"max_generation_time_s": 120.0,
|
||||
"max_peak_memory_mb": 30000.0
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,438 @@
|
||||
env:
|
||||
IMAGE_VERSION: "py3.12-latest"
|
||||
BUILDKITE_CLEAN_CHECKOUT: true
|
||||
|
||||
notify:
|
||||
- github_commit_status:
|
||||
context: "fastcheck-passed"
|
||||
if: build.env("TEST_SCOPE") == "fastcheck" || build.env("TEST_SCOPE") == null
|
||||
- github_commit_status:
|
||||
context: "full-suite-passed"
|
||||
if: build.env("TEST_SCOPE") == "full"
|
||||
- github_commit_status:
|
||||
context: "direct-test-completed"
|
||||
if: build.env("TEST_SCOPE") == "direct"
|
||||
|
||||
steps:
|
||||
# ============================================================
|
||||
# Direct test: triggered by /test <name> slash command.
|
||||
# Labels match fastcheck/full-suite counterparts so the GitHub
|
||||
# check status overwrites the original failed check.
|
||||
# Only ONE step executes per build (gated by TEST_TYPE).
|
||||
# ============================================================
|
||||
|
||||
# --- Fastcheck-scope direct tests ---
|
||||
- label: ":microscope: Encoder Tests"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "encoder"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":microscope: VAE Tests"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "vae"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":microscope: Transformer Tests"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "transformer"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":microscope: Kernel Tests"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "kernel_tests"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":microscope: Unit Tests"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "unit_test"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
|
||||
# --- Full-suite-scope direct tests ---
|
||||
- label: ":bar_chart: SSIM Tests"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "ssim"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
- exit_status: 1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":test_tube: LoRA Inference Tests"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "inference_lora"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":test_tube: Training Tests"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "training"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":test_tube: Distillation DMD Tests"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "distillation_dmd"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":test_tube: Self-Forcing Tests"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "self_forcing"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":test_tube: LoRA Training Tests"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "training_lora"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
- exit_status: 1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":test_tube: Training Tests VSA"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "training_vsa"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
- exit_status: 1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":test_tube: Inference Tests VMoBA"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "inference_vmoba"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":test_tube: Performance Tests"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "performance"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":test_tube: API Server Tests"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "api_server"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
|
||||
# ============================================================
|
||||
# Fastcheck: Runs on every PR (~10-15 min parallel)
|
||||
# Core component validation: encoders, VAEs, transformers,
|
||||
# CUDA kernels, and unit tests.
|
||||
# ============================================================
|
||||
- label: "Trigger Fastcheck"
|
||||
if: build.env("TEST_SCOPE") == "fastcheck" || build.env("TEST_SCOPE") == null
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
plugins:
|
||||
- monorepo-diff#v1.4.0:
|
||||
diff: 'git fetch origin "${BUILDKITE_PULL_REQUEST_BASE_BRANCH:-main}" && git diff --name-only "origin/${BUILDKITE_PULL_REQUEST_BASE_BRANCH:-main}...HEAD"'
|
||||
watch:
|
||||
- path:
|
||||
- "fastvideo/models/encoders/**"
|
||||
- "fastvideo/models/loader/**"
|
||||
- "fastvideo/tests/encoders/**"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile.python3.12"
|
||||
config:
|
||||
command: "timeout 20m .buildkite/scripts/pr_test.sh"
|
||||
label: ":microscope: Encoder Tests"
|
||||
env:
|
||||
- TEST_TYPE=encoder
|
||||
agents:
|
||||
queue: "default"
|
||||
- path:
|
||||
- "fastvideo/models/vaes/**"
|
||||
- "fastvideo/models/loader/**"
|
||||
- "fastvideo/tests/vaes/**"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile.python3.12"
|
||||
config:
|
||||
command: "timeout 20m .buildkite/scripts/pr_test.sh"
|
||||
label: ":microscope: VAE Tests"
|
||||
env:
|
||||
- TEST_TYPE=vae
|
||||
agents:
|
||||
queue: "default"
|
||||
- path:
|
||||
- "fastvideo/models/dits/**"
|
||||
- "fastvideo/models/loader/**"
|
||||
- "fastvideo/tests/transformers/**"
|
||||
- "fastvideo/layers/**"
|
||||
- "fastvideo/attention/**"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile.python3.12"
|
||||
config:
|
||||
command: "timeout 15m .buildkite/scripts/pr_test.sh"
|
||||
label: ":microscope: Transformer Tests"
|
||||
env:
|
||||
- TEST_TYPE=transformer
|
||||
agents:
|
||||
queue: "default"
|
||||
- path:
|
||||
- "fastvideo-kernel/**"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile.python3.12"
|
||||
config:
|
||||
command: "timeout 15m .buildkite/scripts/pr_test.sh"
|
||||
label: ":microscope: Kernel Tests"
|
||||
env:
|
||||
- TEST_TYPE=kernel_tests
|
||||
agents:
|
||||
queue: "default"
|
||||
- path:
|
||||
- "fastvideo/**"
|
||||
- ".buildkite/**"
|
||||
- ".github/**"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile.python3.12"
|
||||
config:
|
||||
command: "timeout 15m .buildkite/scripts/pr_test.sh"
|
||||
label: ":microscope: Unit Tests"
|
||||
env:
|
||||
- TEST_TYPE=unit_test
|
||||
agents:
|
||||
queue: "default"
|
||||
|
||||
# ============================================================
|
||||
# Full Suite: Runs when TEST_SCOPE=full
|
||||
# Triggered by adding the 'ready' label (via ci-trigger-full-suite.yml)
|
||||
# or on-demand via /test full slash command.
|
||||
# Includes integration tests, SSIM regression, training pipelines,
|
||||
# and performance benchmarks.
|
||||
# ============================================================
|
||||
- label: "Trigger Full Suite"
|
||||
if: build.env("TEST_SCOPE") == "full"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
plugins:
|
||||
- monorepo-diff#v1.4.0:
|
||||
diff: 'git fetch origin "${BUILDKITE_PULL_REQUEST_BASE_BRANCH:-main}" && git diff --name-only "origin/${BUILDKITE_PULL_REQUEST_BASE_BRANCH:-main}...HEAD"'
|
||||
watch:
|
||||
- path:
|
||||
- "fastvideo/**/*.py"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile.python3.12"
|
||||
config:
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
label: ":bar_chart: SSIM Tests"
|
||||
env:
|
||||
- TEST_TYPE=ssim
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- path:
|
||||
- "fastvideo/tests/lora/**"
|
||||
- "fastvideo/models/loader/**"
|
||||
- "fastvideo/tests/transformers/**"
|
||||
- "fastvideo/pipelines/**"
|
||||
- "fastvideo/layers/lora/**"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile.python3.12"
|
||||
config:
|
||||
command: "timeout 20m .buildkite/scripts/pr_test.sh"
|
||||
label: ":test_tube: LoRA Inference Tests"
|
||||
env:
|
||||
- TEST_TYPE=inference_lora
|
||||
agents:
|
||||
queue: "default"
|
||||
- path:
|
||||
- "fastvideo/**"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile.python3.12"
|
||||
config:
|
||||
command: "timeout 15m .buildkite/scripts/pr_test.sh"
|
||||
label: ":test_tube: Training Tests"
|
||||
env:
|
||||
- TEST_TYPE=training
|
||||
agents:
|
||||
queue: "default"
|
||||
- path:
|
||||
- "fastvideo/training/*distillation_pipeline.py"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile.python3.12"
|
||||
config:
|
||||
command: "timeout 15m .buildkite/scripts/pr_test.sh"
|
||||
label: ":test_tube: Distillation DMD Tests"
|
||||
env:
|
||||
- TEST_TYPE=distillation_dmd
|
||||
agents:
|
||||
queue: "default"
|
||||
- path:
|
||||
- "fastvideo/training/*self_forcing_distillation_pipeline.py"
|
||||
- "fastvideo/tests/training/self-forcing/**"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile.python3.12"
|
||||
config:
|
||||
command: "timeout 15m .buildkite/scripts/pr_test.sh"
|
||||
label: ":test_tube: Self-Forcing Tests"
|
||||
env:
|
||||
- TEST_TYPE=self_forcing
|
||||
agents:
|
||||
queue: "default"
|
||||
- path:
|
||||
- "fastvideo/**"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile.python3.12"
|
||||
config:
|
||||
command: "timeout 15m .buildkite/scripts/pr_test.sh"
|
||||
label: ":test_tube: LoRA Training Tests"
|
||||
env:
|
||||
- TEST_TYPE=training_lora
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- path:
|
||||
- "fastvideo/**"
|
||||
- "fastvideo-kernel/**"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile.python3.12"
|
||||
config:
|
||||
command: "timeout 15m .buildkite/scripts/pr_test.sh"
|
||||
label: ":test_tube: Training Tests VSA"
|
||||
env:
|
||||
- TEST_TYPE=training_vsa
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- path:
|
||||
- "fastvideo-kernel/**"
|
||||
- "fastvideo/attention/backends/vmoba.py"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile.python3.12"
|
||||
config:
|
||||
command: "timeout 15m .buildkite/scripts/pr_test.sh"
|
||||
label: ":test_tube: Inference Tests VMoBA"
|
||||
env:
|
||||
- TEST_TYPE=inference_vmoba
|
||||
agents:
|
||||
queue: "default"
|
||||
- path:
|
||||
- "fastvideo/models/dits/**"
|
||||
- "fastvideo/pipelines/**"
|
||||
- "fastvideo/attention/**"
|
||||
- "fastvideo/layers/**"
|
||||
- "fastvideo/worker/**"
|
||||
- "fastvideo/entrypoints/**"
|
||||
- "fastvideo/tests/performance/**"
|
||||
- ".buildkite/performance-benchmarks/**"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile.python3.12"
|
||||
config:
|
||||
command: "timeout 30m .buildkite/scripts/pr_test.sh"
|
||||
label: ":test_tube: Performance Tests"
|
||||
env:
|
||||
- TEST_TYPE=performance
|
||||
agents:
|
||||
queue: "default"
|
||||
- path:
|
||||
- "fastvideo/entrypoints/openai/**"
|
||||
- "fastvideo/entrypoints/cli/serve.py"
|
||||
- "fastvideo/tests/entrypoints/test_openai_api_integration.py"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile.python3.12"
|
||||
config:
|
||||
command: "timeout 30m .buildkite/scripts/pr_test.sh"
|
||||
label: ":test_tube: API Server Tests"
|
||||
env:
|
||||
- TEST_TYPE=api_server
|
||||
agents:
|
||||
queue: "default"
|
||||
@@ -0,0 +1,249 @@
|
||||
#!/bin/bash
|
||||
set -uo pipefail
|
||||
|
||||
log() {
|
||||
echo "[$(date '+%Y-%m-%d %H:%M:%S')] $1"
|
||||
}
|
||||
|
||||
log "=== Starting Modal test execution ==="
|
||||
|
||||
# Change to the project directory
|
||||
cd "$(dirname "$0")/../.."
|
||||
PROJECT_ROOT=$(pwd)
|
||||
log "Project root: $PROJECT_ROOT"
|
||||
|
||||
# Install Modal if not available
|
||||
if ! python3 -m modal --version &> /dev/null; then
|
||||
log "Modal not found, installing..."
|
||||
if ! command -v uv &> /dev/null; then
|
||||
log "uv not found, bootstrapping..."
|
||||
if ! curl -LsSf https://astral.sh/uv/install.sh | sh; then
|
||||
log "Error: Failed to bootstrap uv via astral.sh installer."
|
||||
exit 1
|
||||
fi
|
||||
export PATH="$HOME/.local/bin:$PATH"
|
||||
if ! command -v uv &> /dev/null; then
|
||||
log "Error: uv still not on PATH after bootstrap."
|
||||
exit 1
|
||||
fi
|
||||
fi
|
||||
# --break-system-packages preserves prior `pip install --user` semantics on PEP 668 agents.
|
||||
uv pip install --system --break-system-packages modal
|
||||
|
||||
# Verify installation
|
||||
if ! python3 -m modal --version &> /dev/null; then
|
||||
log "Error: Failed to install modal. Please install it manually."
|
||||
exit 1
|
||||
fi
|
||||
fi
|
||||
|
||||
log "modal version: $(python3 -m modal --version)"
|
||||
|
||||
# Set up Modal authentication using Buildkite secrets
|
||||
log "Setting up Modal authentication from Buildkite secrets..."
|
||||
MODAL_TOKEN_ID=$(buildkite-agent secret get modal_token_id)
|
||||
MODAL_TOKEN_SECRET=$(buildkite-agent secret get modal_token_secret)
|
||||
|
||||
# Retrieve other secrets
|
||||
WANDB_API_KEY=$(buildkite-agent secret get wandb_api_key)
|
||||
HF_API_KEY=$(buildkite-agent secret get hf_api_key)
|
||||
|
||||
if [ -n "$MODAL_TOKEN_ID" ] && [ -n "$MODAL_TOKEN_SECRET" ]; then
|
||||
log "Retrieved Modal credentials from Buildkite secrets"
|
||||
python3 -m modal token set --token-id "$MODAL_TOKEN_ID" --token-secret "$MODAL_TOKEN_SECRET" --profile buildkite-ci --activate --verify
|
||||
if [ $? -eq 0 ]; then
|
||||
log "Modal authentication successful"
|
||||
else
|
||||
log "Error: Failed to set Modal credentials"
|
||||
exit 1
|
||||
fi
|
||||
else
|
||||
log "Error: Could not retrieve Modal credentials from Buildkite secrets."
|
||||
log "Please ensure 'modal_token_id' and 'modal_token_secret' secrets are set in Buildkite."
|
||||
exit 1
|
||||
fi
|
||||
|
||||
MODAL_TEST_FILE="fastvideo/tests/modal/pr_test.py"
|
||||
MODAL_SSIM_TEST_FILE="fastvideo/tests/modal/ssim_test.py"
|
||||
|
||||
if [ -z "${TEST_TYPE:-}" ]; then
|
||||
log "Error: TEST_TYPE environment variable is not set"
|
||||
exit 1
|
||||
fi
|
||||
log "Test type: $TEST_TYPE"
|
||||
|
||||
EFFECTIVE_PR=${BUILDKITE_PULL_REQUEST:-false}
|
||||
if [ "$EFFECTIVE_PR" = "false" ] && [ -n "${PR_NUMBER:-}" ]; then
|
||||
EFFECTIVE_PR=$PR_NUMBER
|
||||
fi
|
||||
MODAL_ENV="BUILDKITE_REPO=$BUILDKITE_REPO BUILDKITE_COMMIT=$BUILDKITE_COMMIT BUILDKITE_PULL_REQUEST=$EFFECTIVE_PR BUILDKITE_BRANCH=${BUILDKITE_BRANCH:-} TEST_SCOPE=${TEST_SCOPE:-} IMAGE_VERSION=$IMAGE_VERSION"
|
||||
|
||||
POST_RUN_HOOK=""
|
||||
|
||||
upload_performance_artifacts() {
|
||||
SHORT_SHA=${BUILDKITE_COMMIT:0:7}
|
||||
LOCAL_DIR="downloaded_reports"
|
||||
|
||||
_download_reports() {
|
||||
log "Downloading perf_reports/ from Modal Volume..."
|
||||
mkdir -p "$LOCAL_DIR"
|
||||
if ! modal volume get hf-model-weights "perf_reports/" "$LOCAL_DIR"; then
|
||||
log "Error: Failed to download perf_reports/ from Modal Volume."
|
||||
return 1
|
||||
fi
|
||||
}
|
||||
|
||||
_upload_dashboard() {
|
||||
local target
|
||||
target=$(find "$LOCAL_DIR" -name "dashboard_${SHORT_SHA}_*" | head -n 1)
|
||||
log "TARGET dashboard: '$target'"
|
||||
|
||||
if [ -n "$target" ]; then
|
||||
log "Found dashboard: $target. Uploading to Buildkite..."
|
||||
buildkite-agent artifact upload "$target"
|
||||
buildkite-agent annotate --style info --context "perf-dashboard" < "$target"
|
||||
else
|
||||
log "Warning: Could not find a dashboard file matching $SHORT_SHA"
|
||||
fi
|
||||
}
|
||||
|
||||
_upload_perf_summary() {
|
||||
local target
|
||||
target=$(find "$LOCAL_DIR" -name "perf_${SHORT_SHA}_*" | head -n 1)
|
||||
log "TARGET perf summary: '$target'"
|
||||
|
||||
if [ -n "$target" ]; then
|
||||
log "Found perf summary: $target. Uploading to Buildkite..."
|
||||
buildkite-agent artifact upload "$target"
|
||||
buildkite-agent annotate --style info --context "perf-summary" < "$target"
|
||||
else
|
||||
log "Warning: Could not find a perf summary file matching $SHORT_SHA"
|
||||
fi
|
||||
}
|
||||
|
||||
_upload_normalized_perf_results() {
|
||||
local found=0
|
||||
while IFS= read -r -d '' target; do
|
||||
found=1
|
||||
log "Found normalized performance result: $target. Uploading to Buildkite..."
|
||||
buildkite-agent artifact upload "$target"
|
||||
done < <(find "$LOCAL_DIR" -path "*/results/normalized_perf_*.json" -print0)
|
||||
|
||||
if [ "$found" -eq 0 ]; then
|
||||
log "No normalized performance result artifacts found. This is expected when the rolling performance comparison did not run."
|
||||
fi
|
||||
}
|
||||
|
||||
_cleanup_modal_volume() {
|
||||
log "Cleaning up perf_reports/ from Modal Volume..."
|
||||
if modal volume rm hf-model-weights "perf_reports/" --recursive; then
|
||||
log "Successfully deleted perf_reports/ from Modal Volume."
|
||||
else
|
||||
log "Warning: Failed to delete perf_reports/ from Modal Volume. Manual cleanup may be required."
|
||||
fi
|
||||
}
|
||||
|
||||
_cleanup_local() {
|
||||
log "Cleaning up local download directory..."
|
||||
rm -rf "$LOCAL_DIR"
|
||||
}
|
||||
|
||||
# --- Main flow ---
|
||||
_download_reports || { _cleanup_local; return 1; }
|
||||
_upload_dashboard
|
||||
_upload_perf_summary
|
||||
_upload_normalized_perf_results
|
||||
_cleanup_modal_volume
|
||||
_cleanup_local
|
||||
}
|
||||
|
||||
case "$TEST_TYPE" in
|
||||
"encoder")
|
||||
log "Running encoder tests..."
|
||||
MODAL_COMMAND="$MODAL_ENV HF_API_KEY=$HF_API_KEY python3 -m modal run $MODAL_TEST_FILE::run_encoder_tests"
|
||||
;;
|
||||
"vae")
|
||||
log "Running VAE tests..."
|
||||
MODAL_COMMAND="$MODAL_ENV HF_API_KEY=$HF_API_KEY python3 -m modal run $MODAL_TEST_FILE::run_vae_tests"
|
||||
;;
|
||||
"transformer")
|
||||
log "Running transformer tests..."
|
||||
MODAL_COMMAND="$MODAL_ENV HF_API_KEY=$HF_API_KEY python3 -m modal run $MODAL_TEST_FILE::run_transformer_tests"
|
||||
;;
|
||||
"ssim")
|
||||
log "Running SSIM tests..."
|
||||
MODAL_COMMAND="$MODAL_ENV HF_API_KEY=$HF_API_KEY python3 -m modal run $MODAL_SSIM_TEST_FILE::run_ssim_tests"
|
||||
;;
|
||||
"training")
|
||||
log "Running training tests..."
|
||||
MODAL_COMMAND="$MODAL_ENV WANDB_API_KEY=$WANDB_API_KEY python3 -m modal run $MODAL_TEST_FILE::run_training_tests"
|
||||
;;
|
||||
"training_lora")
|
||||
log "Running LoRA training tests..."
|
||||
MODAL_COMMAND="$MODAL_ENV WANDB_API_KEY=$WANDB_API_KEY python3 -m modal run $MODAL_TEST_FILE::run_training_lora_tests"
|
||||
;;
|
||||
"training_vsa")
|
||||
log "Running training VSA tests..."
|
||||
MODAL_COMMAND="$MODAL_ENV WANDB_API_KEY=$WANDB_API_KEY python3 -m modal run $MODAL_TEST_FILE::run_training_tests_VSA"
|
||||
;;
|
||||
"kernel_tests")
|
||||
log "Running kernel tests..."
|
||||
MODAL_COMMAND="$MODAL_ENV python3 -m modal run $MODAL_TEST_FILE::run_kernel_tests"
|
||||
;;
|
||||
"inference_lora")
|
||||
log "Running LoRA tests..."
|
||||
MODAL_COMMAND="$MODAL_ENV python3 -m modal run $MODAL_TEST_FILE::run_inference_lora_tests"
|
||||
;;
|
||||
"distillation_dmd")
|
||||
log "Running distillation DMD tests..."
|
||||
MODAL_COMMAND="$MODAL_ENV WANDB_API_KEY=$WANDB_API_KEY python3 -m modal run $MODAL_TEST_FILE::run_distill_dmd_tests"
|
||||
;;
|
||||
# run_inference_tests_vmoba
|
||||
"self_forcing")
|
||||
log "Running self-forcing tests..."
|
||||
MODAL_COMMAND="$MODAL_ENV WANDB_API_KEY=$WANDB_API_KEY python3 -m modal run $MODAL_TEST_FILE::run_self_forcing_tests"
|
||||
;;
|
||||
"inference_vmoba")
|
||||
log "Running V-MoBA inference tests..."
|
||||
MODAL_COMMAND="$MODAL_ENV python3 -m modal run $MODAL_TEST_FILE::run_inference_tests_vmoba"
|
||||
;;
|
||||
"unit_test")
|
||||
log "Running unit tests..."
|
||||
MODAL_COMMAND="$MODAL_ENV python3 -m modal run $MODAL_TEST_FILE::run_unit_test"
|
||||
;;
|
||||
"lora_extraction")
|
||||
log "Running LoRA extraction tests..."
|
||||
MODAL_COMMAND="$MODAL_ENV HF_API_KEY=$HF_API_KEY python3 -m modal run $MODAL_TEST_FILE::run_lora_extraction_tests"
|
||||
;;
|
||||
"performance")
|
||||
log "Running performance tests on Modal..."
|
||||
MODAL_COMMAND="$MODAL_ENV HF_API_KEY=$HF_API_KEY python3 -m modal run $MODAL_TEST_FILE::run_performance_tests"
|
||||
POST_RUN_HOOK="upload_performance_artifacts"
|
||||
;;
|
||||
"api_server")
|
||||
log "Running API server integration tests..."
|
||||
MODAL_COMMAND="$MODAL_ENV HF_API_KEY=$HF_API_KEY python3 -m modal run $MODAL_TEST_FILE::run_api_server_tests"
|
||||
;;
|
||||
*)
|
||||
log "Error: Unknown test type: $TEST_TYPE"
|
||||
exit 1
|
||||
;;
|
||||
esac
|
||||
|
||||
log "Executing: $MODAL_COMMAND"
|
||||
eval "$MODAL_COMMAND"
|
||||
TEST_EXIT_CODE=$?
|
||||
|
||||
if [ $TEST_EXIT_CODE -eq 0 ]; then
|
||||
log "Modal test completed successfully"
|
||||
else
|
||||
log "Error: Modal test failed with exit code: $TEST_EXIT_CODE"
|
||||
fi
|
||||
|
||||
if [ -n "$POST_RUN_HOOK" ]; then
|
||||
log "Executing post-run hook: $POST_RUN_HOOK"
|
||||
"$POST_RUN_HOOK"
|
||||
fi
|
||||
|
||||
log "=== Test execution completed with exit code: $TEST_EXIT_CODE ==="
|
||||
exit $TEST_EXIT_CODE
|
||||
@@ -0,0 +1,53 @@
|
||||
#!/bin/bash
|
||||
set -uo pipefail
|
||||
|
||||
log() {
|
||||
echo "[$(date '+%Y-%m-%d %H:%M:%S')] $1"
|
||||
}
|
||||
|
||||
log "=== Starting pre-commit checks ==="
|
||||
|
||||
cd "$(dirname "$0")/../.."
|
||||
PROJECT_ROOT=$(pwd)
|
||||
log "Project root: $PROJECT_ROOT"
|
||||
|
||||
if ! python3 -m pre_commit --version &> /dev/null; then
|
||||
log "pre-commit not found, installing..."
|
||||
if ! command -v uv &> /dev/null; then
|
||||
log "uv not found, bootstrapping..."
|
||||
if ! curl -LsSf https://astral.sh/uv/install.sh | sh; then
|
||||
log "Error: Failed to bootstrap uv via astral.sh installer."
|
||||
exit 1
|
||||
fi
|
||||
export PATH="$HOME/.local/bin:$PATH"
|
||||
if ! command -v uv &> /dev/null; then
|
||||
log "Error: uv still not on PATH after bootstrap."
|
||||
exit 1
|
||||
fi
|
||||
fi
|
||||
# --break-system-packages preserves prior `pip install --user` semantics on PEP 668 agents.
|
||||
uv pip install --system --break-system-packages pre-commit==4.0.1
|
||||
|
||||
if ! python3 -m pre_commit --version &> /dev/null; then
|
||||
log "Error: Failed to install pre-commit."
|
||||
exit 1
|
||||
fi
|
||||
fi
|
||||
|
||||
log "Pre-commit version: $(python3 -m pre_commit --version)"
|
||||
|
||||
log "Installing/updating pre-commit hooks..."
|
||||
python3 -m pre_commit install --install-hooks
|
||||
|
||||
log "Running pre-commit checks on all files..."
|
||||
python3 -m pre_commit run --all-files
|
||||
PRE_COMMIT_EXIT_CODE=$?
|
||||
|
||||
if [ $PRE_COMMIT_EXIT_CODE -eq 0 ]; then
|
||||
log "Pre-commit checks completed successfully"
|
||||
else
|
||||
log "Error: Pre-commit checks failed with exit code: $PRE_COMMIT_EXIT_CODE"
|
||||
fi
|
||||
|
||||
log "=== Pre-commit checks completed with exit code: $PRE_COMMIT_EXIT_CODE ==="
|
||||
exit $PRE_COMMIT_EXIT_CODE
|
||||
@@ -4,14 +4,6 @@ title: "[Bug] "
|
||||
labels: ['Bug']
|
||||
|
||||
body:
|
||||
- type: textarea
|
||||
attributes:
|
||||
label: Environment
|
||||
description: |
|
||||
Please share your environment with us. You can run the command **python fastvideo/utils/env_utils.py** and copy-paste its output below.
|
||||
placeholder: FastVideo version, platform, python version, cuda version...
|
||||
validations:
|
||||
required: true
|
||||
- type: textarea
|
||||
attributes:
|
||||
label: Describe the bug
|
||||
@@ -25,5 +17,13 @@ body:
|
||||
What command or script did you run? Which **model** are you using?
|
||||
placeholder: |
|
||||
A placeholder for the command.
|
||||
validations:
|
||||
required: true
|
||||
- type: textarea
|
||||
attributes:
|
||||
label: Environment
|
||||
description: |
|
||||
Please share your environment with us. You can run the command **python collect_env.py** and copy-paste its output below.
|
||||
placeholder: FastVideo version, platform, python version, cuda version...
|
||||
validations:
|
||||
required: true
|
||||
@@ -0,0 +1,56 @@
|
||||
name: 💬 Request for comments (RFC).
|
||||
description: Ask for feedback on major architectural changes or design choices.
|
||||
title: "[RFC]: "
|
||||
labels: ["RFC"]
|
||||
|
||||
body:
|
||||
- type: markdown
|
||||
attributes:
|
||||
value: >
|
||||
#### Please take a look at previous [RFCs](https://github.com/hao-ai-lab/FastVideo/issues?q=label%3ARFC+sort%3Aupdated-desc) for reference.
|
||||
- type: textarea
|
||||
attributes:
|
||||
label: Motivation.
|
||||
description: >
|
||||
The motivation of the RFC.
|
||||
validations:
|
||||
required: true
|
||||
- type: textarea
|
||||
attributes:
|
||||
label: Proposed Change.
|
||||
description: >
|
||||
The proposed change of the RFC.
|
||||
validations:
|
||||
required: true
|
||||
- type: textarea
|
||||
attributes:
|
||||
label: Feedback Period.
|
||||
description: >
|
||||
The feedback period of the RFC. Usually at least one week.
|
||||
validations:
|
||||
required: false
|
||||
- type: textarea
|
||||
attributes:
|
||||
label: CC List.
|
||||
description: >
|
||||
The list of people you want to CC.
|
||||
validations:
|
||||
required: false
|
||||
- type: textarea
|
||||
attributes:
|
||||
label: Any Other Things.
|
||||
description: >
|
||||
Any other things you would like to mention.
|
||||
validations:
|
||||
required: false
|
||||
- type: markdown
|
||||
attributes:
|
||||
value: >
|
||||
Thanks for contributing 🎉!
|
||||
- type: checkboxes
|
||||
id: askllm
|
||||
attributes:
|
||||
label: Before submitting a new issue...
|
||||
options:
|
||||
- label: Make sure you already searched for relevant issues.
|
||||
required: true
|
||||
@@ -0,0 +1,62 @@
|
||||
<!--
|
||||
PR TITLE: Must start with a type tag, e.g.:
|
||||
[feat] Add new model [bugfix] Fix VAE tiling [refactor] Restructure pipeline
|
||||
[perf] Optimize kernel [ci] Update tests [docs] Add guide
|
||||
[misc] Cleanup configs [new-model] Port Flux2
|
||||
|
||||
MERGE WORKFLOW:
|
||||
1. Ensure pre-commit passes and you have at least 1 approval
|
||||
2. Comment /merge (or add the "ready" label) to enter the Merge Queue
|
||||
3. Full Test Suite runs automatically on a staging branch → auto-merge on success
|
||||
|
||||
ON-DEMAND TESTING (write access required):
|
||||
/test full — Full Test Suite /test ssim — SSIM regression
|
||||
/test training — Training pipeline /test encoder — Encoder tests
|
||||
/test transformer — Transformer tests /test vae — VAE tests
|
||||
/test kernel — CUDA kernel tests /test unit — Unit tests
|
||||
See docs/contributing/pull_requests.md for all 17 test commands
|
||||
-->
|
||||
|
||||
## Purpose
|
||||
|
||||
<!-- What does this PR do? Link the related issue if applicable. -->
|
||||
|
||||
Fixes #
|
||||
|
||||
## Changes
|
||||
|
||||
<!-- Describe your changes concisely. What approach did you take? -->
|
||||
|
||||
-
|
||||
|
||||
## Test Plan
|
||||
|
||||
<!-- How did you verify your changes? Paste exact commands and output. -->
|
||||
|
||||
```bash
|
||||
# Commands you ran
|
||||
```
|
||||
|
||||
## Test Results
|
||||
|
||||
<!-- Paste test output, before/after comparisons, or SSIM scores for model changes. -->
|
||||
|
||||
<details>
|
||||
<summary>Test output</summary>
|
||||
|
||||
```
|
||||
# Paste output here
|
||||
```
|
||||
|
||||
</details>
|
||||
|
||||
## Checklist
|
||||
|
||||
- [ ] I ran `pre-commit run --all-files` and fixed all issues
|
||||
- [ ] I added or updated tests for my changes
|
||||
- [ ] I updated documentation if needed
|
||||
- [ ] I considered GPU memory impact of my changes
|
||||
|
||||
**For model/pipeline changes, also check:**
|
||||
- [ ] I verified SSIM regression tests pass
|
||||
- [ ] I updated the support matrix if adding a new model
|
||||
@@ -0,0 +1,316 @@
|
||||
merge_protections:
|
||||
- name: PR merge requirements
|
||||
if:
|
||||
- base = main
|
||||
success_conditions:
|
||||
- "title~=(?i)^\\[(feat|feature|bugfix|fix|refactor|perf|ci|doc|docs|misc|chore|kernel|new.?model)\\]"
|
||||
- "#approved-reviews-by>=1"
|
||||
- check-success~=pre-commit
|
||||
- check-success=fastcheck-passed
|
||||
- check-success=full-suite-passed
|
||||
|
||||
pull_request_rules:
|
||||
|
||||
# ============================================================
|
||||
# Type labels (from PR title prefix)
|
||||
# ============================================================
|
||||
|
||||
- name: "label type: feat"
|
||||
conditions:
|
||||
- "title~=(?i)^\\[(feat|feature)\\]"
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["type: feat"]
|
||||
|
||||
- name: "label type: bugfix"
|
||||
conditions:
|
||||
- "title~=(?i)^\\[(bug)?fix\\]"
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["type: bugfix"]
|
||||
|
||||
- name: "label type: refactor"
|
||||
conditions:
|
||||
- "title~=(?i)^\\[refactor\\]"
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["type: refactor"]
|
||||
|
||||
- name: "label type: perf"
|
||||
conditions:
|
||||
- "title~=(?i)^\\[perf\\]"
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["type: perf"]
|
||||
|
||||
- name: "label type: ci"
|
||||
conditions:
|
||||
- "title~=(?i)^\\[ci\\]"
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["type: ci"]
|
||||
|
||||
- name: "label type: docs"
|
||||
conditions:
|
||||
- "title~=(?i)^\\[(doc|docs)\\]"
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["type: docs"]
|
||||
|
||||
- name: "label type: misc"
|
||||
conditions:
|
||||
- "title~=(?i)^\\[(misc|chore)\\]"
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["type: misc"]
|
||||
|
||||
- name: "label type: new-model"
|
||||
conditions:
|
||||
- "title~=(?i)^\\[new.?model\\]"
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["type: new-model"]
|
||||
|
||||
# ============================================================
|
||||
# Scope labels (from changed files)
|
||||
# ============================================================
|
||||
|
||||
- name: "label scope: training"
|
||||
conditions:
|
||||
- or:
|
||||
- files~=^fastvideo/train/
|
||||
- files~=^fastvideo/training/
|
||||
- files~=^fastvideo/distillation/
|
||||
- files~=^examples/train/
|
||||
- files~=^examples/training/
|
||||
- files~=^examples/distill/
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["scope: training"]
|
||||
|
||||
- name: "label scope: inference"
|
||||
conditions:
|
||||
- or:
|
||||
- files~=^fastvideo/pipelines/basic/
|
||||
- files~=^fastvideo/pipelines/stages/
|
||||
- files~=^fastvideo/pipelines/samplers/
|
||||
- files~=^fastvideo/entrypoints/
|
||||
- files~=^fastvideo/worker/
|
||||
- files~=^fastvideo/api/sampling_param
|
||||
- files~=^fastvideo/configs/pipelines/
|
||||
- files~=^examples/inference/
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["scope: inference"]
|
||||
|
||||
- name: "label scope: attention"
|
||||
conditions:
|
||||
- files~=^fastvideo/attention/
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["scope: attention"]
|
||||
|
||||
- name: "label scope: kernel"
|
||||
conditions:
|
||||
- or:
|
||||
- files~=^fastvideo-kernel/
|
||||
- files~=^csrc/
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["scope: kernel"]
|
||||
|
||||
- name: "label scope: data"
|
||||
conditions:
|
||||
- or:
|
||||
- files~=^fastvideo/dataset/
|
||||
- files~=^fastvideo/pipelines/preprocess/
|
||||
- files~=^examples/preprocessing/
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["scope: data"]
|
||||
|
||||
- name: "label scope: infra"
|
||||
conditions:
|
||||
- or:
|
||||
- files~=^\.github/
|
||||
- files~=^\.buildkite/
|
||||
- files~=^fastvideo/tests/
|
||||
- files~=^docker/
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["scope: infra"]
|
||||
|
||||
- name: "label scope: distributed"
|
||||
conditions:
|
||||
- files~=^fastvideo/distributed/
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["scope: distributed"]
|
||||
|
||||
- name: "label scope: docs"
|
||||
conditions:
|
||||
- files~=^docs/
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["scope: docs"]
|
||||
|
||||
- name: "label scope: ui"
|
||||
conditions:
|
||||
- files~=^ui/
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["scope: ui"]
|
||||
|
||||
- name: "label scope: model"
|
||||
conditions:
|
||||
- or:
|
||||
- files~=^fastvideo/models/
|
||||
- files~=^fastvideo/layers/
|
||||
- files~=^fastvideo/configs/models/
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["scope: model"]
|
||||
|
||||
# ============================================================
|
||||
# Pre-commit failure help comment
|
||||
# ============================================================
|
||||
|
||||
- name: comment on pre-commit failure
|
||||
conditions:
|
||||
- check-failure~=pre-commit
|
||||
- -closed
|
||||
actions:
|
||||
comment:
|
||||
message: |
|
||||
## Pre-commit checks failed
|
||||
|
||||
Hi @{{author}}, the pre-commit checks have failed. To fix them locally:
|
||||
|
||||
```bash
|
||||
# Install pre-commit if you haven't already
|
||||
uv pip install pre-commit
|
||||
pre-commit install
|
||||
|
||||
# Run all checks and auto-fix what's possible
|
||||
pre-commit run --all-files
|
||||
```
|
||||
|
||||
Common fixes:
|
||||
- **yapf**: `yapf -i <file>` (formatting)
|
||||
- **ruff**: `ruff check --fix <file>` (linting)
|
||||
- **codespell**: `codespell --write-changes <file>` (spelling)
|
||||
|
||||
After fixing, commit and push the changes. The checks will re-run automatically.
|
||||
|
||||
For future commits, `pre-commit` will run automatically on changed files before each commit.
|
||||
|
||||
|
||||
# ============================================================
|
||||
# Merge conflict detection
|
||||
# ============================================================
|
||||
|
||||
- name: label conflicting PRs
|
||||
conditions:
|
||||
- conflict
|
||||
- -closed
|
||||
- label!=stale
|
||||
actions:
|
||||
label:
|
||||
add: [needs-rebase]
|
||||
comment:
|
||||
message: |
|
||||
This PR has merge conflicts with the base branch. Please rebase:
|
||||
|
||||
```bash
|
||||
git fetch origin main
|
||||
git rebase origin/main
|
||||
# Resolve any conflicts, then:
|
||||
git push --force-with-lease
|
||||
```
|
||||
|
||||
- name: remove conflict label when resolved
|
||||
conditions:
|
||||
- -conflict
|
||||
- -closed
|
||||
- label=needs-rebase
|
||||
actions:
|
||||
label:
|
||||
remove: [needs-rebase]
|
||||
|
||||
# ============================================================
|
||||
# Auto-merge and auto-rebase
|
||||
# ============================================================
|
||||
|
||||
- name: auto-merge when ready and all checks pass
|
||||
conditions:
|
||||
- label=ready
|
||||
- "title~=(?i)^\\[(feat|feature|bugfix|fix|refactor|perf|ci|doc|docs|misc|chore|kernel|new.?model)\\]"
|
||||
- "#approved-reviews-by>=1"
|
||||
- check-success~=pre-commit
|
||||
- check-success=fastcheck-passed
|
||||
- check-success=full-suite-passed
|
||||
- -conflict
|
||||
- -closed
|
||||
- -draft
|
||||
actions:
|
||||
merge:
|
||||
method: squash
|
||||
|
||||
- name: auto-update when ready
|
||||
conditions:
|
||||
- label=ready
|
||||
- "#approved-reviews-by>=1"
|
||||
- -conflict
|
||||
- -closed
|
||||
- -draft
|
||||
actions:
|
||||
update: {}
|
||||
|
||||
# ============================================================
|
||||
# PR title format help
|
||||
# ============================================================
|
||||
|
||||
- name: comment on invalid PR title format
|
||||
conditions:
|
||||
- -closed
|
||||
- -draft
|
||||
- "-title~=(?i)^\\[(feat|feature|bugfix|fix|refactor|perf|ci|doc|docs|misc|chore|kernel|new.?model)\\]"
|
||||
actions:
|
||||
comment:
|
||||
message: |
|
||||
## ⚠️ PR title format required
|
||||
|
||||
Your PR title must start with a type tag in brackets. Examples:
|
||||
- `[feat] Add new model support`
|
||||
- `[bugfix] Fix VAE tiling corruption`
|
||||
- `[refactor] Restructure training pipeline`
|
||||
- `[perf] Optimize attention kernel`
|
||||
- `[ci] Update test infrastructure`
|
||||
- `[docs] Add inference guide`
|
||||
- `[misc] Clean up configs`
|
||||
- `[new-model] Port Flux2 to FastVideo`
|
||||
|
||||
Valid tags: `feat`, `feature`, `bugfix`, `fix`, `refactor`, `perf`, `ci`, `doc`, `docs`, `misc`, `chore`, `kernel`, `new-model`
|
||||
|
||||
Please update your PR title and the merge protection check will pass automatically.
|
||||
|
||||
merge_protections_settings:
|
||||
reporting_method: check-runs
|
||||
@@ -1,250 +0,0 @@
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
|
||||
import requests
|
||||
|
||||
|
||||
def parse_arguments():
|
||||
"""Parse command line arguments"""
|
||||
parser = argparse.ArgumentParser(description='Run tests on RunPod GPU')
|
||||
parser.add_argument('--gpu-type', type=str, help='GPU type to use')
|
||||
parser.add_argument('--gpu-count',
|
||||
type=int,
|
||||
help='Number of GPUs to use',
|
||||
default=1)
|
||||
parser.add_argument('--test-command', type=str, help='Test command to run')
|
||||
parser.add_argument('--disk-size',
|
||||
type=int,
|
||||
default=20,
|
||||
help='Container disk size in GB (default: 20)')
|
||||
parser.add_argument('--volume-size',
|
||||
type=int,
|
||||
default=20,
|
||||
help='Persistent volume size in GB (default: 20)')
|
||||
parser.add_argument(
|
||||
'--image',
|
||||
type=str,
|
||||
required=True,
|
||||
help='Docker image to use')
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
args = parse_arguments()
|
||||
API_KEY = os.environ['RUNPOD_API_KEY']
|
||||
RUN_ID = os.environ['GITHUB_RUN_ID']
|
||||
JOB_ID = os.environ['JOB_ID']
|
||||
PODS_API = "https://rest.runpod.io/v1/pods"
|
||||
HEADERS = {
|
||||
"Content-Type": "application/json",
|
||||
"Authorization": f"Bearer {API_KEY}"
|
||||
}
|
||||
|
||||
|
||||
def create_pod():
|
||||
"""Create a RunPod instance"""
|
||||
# Ensure image name is lowercase (Docker requirement)
|
||||
image_name = args.image.lower()
|
||||
print(f"Using specified image: {image_name}")
|
||||
|
||||
docker_start_cmd = [
|
||||
"bash",
|
||||
"-c",
|
||||
"apt update;DEBIAN_FRONTEND=noninteractive apt-get install openssh-server -y;mkdir -p ~/.ssh;cd $_;chmod 700 ~/.ssh;echo \"$PUBLIC_KEY\" >> authorized_keys;chmod 700 authorized_keys;service ssh start;sleep infinity"
|
||||
]
|
||||
|
||||
print(f"Creating RunPod instance with GPU: {args.gpu_type}...")
|
||||
payload = {
|
||||
"name": f"fastvideo-{JOB_ID}-{RUN_ID}",
|
||||
"containerDiskInGb": args.disk_size,
|
||||
"volumeInGb": args.volume_size,
|
||||
"gpuTypeIds": [args.gpu_type],
|
||||
"gpuCount": args.gpu_count,
|
||||
"imageName": image_name,
|
||||
"allowedCudaVersions": ["12.4"],
|
||||
"dockerStartCmd": docker_start_cmd
|
||||
}
|
||||
|
||||
response = requests.post(PODS_API, headers=HEADERS, json=payload)
|
||||
response_data = response.json()
|
||||
print(f"Response: {json.dumps(response_data, indent=2)}")
|
||||
|
||||
return response_data["id"]
|
||||
|
||||
|
||||
def wait_for_pod(pod_id):
|
||||
"""Wait for pod to be in RUNNING state and fully ready with SSH access"""
|
||||
print("Waiting for RunPod to be ready...")
|
||||
|
||||
# First wait for RUNNING status
|
||||
max_attempts = 10
|
||||
attempts = 0
|
||||
while attempts < max_attempts:
|
||||
response = requests.get(f"{PODS_API}/{pod_id}", headers=HEADERS)
|
||||
pod_data = response.json()
|
||||
status = pod_data["desiredStatus"]
|
||||
|
||||
if status == "RUNNING":
|
||||
print("RunPod is running! Now waiting for ports to be assigned...")
|
||||
break
|
||||
|
||||
print(
|
||||
f"Current status: {status}, waiting... (attempt {attempts+1}/{max_attempts})"
|
||||
)
|
||||
time.sleep(2)
|
||||
attempts += 1
|
||||
|
||||
if attempts >= max_attempts:
|
||||
raise TimeoutError(
|
||||
"Timed out waiting for RunPod to reach RUNNING state")
|
||||
|
||||
# Wait for ports to be assigned
|
||||
max_attempts = 50
|
||||
attempts = 0
|
||||
while attempts < max_attempts:
|
||||
response = requests.get(f"{PODS_API}/{pod_id}", headers=HEADERS)
|
||||
pod_data = response.json()
|
||||
port_mappings = pod_data.get("portMappings")
|
||||
|
||||
if (port_mappings is not None and "22" in port_mappings
|
||||
and pod_data.get("publicIp", "") != ""):
|
||||
print("RunPod is ready with SSH access!")
|
||||
print(f"SSH IP: {pod_data['publicIp']}")
|
||||
print(f"SSH Port: {port_mappings['22']}")
|
||||
break
|
||||
|
||||
print(
|
||||
f"Waiting for SSH port and public IP to be available... (attempt {attempts+1}/{max_attempts})"
|
||||
)
|
||||
time.sleep(20)
|
||||
attempts += 1
|
||||
|
||||
if attempts >= max_attempts:
|
||||
raise TimeoutError("Timed out waiting for RunPod SSH access")
|
||||
|
||||
|
||||
def execute_command(pod_id):
|
||||
"""Execute command on the pod via SSH using system SSH client"""
|
||||
print(f"Running command: {args.test_command}")
|
||||
|
||||
response = requests.get(f"{PODS_API}/{pod_id}", headers=HEADERS)
|
||||
pod_data = response.json()
|
||||
ssh_ip = pod_data["publicIp"]
|
||||
ssh_port = pod_data["portMappings"]["22"]
|
||||
|
||||
# Copy the repository to the pod using scp
|
||||
repo_dir = os.path.abspath(os.getcwd())
|
||||
repo_name = os.path.basename(repo_dir)
|
||||
|
||||
print(f"Copying repository from {repo_dir} to RunPod...")
|
||||
|
||||
tar_command = [
|
||||
"tar", "-czf", "/tmp/repo.tar.gz", "-C",
|
||||
os.path.dirname(repo_dir), repo_name
|
||||
]
|
||||
subprocess.run(tar_command, check=True)
|
||||
|
||||
# Copy the tarball to the pod
|
||||
scp_command = [
|
||||
"scp", "-o", "StrictHostKeyChecking=no", "-o",
|
||||
"UserKnownHostsFile=/dev/null", "-o", "ServerAliveInterval=60", "-o",
|
||||
"ServerAliveCountMax=10", "-P",
|
||||
str(ssh_port), "/tmp/repo.tar.gz", f"root@{ssh_ip}:/tmp/"
|
||||
]
|
||||
subprocess.run(scp_command, check=True)
|
||||
|
||||
# For custom image, we can use the pre-configured environment
|
||||
setup_steps = [
|
||||
"tar -xzf /tmp/repo.tar.gz --no-same-owner -C /workspace/",
|
||||
f"cd /workspace/{repo_name}",
|
||||
"source /opt/conda/etc/profile.d/conda.sh",
|
||||
"conda activate fastvideo-dev",
|
||||
args.test_command
|
||||
]
|
||||
|
||||
remote_command = " && ".join(setup_steps)
|
||||
|
||||
ssh_command = [
|
||||
"ssh", "-o", "StrictHostKeyChecking=no", "-o",
|
||||
"UserKnownHostsFile=/dev/null", "-o", "ServerAliveInterval=60", "-o",
|
||||
"ServerAliveCountMax=10", "-p",
|
||||
str(ssh_port), f"root@{ssh_ip}", remote_command
|
||||
]
|
||||
|
||||
print(f"Connecting to {ssh_ip}:{ssh_port}...")
|
||||
|
||||
try:
|
||||
process = subprocess.Popen(ssh_command,
|
||||
stdout=subprocess.PIPE,
|
||||
stderr=subprocess.STDOUT,
|
||||
universal_newlines=True,
|
||||
bufsize=0)
|
||||
|
||||
stdout_lines = []
|
||||
|
||||
print("Command output:")
|
||||
|
||||
for line in iter(process.stdout.readline, ''):
|
||||
print(line.strip())
|
||||
stdout_lines.append(line)
|
||||
|
||||
process.wait()
|
||||
|
||||
return_code = process.returncode
|
||||
success = return_code == 0
|
||||
|
||||
stdout_str = "".join(stdout_lines)
|
||||
|
||||
if success:
|
||||
print("Command executed successfully")
|
||||
else:
|
||||
print(f"Command failed with exit code {return_code}")
|
||||
|
||||
result = {
|
||||
"success": success,
|
||||
"return_code": return_code,
|
||||
"stdout": stdout_str,
|
||||
"stderr": ""
|
||||
}
|
||||
return result
|
||||
|
||||
except Exception as e:
|
||||
print(f"Error executing SSH command: {str(e)}")
|
||||
result = {"success": False, "error": str(e), "stdout": "", "stderr": ""}
|
||||
return result
|
||||
|
||||
|
||||
def terminate_pod(pod_id):
|
||||
"""Terminate the pod"""
|
||||
print("Terminating RunPod...")
|
||||
requests.delete(f"{PODS_API}/{pod_id}", headers=HEADERS)
|
||||
print(f"Terminated pod {pod_id}")
|
||||
|
||||
|
||||
def main():
|
||||
pod_id = None
|
||||
try:
|
||||
pod_id = create_pod()
|
||||
wait_for_pod(pod_id)
|
||||
result = execute_command(pod_id)
|
||||
|
||||
if result.get("error") is not None:
|
||||
print(f"Error executing command: {result['error']}")
|
||||
sys.exit(1)
|
||||
|
||||
if not result.get("success", False):
|
||||
print(
|
||||
"Tests failed - check the output above for details on which tests failed"
|
||||
)
|
||||
sys.exit(1)
|
||||
|
||||
finally:
|
||||
if pod_id:
|
||||
terminate_pod(pod_id)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -1,90 +0,0 @@
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
import uuid
|
||||
|
||||
import requests
|
||||
|
||||
API_KEY = os.environ['RUNPOD_API_KEY']
|
||||
RUN_ID = os.environ.get('GITHUB_RUN_ID', str(uuid.uuid4()))
|
||||
PODS_API = "https://rest.runpod.io/v1/pods"
|
||||
HEADERS = {
|
||||
"Content-Type": "application/json",
|
||||
"Authorization": f"Bearer {API_KEY}"
|
||||
}
|
||||
|
||||
|
||||
def get_job_ids():
|
||||
"""Parse job IDs from environment variable"""
|
||||
job_ids_str = os.environ.get('JOB_IDS')
|
||||
try:
|
||||
job_ids = json.loads(job_ids_str)
|
||||
if not isinstance(job_ids, list):
|
||||
print("Error: JOB_IDS is not a list.")
|
||||
sys.exit(1)
|
||||
return job_ids
|
||||
except json.JSONDecodeError as e:
|
||||
print(f"Error parsing JOB_IDS: {e}")
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
def cleanup_pods():
|
||||
"""Find and terminate RunPod instances"""
|
||||
print(f"Run ID: {RUN_ID}")
|
||||
|
||||
single_job_id = os.environ.get('JOB_ID')
|
||||
|
||||
if single_job_id:
|
||||
job_ids = [single_job_id]
|
||||
print(f"Job ID: {single_job_id}")
|
||||
else:
|
||||
job_ids = get_job_ids()
|
||||
print(f"Job IDs: {job_ids}")
|
||||
|
||||
# Get all pods associated with RunPod API_KEY
|
||||
try:
|
||||
response = requests.get(PODS_API, headers=HEADERS)
|
||||
response.raise_for_status()
|
||||
pods = response.json()
|
||||
except requests.exceptions.RequestException as e:
|
||||
print(f"Error getting pods: {e}")
|
||||
sys.exit(1)
|
||||
|
||||
# Find and terminate pods created by this workflow run
|
||||
terminated_pods = []
|
||||
for pod in pods:
|
||||
pod_name = pod.get("name", "")
|
||||
pod_id = pod.get("id")
|
||||
|
||||
# Check if this pod was created by one of our jobs
|
||||
if any(f"{job_id}-{RUN_ID}" in pod_name for job_id in job_ids):
|
||||
print(f"Found pod: {pod_id} ({pod_name})")
|
||||
try:
|
||||
print(f"Terminating pod {pod_id}...")
|
||||
term_response = requests.delete(f"{PODS_API}/{pod_id}",
|
||||
headers=HEADERS)
|
||||
term_response.raise_for_status()
|
||||
terminated_pods.append(pod_id)
|
||||
print(f"Successfully terminated pod {pod_id}")
|
||||
except requests.exceptions.RequestException as e:
|
||||
print(f"Error terminating pod {pod_id}: {e}")
|
||||
sys.exit(1)
|
||||
|
||||
if terminated_pods:
|
||||
if single_job_id:
|
||||
print(f"Terminated pod: {terminated_pods[0]}")
|
||||
else:
|
||||
print(f"Terminated {len(terminated_pods)} pods: {terminated_pods}")
|
||||
else:
|
||||
if single_job_id:
|
||||
print(f"No pod found matching pattern: {single_job_id}-{RUN_ID}")
|
||||
else:
|
||||
print("No pods found to terminate.")
|
||||
|
||||
|
||||
def main():
|
||||
cleanup_pods()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -1,7 +1,17 @@
|
||||
name: Build and Push Docker Image
|
||||
name: Build Image Template
|
||||
|
||||
on:
|
||||
workflow_dispatch: # Only manual triggers
|
||||
workflow_call:
|
||||
inputs:
|
||||
python_version:
|
||||
required: true
|
||||
type: string
|
||||
dockerfile_path:
|
||||
required: true
|
||||
type: string
|
||||
tag_suffix:
|
||||
required: true
|
||||
type: string
|
||||
|
||||
jobs:
|
||||
build-and-push:
|
||||
@@ -52,20 +62,38 @@ jobs:
|
||||
username: ${{ github.repository_owner }}
|
||||
password: ${{ secrets.GITHUB_TOKEN }}
|
||||
|
||||
- name: Prepare tags
|
||||
id: prepare-tags
|
||||
run: |
|
||||
SHORT_SHA=$(echo ${{ github.sha }} | cut -c1-7)
|
||||
|
||||
TAGS="type=raw,value=${{ inputs.tag_suffix }}-latest"
|
||||
TAGS="${TAGS}\ntype=raw,value=${{ inputs.tag_suffix }}-sha-${SHORT_SHA}"
|
||||
|
||||
# Set Python 3.10 as the default image
|
||||
if [[ "${{ inputs.python_version }}" == "3.10" ]]; then
|
||||
TAGS="${TAGS}\ntype=raw,value=latest"
|
||||
fi
|
||||
|
||||
{
|
||||
echo "tags<<EOF"
|
||||
echo -e "$TAGS"
|
||||
echo "EOF"
|
||||
} >> $GITHUB_OUTPUT
|
||||
|
||||
- name: Extract metadata for Docker
|
||||
id: meta
|
||||
uses: docker/metadata-action@v5
|
||||
with:
|
||||
images: ghcr.io/${{ github.repository }}/fastvideo-dev
|
||||
tags: |
|
||||
type=raw,value=latest
|
||||
type=sha,format=short
|
||||
tags: ${{ steps.prepare-tags.outputs.tags }}
|
||||
|
||||
- name: Build and push Docker image
|
||||
id: build-push
|
||||
uses: docker/build-push-action@v6
|
||||
with:
|
||||
context: .
|
||||
file: ${{ inputs.dockerfile_path }}
|
||||
push: true
|
||||
tags: ${{ steps.meta.outputs.tags }}
|
||||
labels: ${{ steps.meta.outputs.labels }}
|
||||
@@ -74,5 +102,5 @@ jobs:
|
||||
|
||||
- name: Success message
|
||||
run: |
|
||||
echo "✅ Image successfully built and pushed to ghcr.io/${{ github.repository }}/fastvideo-dev:latest"
|
||||
echo "✅ Python ${{ inputs.python_version }} image successfully built and pushed to ghcr.io/${{ github.repository }}/fastvideo-dev:${{ inputs.tag_suffix }}-latest"
|
||||
echo "To run tests with this image, manually trigger the 'Run Tests' workflow."
|
||||
@@ -0,0 +1,80 @@
|
||||
name: Aggregate Test Status
|
||||
|
||||
on:
|
||||
status:
|
||||
|
||||
permissions:
|
||||
statuses: write
|
||||
|
||||
jobs:
|
||||
aggregate:
|
||||
if: >-
|
||||
github.event.context == 'direct-test-completed'
|
||||
&& github.event.state == 'success'
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Check and update aggregate status
|
||||
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7.0.1
|
||||
with:
|
||||
script: |
|
||||
const sha = context.payload.sha;
|
||||
|
||||
const { data } = await github.rest.repos.getCombinedStatusForRef({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
ref: sha,
|
||||
per_page: 100,
|
||||
});
|
||||
|
||||
const bkStatuses = data.statuses.filter(
|
||||
s => s.context.startsWith('buildkite/ci/')
|
||||
);
|
||||
|
||||
const FASTCHECK_PREFIX = 'buildkite/ci/microscope-';
|
||||
const FULL_SUITE_PREFIXES = [
|
||||
'buildkite/ci/test-tube-',
|
||||
'buildkite/ci/bar-chart-',
|
||||
];
|
||||
|
||||
const fastcheck = bkStatuses.filter(
|
||||
s => s.context.startsWith(FASTCHECK_PREFIX)
|
||||
);
|
||||
const fullSuite = bkStatuses.filter(
|
||||
s => FULL_SUITE_PREFIXES.some(p => s.context.startsWith(p))
|
||||
);
|
||||
|
||||
if (
|
||||
fastcheck.length > 0
|
||||
&& fastcheck.every(s => s.state === 'success')
|
||||
) {
|
||||
core.info(
|
||||
`All ${fastcheck.length} fastcheck tests passed — updating fastcheck-passed`
|
||||
);
|
||||
await github.rest.repos.createCommitStatus({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
sha,
|
||||
state: 'success',
|
||||
context: 'fastcheck-passed',
|
||||
description:
|
||||
`All ${fastcheck.length} fastcheck tests passed`,
|
||||
});
|
||||
}
|
||||
|
||||
if (
|
||||
fullSuite.length > 0
|
||||
&& fullSuite.every(s => s.state === 'success')
|
||||
) {
|
||||
core.info(
|
||||
`All ${fullSuite.length} full suite tests passed — updating full-suite-passed`
|
||||
);
|
||||
await github.rest.repos.createCommitStatus({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
sha,
|
||||
state: 'success',
|
||||
context: 'full-suite-passed',
|
||||
description:
|
||||
`All ${fullSuite.length} full suite tests passed`,
|
||||
});
|
||||
}
|
||||
@@ -0,0 +1,32 @@
|
||||
name: pre-commit
|
||||
|
||||
on:
|
||||
pull_request:
|
||||
branches: [main]
|
||||
workflow_call:
|
||||
inputs:
|
||||
ref:
|
||||
description: 'Git ref to checkout (defaults to github.ref)'
|
||||
required: false
|
||||
type: string
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
pre-commit:
|
||||
if: github.event_name == 'workflow_call' || github.event.pull_request.draft != true
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
with:
|
||||
ref: ${{ inputs.ref || '' }}
|
||||
- uses: actions/setup-python@v5
|
||||
with:
|
||||
python-version: "3.12"
|
||||
- run: echo "::add-matcher::.github/workflows/matchers/actionlint.json"
|
||||
- run: echo "::add-matcher::.github/workflows/matchers/mypy.json"
|
||||
- run: echo "::add-matcher::.github/workflows/matchers/ruff.json"
|
||||
- uses: pre-commit/action@v3.0.1
|
||||
with:
|
||||
extra_args: --all-files --hook-stage manual
|
||||
@@ -0,0 +1,271 @@
|
||||
name: Slash Commands
|
||||
|
||||
on:
|
||||
issue_comment:
|
||||
types: [created]
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
pull-requests: write
|
||||
statuses: write
|
||||
|
||||
jobs:
|
||||
handle-merge:
|
||||
if: >-
|
||||
github.event.issue.pull_request != null
|
||||
&& startsWith(github.event.comment.body, '/merge')
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Check write permission
|
||||
id: perm
|
||||
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7.0.1
|
||||
with:
|
||||
script: |
|
||||
const { data: perm } = await github.rest.repos.getCollaboratorPermissionLevel({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
username: context.payload.comment.user.login,
|
||||
});
|
||||
const hasWrite = ['admin', 'write'].includes(perm.permission);
|
||||
if (!hasWrite) {
|
||||
core.setFailed(`User ${context.payload.comment.user.login} lacks write permission (has: ${perm.permission}).`);
|
||||
}
|
||||
core.setOutput('has_write', String(hasWrite));
|
||||
|
||||
- name: Add ready label and react
|
||||
id: label
|
||||
if: steps.perm.outputs.has_write == 'true'
|
||||
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7.0.1
|
||||
with:
|
||||
script: |
|
||||
const owner = context.repo.owner;
|
||||
const repo = context.repo.repo;
|
||||
const prNumber = context.payload.issue.number;
|
||||
try { await github.rest.issues.removeLabel({ owner, repo, issue_number: prNumber, name: 'ready' }); } catch {}
|
||||
await github.rest.issues.addLabels({ owner, repo, issue_number: prNumber, labels: ['ready'] });
|
||||
await github.rest.reactions.createForIssueComment({
|
||||
owner, repo,
|
||||
comment_id: context.payload.comment.id,
|
||||
content: 'rocket',
|
||||
});
|
||||
const { data: pr } = await github.rest.pulls.get({ owner, repo, pull_number: prNumber });
|
||||
core.setOutput('pr_sha', pr.head.sha);
|
||||
core.setOutput('pr_branch', pr.head.ref);
|
||||
core.setOutput('pr_number', String(prNumber));
|
||||
|
||||
- name: Trigger Full Suite
|
||||
if: steps.perm.outputs.has_write == 'true'
|
||||
env:
|
||||
BUILDKITE_API_TOKEN: ${{ secrets.BUILDKITE_API_TOKEN }}
|
||||
PR_SHA: ${{ steps.label.outputs.pr_sha }}
|
||||
PR_BRANCH: ${{ steps.label.outputs.pr_branch }}
|
||||
PR_NUMBER: ${{ steps.label.outputs.pr_number }}
|
||||
BK_ORG: ${{ vars.BUILDKITE_ORG_SLUG }}
|
||||
BK_PIPELINE: ${{ vars.BUILDKITE_PIPELINE_SLUG }}
|
||||
run: |
|
||||
curl -sS --fail-with-body -X POST \
|
||||
"https://api.buildkite.com/v2/organizations/${BK_ORG}/pipelines/${BK_PIPELINE}/builds" \
|
||||
-H "Authorization: Bearer $BUILDKITE_API_TOKEN" \
|
||||
-H "Content-Type: application/json" \
|
||||
--data-raw "$(jq -n \
|
||||
--arg commit "$PR_SHA" \
|
||||
--arg branch "$PR_BRANCH" \
|
||||
--arg message "Full Suite for PR #${PR_NUMBER} (via /merge)" \
|
||||
--argjson pr_id "$PR_NUMBER" \
|
||||
'{
|
||||
commit: $commit,
|
||||
branch: $branch,
|
||||
message: $message,
|
||||
ignore_pipeline_branch_filters: true,
|
||||
pull_request_id: $pr_id,
|
||||
pull_request_base_branch: "main",
|
||||
env: {
|
||||
TEST_SCOPE: "full",
|
||||
FULL_SUITE: "true",
|
||||
PR_NUMBER: ($pr_id | tostring)
|
||||
}
|
||||
}')"
|
||||
|
||||
parse-command:
|
||||
if: >-
|
||||
github.event.issue.pull_request != null
|
||||
&& startsWith(github.event.comment.body, '/test')
|
||||
runs-on: ubuntu-latest
|
||||
outputs:
|
||||
test_type: ${{ steps.parse.outputs.test_type }}
|
||||
test_scope: ${{ steps.parse.outputs.test_scope }}
|
||||
full_suite: ${{ steps.parse.outputs.full_suite }}
|
||||
pr_sha: ${{ steps.pr.outputs.sha }}
|
||||
pr_branch: ${{ steps.pr.outputs.branch }}
|
||||
has_write: ${{ steps.perm.outputs.has_write }}
|
||||
steps:
|
||||
- name: Check write permission
|
||||
id: perm
|
||||
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7.0.1
|
||||
with:
|
||||
script: |
|
||||
const { data: perm } = await github.rest.repos.getCollaboratorPermissionLevel({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
username: context.payload.comment.user.login,
|
||||
});
|
||||
const hasWrite = ['admin', 'write'].includes(perm.permission);
|
||||
core.setOutput('has_write', String(hasWrite));
|
||||
if (!hasWrite) {
|
||||
core.info(`User ${context.payload.comment.user.login} lacks write permission — ignoring.`);
|
||||
}
|
||||
|
||||
- name: Parse /test command
|
||||
id: parse
|
||||
if: steps.perm.outputs.has_write == 'true'
|
||||
shell: bash
|
||||
env:
|
||||
COMMENT: ${{ github.event.comment.body }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
TEST_NAME=$(echo "$COMMENT" | grep -oP '(?<=/test\s)\S+' | head -1 || true)
|
||||
|
||||
VALID="encoder vae transformer kernel unit ssim training lora-inference lora-training distillation self-forcing vsa vmoba performance api full fastcheck pre-commit"
|
||||
if [ -z "$TEST_NAME" ] || ! echo "$VALID" | grep -qw "$TEST_NAME"; then
|
||||
echo "Unknown test: '$TEST_NAME'. Valid: $VALID"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
declare -A MAP=(
|
||||
[encoder]=encoder [vae]=vae [transformer]=transformer
|
||||
[kernel]=kernel_tests [unit]=unit_test
|
||||
[ssim]=ssim [training]=training
|
||||
[lora-inference]=inference_lora [lora-training]=training_lora
|
||||
[distillation]=distillation_dmd [self-forcing]=self_forcing
|
||||
[vsa]=training_vsa [vmoba]=inference_vmoba
|
||||
[performance]=performance [api]=api_server
|
||||
)
|
||||
|
||||
if [ "$TEST_NAME" = "full" ]; then
|
||||
{
|
||||
echo "test_type=all"
|
||||
echo "test_scope=full"
|
||||
echo "full_suite=true"
|
||||
} >> "$GITHUB_OUTPUT"
|
||||
elif [ "$TEST_NAME" = "fastcheck" ]; then
|
||||
{
|
||||
echo "test_type=fastcheck"
|
||||
echo "test_scope=fastcheck"
|
||||
echo "full_suite=false"
|
||||
} >> "$GITHUB_OUTPUT"
|
||||
elif [ "$TEST_NAME" = "pre-commit" ]; then
|
||||
{
|
||||
echo "test_type="
|
||||
echo "test_scope=precommit"
|
||||
echo "full_suite=false"
|
||||
} >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
{
|
||||
echo "test_type=${MAP[$TEST_NAME]}"
|
||||
echo "test_scope=direct"
|
||||
echo "full_suite=false"
|
||||
} >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
|
||||
- name: Get PR details
|
||||
id: pr
|
||||
if: steps.perm.outputs.has_write == 'true'
|
||||
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7.0.1
|
||||
with:
|
||||
script: |
|
||||
const { data: pr } = await github.rest.pulls.get({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
pull_number: context.payload.issue.number,
|
||||
});
|
||||
core.setOutput('sha', pr.head.sha);
|
||||
core.setOutput('branch', pr.head.ref);
|
||||
|
||||
- name: React to comment
|
||||
if: steps.perm.outputs.has_write == 'true'
|
||||
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7.0.1
|
||||
with:
|
||||
script: |
|
||||
await github.rest.reactions.createForIssueComment({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
comment_id: context.payload.comment.id,
|
||||
content: 'rocket',
|
||||
});
|
||||
|
||||
pre-commit:
|
||||
needs: parse-command
|
||||
if: >-
|
||||
needs.parse-command.outputs.has_write == 'true'
|
||||
&& needs.parse-command.outputs.test_scope == 'precommit'
|
||||
uses: ./.github/workflows/ci-precommit.yml
|
||||
with:
|
||||
ref: refs/pull/${{ github.event.issue.number }}/merge
|
||||
|
||||
post-precommit-status:
|
||||
needs: [parse-command, pre-commit]
|
||||
if: always() && needs.parse-command.outputs.test_scope == 'precommit'
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7.0.1
|
||||
env:
|
||||
PR_SHA: ${{ needs.parse-command.outputs.pr_sha }}
|
||||
RESULT: ${{ needs.pre-commit.result }}
|
||||
with:
|
||||
script: |
|
||||
const state = process.env.RESULT === 'success' ? 'success' : 'failure';
|
||||
await github.rest.repos.createCommitStatus({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
sha: process.env.PR_SHA,
|
||||
state,
|
||||
context: 'pre-commit',
|
||||
description: `Triggered via /test pre-commit (${state})`,
|
||||
});
|
||||
|
||||
trigger-buildkite:
|
||||
needs: parse-command
|
||||
if: >-
|
||||
needs.parse-command.outputs.has_write == 'true'
|
||||
&& needs.parse-command.outputs.test_type != ''
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Trigger Buildkite
|
||||
env:
|
||||
BUILDKITE_API_TOKEN: ${{ secrets.BUILDKITE_API_TOKEN }}
|
||||
PR_SHA: ${{ needs.parse-command.outputs.pr_sha }}
|
||||
PR_BRANCH: ${{ needs.parse-command.outputs.pr_branch }}
|
||||
PR_NUMBER: ${{ github.event.issue.number }}
|
||||
TEST_SCOPE: ${{ needs.parse-command.outputs.test_scope }}
|
||||
FULL_SUITE: ${{ needs.parse-command.outputs.full_suite }}
|
||||
TEST_TYPE: ${{ needs.parse-command.outputs.test_type }}
|
||||
BK_ORG: ${{ vars.BUILDKITE_ORG_SLUG }}
|
||||
BK_PIPELINE: ${{ vars.BUILDKITE_PIPELINE_SLUG }}
|
||||
run: |
|
||||
curl -sS --fail-with-body -X POST \
|
||||
"https://api.buildkite.com/v2/organizations/${BK_ORG}/pipelines/${BK_PIPELINE}/builds" \
|
||||
-H "Authorization: Bearer $BUILDKITE_API_TOKEN" \
|
||||
-H "Content-Type: application/json" \
|
||||
--data-raw "$(jq -n \
|
||||
--arg commit "$PR_SHA" \
|
||||
--arg branch "$PR_BRANCH" \
|
||||
--arg message "/test ${TEST_TYPE} on PR #${PR_NUMBER}" \
|
||||
--argjson pr_id "$PR_NUMBER" \
|
||||
--arg test_scope "$TEST_SCOPE" \
|
||||
--arg full_suite "$FULL_SUITE" \
|
||||
--arg test_type "$TEST_TYPE" \
|
||||
--arg pr_number "$PR_NUMBER" \
|
||||
'{
|
||||
commit: $commit,
|
||||
branch: $branch,
|
||||
message: $message,
|
||||
ignore_pipeline_branch_filters: true,
|
||||
pull_request_id: $pr_id,
|
||||
pull_request_base_branch: "main",
|
||||
env: {
|
||||
TEST_SCOPE: $test_scope,
|
||||
FULL_SUITE: $full_suite,
|
||||
TEST_TYPE: $test_type,
|
||||
PR_NUMBER: $pr_number
|
||||
}
|
||||
}')"
|
||||
@@ -0,0 +1,83 @@
|
||||
name: Trigger Full Suite
|
||||
|
||||
on:
|
||||
pull_request_target:
|
||||
types: [labeled, synchronize]
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
pull-requests: read
|
||||
|
||||
concurrency:
|
||||
group: full-suite-${{ github.event.pull_request.number }}
|
||||
cancel-in-progress: false
|
||||
|
||||
jobs:
|
||||
trigger:
|
||||
if: >-
|
||||
(github.event.action == 'labeled' && github.event.label.name == 'ready')
|
||||
|| github.event.action == 'synchronize'
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Check ready label
|
||||
id: check
|
||||
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7.0.1
|
||||
with:
|
||||
script: |
|
||||
const { data: pr } = await github.rest.pulls.get({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
pull_number: context.payload.pull_request.number,
|
||||
});
|
||||
const hasReady = pr.labels.some(l => l.name === 'ready');
|
||||
core.setOutput('has_ready', String(hasReady));
|
||||
if (!hasReady) core.info('No ready label — skipping Full Suite trigger.');
|
||||
|
||||
- name: Cancel previous Buildkite builds
|
||||
if: steps.check.outputs.has_ready == 'true'
|
||||
env:
|
||||
BUILDKITE_API_TOKEN: ${{ secrets.BUILDKITE_API_TOKEN }}
|
||||
PR_BRANCH: ${{ github.event.pull_request.head.ref }}
|
||||
run: |
|
||||
# Find running builds for this branch with TEST_SCOPE=full and cancel them
|
||||
builds=$(curl -sS -H "Authorization: Bearer $BUILDKITE_API_TOKEN" \
|
||||
"https://api.buildkite.com/v2/organizations/${{ vars.BUILDKITE_ORG_SLUG }}/pipelines/${{ vars.BUILDKITE_PIPELINE_SLUG }}/builds?branch=${PR_BRANCH}&state=running,scheduled" \
|
||||
| jq -r '.[] | select(try (.env.TEST_SCOPE == "full") catch false) | .number')
|
||||
for build_num in $builds; do
|
||||
echo "Cancelling Buildkite build #$build_num"
|
||||
curl -sS -X PUT -H "Authorization: Bearer $BUILDKITE_API_TOKEN" \
|
||||
"https://api.buildkite.com/v2/organizations/${{ vars.BUILDKITE_ORG_SLUG }}/pipelines/${{ vars.BUILDKITE_PIPELINE_SLUG }}/builds/${build_num}/cancel"
|
||||
done
|
||||
|
||||
- name: Trigger Buildkite Full Suite
|
||||
if: steps.check.outputs.has_ready == 'true'
|
||||
env:
|
||||
BUILDKITE_API_TOKEN: ${{ secrets.BUILDKITE_API_TOKEN }}
|
||||
PR_SHA: ${{ github.event.pull_request.head.sha }}
|
||||
PR_BRANCH: ${{ github.event.pull_request.head.ref }}
|
||||
PR_NUMBER: ${{ github.event.pull_request.number }}
|
||||
BK_ORG: ${{ vars.BUILDKITE_ORG_SLUG }}
|
||||
BK_PIPELINE: ${{ vars.BUILDKITE_PIPELINE_SLUG }}
|
||||
run: |
|
||||
curl -sS --fail-with-body -X POST \
|
||||
"https://api.buildkite.com/v2/organizations/${BK_ORG}/pipelines/${BK_PIPELINE}/builds" \
|
||||
-H "Authorization: Bearer $BUILDKITE_API_TOKEN" \
|
||||
-H "Content-Type: application/json" \
|
||||
--data-raw "$(jq -n \
|
||||
--arg commit "$PR_SHA" \
|
||||
--arg branch "$PR_BRANCH" \
|
||||
--arg message "Full Suite for PR #${PR_NUMBER}" \
|
||||
--argjson pr_id "$PR_NUMBER" \
|
||||
'{
|
||||
commit: $commit,
|
||||
branch: $branch,
|
||||
message: $message,
|
||||
ignore_pipeline_branch_filters: true,
|
||||
pull_request_id: $pr_id,
|
||||
pull_request_base_branch: "main",
|
||||
env: {
|
||||
TEST_SCOPE: "full",
|
||||
FULL_SUITE: "true",
|
||||
PR_NUMBER: ($pr_id | tostring)
|
||||
}
|
||||
}')"
|
||||
@@ -0,0 +1,65 @@
|
||||
name: Auto-Label Issues
|
||||
|
||||
on:
|
||||
issues:
|
||||
types: [opened, edited]
|
||||
|
||||
permissions:
|
||||
issues: write
|
||||
|
||||
jobs:
|
||||
label-issues:
|
||||
if: github.repository == 'hao-ai-lab/FastVideo'
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Label by keywords
|
||||
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7.0.1
|
||||
with:
|
||||
script: |
|
||||
const title = context.payload.issue.title.toLowerCase();
|
||||
const body = (context.payload.issue.body || '').toLowerCase();
|
||||
const text = title + ' ' + body;
|
||||
const labels = [];
|
||||
|
||||
const rules = [
|
||||
// scope labels (shared with PR labeling via Mergify)
|
||||
// Mapping: label → repo directories
|
||||
// scope: training → fastvideo/train/, fastvideo/training/, fastvideo/distillation/
|
||||
// scope: inference → fastvideo/pipelines/, fastvideo/entrypoints/, fastvideo/worker/
|
||||
// scope: attention → fastvideo/attention/
|
||||
// scope: kernel → fastvideo-kernel/, csrc/
|
||||
// scope: model → fastvideo/models/, fastvideo/layers/, fastvideo/configs/models/
|
||||
// scope: data → fastvideo/dataset/, fastvideo/pipelines/preprocess/
|
||||
// scope: distributed → fastvideo/distributed/
|
||||
// scope: docs → docs/
|
||||
{ keywords: ['training', 'finetune', 'fine-tune', 'lora', 'fsdp', 'distill'], label: 'scope: training' },
|
||||
{ keywords: ['inference', 'generate', 'pipeline', 'slow', 'latency'], label: 'scope: inference' },
|
||||
{ keywords: ['attention', 'vsa', 'flash', 'sta', 'vmoba', 'sparse attn'], label: 'scope: attention' },
|
||||
{ keywords: ['kernel', 'csrc', 'cuda kernel', 'thunderkittens'], label: 'scope: kernel' },
|
||||
{ keywords: ['wan', 'hunyuan', 'mochi', 'ltx', 'cogvideo', 'flux', 'sd3', 'cosmos'], label: 'scope: model' },
|
||||
{ keywords: ['dataset', 'dataloader', 'preprocessing', 'preprocess'], label: 'scope: data' },
|
||||
{ keywords: ['distributed', 'sequence parallel', 'fsdp', 'tensor parallel', 'multi-node', 'multi-gpu'], label: 'scope: distributed' },
|
||||
{ keywords: ['docs', 'documentation', 'tutorial', 'example'], label: 'scope: docs' },
|
||||
// issue-only labels (cross-module, no single repo directory)
|
||||
{ keywords: ['install', 'setup', 'pip', 'cuda', 'uv ', 'import error', 'modulenotfound'], label: 'installation' },
|
||||
{ keywords: ['memory', 'oom', 'out of memory', 'gpu memory', 'vram'], label: 'performance' },
|
||||
{ keywords: ['windows', 'macos', 'mac os', 'apple', 'mps', 'rocm', 'amd', 'npu'], label: 'platform' },
|
||||
];
|
||||
|
||||
for (const rule of rules) {
|
||||
if (rule.keywords.some(kw => text.includes(kw))) {
|
||||
labels.push(rule.label);
|
||||
}
|
||||
}
|
||||
|
||||
if (labels.length > 0) {
|
||||
await github.rest.issues.addLabels({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
issue_number: context.payload.issue.number,
|
||||
labels: labels,
|
||||
});
|
||||
console.log(`Added labels: ${labels.join(', ')}`);
|
||||
} else {
|
||||
console.log('No keyword matches found');
|
||||
}
|
||||
@@ -0,0 +1,51 @@
|
||||
name: Close Stale Issues and PRs
|
||||
|
||||
on:
|
||||
schedule:
|
||||
# Daily at 1:30 AM UTC
|
||||
- cron: '30 1 * * *'
|
||||
|
||||
jobs:
|
||||
stale:
|
||||
if: github.repository == 'hao-ai-lab/FastVideo'
|
||||
permissions:
|
||||
issues: write
|
||||
pull-requests: write
|
||||
actions: write
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/stale@997185467fa4f803885201cee163a9f38240193d # v10.1.1
|
||||
with:
|
||||
operations-per-run: 500
|
||||
|
||||
exempt-draft-pr: true
|
||||
exempt-issue-labels: 'keep-open,pinned,security,Bug,RFC'
|
||||
exempt-pr-labels: 'keep-open,pinned'
|
||||
|
||||
labels-to-add-when-unstale: 'unstale'
|
||||
labels-to-remove-when-stale: 'unstale'
|
||||
|
||||
days-before-issue-stale: 90
|
||||
days-before-issue-close: 30
|
||||
stale-issue-label: 'stale'
|
||||
stale-issue-message: >
|
||||
This issue has been automatically marked as stale because it has not
|
||||
had any activity within 90 days. It will be automatically closed if
|
||||
no further activity occurs within 30 days. Leave a comment if you
|
||||
feel this issue should remain open. Thank you!
|
||||
close-issue-message: >
|
||||
This issue has been automatically closed due to inactivity. Please
|
||||
feel free to reopen if you feel it is still relevant. Thank you!
|
||||
|
||||
days-before-pr-stale: 60
|
||||
days-before-pr-close: 14
|
||||
stale-pr-label: 'stale'
|
||||
stale-pr-message: >
|
||||
This pull request has been automatically marked as stale because it
|
||||
has not had any activity within 60 days. It will be automatically
|
||||
closed if no further activity occurs within 14 days. Leave a comment
|
||||
if you feel this pull request should remain open. Thank you!
|
||||
close-pr-message: >
|
||||
This pull request has been automatically closed due to inactivity.
|
||||
Please feel free to reopen if you intend to continue working on it.
|
||||
Thank you!
|
||||
@@ -0,0 +1,56 @@
|
||||
name: Welcome First-Time Contributors
|
||||
|
||||
on:
|
||||
issues:
|
||||
types: [opened]
|
||||
pull_request_target:
|
||||
types: [opened]
|
||||
|
||||
permissions:
|
||||
issues: write
|
||||
pull-requests: write
|
||||
|
||||
jobs:
|
||||
welcome:
|
||||
if: github.repository == 'hao-ai-lab/FastVideo'
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/first-interaction@34f15e814fe48ac9312ccf29db4e74fa767cbab7 # v1.3.0
|
||||
with:
|
||||
repo-token: ${{ secrets.GITHUB_TOKEN }}
|
||||
issue-message: |
|
||||
Welcome to FastVideo! Thanks for opening your first issue.
|
||||
|
||||
To help us investigate, please include:
|
||||
- **FastVideo version**: `pip show fastvideo`
|
||||
- **GPU**: `nvidia-smi` output (GPU model, driver, CUDA version)
|
||||
- **Python version**: `python --version`
|
||||
- **OS**: e.g., Ubuntu 22.04
|
||||
|
||||
If this is a bug, a minimal reproduction script helps us fix it faster.
|
||||
|
||||
Useful links:
|
||||
- [Documentation](https://hao-ai-lab.github.io/FastVideo)
|
||||
- [Contributing Guide](https://hao-ai-lab.github.io/FastVideo/contributing/overview/)
|
||||
- [Slack](https://join.slack.com/t/fastvideo/shared_invite/zt-3f4lao1uq-u~Ipx6Lt4J27AlD2y~IdLQ)
|
||||
pr-message: |
|
||||
Welcome to FastVideo! Thanks for your first pull request.
|
||||
|
||||
**How our CI works:**
|
||||
|
||||
PRs run a two-tier CI system:
|
||||
1. **Pre-commit** — formatting (yapf), linting (ruff), type checking (mypy). Runs immediately on every PR.
|
||||
2. **Fastcheck** — core GPU tests (encoders, VAEs, transformers, kernels, unit tests). Runs automatically via Buildkite on relevant file changes (~10-15 min).
|
||||
3. **Full Suite** — integration tests, training pipelines, SSIM regression. Runs only when a reviewer adds the `ready` label.
|
||||
|
||||
**Before your PR is reviewed:**
|
||||
- [ ] `pre-commit run --all-files` passes locally
|
||||
- [ ] You've added or updated tests for your changes
|
||||
- [ ] The PR description explains what and why
|
||||
|
||||
If pre-commit fails, a bot comment will explain how to fix it. Fastcheck and Full Suite results appear in the Checks section below.
|
||||
|
||||
**Useful links:**
|
||||
- [Contributing Guide](https://hao-ai-lab.github.io/FastVideo/contributing/overview/)
|
||||
- [Development Roadmap](https://github.com/hao-ai-lab/FastVideo/issues/899)
|
||||
- [Slack](https://join.slack.com/t/fastvideo/shared_invite/zt-3f4lao1uq-u~Ipx6Lt4J27AlD2y~IdLQ)
|
||||
@@ -1,83 +0,0 @@
|
||||
# Sample workflow for building and deploying a Hugo site to GitHub Pages
|
||||
name: Deploy FastVideo Docs to Pages
|
||||
|
||||
on:
|
||||
# Runs on pushes targeting the default branch
|
||||
push:
|
||||
branches:
|
||||
- main
|
||||
paths:
|
||||
- "docs/**/*.md"
|
||||
- "fastvideo/v1/examples/**/*.py"
|
||||
pull_request:
|
||||
branches:
|
||||
- main
|
||||
types: [opened, ready_for_review, synchronize, reopened]
|
||||
paths:
|
||||
- "docs/**/*.md"
|
||||
- "fastvideo/v1/examples/**/*.py"
|
||||
|
||||
# Allows you to run this workflow manually from the Actions tab
|
||||
workflow_dispatch:
|
||||
|
||||
# Sets permissions of the GITHUB_TOKEN to allow deployment to GitHub Pages
|
||||
permissions:
|
||||
contents: read
|
||||
pages: write
|
||||
id-token: write
|
||||
|
||||
# Allow only one concurrent deployment, skipping runs queued between the run in-progress and latest queued.
|
||||
# However, do NOT cancel in-progress runs as we want to allow these production deployments to complete.
|
||||
concurrency:
|
||||
group: "pages"
|
||||
cancel-in-progress: false
|
||||
|
||||
# Default to bash
|
||||
defaults:
|
||||
run:
|
||||
shell: bash
|
||||
|
||||
jobs:
|
||||
pre-commit:
|
||||
uses: ./.github/workflows/pre-commit.yml
|
||||
|
||||
# Build job
|
||||
build:
|
||||
runs-on: ubuntu-latest
|
||||
needs: pre-commit
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
- name: Setup Pages
|
||||
id: pages
|
||||
uses: actions/configure-pages@v5
|
||||
- name: Set up Python
|
||||
uses: actions/setup-python@v5
|
||||
with:
|
||||
python-version: "3.10"
|
||||
- name: Install dependencies
|
||||
run: |
|
||||
cd docs
|
||||
pip install -r requirements-docs.txt
|
||||
- name: Build docs
|
||||
run: |
|
||||
cd docs
|
||||
make clean
|
||||
make html
|
||||
- name: Upload artifact
|
||||
uses: actions/upload-pages-artifact@v3
|
||||
with:
|
||||
path: ./docs/build/html
|
||||
|
||||
# Deployment job
|
||||
deploy:
|
||||
environment:
|
||||
name: github-pages
|
||||
url: ${{ steps.deployment.outputs.page_url }}
|
||||
if: ${{ github.event_name == 'push' }}
|
||||
runs-on: ubuntu-latest
|
||||
needs: build
|
||||
steps:
|
||||
- name: Deploy to GitHub Pages
|
||||
id: deployment
|
||||
uses: actions/deploy-pages@v4
|
||||
@@ -0,0 +1,67 @@
|
||||
name: Build and Push Docker Images
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
python_3_10:
|
||||
description: 'Build Python 3.10 image'
|
||||
required: false
|
||||
default: false
|
||||
type: boolean
|
||||
python_3_11:
|
||||
description: 'Build Python 3.11 image'
|
||||
required: false
|
||||
default: false
|
||||
type: boolean
|
||||
python_3_12:
|
||||
description: 'Build Python 3.12 image'
|
||||
required: false
|
||||
default: false
|
||||
type: boolean
|
||||
python_3_12_cuda_12_9:
|
||||
description: 'Build Python 3.12 image Cuda 12.9'
|
||||
required: false
|
||||
default: false
|
||||
type: boolean
|
||||
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
packages: write
|
||||
|
||||
jobs:
|
||||
build-python-3-10:
|
||||
if: ${{ github.event.inputs.python_3_10 == 'true' }}
|
||||
uses: ./.github/workflows/_template-build-image.yml
|
||||
with:
|
||||
python_version: '3.10'
|
||||
dockerfile_path: docker/Dockerfile.python3.10
|
||||
tag_suffix: py3.10
|
||||
secrets: inherit
|
||||
|
||||
build-python-3-11:
|
||||
if: ${{ github.event.inputs.python_3_11 == 'true' }}
|
||||
uses: ./.github/workflows/_template-build-image.yml
|
||||
with:
|
||||
python_version: '3.11'
|
||||
dockerfile_path: docker/Dockerfile.python3.11
|
||||
tag_suffix: py3.11
|
||||
secrets: inherit
|
||||
|
||||
build-python-3-12:
|
||||
if: ${{ github.event.inputs.python_3_12 == 'true' }}
|
||||
uses: ./.github/workflows/_template-build-image.yml
|
||||
with:
|
||||
python_version: '3.12'
|
||||
dockerfile_path: docker/Dockerfile.python3.12
|
||||
tag_suffix: py3.12
|
||||
secrets: inherit
|
||||
|
||||
build-python-3-12-cuda-12-9:
|
||||
if: ${{ github.event.inputs.python_3_12_cuda_12_9 == 'true' }}
|
||||
uses: ./.github/workflows/_template-build-image.yml
|
||||
with:
|
||||
python_version: '3.12'
|
||||
dockerfile_path: docker/Dockerfile.python3.12.cuda12.9.1
|
||||
tag_suffix: py3.12-cuda12.9.1
|
||||
secrets: inherit
|
||||
@@ -0,0 +1,73 @@
|
||||
name: Deploy Documentation
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [ main ]
|
||||
paths:
|
||||
- 'docs/**'
|
||||
- 'mkdocs.yml'
|
||||
- 'requirements-mkdocs.txt'
|
||||
- '.github/workflows/infra-docs.yml'
|
||||
pull_request:
|
||||
branches: [ main ]
|
||||
paths:
|
||||
- 'docs/**'
|
||||
- 'mkdocs.yml'
|
||||
- 'requirements-mkdocs.txt'
|
||||
- '.github/workflows/infra-docs.yml'
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
pages: write
|
||||
id-token: write
|
||||
|
||||
concurrency:
|
||||
group: "pages"
|
||||
cancel-in-progress: false
|
||||
|
||||
jobs:
|
||||
build:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Setup Python
|
||||
uses: actions/setup-python@v5
|
||||
with:
|
||||
python-version: '3.12'
|
||||
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@v3
|
||||
|
||||
- name: Install dependencies
|
||||
run: uv pip install --system -r requirements-mkdocs.txt
|
||||
|
||||
- name: Setup Pages
|
||||
uses: actions/configure-pages@v4
|
||||
|
||||
- name: Generate docs examples
|
||||
run: python docs/generate_examples.py
|
||||
|
||||
- name: Check docs links
|
||||
run: python scripts/check_docs_links.py
|
||||
|
||||
- name: Build documentation
|
||||
run: mkdocs build
|
||||
|
||||
- name: Upload artifact
|
||||
uses: actions/upload-pages-artifact@v3
|
||||
with:
|
||||
path: ./site
|
||||
|
||||
deploy:
|
||||
environment:
|
||||
name: github-pages
|
||||
url: ${{ steps.deployment.outputs.page_url }}
|
||||
runs-on: ubuntu-latest
|
||||
needs: build
|
||||
if: github.ref == 'refs/heads/main'
|
||||
steps:
|
||||
- name: Deploy to GitHub Pages
|
||||
id: deployment
|
||||
uses: actions/deploy-pages@v4
|
||||
@@ -13,4 +13,4 @@
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,17 @@
|
||||
{
|
||||
"problemMatcher": [
|
||||
{
|
||||
"owner": "ruff",
|
||||
"pattern": [
|
||||
{
|
||||
"regexp": "^(.+):(\\d+):(\\d+): (\\w+) (.+)$",
|
||||
"file": 1,
|
||||
"line": 2,
|
||||
"column": 3,
|
||||
"code": 4,
|
||||
"message": 5
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -1,295 +0,0 @@
|
||||
name: PR Test
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [main]
|
||||
paths:
|
||||
- "fastvideo/**/*.py"
|
||||
- ".github/workflows/pr-test.yml"
|
||||
pull_request:
|
||||
branches: [main]
|
||||
types: [opened, ready_for_review, synchronize, reopened]
|
||||
paths:
|
||||
- "fastvideo/**/*.py"
|
||||
- ".github/workflows/pr-test.yml"
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
custom_image:
|
||||
description: "Custom image from this repository (default: fastvideo-dev:latest)"
|
||||
required: false
|
||||
default: "fastvideo-dev:latest"
|
||||
type: string
|
||||
run_encoder_test:
|
||||
description: "Run encoder-test"
|
||||
required: false
|
||||
default: false
|
||||
type: boolean
|
||||
run_vae_test:
|
||||
description: "Run vae-test"
|
||||
required: false
|
||||
default: false
|
||||
type: boolean
|
||||
run_transformer_test:
|
||||
description: "Run transformer-test"
|
||||
required: false
|
||||
default: false
|
||||
type: boolean
|
||||
run_ssim_test:
|
||||
description: "Run ssim-test"
|
||||
required: false
|
||||
default: false
|
||||
type: boolean
|
||||
|
||||
env:
|
||||
PYTHONUNBUFFERED: "1"
|
||||
|
||||
concurrency:
|
||||
group: pr-test-${{ github.ref }}
|
||||
cancel-in-progress: true
|
||||
|
||||
jobs:
|
||||
pre-commit:
|
||||
uses: ./.github/workflows/pre-commit.yml
|
||||
|
||||
change-filter:
|
||||
runs-on: ubuntu-latest
|
||||
needs: pre-commit
|
||||
if: ${{ github.event.pull_request.draft == false || github.event_name == 'workflow_dispatch' }}
|
||||
outputs:
|
||||
encoder-test: ${{ steps.filter.outputs.encoder-test }}
|
||||
vae-test: ${{ steps.filter.outputs.vae-test }}
|
||||
transformer-test: ${{ steps.filter.outputs.transformer-test }}
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: dorny/paths-filter@v3
|
||||
id: filter
|
||||
with:
|
||||
filters: |
|
||||
encoder-test:
|
||||
- 'fastvideo/v1/models/encoders/**'
|
||||
- 'fastvideo/v1/models/loaders/**'
|
||||
- 'fastvideo/v1/tests/encoders/**'
|
||||
vae-test:
|
||||
- 'fastvideo/v1/models/vaes/**'
|
||||
- 'fastvideo/v1/models/loaders/**'
|
||||
- 'fastvideo/v1/tests/vaes/**'
|
||||
transformer-test:
|
||||
- 'fastvideo/v1/models/dits/**'
|
||||
- 'fastvideo/v1/models/loaders/**'
|
||||
- 'fastvideo/v1/tests/transformers/**'
|
||||
|
||||
encoder-test:
|
||||
needs: change-filter
|
||||
if: >-
|
||||
(github.event_name != 'workflow_dispatch' && needs.change-filter.outputs.encoder-test == 'true') ||
|
||||
(github.event_name == 'workflow_dispatch' && github.event.inputs.run_encoder_test == 'true')
|
||||
runs-on: ubuntu-latest
|
||||
environment: runpod-runners
|
||||
steps:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Set up Python
|
||||
uses: actions/setup-python@v5
|
||||
with:
|
||||
python-version: "3.10"
|
||||
|
||||
- name: Set up SSH key
|
||||
run: |
|
||||
mkdir -p ~/.ssh
|
||||
echo "${{ secrets.RUNPOD_PRIVATE_KEY }}" > ~/.ssh/id_rsa
|
||||
chmod 600 ~/.ssh/id_rsa
|
||||
ssh-keygen -y -f ~/.ssh/id_rsa > ~/.ssh/id_rsa.pub
|
||||
|
||||
- name: Install dependencies
|
||||
run: pip install requests
|
||||
|
||||
- name: Run tests on RunPod
|
||||
env:
|
||||
JOB_ID: "encoder-test"
|
||||
RUNPOD_API_KEY: ${{ secrets.RUNPOD_API_KEY }}
|
||||
GITHUB_RUN_ID: ${{ github.run_id }}
|
||||
timeout-minutes: 30
|
||||
run: >-
|
||||
python .github/scripts/runpod_api.py
|
||||
--gpu-type "NVIDIA A40"
|
||||
--gpu-count 1
|
||||
--volume-size 100
|
||||
--image "ghcr.io/${{ github.repository }}/${{ github.event.inputs.custom_image || 'fastvideo-dev:latest' }}"
|
||||
--test-command "pip install -e .[test] && pytest ./fastvideo/v1/tests/encoders -s"
|
||||
|
||||
- name: Terminate RunPod Instances
|
||||
if: ${{ always() }}
|
||||
env:
|
||||
RUNPOD_API_KEY: ${{ secrets.RUNPOD_API_KEY }}
|
||||
GITHUB_RUN_ID: ${{ github.run_id }}
|
||||
JOB_ID: "encoder-test"
|
||||
run: python .github/scripts/runpod_cleanup.py
|
||||
|
||||
vae-test:
|
||||
needs: change-filter
|
||||
if: >-
|
||||
(github.event_name != 'workflow_dispatch' && needs.change-filter.outputs.vae-test == 'true') ||
|
||||
(github.event_name == 'workflow_dispatch' && github.event.inputs.run_vae_test == 'true')
|
||||
runs-on: ubuntu-latest
|
||||
environment: runpod-runners
|
||||
steps:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Set up Python
|
||||
uses: actions/setup-python@v5
|
||||
with:
|
||||
python-version: "3.10"
|
||||
|
||||
- name: Set up SSH key
|
||||
run: |
|
||||
mkdir -p ~/.ssh
|
||||
echo "${{ secrets.RUNPOD_PRIVATE_KEY }}" > ~/.ssh/id_rsa
|
||||
chmod 600 ~/.ssh/id_rsa
|
||||
ssh-keygen -y -f ~/.ssh/id_rsa > ~/.ssh/id_rsa.pub
|
||||
|
||||
- name: Install dependencies
|
||||
run: pip install requests
|
||||
|
||||
- name: Run tests on RunPod
|
||||
env:
|
||||
JOB_ID: "vae-test"
|
||||
RUNPOD_API_KEY: ${{ secrets.RUNPOD_API_KEY }}
|
||||
GITHUB_RUN_ID: ${{ github.run_id }}
|
||||
timeout-minutes: 30
|
||||
run: >-
|
||||
python .github/scripts/runpod_api.py
|
||||
--gpu-type "NVIDIA A40"
|
||||
--gpu-count 1
|
||||
--volume-size 100
|
||||
--image "ghcr.io/${{ github.repository }}/${{ github.event.inputs.custom_image || 'fastvideo-dev:latest' }}"
|
||||
--test-command "pip install -e .[test] && pytest ./fastvideo/v1/tests/vaes -s"
|
||||
|
||||
- name: Terminate RunPod Instances
|
||||
if: ${{ always() }}
|
||||
env:
|
||||
RUNPOD_API_KEY: ${{ secrets.RUNPOD_API_KEY }}
|
||||
GITHUB_RUN_ID: ${{ github.run_id }}
|
||||
JOB_ID: "vae-test"
|
||||
run: python .github/scripts/runpod_cleanup.py
|
||||
|
||||
transformer-test:
|
||||
needs: change-filter
|
||||
if: >-
|
||||
(github.event_name != 'workflow_dispatch' && needs.change-filter.outputs.transformer-test == 'true') ||
|
||||
(github.event_name == 'workflow_dispatch' && github.event.inputs.run_transformer_test == 'true')
|
||||
runs-on: ubuntu-latest
|
||||
environment: runpod-runners
|
||||
steps:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Set up Python
|
||||
uses: actions/setup-python@v5
|
||||
with:
|
||||
python-version: "3.10"
|
||||
|
||||
- name: Set up SSH key
|
||||
run: |
|
||||
mkdir -p ~/.ssh
|
||||
echo "${{ secrets.RUNPOD_PRIVATE_KEY }}" > ~/.ssh/id_rsa
|
||||
chmod 600 ~/.ssh/id_rsa
|
||||
ssh-keygen -y -f ~/.ssh/id_rsa > ~/.ssh/id_rsa.pub
|
||||
|
||||
- name: Install dependencies
|
||||
run: pip install requests
|
||||
|
||||
- name: Run tests on RunPod
|
||||
env:
|
||||
JOB_ID: "transformer-test"
|
||||
RUNPOD_API_KEY: ${{ secrets.RUNPOD_API_KEY }}
|
||||
GITHUB_RUN_ID: ${{ github.run_id }}
|
||||
timeout-minutes: 30
|
||||
run: >-
|
||||
python .github/scripts/runpod_api.py
|
||||
--gpu-type "NVIDIA L40S"
|
||||
--gpu-count 1
|
||||
--volume-size 100
|
||||
--image "ghcr.io/${{ github.repository }}/${{ github.event.inputs.custom_image || 'fastvideo-dev:latest' }}"
|
||||
--test-command "pip install -e .[test] && pytest ./fastvideo/v1/tests/transformers -s"
|
||||
|
||||
- name: Terminate RunPod Instances
|
||||
if: ${{ always() }}
|
||||
env:
|
||||
RUNPOD_API_KEY: ${{ secrets.RUNPOD_API_KEY }}
|
||||
GITHUB_RUN_ID: ${{ github.run_id }}
|
||||
JOB_ID: "transformer-test"
|
||||
run: python .github/scripts/runpod_cleanup.py
|
||||
|
||||
ssim-test:
|
||||
needs: change-filter
|
||||
if: >-
|
||||
(github.event_name != 'workflow_dispatch' && github.event.pull_request.draft == false) ||
|
||||
(github.event_name == 'workflow_dispatch' && github.event.inputs.run_ssim_test == 'true')
|
||||
runs-on: ubuntu-latest
|
||||
environment: runpod-runners
|
||||
steps:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Set up Python
|
||||
uses: actions/setup-python@v5
|
||||
with:
|
||||
python-version: "3.10"
|
||||
|
||||
- name: Set up SSH key
|
||||
run: |
|
||||
mkdir -p ~/.ssh
|
||||
echo "${{ secrets.RUNPOD_PRIVATE_KEY }}" > ~/.ssh/id_rsa
|
||||
chmod 600 ~/.ssh/id_rsa
|
||||
ssh-keygen -y -f ~/.ssh/id_rsa > ~/.ssh/id_rsa.pub
|
||||
|
||||
- name: Install dependencies
|
||||
run: pip install requests
|
||||
|
||||
- name: Run tests on RunPod
|
||||
env:
|
||||
JOB_ID: "ssim-test"
|
||||
RUNPOD_API_KEY: ${{ secrets.RUNPOD_API_KEY }}
|
||||
GITHUB_RUN_ID: ${{ github.run_id }}
|
||||
timeout-minutes: 45
|
||||
run: >-
|
||||
python .github/scripts/runpod_api.py
|
||||
--gpu-type "NVIDIA A40"
|
||||
--gpu-count 2
|
||||
--disk-size 200
|
||||
--volume-size 200
|
||||
--image "ghcr.io/${{ github.repository }}/${{ github.event.inputs.custom_image || 'fastvideo-dev:latest' }}"
|
||||
--test-command "pip install -e .[test] && pytest ./fastvideo/v1/tests/ssim -vs"
|
||||
|
||||
- name: Terminate RunPod Instances
|
||||
if: ${{ always() }}
|
||||
env:
|
||||
RUNPOD_API_KEY: ${{ secrets.RUNPOD_API_KEY }}
|
||||
GITHUB_RUN_ID: ${{ github.run_id }}
|
||||
JOB_ID: "ssim-test"
|
||||
run: python .github/scripts/runpod_cleanup.py
|
||||
|
||||
runpod-cleanup:
|
||||
needs: [encoder-test, vae-test, transformer-test, ssim-test] # Add other jobs to this list as you create them
|
||||
if: ${{ always() && ((github.event_name != 'workflow_dispatch' && github.event.pull_request.draft == false) || github.event_name == 'workflow_dispatch') }}
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Set up Python
|
||||
uses: actions/setup-python@v5
|
||||
with:
|
||||
python-version: "3.10"
|
||||
|
||||
- name: Install dependencies
|
||||
run: pip install requests
|
||||
|
||||
- name: Cleanup all RunPod instances
|
||||
env:
|
||||
JOB_IDS: '["encoder-test", "vae-test", "transformer-test", "ssim-test"]' # JSON array of job IDs
|
||||
RUNPOD_API_KEY: ${{ secrets.RUNPOD_API_KEY }}
|
||||
GITHUB_RUN_ID: ${{ github.run_id }}
|
||||
run: python .github/scripts/runpod_cleanup.py
|
||||
@@ -1,18 +0,0 @@
|
||||
name: pre-commit
|
||||
|
||||
on:
|
||||
workflow_call:
|
||||
|
||||
jobs:
|
||||
pre-commit:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/setup-python@v5
|
||||
with:
|
||||
python-version: "3.10"
|
||||
- run: echo "::add-matcher::.github/workflows/matchers/actionlint.json"
|
||||
- run: echo "::add-matcher::.github/workflows/matchers/mypy.json"
|
||||
- uses: pre-commit/action@v3.0.1
|
||||
with:
|
||||
extra_args: --all-files --hook-stage manual
|
||||
@@ -0,0 +1,28 @@
|
||||
name: Publish to Comfy registry
|
||||
on:
|
||||
workflow_dispatch:
|
||||
push:
|
||||
branches:
|
||||
- main
|
||||
- master
|
||||
paths:
|
||||
- "pyproject.toml"
|
||||
|
||||
permissions:
|
||||
issues: write
|
||||
|
||||
jobs:
|
||||
publish-node:
|
||||
name: Publish Custom Node to registry
|
||||
runs-on: ubuntu-latest
|
||||
if: ${{ github.repository_owner == 'hao-ai-lab' }}
|
||||
steps:
|
||||
- name: Check out code
|
||||
uses: actions/checkout@v4
|
||||
with:
|
||||
submodules: true
|
||||
- name: Publish Custom Node
|
||||
uses: Comfy-Org/publish-node-action@v1
|
||||
with:
|
||||
## Add your own personal access token to your Github Repository secrets and reference it here.
|
||||
personal_access_token: ${{ secrets.REGISTRY_ACCESS_TOKEN }}
|
||||
@@ -56,10 +56,11 @@ jobs:
|
||||
with:
|
||||
python-version: '3.10'
|
||||
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@v3
|
||||
|
||||
- name: Install build dependencies
|
||||
run: |
|
||||
python -m pip install --upgrade pip
|
||||
pip install build twine wheel
|
||||
run: uv pip install --system build twine wheel
|
||||
|
||||
- name: Build package
|
||||
run: |
|
||||
@@ -0,0 +1,230 @@
|
||||
name: Publish FastVideo Kernel to PyPI on Version Change
|
||||
|
||||
on:
|
||||
push:
|
||||
branches:
|
||||
- main
|
||||
paths:
|
||||
- "fastvideo-kernel/pyproject.toml"
|
||||
workflow_dispatch:
|
||||
|
||||
jobs:
|
||||
check-version-change:
|
||||
runs-on: ubuntu-latest
|
||||
outputs:
|
||||
version-changed: ${{ steps.check-version.outputs.changed }}
|
||||
new-version: ${{ steps.check-version.outputs.new-version }}
|
||||
steps:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@v4
|
||||
with:
|
||||
fetch-depth: 2
|
||||
|
||||
- name: Check if version changed
|
||||
id: check-version
|
||||
run: |
|
||||
cd fastvideo-kernel
|
||||
# Get current commit's version from pyproject.toml
|
||||
# Use ^ to match start of line to avoid matching minimum-version
|
||||
NEW_VERSION=$(grep -oP '^version\s*=\s*"\K[^"]+' pyproject.toml)
|
||||
echo "New version: $NEW_VERSION"
|
||||
|
||||
# Get previous version from git history
|
||||
# Note: git show expects path relative to repo root
|
||||
OLD_VERSION=$(git show HEAD~1:fastvideo-kernel/pyproject.toml | grep -oP '^version\s*=\s*"\K[^"]+' || echo "0.0.0")
|
||||
echo "Old version: $OLD_VERSION"
|
||||
|
||||
if [ "$NEW_VERSION" != "$OLD_VERSION" ]; then
|
||||
echo "Version changed from $OLD_VERSION to $NEW_VERSION"
|
||||
echo "changed=true" >> "$GITHUB_OUTPUT"
|
||||
echo "new-version=$NEW_VERSION" >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
echo "Version did not change"
|
||||
echo "changed=false" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
|
||||
build_wheels:
|
||||
name: Build Wheel
|
||||
needs: check-version-change
|
||||
if: ${{ needs.check-version-change.outputs.version-changed == 'true' || github.event_name == 'workflow_dispatch' }}
|
||||
runs-on: ${{ matrix.os }}
|
||||
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
os: [ubuntu-22.04]
|
||||
python-version: ['3.10', '3.11', '3.12']
|
||||
torch-cuda:
|
||||
# - torch-version: '2.5.1'
|
||||
# cuda-version: '12.4.1'
|
||||
# torch-cuda-short: 'cu124'
|
||||
# - torch-version: '2.6.0'
|
||||
# cuda-version: '12.6.3'
|
||||
# torch-cuda-short: 'cu126'
|
||||
# - torch-version: '2.7.1'
|
||||
# cuda-version: '12.8.0'
|
||||
# torch-cuda-short: 'cu128'
|
||||
# - torch-version: '2.9.1'
|
||||
# cuda-version: '12.8.0'
|
||||
# torch-cuda-short: 'cu128'
|
||||
- torch-version: '2.10.0'
|
||||
cuda-version: '12.8.0'
|
||||
torch-cuda-short: 'cu128'
|
||||
|
||||
steps:
|
||||
- name: Free up disk space
|
||||
run: |
|
||||
echo "Initial disk space:"
|
||||
df -h
|
||||
|
||||
# Remove large directories
|
||||
sudo rm -rf /usr/share/dotnet
|
||||
sudo rm -rf /usr/local/lib/android
|
||||
sudo rm -rf /opt/ghc
|
||||
sudo rm -rf /usr/local/share/boost
|
||||
sudo rm -rf /usr/share/swift
|
||||
sudo rm -rf /usr/local/lib/node_modules
|
||||
sudo rm -rf /usr/local/share/powershell
|
||||
sudo rm -rf /usr/share/rust
|
||||
sudo rm -rf /usr/local/.ghcup
|
||||
|
||||
# Remove cached files
|
||||
sudo rm -rf /var/lib/apt/lists/*
|
||||
sudo rm -rf /var/cache/apt/archives/*
|
||||
|
||||
echo "Disk space after cleanup:"
|
||||
df -h
|
||||
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Set up Python
|
||||
uses: actions/setup-python@v5
|
||||
with:
|
||||
python-version: ${{ matrix.python-version }}
|
||||
|
||||
- name: Install CUDA ${{ matrix.torch-cuda.cuda-version }}
|
||||
uses: Jimver/cuda-toolkit@v0.2.21
|
||||
id: cuda-toolkit
|
||||
with:
|
||||
cuda: ${{ matrix.torch-cuda.cuda-version }}
|
||||
linux-local-args: '["--toolkit"]'
|
||||
method: 'network'
|
||||
|
||||
- name: Install dependencies (GCC, Clang, CUDA Paths, Git)
|
||||
run: |
|
||||
sudo apt update
|
||||
sudo apt install -y git patchelf gcc-11 g++-11 clang-11
|
||||
sudo update-alternatives --install /usr/bin/gcc gcc /usr/bin/gcc-11 100 --slave /usr/bin/g++ g++ /usr/bin/g++-11
|
||||
|
||||
# Allow Git to Access Safe Directory
|
||||
git config --global --add safe.directory /__w/FastVideo/FastVideo
|
||||
|
||||
# Set CUDA environment variables
|
||||
export CUDA_HOME=/usr/local/cuda-${{ matrix.torch-cuda.cuda-version }}
|
||||
export PATH=${CUDA_HOME}/bin:${PATH}
|
||||
export LD_LIBRARY_PATH=${CUDA_HOME}/lib64:$LD_LIBRARY_PATH
|
||||
|
||||
# Verify installation
|
||||
gcc --version
|
||||
g++ --version
|
||||
clang-11 --version
|
||||
nvcc --version
|
||||
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@v3
|
||||
|
||||
- name: Install PyTorch ${{ matrix.torch-cuda.torch-version }}+cu${{ matrix.torch-cuda.cuda-version }}
|
||||
run: |
|
||||
uv pip install --system typing-extensions==4.12.2
|
||||
uv pip install --system --no-cache-dir torch==${{ matrix.torch-cuda.torch-version }} --index-url https://download.pytorch.org/whl/${{matrix.torch-cuda.torch-cuda-short}}
|
||||
nvcc --version
|
||||
python --version
|
||||
python -c "import torch; print('PyTorch:', torch.__version__)"
|
||||
python -c "import torch; print('CUDA:', torch.version.cuda)"
|
||||
python -c "from torch.utils import cpp_extension; print (cpp_extension.CUDA_HOME)"
|
||||
|
||||
- name: Build wheel
|
||||
run: |
|
||||
export PYTHONPATH=$GITHUB_WORKSPACE:$PYTHONPATH
|
||||
|
||||
uv pip install --system setuptools ninja packaging wheel triton scikit-build-core cmake build
|
||||
|
||||
cd fastvideo-kernel
|
||||
git submodule update --init --recursive # Ensure ThunderKittens submodule is initialized
|
||||
# Release builds are produced on GPU-less runners, so force-enable TK and target Hopper.
|
||||
export TORCH_CUDA_ARCH_LIST="9.0a"
|
||||
export CMAKE_ARGS="${CMAKE_ARGS:-} -DFASTVIDEO_KERNEL_BUILD_TK=ON -DCMAKE_CUDA_ARCHITECTURES=90a"
|
||||
|
||||
# Build standard wheel (no local version suffix) for PyPI
|
||||
python -m build --wheel --outdir dist
|
||||
|
||||
# Fix the wheel to be manylinux compliant
|
||||
uv pip install --system auditwheel
|
||||
# Point auditwheel at torch libs, but do not vendor them into the wheel.
|
||||
TORCH_LIB_DIR=$(python - <<'PY'
|
||||
import os
|
||||
import torch
|
||||
|
||||
print(os.path.join(os.path.dirname(torch.__file__), "lib"))
|
||||
PY
|
||||
)
|
||||
export LD_LIBRARY_PATH="${TORCH_LIB_DIR}:${LD_LIBRARY_PATH}"
|
||||
# Target manylinux_2_35 (Ubuntu 22.04 native)
|
||||
auditwheel repair dist/*.whl --plat manylinux_2_35_x86_64 -w fixed_dist \
|
||||
--exclude libtorch_cuda.so \
|
||||
--exclude libtorch_cpu.so \
|
||||
--exclude libtorch.so \
|
||||
--exclude libc10.so \
|
||||
--exclude libc10_cuda.so \
|
||||
--exclude libtorch_python.so
|
||||
# Move fixed wheels back to dist for upload consistency
|
||||
rm dist/*.whl
|
||||
mv fixed_dist/*.whl dist/
|
||||
|
||||
- name: Upload wheel artifact
|
||||
# Only upload if it's the "main" CUDA version we want on PyPI
|
||||
# We upload all to artifacts for inspection/GH releases, but give them distinct artifact names
|
||||
uses: actions/upload-artifact@v4
|
||||
with:
|
||||
name: fastvideo_kernel-py${{ matrix.python-version }}-${{ matrix.torch-cuda.torch-cuda-short }}-torch${{ matrix.torch-cuda.torch-version }}
|
||||
path: fastvideo-kernel/dist/*.whl
|
||||
retention-days: 90
|
||||
|
||||
publish_package:
|
||||
name: Publish package
|
||||
needs: [build_wheels, check-version-change]
|
||||
if: ${{ needs.check-version-change.outputs.version-changed == 'true' || github.event_name == 'workflow_dispatch' }}
|
||||
runs-on: ubuntu-22.04
|
||||
permissions:
|
||||
id-token: write # Needed for OIDC Trusted Publishing
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- uses: actions/setup-python@v5
|
||||
with:
|
||||
python-version: '3.10'
|
||||
|
||||
- name: Download PyPI wheels
|
||||
uses: actions/download-artifact@v4
|
||||
with:
|
||||
path: fastvideo-kernel/dist/
|
||||
pattern: 'fastvideo_kernel-py*'
|
||||
merge-multiple: true
|
||||
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@v3
|
||||
|
||||
- name: Build source distribution
|
||||
run: |
|
||||
uv pip install --system build scikit-build-core cmake ninja
|
||||
|
||||
cd fastvideo-kernel
|
||||
# We don't need full CUDA/Torch to just package the source (sdist)
|
||||
python -m build --sdist --outdir dist
|
||||
|
||||
- name: Publish release distributions to PyPI
|
||||
uses: pypa/gh-action-pypi-publish@release/v1
|
||||
with:
|
||||
packages-dir: fastvideo-kernel/dist/
|
||||
@@ -1,245 +0,0 @@
|
||||
name: Publish Sliding Tile Attention Kernel to PyPI on Version Change
|
||||
|
||||
on:
|
||||
push:
|
||||
branches:
|
||||
- main
|
||||
paths:
|
||||
- "csrc/sliding_tile_attention/setup.py"
|
||||
workflow_dispatch:
|
||||
|
||||
jobs:
|
||||
check-version-change:
|
||||
runs-on: ubuntu-latest
|
||||
outputs:
|
||||
version-changed: ${{ steps.check-version.outputs.changed }}
|
||||
new-version: ${{ steps.check-version.outputs.new-version }}
|
||||
steps:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@v4
|
||||
with:
|
||||
fetch-depth: 2
|
||||
|
||||
- name: Check if version changed
|
||||
id: check-version
|
||||
run: |
|
||||
cd csrc/sliding_tile_attention
|
||||
# Get current commit's version
|
||||
NEW_VERSION=$(grep -oP 'VERSION\s*=\s*"\K[^"]+' setup.py)
|
||||
echo "New version: $NEW_VERSION"
|
||||
|
||||
# Get previous version from git history
|
||||
OLD_VERSION=$(git show HEAD~1:./setup.py | grep -oP 'VERSION\s*=\s*"\K[^"]+' || echo "0.0.0")
|
||||
echo "Old version: $OLD_VERSION"
|
||||
|
||||
if [ "$NEW_VERSION" != "$OLD_VERSION" ]; then
|
||||
echo "Version changed from $OLD_VERSION to $NEW_VERSION"
|
||||
echo "changed=true" >> $GITHUB_OUTPUT
|
||||
echo "new-version=$NEW_VERSION" >> $GITHUB_OUTPUT
|
||||
else
|
||||
echo "Version did not change"
|
||||
echo "changed=false" >> $GITHUB_OUTPUT
|
||||
fi
|
||||
|
||||
build_wheels:
|
||||
name: Build Wheel
|
||||
needs: check-version-change
|
||||
if: ${{ needs.check-version-change.outputs.version-changed == 'true' || github.event_name == 'workflow_dispatch' }}
|
||||
runs-on: ${{ matrix.os }}
|
||||
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
# Using ubuntu-20.04 instead of 22.04 for more compatibility (glibc). Ideally we'd use the
|
||||
# manylinux docker image, but I haven't figured out how to install CUDA on manylinux.
|
||||
os: [ubuntu-22.04]
|
||||
python-version: ['3.10', '3.11', '3.12', '3.13']
|
||||
torch-version: ['2.5.1', '2.6.0']
|
||||
cuda-version: ['12.4.1', '12.5.1', '12.6.3']
|
||||
|
||||
steps:
|
||||
- name: Free up disk space
|
||||
run: |
|
||||
echo "Initial disk space:"
|
||||
df -h
|
||||
|
||||
# Remove large directories
|
||||
sudo rm -rf /usr/share/dotnet
|
||||
sudo rm -rf /usr/local/lib/android
|
||||
sudo rm -rf /opt/ghc
|
||||
sudo rm -rf /usr/local/share/boost
|
||||
sudo rm -rf /usr/share/swift
|
||||
sudo rm -rf /usr/local/lib/node_modules
|
||||
sudo rm -rf /usr/local/share/powershell
|
||||
sudo rm -rf /usr/share/rust
|
||||
sudo rm -rf /usr/local/.ghcup
|
||||
|
||||
# Remove cached files
|
||||
sudo rm -rf /var/lib/apt/lists/*
|
||||
sudo rm -rf /var/cache/apt/archives/*
|
||||
|
||||
echo "Disk space after cleanup:"
|
||||
df -h
|
||||
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Set up Python
|
||||
uses: actions/setup-python@v5
|
||||
with:
|
||||
python-version: ${{ matrix.python-version }}
|
||||
|
||||
- name: Install CUDA ${{ matrix.cuda-version }}
|
||||
uses: Jimver/cuda-toolkit@v0.2.21
|
||||
id: cuda-toolkit
|
||||
with:
|
||||
cuda: ${{ matrix.cuda-version }}
|
||||
linux-local-args: '["--toolkit"]'
|
||||
method: 'network'
|
||||
|
||||
- name: Install dependencies (GCC, Clang, CUDA Paths, Git)
|
||||
run: |
|
||||
sudo apt update
|
||||
sudo apt install -y git patchelf gcc-11 g++-11 clang-11
|
||||
sudo update-alternatives --install /usr/bin/gcc gcc /usr/bin/gcc-11 100 --slave /usr/bin/g++ g++ /usr/bin/g++-11
|
||||
|
||||
# Allow Git to Access Safe Directory
|
||||
git config --global --add safe.directory /__w/FastVideo/FastVideo
|
||||
|
||||
# Set CUDA environment variables
|
||||
export CUDA_HOME=/usr/local/cuda-${{ matrix.cuda-version }}
|
||||
export PATH=${CUDA_HOME}/bin:${PATH}
|
||||
export LD_LIBRARY_PATH=${CUDA_HOME}/lib64:$LD_LIBRARY_PATH
|
||||
|
||||
# Verify installation
|
||||
gcc --version
|
||||
g++ --version
|
||||
clang-11 --version
|
||||
nvcc --version
|
||||
|
||||
- name: Install PyTorch ${{ matrix.torch-version }}+cu${{ matrix.cuda-version }}
|
||||
run: |
|
||||
pip install --upgrade pip
|
||||
# With python 3.13 and torch 2.5.1, unless we update typing-extensions, we get error
|
||||
# AttributeError: attribute '__default__' of 'typing.ParamSpec' objects is not writable
|
||||
pip install typing-extensions==4.12.2
|
||||
# We want to figure out the CUDA version to download pytorch
|
||||
# e.g. we can have system CUDA version being 11.7 but if torch==1.12 then we need to download the wheel from cu116
|
||||
# see https://github.com/pytorch/pytorch/blob/main/RELEASE.md#release-compatibility-matrix
|
||||
export TORCH_CUDA_VERSION=124
|
||||
pip install --no-cache-dir torch==${{ matrix.torch-version }} --index-url https://download.pytorch.org/whl/cu${TORCH_CUDA_VERSION}
|
||||
nvcc --version
|
||||
python --version
|
||||
python -c "import torch; print('PyTorch:', torch.__version__)"
|
||||
python -c "import torch; print('CUDA:', torch.version.cuda)"
|
||||
python -c "from torch.utils import cpp_extension; print (cpp_extension.CUDA_HOME)"
|
||||
|
||||
- name: Build wheel
|
||||
run: |
|
||||
# We want setuptools >= 49.6.0 otherwise we can't compile the extension if system CUDA version is 11.7 and pytorch cuda version is 11.6
|
||||
# https://github.com/pytorch/pytorch/blob/664058fa83f1d8eede5d66418abff6e20bd76ca8/torch/utils/cpp_extension.py#L810
|
||||
# However this still fails so I'm using a newer version of setuptools
|
||||
pip install setuptools
|
||||
pip install ninja packaging wheel
|
||||
|
||||
cd csrc/sliding_tile_attention # Move into the correct folder
|
||||
git submodule update --init --recursive tk # Ensure ThunderKittens submodule is initialized
|
||||
python setup.py bdist_wheel --dist-dir=dist
|
||||
|
||||
- name: Rename wheel file
|
||||
run: |
|
||||
cd csrc/sliding_tile_attention
|
||||
|
||||
CUDA_SHORT_VERSION=$(echo ${{ matrix.cuda-version }} | cut -d. -f1,2 | sed 's/\.//g')
|
||||
TORCH_SHORT_VERSION=$(echo ${{ matrix.torch-version }} | cut -d. -f1,2)
|
||||
# Get the correct version format
|
||||
tmpname=cu${CUDA_SHORT_VERSION}torch${TORCH_SHORT_VERSION}
|
||||
wheel_name=$(ls dist/*whl | xargs -n 1 basename | sed "s/-/+$tmpname-/2")
|
||||
# Rename with version information
|
||||
ls dist/*whl |xargs -I {} mv {} dist/${wheel_name}
|
||||
echo "wheel_name=${wheel_name}" >> $GITHUB_ENV
|
||||
|
||||
- name: Upload wheel artifact
|
||||
uses: actions/upload-artifact@v4
|
||||
with:
|
||||
name: ${{ env.wheel_name }}
|
||||
path: csrc/sliding_tile_attention/dist/*.whl
|
||||
retention-days: 90
|
||||
|
||||
publish_package:
|
||||
name: Publish package
|
||||
needs: [build_wheels, check-version-change]
|
||||
if: ${{ needs.check-version-change.outputs.version-changed == 'true' || github.event_name == 'workflow_dispatch' }}
|
||||
runs-on: ubuntu-22.04
|
||||
permissions:
|
||||
id-token: write # Needed for OIDC Trusted Publishing
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- uses: actions/setup-python@v5
|
||||
with:
|
||||
python-version: '3.10'
|
||||
|
||||
- name: Install CUDA 12.4.1
|
||||
uses: Jimver/cuda-toolkit@v0.2.21
|
||||
id: cuda-toolkit
|
||||
with:
|
||||
cuda: 12.4.1
|
||||
linux-local-args: '["--toolkit"]'
|
||||
method: 'network'
|
||||
sub-packages: '["nvcc"]'
|
||||
|
||||
- name: Install dependencies (GCC, Clang, CUDA Paths, Git)
|
||||
run: |
|
||||
sudo apt update
|
||||
sudo apt install -y git patchelf gcc-11 g++-11 clang-11
|
||||
sudo update-alternatives --install /usr/bin/gcc gcc /usr/bin/gcc-11 100 --slave /usr/bin/g++ g++ /usr/bin/g++-11
|
||||
|
||||
# Allow Git to Access Safe Directory
|
||||
git config --global --add safe.directory /__w/FastVideo/FastVideo
|
||||
|
||||
# Set CUDA environment variables
|
||||
export CUDA_HOME=/usr/local/cuda-12.4.1
|
||||
export PATH=${CUDA_HOME}/bin:${PATH}
|
||||
export LD_LIBRARY_PATH=${CUDA_HOME}/lib64:$LD_LIBRARY_PATH
|
||||
|
||||
# Verify installation
|
||||
gcc --version
|
||||
g++ --version
|
||||
clang-11 --version
|
||||
nvcc --version
|
||||
|
||||
- name: Install PyTorch 2.5.1+cu12.4.1
|
||||
run: |
|
||||
pip install --upgrade pip
|
||||
# With python 3.13 and torch 2.5.1, unless we update typing-extensions, we get error
|
||||
# AttributeError: attribute '__default__' of 'typing.ParamSpec' objects is not writable
|
||||
pip install typing-extensions==4.12.2
|
||||
# We want to figure out the CUDA version to download pytorch
|
||||
# e.g. we can have system CUDA version being 11.7 but if torch==1.12 then we need to download the wheel from cu116
|
||||
# see https://github.com/pytorch/pytorch/blob/main/RELEASE.md#release-compatibility-matrix
|
||||
export TORCH_CUDA_VERSION=124
|
||||
pip install --no-cache-dir torch==2.5.1 --index-url https://download.pytorch.org/whl/cu${TORCH_CUDA_VERSION}
|
||||
nvcc --version
|
||||
python --version
|
||||
python -c "import torch; print('PyTorch:', torch.__version__)"
|
||||
python -c "import torch; print('CUDA:', torch.version.cuda)"
|
||||
python -c "from torch.utils import cpp_extension; print (cpp_extension.CUDA_HOME)"
|
||||
|
||||
- name: Build source distribution
|
||||
run: |
|
||||
# We want setuptools >= 49.6.0 otherwise we can't compile the extension if system CUDA version is 11.7 and pytorch cuda version is 11.6
|
||||
# https://github.com/pytorch/pytorch/blob/664058fa83f1d8eede5d66418abff6e20bd76ca8/torch/utils/cpp_extension.py#L810
|
||||
# However this still fails so I'm using a newer version of setuptools
|
||||
pip install setuptools
|
||||
pip install ninja packaging wheel
|
||||
|
||||
cd csrc/sliding_tile_attention # Move into the correct folder
|
||||
git submodule update --init --recursive tk # Ensure ThunderKittens submodule is initialized
|
||||
python setup.py sdist --dist-dir=dist
|
||||
|
||||
- name: Publish release distributions to PyPI
|
||||
uses: pypa/gh-action-pypi-publish@release/v1
|
||||
with:
|
||||
packages-dir: csrc/sliding_tile_attention/dist/
|
||||
@@ -1,31 +0,0 @@
|
||||
name: Run Tests
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [ main ]
|
||||
pull_request:
|
||||
branches: [ main ]
|
||||
|
||||
jobs:
|
||||
test:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Check out repository
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Set up Python
|
||||
uses: actions/setup-python@v5
|
||||
with:
|
||||
python-version: '3.12' # or any version you need
|
||||
|
||||
- name: Install dependencies
|
||||
run: |
|
||||
python -m pip install --upgrade pip setuptools wheel
|
||||
pip install torch
|
||||
pip install packaging ninja
|
||||
pip install -e .
|
||||
pip install pytest
|
||||
|
||||
- name: Run Pytest
|
||||
run: |
|
||||
pytest --ignore csrc/sliding_tile_attention/test
|
||||
@@ -14,12 +14,16 @@ wandb/
|
||||
*.pt
|
||||
cache_dir/
|
||||
wandb/
|
||||
venv/
|
||||
.venv/
|
||||
runs/
|
||||
samples/
|
||||
Miniconda3-latest-Linux-x86_64.sh
|
||||
*validation/
|
||||
data/
|
||||
outputs/
|
||||
outputs_video
|
||||
checkpoints/
|
||||
sbatch.sh
|
||||
*.out
|
||||
env
|
||||
@@ -27,7 +31,14 @@ env
|
||||
**/build/
|
||||
**.pyc
|
||||
**.txt
|
||||
**.json
|
||||
*.log
|
||||
weights/
|
||||
logs/
|
||||
|
||||
# SSIM test outputs
|
||||
fastvideo/tests/ssim/generated_videos/
|
||||
**/.cache/**
|
||||
|
||||
|
||||
# Distribution / packaging
|
||||
build/
|
||||
@@ -37,10 +48,13 @@ dist/
|
||||
eggs/
|
||||
.eggs/
|
||||
|
||||
# Sphinx documentation
|
||||
docs/_build/
|
||||
docs/source/getting_started/examples/
|
||||
docs/source/inference/examples/
|
||||
# MkDocs documentation
|
||||
site/
|
||||
docs/getting_started/examples/
|
||||
docs/inference/examples/
|
||||
docs/training/examples/
|
||||
docs/distillation/examples/
|
||||
!requirements-mkdocs.txt
|
||||
|
||||
# VSCode
|
||||
.vscode/
|
||||
@@ -56,7 +70,25 @@ docs/source/inference/examples/
|
||||
*.pkl
|
||||
|
||||
# Reference videos
|
||||
!fastvideo/v1/tests/ssim/reference_videos/**/*.mp4
|
||||
!fastvideo/tests/ssim/reference_videos/**/*.mp4
|
||||
|
||||
# Static images
|
||||
!docs/source/_static/images/**/*.png
|
||||
!docs/assets/images/**/*.png
|
||||
!comfyui/assets/**/*.png
|
||||
!comfyui/assets/**/*.gif
|
||||
!assets/images/**/*.png
|
||||
!assets/images/**/*.jpg
|
||||
!assets/images/**/*.jpeg
|
||||
!assets/images/**/*.gif
|
||||
!assets/videos/**/*.mp4
|
||||
|
||||
dmd_t2v_output/
|
||||
preprocess_output_text/
|
||||
|
||||
# Next.js / Node artifacts under ui/: see ui/.gitignore
|
||||
|
||||
.claude/
|
||||
.codex/
|
||||
.sisyphus/
|
||||
openspec/
|
||||
fastvideo/tests/ssim/reference_videos/**
|
||||
|
||||
@@ -1,3 +1,6 @@
|
||||
[submodule "csrc/sliding_tile_attention/tk"]
|
||||
path = csrc/sliding_tile_attention/tk
|
||||
[submodule "fastvideo-kernel/include/tk"]
|
||||
path = fastvideo-kernel/include/tk
|
||||
url = https://github.com/HazyResearch/ThunderKittens.git
|
||||
[submodule "fastvideo-kernel/include/cutlass"]
|
||||
path = fastvideo-kernel/include/cutlass
|
||||
url = https://github.com/NVIDIA/cutlass.git
|
||||
|
||||
@@ -0,0 +1 @@
|
||||
WRN 2026-03-26T13:46:33.469 ?.19646 server_start:193: Failed to start server: operation not permitted: /var/folders/z_/h_6myyk14d1b7z87z3vy4mjh0000gn/T/nvim.dsynkd/iSe0el/nvim.19646.0
|
||||
@@ -3,25 +3,17 @@ default_stages:
|
||||
- manual # Run in CI
|
||||
exclude: |
|
||||
(?x)(
|
||||
fastvideo/v1/third_party/.*|
|
||||
csrc/.*|
|
||||
fastvideo/third_party/.*|
|
||||
fastvideo-kernel/.*|
|
||||
assets/.*|
|
||||
tests/.*|
|
||||
demo/.*|
|
||||
predict\.py|
|
||||
scripts/.*|
|
||||
fastvideo/data_preprocess/.*|
|
||||
fastvideo/dataset/.*|
|
||||
fastvideo/distill/.*|
|
||||
fastvideo/distill\.py|
|
||||
fastvideo/distill_adv\.py|
|
||||
fastvideo/models/.*|
|
||||
fastvideo/sample/.*|
|
||||
fastvideo/train\.py|
|
||||
fastvideo/utils/.*|
|
||||
fastvideo/v1/examples/.*|
|
||||
.github/workflows/fastvideo-publish.yml|
|
||||
.github/workflows/sta-publish.yml
|
||||
examples/.*|
|
||||
\.agents/.*|
|
||||
.github/workflows/publish-fastvideo.yml|
|
||||
.github/workflows/_template-build-image.yml
|
||||
)
|
||||
repos:
|
||||
- repo: https://github.com/google/yapf
|
||||
@@ -31,7 +23,7 @@ repos:
|
||||
args: [--in-place, --verbose]
|
||||
additional_dependencies: [toml] # TODO: Remove when yapf is upgraded
|
||||
- repo: https://github.com/astral-sh/ruff-pre-commit
|
||||
rev: v0.11.4
|
||||
rev: v0.11.12
|
||||
hooks:
|
||||
- id: ruff
|
||||
args: [--output-format, github, --fix]
|
||||
@@ -41,12 +33,12 @@ repos:
|
||||
- id: codespell
|
||||
additional_dependencies: ['tomli']
|
||||
args: ['--toml', 'pyproject.toml']
|
||||
- repo: https://github.com/PyCQA/isort
|
||||
rev: 6.0.1
|
||||
hooks:
|
||||
- id: isort
|
||||
# - repo: https://github.com/PyCQA/isort
|
||||
# rev: 6.0.1
|
||||
# hooks:
|
||||
# - id: isort
|
||||
- repo: https://github.com/jackdewinter/pymarkdown
|
||||
rev: v0.9.29
|
||||
rev: v0.9.30
|
||||
hooks:
|
||||
- id: pymarkdown
|
||||
args: [fix]
|
||||
@@ -58,8 +50,8 @@ repos:
|
||||
rev: v1.15.0
|
||||
hooks:
|
||||
- id: mypy
|
||||
args: [--python-version, '3.10', --follow-imports, "skip", ]
|
||||
additional_dependencies: [types-cachetools, types-setuptools, types-PyYAML, types-requests]
|
||||
args: [--python-version, '3.10', --follow-imports, "skip", "--disable-error-code", "union-attr", "--disable-error-code", "override" ]
|
||||
additional_dependencies: [types-aiofiles, types-cachetools, types-setuptools, types-PyYAML, types-requests]
|
||||
- repo: local
|
||||
hooks:
|
||||
- id: check-filenames
|
||||
@@ -67,7 +59,7 @@ repos:
|
||||
entry: bash
|
||||
args:
|
||||
- -c
|
||||
- 'git ls-files | grep -v "^fastvideo/v1/tests/ssim/" | grep " " && echo "Filenames should not contain spaces!" && exit 1 || exit 0'
|
||||
- 'git ls-files | grep -v "^\"*fastvideo/tests/ssim/" | grep -v "^\"*fastvideo/tests/inference/lora/L40S_reference_videos/" | grep " " && echo "Filenames should not contain spaces!" && exit 1 || exit 0'
|
||||
language: system
|
||||
always_run: true
|
||||
pass_filenames: false
|
||||
|
||||
@@ -0,0 +1 @@
|
||||
3.12
|
||||
@@ -0,0 +1,85 @@
|
||||
# Repository Guidelines
|
||||
|
||||
## Project Structure & Module Organization
|
||||
- Core Python package: `fastvideo/` (models, pipelines, training, distributed runtime, CLI entrypoints).
|
||||
- CUDA/custom kernels: `fastvideo-kernel/` (separate build/test flow).
|
||||
- Tests:
|
||||
- `fastvideo/tests/` for package-level tests (dataset, encoders, inference, training, SSIM, workflow).
|
||||
- `tests/local_tests/` for additional local/component checks.
|
||||
- Docs and guides: `docs/` (MkDocs source), with contributor docs in `docs/contributing/`.
|
||||
- Runnable examples and scripts: `examples/` and `scripts/`.
|
||||
- Static assets: `assets/` (including `assets/images/`, `assets/videos/`, and `assets/prompts/`) and `comfyui/assets/`.
|
||||
|
||||
## Build, Test, and Development Commands
|
||||
- `uv pip install -e ".[dev]"`: editable install with lint/test extras.
|
||||
- `pre-commit install --hook-type pre-commit --hook-type commit-msg`: enable local hooks.
|
||||
- `pre-commit run --all-files`: run formatter/lint/type/spelling checks.
|
||||
- `pytest tests/`: run top-level test suite.
|
||||
- `pytest fastvideo/tests/ -v`: run package tests.
|
||||
- `pytest fastvideo/tests/ssim/ -vs`: run SSIM regression tests (GPU-heavy).
|
||||
- `cd fastvideo-kernel && ./build.sh`: build kernel extensions.
|
||||
|
||||
## Coding Style & Naming Conventions
|
||||
- Python 3.10+; 4-space indentation; keep code and imports readable and explicit.
|
||||
- Style tools are configured in `pyproject.toml` and `.pre-commit-config.yaml`:
|
||||
- `yapf` (format), `ruff` (lint, auto-fix), `mypy` (typing), `codespell`.
|
||||
- Lint via `pre-commit run --files <changed paths>` (or `pre-commit run --all-files` for a full sweep) before committing. Do not shell out to `yapf`/`ruff`/`codespell`/`mypy` directly — pre-commit chains them with the project's config and respects the `.pre-commit-config.yaml` excludes (e.g. `fastvideo/tests/` is intentionally skipped). If pre-commit reports `(no files to check)` for your paths, that exclude is deliberate — don't bypass it.
|
||||
- Target line length is 120 (configured in `pyproject.toml` for ruff, yapf, and isort).
|
||||
- Naming: `snake_case` for functions/files, `PascalCase` for classes, `UPPER_SNAKE_CASE` for constants.
|
||||
|
||||
## Testing Guidelines
|
||||
- Use `pytest` and place tests near relevant domains (e.g., `fastvideo/tests/encoders/`).
|
||||
- Prefer descriptive names like `test_<feature>_<expected_behavior>.py`.
|
||||
- For new pipelines/backends, include at least one regression-oriented test; add SSIM coverage when output quality must be preserved.
|
||||
- Document GPU assumptions in tests that require specific hardware.
|
||||
|
||||
## Commit & Pull Request Guidelines
|
||||
- Follow existing commit style: short subject with optional tag prefix, e.g. `[bugfix]: ...`, `[feat]: ...`, `[misc]: ...`, and include PR reference like `(#1234)` when applicable.
|
||||
- Keep commits focused by concern (feature, refactor, fix).
|
||||
- PRs should include:
|
||||
- clear problem/solution summary,
|
||||
- test evidence (`pytest`/SSIM outputs or rationale if skipped),
|
||||
- linked issue/PR context,
|
||||
- screenshots or sample outputs for UI/demo/docs changes.
|
||||
|
||||
## Agent Infrastructure
|
||||
|
||||
This repository is agent-friendly. Before doing any work, read:
|
||||
|
||||
1. `.agents/onboarding/README.md` — full onboarding guide with step-by-step instructions.
|
||||
2. `.agents/memory/codebase-map/README.md` — structural index of the entire repository.
|
||||
3. `.agents/skills/` — available agent skills (check if one exists before writing code).
|
||||
4. `.agents/workflows/` — SOPs for common procedures (experiment lifecycle, evaluation, etc.).
|
||||
5. `.agents/lessons/` — known pitfalls and their documented fixes.
|
||||
|
||||
If you are exploring a new procedure that has no existing SOP, document your
|
||||
progress in `.agents/exploration/` and flag it for review at the end of your
|
||||
session.
|
||||
|
||||
## Per-Directory AGENTS.md
|
||||
|
||||
Local guidance lives next to the code. Read the in-scope file before editing:
|
||||
|
||||
| Directory | What it covers |
|
||||
|-----------|----------------|
|
||||
| `fastvideo/AGENTS.md` | Core package map, public API, registry-driven model dispatch |
|
||||
| `fastvideo/configs/AGENTS.md` | Arch + pipeline config dataclasses, `param_names_mapping` |
|
||||
| `fastvideo/models/AGENTS.md` | DiT / VAE / encoder / scheduler / loader layout (pre-commit excluded) |
|
||||
| `fastvideo/layers/AGENTS.md` | Tensor-parallel linear/attention layer rules for ports |
|
||||
| `fastvideo/attention/AGENTS.md` | Backend registry + env-var override |
|
||||
| `fastvideo/pipelines/AGENTS.md` | Stage ABC, `basic/<model>/`, `preprocess/`, presets |
|
||||
| `fastvideo/training/AGENTS.md` | Legacy monolithic pipelines (frozen for existing models) |
|
||||
| `fastvideo/train/AGENTS.md` | New modular trainer (methods × models × callbacks, YAML) |
|
||||
| `fastvideo/tests/AGENTS.md` | Test taxonomy, conftest, pre-commit-excluded path |
|
||||
| `fastvideo/tests/ssim/AGENTS.md` | GPU SSIM regression authoring + reference video sync |
|
||||
| `scripts/checkpoint_conversion/AGENTS.md` | Adding a converter for a new HF/official checkpoint |
|
||||
|
||||
## Critical: Two Training Stacks Coexist
|
||||
|
||||
- `fastvideo/training/` — legacy, monolithic per-model `*_training_pipeline.py` and
|
||||
`*_distillation_pipeline.py`. Still authoritative for shipped models.
|
||||
- `fastvideo/train/` — new modular framework (composable methods × models × callbacks
|
||||
driven by YAML). Preferred for new training work.
|
||||
|
||||
Pick the matching stack before editing. Do not migrate a pipeline between them
|
||||
without an explicit ask — the conventions and config surfaces differ.
|
||||
@@ -1,48 +0,0 @@
|
||||
FROM nvidia/cuda:12.4.1-devel-ubuntu20.04
|
||||
|
||||
ENV DEBIAN_FRONTEND=noninteractive
|
||||
|
||||
WORKDIR /FastVideo
|
||||
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
wget \
|
||||
git \
|
||||
ca-certificates \
|
||||
openssh-server \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
RUN wget https://repo.anaconda.com/miniconda/Miniconda3-latest-Linux-x86_64.sh && \
|
||||
bash Miniconda3-latest-Linux-x86_64.sh -b -p /opt/conda && \
|
||||
rm Miniconda3-latest-Linux-x86_64.sh
|
||||
|
||||
ENV PATH=/opt/conda/bin:$PATH
|
||||
|
||||
RUN conda create --name fastvideo-dev python=3.10.0 -y
|
||||
|
||||
SHELL ["/bin/bash", "-c"]
|
||||
|
||||
# Copy just the pyproject.toml first to leverage Docker cache
|
||||
COPY pyproject.toml ./
|
||||
|
||||
# Create a dummy README to satisfy the installation
|
||||
RUN echo "# Placeholder" > README.md
|
||||
|
||||
RUN conda run -n fastvideo-dev pip install --no-cache-dir --upgrade pip && \
|
||||
conda run -n fastvideo-dev pip install --no-cache-dir .[dev] && \
|
||||
conda run -n fastvideo-dev pip install --no-cache-dir flash-attn==2.7.0.post2 --no-build-isolation && \
|
||||
conda clean -afy
|
||||
|
||||
COPY . .
|
||||
|
||||
RUN conda run -n fastvideo-dev pip install --no-cache-dir -e .[dev]
|
||||
|
||||
# Remove authentication headers
|
||||
RUN git config --unset-all http.https://github.com/.extraheader || true
|
||||
|
||||
# Set up automatic conda environment activation for all shells
|
||||
RUN echo 'source /opt/conda/etc/profile.d/conda.sh' >> /root/.bashrc && \
|
||||
echo 'conda activate fastvideo-dev' >> /root/.bashrc && \
|
||||
# Ensure .bashrc is sourced for SSH login shells
|
||||
echo 'if [ -f ~/.bashrc ]; then . ~/.bashrc; fi' > /root/.profile
|
||||
|
||||
EXPOSE 22
|
||||
@@ -1,102 +1,157 @@
|
||||
<div align="center">
|
||||
<img src=assets/logo.jpg width="30%"/>
|
||||
<img src=assets/logos/logo.svg width="30%"/>
|
||||
</div>
|
||||
|
||||
FastVideo is a lightweight framework for accelerating large video diffusion models.
|
||||
|
||||
<p align="center">
|
||||
| <a href="https://hao-ai-lab.github.io/FastVideo"><b>Documentation</b></a> | 🤗 <a href="https://huggingface.co/FastVideo/FastHunyuan" target="_blank"><b>FastHunyuan</b></a> | 🤗 <a href="https://huggingface.co/FastVideo/FastMochi-diffusers" target="_blank"><b>FastMochi</b></a> | 🟣💬 <a href="https://join.slack.com/t/fastvideo/shared_invite/zt-2zf6ru791-sRwI9lPIUJQq1mIeB_yjJg" target="_blank"> <b>Slack</b> </a> |
|
||||
| <a href="https://hao-ai-lab.github.io/FastVideo"><b>Documentation</b></a> | <a href="https://hao-ai-lab.github.io/FastVideo/inference/inference_quick_start/"><b> Quick Start</b></a> | <a href="https://github.com/hao-ai-lab/FastVideo/discussions/982" target="_blank"><b>Weekly Dev Meeting</b></a> | 🟣💬 <a href="https://join.slack.com/t/fastvideo/shared_invite/zt-3f4lao1uq-u~Ipx6Lt4J27AlD2y~IdLQ" target="_blank"> <b>Slack</b> </a> | 🟣💬 <a href="https://github.com/hao-ai-lab/FastVideo/discussions/1097" target="_blank"> <b> WeChat </b> </a> |
|
||||
</p>
|
||||
|
||||
https://github.com/user-attachments/assets/79af5fb8-707c-4263-b153-9ab2a01d3ac1
|
||||
**FastVideo is a unified post-training and real-time inference framework for accelerated video generation.**
|
||||
|
||||
FastVideo currently offers: (with more to come)
|
||||
## NEWS
|
||||
- `2026/03/17`: Release Live demo: [Into the Dreamverse: Vibe Directing in FastVideo](https://dreamverse.fastvideo.org/), check out the [Blog](https://haoailab.com/blogs/dreamverse/).
|
||||
- `2026/03/13`: Release Live demo: [Create a 5s 1080p Video in 4.5s with FastVideo on a Single GPU](https://1080p.fastvideo.org/), check out the [Blog](https://haoailab.com/blogs/fastvideo_realtime_1080p/).
|
||||
- `2025/11/19`: Release [CausalWan2.2 I2V A14B Preview](https://huggingface.co/FastVideo/CausalWan2.2-I2V-A14B-Preview-Diffusers) models, [Blog](https://hao-ai-lab.github.io/blogs/fastvideo_causalwan_preview/) and [Inference Code!](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_self_forcing_causal_wan2_2_i2v.py).
|
||||
- `2025/08/04`: Release [FastWan](https://hao-ai-lab.github.io/FastVideo/distillation/dmd) models and [Sparse-Distillation](https://hao-ai-lab.github.io/blogs/fastvideo_post_training/).
|
||||
|
||||
- [NEW!] V1 inference API available. Full announcement coming soon!
|
||||
- [Sliding Tile Attention](https://hao-ai-lab.github.io/blogs/sta/).
|
||||
- FastHunyuan and FastMochi: consistency distilled video diffusion models for 8x inference speedup.
|
||||
- First open distillation recipes for video DiT, based on [PCM](https://github.com/G-U-N/Phased-Consistency-Model).
|
||||
- Support distilling/finetuning/inferencing state-of-the-art open video DiTs: 1. Mochi 2. Hunyuan.
|
||||
- Scalable training with FSDP, sequence parallelism, and selective activation checkpointing, with near linear scaling to 64 GPUs.
|
||||
- Memory efficient finetuning with LoRA, precomputed latent, and precomputed text embeddings.
|
||||
### More News
|
||||
|
||||
Dev in progress and highly experimental.
|
||||
- `2025/06/14`: Release finetuning and inference code for [VSA](https://arxiv.org/pdf/2505.13389).
|
||||
- `2025/04/24`: [FastVideo V1](https://hao-ai-lab.github.io/blogs/fastvideo/) is released!
|
||||
- `2025/02/18`: Release the inference code for [Sliding Tile Attention](https://hao-ai-lab.github.io/blogs/sta/).
|
||||
|
||||
## Change Log
|
||||
- ```2025/02/20```: FastVideo now supports STA on [StepVideo](https://github.com/stepfun-ai/Step-Video-T2V) with 3.4X speedup!
|
||||
- ```2025/02/18```: Release the inference code and kernel for [Sliding Tile Attention](https://hao-ai-lab.github.io/blogs/sta/).
|
||||
- ```2025/01/13```: Support Lora finetuning for HunyuanVideo.
|
||||
- ```2024/12/25```: Enable single 4090 inference for `FastHunyuan`, please rerun the installation steps to update the environment.
|
||||
- ```2024/12/17```: `FastVideo` v0.0.1 is released.
|
||||
## Key Features
|
||||
|
||||
FastVideo has the following features:
|
||||
|
||||
- End-to-end post-training support for bidirectional and autoregressive models:
|
||||
- Support full finetuning and LoRA finetuning for state-of-the-art open video DiTs
|
||||
- Data preprocessing pipeline for video, image, and text data
|
||||
- Distribution Matching Distillation (DMD2) stepwise distillation.
|
||||
- Sparse attention with [Video Sparse Attention](https://arxiv.org/pdf/2505.13389)
|
||||
- [Sparse distillation](https://hao-ai-lab.github.io/blogs/fastvideo_post_training/) to achieve >50x denoising speedup
|
||||
- Scalable training with FSDP2, sequence parallelism, and selective activation checkpointing.
|
||||
- Causal distillation through Self-Forcing
|
||||
- See this [page](https://hao-ai-lab.github.io/FastVideo/training/overview/) for full list of supported models and recipes.
|
||||
- State-of-the-art performance optimizations for inference
|
||||
- Sequence Parallelism for distributed inference
|
||||
- Multiple state-of-the-art attention backends
|
||||
- User-friendly CLI and Python API
|
||||
- See this [page](https://hao-ai-lab.github.io/FastVideo/inference/optimizations/) for full list of supported optimizations.
|
||||
- Diverse hardware and OS support
|
||||
- Support H100, A100, 4090
|
||||
- Support Linux, Windows, MacOS
|
||||
- See this [page](https://hao-ai-lab.github.io/FastVideo/inference/support_matrix/) for full list of supported models, hardware assumptions, and optimization compatibility.
|
||||
|
||||
## Getting Started
|
||||
|
||||
- [Install FastVideo](https://hao-ai-lab.github.io/FastVideo/getting_started/installation.html)
|
||||
- [Design Overview](https://hao-ai-lab.github.io/FastVideo/design/overview.html)
|
||||
- [Contribution Guide](https://hao-ai-lab.github.io/FastVideo/getting_started/installation.html)
|
||||
We recommend using [uv](https://docs.astral.sh/uv/) to create a clean environment. If you previously used Conda, switching to uv generally gives faster and more stable installs.
|
||||
|
||||
### Inference
|
||||
- [Quick Start](https://hao-ai-lab.github.io/FastVideo/inference/examples/basic.html)
|
||||
- V1 Inference API Guide (Coming soon!)
|
||||
```bash
|
||||
# Create and activate a new uv environment
|
||||
uv venv --python 3.12 --seed
|
||||
source .venv/bin/activate
|
||||
|
||||
### Distillation and Finetuning
|
||||
- [Distillation Guide](https://hao-ai-lab.github.io/FastVideo/training/distillation.html)
|
||||
- [Finetuning Guide](https://hao-ai-lab.github.io/FastVideo/training/finetuning.html)
|
||||
# Install FastVideo
|
||||
uv pip install fastvideo
|
||||
```
|
||||
|
||||
### Deprecated APIs
|
||||
- [V0 Inference (Deprecated)](https://hao-ai-lab.github.io/FastVideo/inference/v0_inference.html)
|
||||
Please see our [docs](https://hao-ai-lab.github.io/FastVideo/getting_started/installation/) for more detailed installation instructions.
|
||||
|
||||
## 📑 Development Plan
|
||||
## Sparse Distillation
|
||||
|
||||
<!-- - More distillation methods -->
|
||||
<!-- - [ ] Add Distribution Matching Distillation -->
|
||||
- More models support
|
||||
<!-- - [ ] Add CogvideoX model -->
|
||||
- [ ] Add StepVideo to V1
|
||||
- Optimization features
|
||||
- [ ] Teacache in V1
|
||||
- [ ] SageAttention in V1
|
||||
- Code updates
|
||||
- [ ] V1 Configuration API
|
||||
- [ ] Support Training in V1
|
||||
<!-- - [ ] fp8 support -->
|
||||
<!-- - [ ] faster load model and save model support -->
|
||||
For our sparse distillation techniques, please see our [distillation docs](https://hao-ai-lab.github.io/FastVideo/distillation/dmd/) and check out our [blog](https://hao-ai-lab.github.io/blogs/fastvideo_post_training/).
|
||||
|
||||
See below for recipes and datasets:
|
||||
|
||||
| Model | Sparse Distillation | Dataset |
|
||||
| ------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------- |
|
||||
| [FastWan2.1-T2V-1.3B](https://huggingface.co/FastVideo/FastWan2.1-T2V-1.3B-Diffusers) | [Recipe](https://github.com/hao-ai-lab/FastVideo/tree/main/examples/distill/Wan2.1-T2V/Wan-Syn-Data-480P) | [FastVideo Synthetic Wan2.1 480P](https://huggingface.co/datasets/FastVideo/Wan-Syn_77x448x832_600k) |
|
||||
| [FastWan2.2-TI2V-5B](https://huggingface.co/FastVideo/FastWan2.2-TI2V-5B-Diffusers) | [Recipe](https://github.com/hao-ai-lab/FastVideo/tree/main/examples/distill/Wan2.2-TI2V-5B-Diffusers/Data-free) | [FastVideo Synthetic Wan2.2 720P](https://huggingface.co/datasets/FastVideo/Wan2.2-Syn-121x704x1280_32k) |
|
||||
|
||||
## Inference
|
||||
|
||||
### Generating Your First Video
|
||||
|
||||
Here's a minimal example to generate a video using the default settings. Make sure VSA kernels are [installed](https://hao-ai-lab.github.io/FastVideo/attention/vsa/#installation). Create a file called `example.py` with the following code:
|
||||
|
||||
```python
|
||||
import os
|
||||
from fastvideo import VideoGenerator
|
||||
|
||||
def main():
|
||||
os.environ["FASTVIDEO_ATTENTION_BACKEND"] = "VIDEO_SPARSE_ATTN"
|
||||
|
||||
# Create a video generator with a pre-trained model
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
"FastVideo/FastWan2.1-T2V-1.3B-Diffusers",
|
||||
num_gpus=1, # Adjust based on your hardware
|
||||
)
|
||||
|
||||
# Define a prompt for your video
|
||||
prompt = "A curious raccoon peers through a vibrant field of yellow sunflowers, its eyes wide with interest."
|
||||
|
||||
# Generate the video
|
||||
video = generator.generate_video(
|
||||
prompt,
|
||||
output_path="my_videos/", # Controls where videos are saved
|
||||
save_video=True
|
||||
)
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
```
|
||||
|
||||
Run the script with:
|
||||
|
||||
```bash
|
||||
python example.py
|
||||
```
|
||||
|
||||
For a more detailed guide, please see our [inference quick start](https://hao-ai-lab.github.io/FastVideo/inference/inference_quick_start/).
|
||||
|
||||
## More Guides
|
||||
|
||||
- [Design Overview](https://hao-ai-lab.github.io/FastVideo/design/overview/)
|
||||
- [Distillation Guide](https://hao-ai-lab.github.io/FastVideo/distillation/dmd/)
|
||||
- [Contribution Guide](https://hao-ai-lab.github.io/FastVideo/contributing/overview/)
|
||||
|
||||
## Awesome work using FastVideo or our research projects
|
||||
|
||||
- [SGLang](https://github.com/sgl-project/sglang/tree/main/python/sglang/multimodal_gen): SGLang's diffusion inference functionality is based on a fork of FastVideo on Sept. 24, 2025.
|
||||
- [DanceGRPO](https://github.com/XueZeyue/DanceGRPO): A unified framework to adapt Group Relative Policy Optimization (GRPO) to visual generation paradigms. Code based on FastVideo.
|
||||
- [SRPO](https://github.com/Tencent-Hunyuan/SRPO): A method to directly align the full diffusion trajectory with fine-grained human preference. Code based on FastVideo.
|
||||
- [DCM](https://github.com/Vchitect/DCM): Dual-expert consistency model for efficient and high-quality video generation. Code based on FastVideo.
|
||||
- [HY-WorldPlay](https://github.com/Tencent-Hunyuan/HY-WorldPlay): An action-conditioned world model model trained using FastVideo framework.
|
||||
- [Hunyuan Video 1.5](https://github.com/Tencent-Hunyuan/HunyuanVideo-1.5): A leading lightweight video generation model, where they proposed SSTA based on Sliding Tile Attention.
|
||||
- [Kandinsky-5.0](https://github.com/kandinskylab/kandinsky-5): A family of diffusion models for video & image generation, where their NABLA attention includes a Sliding Tile Attention branch.
|
||||
- [LongCat Video](https://github.com/meituan-longcat/LongCat-Video): A foundational video generation model with 13.6B parameters with block-sparse attention similar to Video Sparse Attention.
|
||||
|
||||
## 🤝 Contributing
|
||||
|
||||
We welcome all contributions. Please check out our guide [here](https://hao-ai-lab.github.io/FastVideo/developer_guide/overview.html)
|
||||
We welcome all contributions. Please check out our guide [here](https://hao-ai-lab.github.io/FastVideo/contributing/overview/).
|
||||
See details in [development roadmap](https://github.com/hao-ai-lab/FastVideo/issues/899).
|
||||
|
||||
## Acknowledgement
|
||||
We learned and reused code from the following projects:
|
||||
- [PCM](https://github.com/G-U-N/Phased-Consistency-Model)
|
||||
- [diffusers](https://github.com/huggingface/diffusers)
|
||||
- [OpenSoraPlan](https://github.com/PKU-YuanGroup/Open-Sora-Plan)
|
||||
- [xDiT](https://github.com/xdit-project/xDiT)
|
||||
- [vLLM](https://github.com/vllm-project/vllm)
|
||||
- [SGLang](https://github.com/sgl-project/sglang)
|
||||
|
||||
We thank MBZUAI and [Anyscale](https://www.anyscale.com/) for their support throughout this project.
|
||||
We learned the design and reused code from the following projects: [Wan-Video](https://github.com/Wan-Video), [ThunderKittens](https://github.com/HazyResearch/ThunderKittens), [DMD2](https://github.com/tianweiy/DMD2), [diffusers](https://github.com/huggingface/diffusers), [xDiT](https://github.com/xdit-project/xDiT), [vLLM](https://github.com/vllm-project/vllm), [SGLang](https://github.com/sgl-project/sglang). We thank [MBZUAI](https://ifm.mbzuai.ac.ae/), [Anyscale](https://www.anyscale.com/), and [GMI Cloud](https://www.gmicloud.ai/) for their support throughout this project.
|
||||
|
||||
## Citation
|
||||
If you use FastVideo for your research, please cite our paper:
|
||||
|
||||
If you find FastVideo useful, please consider citing our research work:
|
||||
|
||||
```bibtex
|
||||
@misc{zhang2025fastvideogenerationsliding,
|
||||
title={Fast Video Generation with Sliding Tile Attention},
|
||||
author={Peiyuan Zhang and Yongqi Chen and Runlong Su and Hangliang Ding and Ion Stoica and Zhenghong Liu and Hao Zhang},
|
||||
year={2025},
|
||||
eprint={2502.04507},
|
||||
archivePrefix={arXiv},
|
||||
primaryClass={cs.CV},
|
||||
url={https://arxiv.org/abs/2502.04507},
|
||||
@article{zhang2025vsa,
|
||||
title={Vsa: Faster video diffusion with trainable sparse attention},
|
||||
author={Zhang, Peiyuan and Chen, Yongqi and Huang, Haofeng and Lin, Will and Liu, Zhengzhong and Stoica, Ion and Xing, Eric and Zhang, Hao},
|
||||
journal={arXiv preprint arXiv:2505.13389},
|
||||
year={2025}
|
||||
}
|
||||
@misc{ding2025efficientvditefficientvideodiffusion,
|
||||
title={Efficient-vDiT: Efficient Video Diffusion Transformers With Attention Tile},
|
||||
author={Hangliang Ding and Dacheng Li and Runlong Su and Peiyuan Zhang and Zhijie Deng and Ion Stoica and Hao Zhang},
|
||||
year={2025},
|
||||
eprint={2502.06155},
|
||||
archivePrefix={arXiv},
|
||||
primaryClass={cs.CV},
|
||||
url={https://arxiv.org/abs/2502.06155},
|
||||
|
||||
@article{zhang2025fast,
|
||||
title={Fast video generation with sliding tile attention},
|
||||
author={Zhang, Peiyuan and Chen, Yongqi and Su, Runlong and Ding, Hangliang and Stoica, Ion and Liu, Zhengzhong and Zhang, Hao},
|
||||
journal={arXiv preprint arXiv:2502.04507},
|
||||
year={2025}
|
||||
}
|
||||
```
|
||||
|
||||
@@ -0,0 +1,10 @@
|
||||
try:
|
||||
from .comfyui.video_generator.nodes import (NODE_CLASS_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS)
|
||||
WEB_DIRECTORY = "./web"
|
||||
__all__ = ['NODE_CLASS_MAPPINGS', 'NODE_DISPLAY_NAME_MAPPINGS', 'WEB_DIRECTORY']
|
||||
except ImportError:
|
||||
# ComfyUI environment not available, skip comfyui imports
|
||||
NODE_CLASS_MAPPINGS = {}
|
||||
NODE_DISPLAY_NAME_MAPPINGS = {}
|
||||
WEB_DIRECTORY = "./web"
|
||||
__all__ = ['NODE_CLASS_MAPPINGS', 'NODE_DISPLAY_NAME_MAPPINGS', 'WEB_DIRECTORY']
|
||||
|
After Width: | Height: | Size: 194 KiB |
@@ -0,0 +1,18 @@
|
||||
<svg width="252" height="105" viewBox="0 0 252 105" fill="none" xmlns="http://www.w3.org/2000/svg">
|
||||
<path d="M89.4843 55.5457H101.361L87.7028 101H74.638L89.4843 55.5457Z" fill="#356CFF"/>
|
||||
<path fill-rule="evenodd" clip-rule="evenodd" d="M96.0167 1.00057H112.645L118.583 48.273H104.924L103.737 39.7882H79.9827L67.5117 55.5457H85.3273L43.1638 101H28.3174L22.3789 55.5457H33.6621L38.4129 91.3031L58.604 68.2729H44.3515L96.0167 1.00057ZM100.768 13.1217L87.7028 29.4852H103.143L100.768 13.1217Z" fill="#356CFF"/>
|
||||
<path d="M37.2252 1.00057L22.3789 48.273H36.0375L40.7884 30.6974L62.6727 30.6974L69.6727 21.0005L43.7576 21.0004L46.7269 11.9096L77.6727 11.9096L86 1.00057L37.2252 1.00057Z" fill="#356CFF"/>
|
||||
<path fill-rule="evenodd" clip-rule="evenodd" d="M108.488 55.5457L94.2351 101C94.2351 101 105.518 101 120.959 101C136.399 101 144.078 93.0133 148.276 79.788C152.432 68.0157 153.027 55.5457 136.399 55.5457C119.771 55.5457 108.488 55.5457 108.488 55.5457ZM109.081 90.697L116.802 65.8487C116.802 65.8487 120.959 65.8487 132.242 65.8487C143.525 65.8487 137.586 78.5759 135.211 84.0304C133.307 88.4021 127.491 90.697 122.74 90.697C117.989 90.697 109.081 90.697 109.081 90.697Z" fill="#356CFF"/>
|
||||
<path d="M173.188 1.00056L168.625 11.9096C168.625 11.9096 149.386 11.9092 142.525 11.9095C135.664 11.9098 136.586 20.3944 141.337 20.3944H159.747C168.654 20.3944 166.961 33.6899 163.904 38.5761C160.467 44.0675 157.371 48.273 148.463 48.273L125.188 48.273L124 37.97L147.87 37.97C153.808 37.97 156.184 29.4852 151.433 29.4852H131.836C120.142 29.4852 125.897 1.00043 141.337 1.00043L173.188 1.00056Z" fill="#356CFF"/>
|
||||
<path d="M179.938 1.00056L175.688 11.9096L191.221 11.9096L179.938 48.273H192.409L203.692 11.9096L219.132 11.9095L223.289 1.00043L179.938 1.00056Z" fill="#356CFF"/>
|
||||
<path d="M161.341 55.5457H202.845L198.5 65.8487H169.654L167.279 73.7268H188.5L184.749 82.8177H164.31L161.934 90.697H190.251L186.624 101H146.494L161.341 55.5457Z" fill="#356CFF"/>
|
||||
<path fill-rule="evenodd" clip-rule="evenodd" d="M230.821 54.9391C255.169 54.9391 251.776 67.0602 249.231 77.9692C246.686 88.8783 240.917 101 217.757 101C194.596 101 195.606 88.8783 199.347 77.9692C203.089 67.0602 206.473 54.9391 230.821 54.9391ZM237.948 77.9692C239.984 70.6965 240.917 65.242 228.446 65.242C215.975 65.242 211.818 71.9087 210.037 77.9692C208.255 84.0298 208.255 91.3025 219.538 91.3025C230.821 91.3025 235.911 85.2419 237.948 77.9692Z" fill="#356CFF"/>
|
||||
<path d="M173.188 1.00056L168.625 11.9096C168.625 11.9096 149.386 11.9092 142.525 11.9095C135.664 11.9098 136.586 20.3944 141.337 20.3944M173.188 1.00056C173.188 1.00056 156.777 1.00043 141.337 1.00043M173.188 1.00056L141.337 1.00043M141.337 20.3944C146.088 20.3944 150.839 20.3944 159.747 20.3944M141.337 20.3944H159.747M159.747 20.3944C168.654 20.3944 166.961 33.6899 163.904 38.5761C160.467 44.0675 157.371 48.273 148.463 48.273M148.463 48.273C139.556 48.273 125.188 48.273 125.188 48.273M148.463 48.273L125.188 48.273M125.188 48.273L124 37.97M124 37.97C124 37.97 141.931 37.97 147.87 37.97M124 37.97L147.87 37.97M147.87 37.97C153.808 37.97 156.184 29.4852 151.433 29.4852M151.433 29.4852C146.682 29.4852 138.962 29.4852 131.836 29.4852M151.433 29.4852H131.836M131.836 29.4852C120.142 29.4852 125.897 1.00043 141.337 1.00043M37.2252 1.00057L22.3789 48.273H36.0375L40.7884 30.6974L62.6727 30.6974L69.6727 21.0005L43.7576 21.0004L46.7269 11.9096L77.6727 11.9096L86 1.00057L37.2252 1.00057ZM96.0167 1.00057H112.645L118.583 48.273H104.924L103.737 39.7882H79.9827L67.5117 55.5457H85.3273L43.1638 101H28.3174L22.3789 55.5457H33.6621L38.4129 91.3031L58.604 68.2729H44.3515L96.0167 1.00057ZM87.7028 29.4852L100.768 13.1217L103.143 29.4852H87.7028ZM89.4843 55.5457H101.361L87.7028 101H74.638L89.4843 55.5457ZM108.488 55.5457L94.2351 101C94.2351 101 105.518 101 120.959 101C136.399 101 144.078 93.0133 148.276 79.788C152.432 68.0157 153.027 55.5457 136.399 55.5457C119.771 55.5457 108.488 55.5457 108.488 55.5457ZM116.802 65.8487L109.081 90.697C109.081 90.697 117.989 90.697 122.74 90.697C127.491 90.697 133.307 88.4021 135.211 84.0304C137.586 78.5759 143.525 65.8487 132.242 65.8487C120.959 65.8487 116.802 65.8487 116.802 65.8487ZM179.938 1.00056L175.688 11.9096L191.221 11.9096L179.938 48.273H192.409L203.692 11.9096L219.132 11.9095L223.289 1.00043L179.938 1.00056ZM161.341 55.5457H202.845L198.5 65.8487H169.654L167.279 73.7268H188.5L184.749 82.8177H164.31L161.934 90.697H190.251L186.624 101H146.494L161.341 55.5457ZM230.821 54.9391C255.169 54.9391 251.776 67.0602 249.231 77.9692C246.686 88.8783 240.917 101 217.757 101C194.596 101 195.606 88.8783 199.347 77.9692C203.089 67.0602 206.473 54.9391 230.821 54.9391ZM228.446 65.242C240.917 65.242 239.984 70.6965 237.948 77.9692C235.911 85.2419 230.821 91.3025 219.538 91.3025C208.255 91.3025 208.255 84.0298 210.037 77.9692C211.818 71.9087 215.975 65.242 228.446 65.242Z" stroke="#356CFF" stroke-width="1.18771"/>
|
||||
<path d="M15.2524 55.5451L21.191 100.999L24.7541 100.999L18.8156 55.5451L15.2524 55.5451Z" fill="#356CFF" stroke="#356CFF" stroke-width="1.18771"/>
|
||||
<path d="M8.12646 55.5451L14.065 100.999L15.2527 100.999L9.31417 55.5451L8.12646 55.5451Z" fill="#356CFF" stroke="#356CFF" stroke-width="1.18771"/>
|
||||
<path d="M1 55.5451L6.93853 100.999L7.53239 100.999L1.59385 55.5451L1 55.5451Z" fill="#356CFF" stroke="#356CFF" stroke-width="0.593853"/>
|
||||
<path d="M15.2524 48.2724L30.0988 1H33.6619L18.8156 48.2724H15.2524Z" fill="#356CFF" stroke="#356CFF" stroke-width="1.18771"/>
|
||||
<path d="M8.12646 48.2724L22.9728 1H24.1605L9.31417 48.2724H8.12646Z" fill="#356CFF" stroke="#356CFF" stroke-width="1.18771"/>
|
||||
<path d="M1 48.2724L15.8463 1H16.4402L1.59385 48.2724H1Z" fill="#356CFF" stroke="#356CFF" stroke-width="0.593853"/>
|
||||
<path d="M85.3271 55.5457H67.5116L87 12.7363L44.3513 68.2729H58.6038L43.1636 101L85.3271 55.5457Z" fill="#FDC717" stroke="#FDC717" stroke-width="1.18771" stroke-miterlimit="16"/>
|
||||
</svg>
|
||||
|
After Width: | Height: | Size: 5.7 KiB |
|
After Width: | Height: | Size: 490 KiB |
@@ -0,0 +1,6 @@
|
||||
<svg width="160" height="93" viewBox="0 0 160 93" fill="none" xmlns="http://www.w3.org/2000/svg">
|
||||
<path d="M28.8511 91.66L57.6319 1.86368H64.5394L35.7585 91.66H28.8511Z" fill="#356CFF" stroke="#356CFF" stroke-width="2.30244"/>
|
||||
<path d="M15.0376 91.66L43.8185 1.86368H46.1209L17.3401 91.66H15.0376Z" fill="#356CFF" stroke="#356CFF" stroke-width="2.30244"/>
|
||||
<path d="M1.22217 91.66L30.003 1.86366H31.1543L2.3734 91.66H1.22217Z" fill="#356CFF" stroke="#356CFF" stroke-width="1.15122"/>
|
||||
<path d="M71.4465 1.86483L42.666 91.6599H69.144L78.3538 58.2746H123.251L129.007 39.855H84.1099L89.866 22.5868H152.032L157.788 1.86483H71.4465Z" fill="#356CFF" stroke="#356CFF" stroke-width="2.30244"/>
|
||||
</svg>
|
||||
|
After Width: | Height: | Size: 691 B |
|
After Width: | Height: | Size: 113 KiB |
|
After Width: | Height: | Size: 1.2 MiB |
|
After Width: | Height: | Size: 229 KiB |
|
After Width: | Height: | Size: 168 KiB |
|
After Width: | Height: | Size: 103 KiB |
|
After Width: | Height: | Size: 148 KiB |
|
After Width: | Height: | Size: 155 KiB |
|
After Width: | Height: | Size: 723 KiB |
|
After Width: | Height: | Size: 723 KiB |
|
After Width: | Height: | Size: 875 KiB |
|
After Width: | Height: | Size: 664 KiB |
|
After Width: | Height: | Size: 62 KiB |
|
After Width: | Height: | Size: 686 KiB |
|
After Width: | Height: | Size: 957 KiB |
|
After Width: | Height: | Size: 585 KiB |
|
After Width: | Height: | Size: 558 KiB |
|
After Width: | Height: | Size: 942 KiB |
|
After Width: | Height: | Size: 890 KiB |
|
After Width: | Height: | Size: 433 KiB |
|
After Width: | Height: | Size: 595 KiB |
|
After Width: | Height: | Size: 781 KiB |
|
After Width: | Height: | Size: 783 KiB |
|
After Width: | Height: | Size: 762 KiB |
|
After Width: | Height: | Size: 68 KiB |
|
After Width: | Height: | Size: 147 KiB |
|
After Width: | Height: | Size: 89 KiB |
|
After Width: | Height: | Size: 133 KiB |
|
After Width: | Height: | Size: 213 KiB |