Compare commits
830
Commits
will/test_pr
...
main
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
c520878fd2 | ||
|
|
99f8f3e7a7 | ||
|
|
14e4b7189b | ||
|
|
8530fdc550 | ||
|
|
315b25d26d | ||
|
|
b8920ba50a | ||
|
|
3467f8befb | ||
|
|
7320d9d350 | ||
|
|
af78c759f1 | ||
|
|
f82c190cbb | ||
|
|
e722023bd6 | ||
|
|
bc0191d775 | ||
|
|
2d6e7d9b6b | ||
|
|
5e61accc37 | ||
|
|
f60514183c | ||
|
|
156611bab3 | ||
|
|
6d6195a9f1 | ||
|
|
8348e83e80 | ||
|
|
2452022f15 | ||
|
|
4b3f99b224 | ||
|
|
d967b99928 | ||
|
|
8f1eb2992c | ||
|
|
683689de4f | ||
|
|
d5287f8352 | ||
|
|
cdfbd64b04 | ||
|
|
299d0f6838 | ||
|
|
c18436125a | ||
|
|
5d515ee617 | ||
|
|
37b9dc1836 | ||
|
|
a0291a57c1 | ||
|
|
6102fac00d | ||
|
|
1ad33c1832 | ||
|
|
da87973412 | ||
|
|
a3fd1d0bab | ||
|
|
3e87d6f4ae | ||
|
|
4bb5627fa8 | ||
|
|
08ab8b5c00 | ||
|
|
fb5efebdab | ||
|
|
6e4813f8b4 | ||
|
|
13ef81e6da | ||
|
|
13b3b36b8b | ||
|
|
72795d0af8 | ||
|
|
af63a92f6a | ||
|
|
a4e7322ba6 | ||
|
|
d6e13f9cc8 | ||
|
|
6c4c18690e | ||
|
|
a8688dddd9 | ||
|
|
cacc8cfcb3 | ||
|
|
ef62ac48d1 | ||
|
|
00cb7d0ba2 | ||
|
|
69a6215e7f | ||
|
|
abe98e87f5 | ||
|
|
274b922d39 | ||
|
|
38966056d0 | ||
|
|
d8ca702dfd | ||
|
|
33d81730e9 | ||
|
|
02a027c49f | ||
|
|
e235b6a332 | ||
|
|
8b51380466 | ||
|
|
527a3165b9 | ||
|
|
f06d824054 | ||
|
|
6ded84ee0f | ||
|
|
a4d6416c52 | ||
|
|
0cc41a22dc | ||
|
|
8444c0897a | ||
|
|
9491c8638a | ||
|
|
6809a751fb | ||
|
|
8322b01815 | ||
|
|
9edc8adf5f | ||
|
|
e1f3904799 | ||
|
|
02f1ce11ae | ||
|
|
7f03e03dc6 | ||
|
|
e3b88bb12a | ||
|
|
cb66acd400 | ||
|
|
442e2d2e18 | ||
|
|
dd35763ad6 | ||
|
|
e90be598e5 | ||
|
|
ba5e81083c | ||
|
|
76ce9c7fd6 | ||
|
|
08d99c089e | ||
|
|
20751a21aa | ||
|
|
9dd2a837f4 | ||
|
|
93aab45ac2 | ||
|
|
017ce6602d | ||
|
|
a575055eec | ||
|
|
81f3fec7fd | ||
|
|
d265a454bf | ||
|
|
c100c66578 | ||
|
|
8760eb7a06 | ||
|
|
361f919c88 | ||
|
|
8b5377aab2 | ||
|
|
d995516da0 | ||
|
|
f47ad3f5b7 | ||
|
|
10bdf5e076 | ||
|
|
3e26db40b0 | ||
|
|
c73dd0ab55 | ||
|
|
430e52154e | ||
|
|
384eee8aef | ||
|
|
c4824c7764 | ||
|
|
9b0e57fe4b | ||
|
|
0100218594 | ||
|
|
39718cd54d | ||
|
|
37d06a832f | ||
|
|
8839ba8d4d | ||
|
|
61b91220c0 | ||
|
|
316f3876c2 | ||
|
|
614b59543c | ||
|
|
1c14afd559 | ||
|
|
9a3c45779c | ||
|
|
bfc9c01797 | ||
|
|
3a3ad3d209 | ||
|
|
aef4e9b3b1 | ||
|
|
a943220c11 | ||
|
|
556ac7088e | ||
|
|
e7456f1b75 | ||
|
|
c993d7393e | ||
|
|
7f83164233 | ||
|
|
4e52f47d1e | ||
|
|
e19913f6e9 | ||
|
|
2413a57651 | ||
|
|
7bb76b5ec9 | ||
|
|
0bd19a976b | ||
|
|
40b93784d2 | ||
|
|
33d3478bad | ||
|
|
3d8ac9d14b | ||
|
|
aaef49bfc6 | ||
|
|
cf6a00b9be | ||
|
|
1ae39562dd | ||
|
|
26064193e2 | ||
|
|
8446fc003e | ||
|
|
a28f2bab4b | ||
|
|
f82d8be4bf | ||
|
|
8e1775183e | ||
|
|
620bc36dc4 | ||
|
|
29ff16ec96 | ||
|
|
8f9d76a80d | ||
|
|
a4d9a75e2c | ||
|
|
b2db0c0a13 | ||
|
|
6aa7d8a278 | ||
|
|
ccc9014430 | ||
|
|
a159b63c67 | ||
|
|
ac48bb3cd1 | ||
|
|
3987b9ddcd | ||
|
|
c7da2f5d60 | ||
|
|
39ae1decc0 | ||
|
|
1aed667377 | ||
|
|
c1612ff397 | ||
|
|
a534ba20a0 | ||
|
|
e9bbaca07d | ||
|
|
9bfa585448 | ||
|
|
b2062556a9 | ||
|
|
c9c5585758 | ||
|
|
9212f4f218 | ||
|
|
6388db815b | ||
|
|
7a4285189f | ||
|
|
a837fe841a | ||
|
|
f9e3680f11 | ||
|
|
98f761ec45 | ||
|
|
c041318f2c | ||
|
|
604e0205a4 | ||
|
|
13213395b4 | ||
|
|
46afee5998 | ||
|
|
c488fa1211 | ||
|
|
d3cff517cd | ||
|
|
2f3d407406 | ||
|
|
bcffa4026e | ||
|
|
6d6a10be7a | ||
|
|
73dd105f3d | ||
|
|
56d4a6074f | ||
|
|
c4ad4227c0 | ||
|
|
a63ccce73d | ||
|
|
0462e1b0e7 | ||
|
|
907f2100ec | ||
|
|
e0a3db5651 | ||
|
|
fca45bc8e1 | ||
|
|
86d639c848 | ||
|
|
00338aa9ca | ||
|
|
8537dcd6de | ||
|
|
8208536cd1 | ||
|
|
0653f8f3af | ||
|
|
e0d702decb | ||
|
|
541ef014ee | ||
|
|
ffc1a7a58b | ||
|
|
9028953625 | ||
|
|
6eb95693a1 | ||
|
|
c3567eb468 | ||
|
|
15568f27db | ||
|
|
126a75ad63 | ||
|
|
a2bfc7cdb2 | ||
|
|
b963a24612 | ||
|
|
fb7be2fe2c | ||
|
|
ab00392664 | ||
|
|
9f1e7c19d2 | ||
|
|
e8b0e4c61e | ||
|
|
9145ffdc46 | ||
|
|
c3d07c870b | ||
|
|
e8812bef0b | ||
|
|
b9be2449dc | ||
|
|
bc7a804618 | ||
|
|
7b094c945b | ||
|
|
eeb3e8a597 | ||
|
|
05406c5d1b | ||
|
|
99d04a7f98 | ||
|
|
1b2b2a0161 | ||
|
|
98d65835b5 | ||
|
|
422585d08f | ||
|
|
e59a1ce16a | ||
|
|
d71acc0eb5 | ||
|
|
af2934dd6b | ||
|
|
1801512818 | ||
|
|
5ae05b032e | ||
|
|
7a592ff09a | ||
|
|
8b23984c79 | ||
|
|
bf18371afe | ||
|
|
69349dd2aa | ||
|
|
8d89f30d3f | ||
|
|
10546353da | ||
|
|
9fb74b9732 | ||
|
|
521dee0e82 | ||
|
|
229419208e | ||
|
|
65f3b946b9 | ||
|
|
191fcbf46c | ||
|
|
755a4e4470 | ||
|
|
9709b7513b | ||
|
|
32cd603515 | ||
|
|
d4bdd3621a | ||
|
|
e2f8322842 | ||
|
|
6966f9e0bc | ||
|
|
743f4ed5f9 | ||
|
|
1c04ace573 | ||
|
|
6cbff73687 | ||
|
|
dec8b10939 | ||
|
|
133a5278af | ||
|
|
da856274cc | ||
|
|
a253856147 | ||
|
|
6e25d94ebc | ||
|
|
cae8fa18dc | ||
|
|
821e5a0832 | ||
|
|
c1abc42782 | ||
|
|
ef15ea2391 | ||
|
|
1ea2517e22 | ||
|
|
0c63528c59 | ||
|
|
b063f8ca41 | ||
|
|
e7fff0173a | ||
|
|
d82abc271e | ||
|
|
970409962f | ||
|
|
055586703d | ||
|
|
5d89f86675 | ||
|
|
19a51a1fe6 | ||
|
|
d3232cea5a | ||
|
|
0c90c8c24d | ||
|
|
4c08ffce49 | ||
|
|
af4a77553c | ||
|
|
c096fda1eb | ||
|
|
8f47e85be0 | ||
|
|
90d3bd19eb | ||
|
|
afb4f7d3c5 | ||
|
|
02e1143f22 | ||
|
|
e2f4d1a7b5 | ||
|
|
f037351146 | ||
|
|
1ee11e08dc | ||
|
|
595f0ea60e | ||
|
|
d921832cd2 | ||
|
|
629697629a | ||
|
|
a25313beec | ||
|
|
dbde64385b | ||
|
|
9d909f5f04 | ||
|
|
76b0550c15 | ||
|
|
384c1e9493 | ||
|
|
b1dbcc93f6 | ||
|
|
b93833772e | ||
|
|
30b523edd6 | ||
|
|
6aab7f3832 | ||
|
|
9cd53fe5f8 | ||
|
|
6a32cf3a5e | ||
|
|
98be9b3da2 | ||
|
|
c53e85b767 | ||
|
|
40a8bd2d3b | ||
|
|
31aa115611 | ||
|
|
98ac10a528 | ||
|
|
a5a6d171e5 | ||
|
|
51ed1ea423 | ||
|
|
00ec3e7388 | ||
|
|
0d626ef2d1 | ||
|
|
b36d0ef085 | ||
|
|
10c8c5df4d | ||
|
|
31e26abec4 | ||
|
|
9e83ba630c | ||
|
|
fc02a9ce8e | ||
|
|
3d36160fc4 | ||
|
|
a11ec43de2 | ||
|
|
2656d6530c | ||
|
|
e3f54e7169 | ||
|
|
4ba5681307 | ||
|
|
8c23c86994 | ||
|
|
78d606a3a7 | ||
|
|
c4e108de78 | ||
|
|
16bf2eaf77 | ||
|
|
1d15d974fa | ||
|
|
5e57868b76 | ||
|
|
6be280c914 | ||
|
|
3ccdec9798 | ||
|
|
8658f774f3 | ||
|
|
4f3ad3f6df | ||
|
|
b454aa56c3 | ||
|
|
f9b8e30ff3 | ||
|
|
0356205b84 | ||
|
|
719a1879bd | ||
|
|
b57180bf97 | ||
|
|
7cebf5f82c | ||
|
|
dd0f4b6753 | ||
|
|
d303b4e03a | ||
|
|
31b719ae49 | ||
|
|
887aaf3d3e | ||
|
|
b1d89eba1f | ||
|
|
995a5fdf97 | ||
|
|
70a70b689e | ||
|
|
4d6ac89b43 | ||
|
|
4171cacd93 | ||
|
|
b2ade71467 | ||
|
|
82ed9fe58d | ||
|
|
3d8cc4f0a0 | ||
|
|
0557f7a7d9 | ||
|
|
dc66cd97ef | ||
|
|
6da206e196 | ||
|
|
87f98c9b8b | ||
|
|
e60601df7f | ||
|
|
eed9c4bfbf | ||
|
|
1dee77f4a4 | ||
|
|
b80148819c | ||
|
|
c3b971488e | ||
|
|
88e753f281 | ||
|
|
77832059cc | ||
|
|
633d393568 | ||
|
|
5854aec2ce | ||
|
|
30e45c2411 | ||
|
|
2a4fe697a6 | ||
|
|
921db7479d | ||
|
|
7f539424cb | ||
|
|
19a838f54f | ||
|
|
d922ab2cbc | ||
|
|
9ea77d37f3 | ||
|
|
2e35b0c6bd | ||
|
|
1c627a3f98 | ||
|
|
a931efe33a | ||
|
|
041e5e9029 | ||
|
|
efcc245c2e | ||
|
|
922e7e0813 | ||
|
|
c62a8514b0 | ||
|
|
3eb8081801 | ||
|
|
3505d09564 | ||
|
|
570607945c | ||
|
|
d3a821cdcf | ||
|
|
3f24578139 | ||
|
|
89fcf08378 | ||
|
|
c2930b2aa1 | ||
|
|
019239690b | ||
|
|
d6119c1f82 | ||
|
|
84214c80bb | ||
|
|
c2d7143c72 | ||
|
|
afdb6fbfa5 | ||
|
|
ba4c02d883 | ||
|
|
2c137931f3 | ||
|
|
36682797a0 | ||
|
|
0ef1357a77 | ||
|
|
a75d19786a | ||
|
|
be548a78ea | ||
|
|
6a610e2bc9 | ||
|
|
321d5112b4 | ||
|
|
ba75ad82db | ||
|
|
58caa5109f | ||
|
|
2f3ca8aaad | ||
|
|
3a67319cb6 | ||
|
|
266fa044b3 | ||
|
|
f3398db868 | ||
|
|
fda02036bc | ||
|
|
68179cd752 | ||
|
|
2dd5760291 | ||
|
|
af2ee9c78a | ||
|
|
1c80371b27 | ||
|
|
eef473225d | ||
|
|
44fb84ef6a | ||
|
|
e8597b7448 | ||
|
|
e2252c0a5e | ||
|
|
72cb427cd9 | ||
|
|
63030cf6ec | ||
|
|
773d44b875 | ||
|
|
1df513922f | ||
|
|
30c45620a2 | ||
|
|
6b2c731596 | ||
|
|
e6022c20b2 | ||
|
|
460f6e398e | ||
|
|
d2ffec5cce | ||
|
|
cb12e88713 | ||
|
|
0958c344b8 | ||
|
|
71dc27ea7b | ||
|
|
1263449d2a | ||
|
|
b4de5a9f1b | ||
|
|
8acd8e21f9 | ||
|
|
d45d82334f | ||
|
|
6392bd40a9 | ||
|
|
17f07bc313 | ||
|
|
325861fb99 | ||
|
|
c5088670c8 | ||
|
|
0403c6f47e | ||
|
|
a55b1cdde4 | ||
|
|
aec9a20a16 | ||
|
|
b4458e5bae | ||
|
|
a790153705 | ||
|
|
0df1445d0d | ||
|
|
038da6e02b | ||
|
|
3fb2fbe1a2 | ||
|
|
acea9d23e1 | ||
|
|
b7a448cf5b | ||
|
|
04e32991c6 | ||
|
|
424c643b6e | ||
|
|
de803cb250 | ||
|
|
effc1d3492 | ||
|
|
b1ddb1ba33 | ||
|
|
9ff65c83b6 | ||
|
|
15473f3c77 | ||
|
|
490641cf47 | ||
|
|
4ad6880ca6 | ||
|
|
9d0af28307 | ||
|
|
d6dfe95466 | ||
|
|
636d3b743e | ||
|
|
e3a5c6954f | ||
|
|
f633e30ebb | ||
|
|
5ce4947ac2 | ||
|
|
323d74c0d2 | ||
|
|
d98aeafc86 | ||
|
|
f6396fb8c6 | ||
|
|
6300329cd5 | ||
|
|
c17d33bf33 | ||
|
|
2aaeee2ab8 | ||
|
|
eb3a394224 | ||
|
|
f673423b51 | ||
|
|
eb0a41528a | ||
|
|
140bd1a6cf | ||
|
|
11f5a8e582 | ||
|
|
71b3cb8c34 | ||
|
|
c85f6a477f | ||
|
|
40d4930d73 | ||
|
|
f9be085243 | ||
|
|
36b53ff350 | ||
|
|
9801037c3d | ||
|
|
74d09b0efd | ||
|
|
38dc8820ac | ||
|
|
c77a76c6af | ||
|
|
d14d5aadea | ||
|
|
4c915b7742 | ||
|
|
9a8bbe18fa | ||
|
|
ea25441ef0 | ||
|
|
48957fcde1 | ||
|
|
7b872cc41e | ||
|
|
37418946c8 | ||
|
|
95fd29e0cb | ||
|
|
e17cd2633c | ||
|
|
e0dc5f2b0c | ||
|
|
70ee5d230c | ||
|
|
24ced500f5 | ||
|
|
4ddcdf541f | ||
|
|
0e3529869c | ||
|
|
e1e0d91c00 | ||
|
|
145a3f166b | ||
|
|
88a5a933ab | ||
|
|
c591d6d2a6 | ||
|
|
65dff806a8 | ||
|
|
b85f0f4c2a | ||
|
|
76c62d7a00 | ||
|
|
f6e65ff668 | ||
|
|
c220aa8000 | ||
|
|
4713fc17ed | ||
|
|
5789955bbe | ||
|
|
2ad84a3b78 | ||
|
|
12d699cd78 | ||
|
|
34f14ded21 | ||
|
|
71d1ab411f | ||
|
|
805e487773 | ||
|
|
8803b4547e | ||
|
|
3b3806b3f6 | ||
|
|
38d962e89d | ||
|
|
3966a365d0 | ||
|
|
d73fd14af0 | ||
|
|
a87cc89916 | ||
|
|
81fd80c8ee | ||
|
|
ff22439f28 | ||
|
|
de0de04212 | ||
|
|
7f2c3e1f64 | ||
|
|
ab55e57c22 | ||
|
|
46f6b43a53 | ||
|
|
833a33b663 | ||
|
|
9ea1307cd4 | ||
|
|
be35003cb1 | ||
|
|
26bd4db253 | ||
|
|
e294ca011c | ||
|
|
2085a4fc4a | ||
|
|
c0c8e39c04 | ||
|
|
f72618dafb | ||
|
|
b3edfacdd8 | ||
|
|
30129a3350 | ||
|
|
4d49f7b0aa | ||
|
|
71bfc13d75 | ||
|
|
74db6e18d1 | ||
|
|
7d263c6a36 | ||
|
|
454c32d1d1 | ||
|
|
d1240b9238 | ||
|
|
4105094fa5 | ||
|
|
f036469d3d | ||
|
|
14261bc98c | ||
|
|
d92858659d | ||
|
|
1a383f3f66 | ||
|
|
bc27a032c5 | ||
|
|
2b13e117f0 | ||
|
|
99c166c381 | ||
|
|
95066245db | ||
|
|
6dcaac768b | ||
|
|
02c1c49b75 | ||
|
|
cd1b7cf139 | ||
|
|
e63b7d8ac4 | ||
|
|
5190c1bb1e | ||
|
|
2cb3bba658 | ||
|
|
f9e1c46c3c | ||
|
|
e1eda47589 | ||
|
|
d902967208 | ||
|
|
fea556269b | ||
|
|
69dd3c68f6 | ||
|
|
5433f6e80b | ||
|
|
e315657066 | ||
|
|
f8d9a0c57f | ||
|
|
fa6d276925 | ||
|
|
fc80d95d7e | ||
|
|
37cab18780 | ||
|
|
128d0b7fc5 | ||
|
|
8092f02e6d | ||
|
|
03d9ce2edb | ||
|
|
10fc92dba5 | ||
|
|
6736dc06a5 | ||
|
|
8c002c62af | ||
|
|
7061313d04 | ||
|
|
76d3ba69e0 | ||
|
|
8e39ce38c9 | ||
|
|
d4bd8bf2c0 | ||
|
|
e83d7bc50c | ||
|
|
ff3d5aff75 | ||
|
|
959dbcc8a2 | ||
|
|
36bf37e9ba | ||
|
|
7a83e0e6fc | ||
|
|
8be1313b86 | ||
|
|
d925ad05f3 | ||
|
|
7f795600c8 | ||
|
|
ec16b6b01d | ||
|
|
31c0f1b341 | ||
|
|
4bee0fa199 | ||
|
|
530e6b8363 | ||
|
|
9ab2725db1 | ||
|
|
0aff68f51d | ||
|
|
ad58f802f3 | ||
|
|
04fa356ee3 | ||
|
|
f9c076fe2b | ||
|
|
f76efe798e | ||
|
|
09f455233e | ||
|
|
aea300f690 | ||
|
|
b92219f6a6 | ||
|
|
a321b95a8a | ||
|
|
98308db7e0 | ||
|
|
c1e18f6722 | ||
|
|
75e193a2c9 | ||
|
|
d6e0a7d0dd | ||
|
|
7fc5f241da | ||
|
|
aae48a7e90 | ||
|
|
88f38eb0f4 | ||
|
|
d750b463dc | ||
|
|
e10b26a3d8 | ||
|
|
74636ba246 | ||
|
|
7e2f3f14e7 | ||
|
|
caa1c402ba | ||
|
|
38a6bd93d3 | ||
|
|
b867ef7e7c | ||
|
|
3ae58c277a | ||
|
|
0c6862ca55 | ||
|
|
e8c854bcf1 | ||
|
|
06860e96fe | ||
|
|
1b503554d1 | ||
|
|
351ceb7c59 | ||
|
|
10875e0d7b | ||
|
|
1eaae8a10b | ||
|
|
59e00f6164 | ||
|
|
745cc05b10 | ||
|
|
c5dc244871 | ||
|
|
dbf3917bf4 | ||
|
|
050f189c95 | ||
|
|
029216029f | ||
|
|
31f44110b5 | ||
|
|
21f3ce6577 | ||
|
|
785d123e36 | ||
|
|
d58c551c11 | ||
|
|
560628709c | ||
|
|
0f53b51e6c | ||
|
|
06093a9c4e | ||
|
|
dbddfab6d2 | ||
|
|
7188170277 | ||
|
|
b7f69c2c1d | ||
|
|
23a4531491 | ||
|
|
7d52ad0118 | ||
|
|
4d7bf35fa3 | ||
|
|
a6a9c9ca07 | ||
|
|
f4704847c2 | ||
|
|
d9c996310b | ||
|
|
d6651afd2e | ||
|
|
cf67618cad | ||
|
|
2f0a2b3c57 | ||
|
|
e7748d9952 | ||
|
|
8eb3140b2f | ||
|
|
d6ddcea682 | ||
|
|
3559ba2377 | ||
|
|
61e63ea0d7 | ||
|
|
4ce4ac4734 | ||
|
|
e7f6db9bd1 | ||
|
|
d83f45a6a0 | ||
|
|
dd91542cd1 | ||
|
|
581e8115fe | ||
|
|
dea69cf651 | ||
|
|
60ac6537df | ||
|
|
5285116e73 | ||
|
|
40ce2d72f5 | ||
|
|
704bc9aaf9 | ||
|
|
de264fcc99 | ||
|
|
7b952e4673 | ||
|
|
551b2d2048 | ||
|
|
7bfaf82fd7 | ||
|
|
9cd6a86b95 | ||
|
|
16e9552778 | ||
|
|
87f8a2782d | ||
|
|
cbbb09d7b8 | ||
|
|
2f6230abcf | ||
|
|
f8bfc76015 | ||
|
|
8f1e6c3336 | ||
|
|
8e7d2e7879 | ||
|
|
6ab2870942 | ||
|
|
e0ad145152 | ||
|
|
da04d08426 | ||
|
|
1f70032af5 | ||
|
|
7f71994653 | ||
|
|
2bb3349da1 | ||
|
|
8fe1689968 | ||
|
|
e53730f324 | ||
|
|
7a4fe9086a | ||
|
|
d277361aae | ||
|
|
734a54e7a9 | ||
|
|
91364982df | ||
|
|
50145e4fcb | ||
|
|
4112507e99 | ||
|
|
424fc2b4ae | ||
|
|
e6066223e6 | ||
|
|
b6fa3d24d8 | ||
|
|
55c2e7cd76 | ||
|
|
5a549af823 | ||
|
|
92fb660c2e | ||
|
|
3ff640b2e6 | ||
|
|
c722429ab5 | ||
|
|
e04a192de6 | ||
|
|
c9ca6d1298 | ||
|
|
754292c419 | ||
|
|
0082bc66fc | ||
|
|
8b1937422e | ||
|
|
fb6cbf23e6 | ||
|
|
c8fdd5ed7b | ||
|
|
1c19a6a00c | ||
|
|
d44409c704 | ||
|
|
5d1c7852b7 | ||
|
|
77a211d006 | ||
|
|
bef8169bb1 | ||
|
|
681f1583f9 | ||
|
|
e3b4564d5a | ||
|
|
c0d03fc43d | ||
|
|
404ee8538e | ||
|
|
e57ac59462 | ||
|
|
8c55fdaf7e | ||
|
|
c30779184f | ||
|
|
9d188c0b6c | ||
|
|
9dd7c54221 | ||
|
|
62b95d8287 | ||
|
|
fdf21702f5 | ||
|
|
2972fc9449 | ||
|
|
8f5712629f | ||
|
|
436c701b9f | ||
|
|
543fea88e3 | ||
|
|
bdec816b31 | ||
|
|
2cd2e57d2e | ||
|
|
9370234294 | ||
|
|
50da62e722 | ||
|
|
4f3e8751db | ||
|
|
f4c58894d9 | ||
|
|
01c94ef385 | ||
|
|
2415226d25 | ||
|
|
404314d00f | ||
|
|
87489f0872 | ||
|
|
9ce7c8039e | ||
|
|
e1e25e95f9 | ||
|
|
490bde90e1 | ||
|
|
dc7596b973 | ||
|
|
335afa4457 | ||
|
|
3f77a6805a | ||
|
|
13d0aae706 | ||
|
|
404cbf4f3c | ||
|
|
958ffec844 | ||
|
|
31f000d1cc | ||
|
|
cd32b3e02f | ||
|
|
bf27908095 | ||
|
|
c5f9ea53b2 | ||
|
|
d32a7184da | ||
|
|
2930abe456 | ||
|
|
b93ef4289d | ||
|
|
401bdbd316 | ||
|
|
1048d79cf8 | ||
|
|
1e8406162d | ||
|
|
03edd35c83 | ||
|
|
ac11127397 | ||
|
|
e028dcc7c0 | ||
|
|
076f45c1ee | ||
|
|
85eb7265db | ||
|
|
d3ceb67e66 | ||
|
|
7ac153a5ca | ||
|
|
d1e7aa0abd | ||
|
|
2d846c55a1 | ||
|
|
b318063c0a | ||
|
|
4aa307be55 | ||
|
|
055e52e5ea | ||
|
|
7d2069596b | ||
|
|
c45009c9a4 | ||
|
|
b91020b407 | ||
|
|
2dcc5ea4f6 | ||
|
|
359151d9a0 | ||
|
|
ce67cd3729 | ||
|
|
7c554e5da8 | ||
|
|
663ea33ff1 | ||
|
|
3ef04f1654 | ||
|
|
0eced76a41 | ||
|
|
3ab6470d1a | ||
|
|
989a03532c | ||
|
|
fa15369a02 | ||
|
|
78a9cb88d8 | ||
|
|
a0bff12746 | ||
|
|
98f2af94e5 | ||
|
|
46f7b6d574 | ||
|
|
911a6a6a35 | ||
|
|
38c7949d5c | ||
|
|
7e7a0dba9d | ||
|
|
f62e210ae6 | ||
|
|
6ceb4942a0 | ||
|
|
8cae5e4708 | ||
|
|
2a773fa34e | ||
|
|
5357f63327 | ||
|
|
60f61c8101 | ||
|
|
3d75ba8251 | ||
|
|
6c6bcd914d | ||
|
|
f79b08de81 | ||
|
|
f2bc037fff | ||
|
|
86604a684b | ||
|
|
47bd1e0178 | ||
|
|
c41305ad18 | ||
|
|
98ce9034f0 | ||
|
|
0ceff110da | ||
|
|
1d018acb3e | ||
|
|
7d8cf38dbe | ||
|
|
8d483fe4aa | ||
|
|
c1191250bf | ||
|
|
4b7266349a | ||
|
|
22f9b7681f | ||
|
|
589d32cc39 | ||
|
|
89199837db | ||
|
|
d6ebaf1b49 | ||
|
|
fac927777c | ||
|
|
ecbd697dae | ||
|
|
7d4acef64d | ||
|
|
9f0ce517cf | ||
|
|
c718e56b0d | ||
|
|
b65f0316d1 | ||
|
|
8d8bcb76b0 | ||
|
|
5f42748ed1 | ||
|
|
c9005045dc | ||
|
|
6c81befc87 | ||
|
|
dfe0b288e1 | ||
|
|
31200fbb83 | ||
|
|
9185978c55 | ||
|
|
fcba463553 | ||
|
|
2c53d3eecf | ||
|
|
516ecd374a | ||
|
|
3b1b54a74d | ||
|
|
6914e7c904 | ||
|
|
5452369749 | ||
|
|
a113311e77 | ||
|
|
44da97da92 | ||
|
|
f759980a58 | ||
|
|
37e0f8c236 | ||
|
|
51711d5906 | ||
|
|
6375223b16 | ||
|
|
4cb046768d | ||
|
|
3322542444 | ||
|
|
65f707354b | ||
|
|
109e2e7e9d | ||
|
|
cbc3a6bb9d | ||
|
|
2fa8d4ae6d | ||
|
|
7b6c8aee99 | ||
|
|
6284eaa363 | ||
|
|
636524e87f | ||
|
|
202b2f3972 | ||
|
|
247fe273d8 | ||
|
|
cb320dfa3a | ||
|
|
d8bb5abc46 | ||
|
|
cc703eca51 | ||
|
|
81c9df629c | ||
|
|
d3c0c52208 | ||
|
|
744e0555c0 | ||
|
|
3a38f7dfdc | ||
|
|
f572319bd9 | ||
|
|
48528f468c | ||
|
|
4264a80ca9 | ||
|
|
8573d4f05e | ||
|
|
210a733515 | ||
|
|
0aef0e6f63 | ||
|
|
dd022ad9be | ||
|
|
832ad61e5b | ||
|
|
9419c04ee3 | ||
|
|
a37b39d83c | ||
|
|
bb8c769c8e | ||
|
|
576c214f28 | ||
|
|
b79d1fc15b | ||
|
|
eb66e1c18d |
@@ -0,0 +1,73 @@
|
||||
---
|
||||
date: 2026-05-07
|
||||
experiment: PR #1280 (daVinci-MagiHuman port), distill DiT parity bring-up
|
||||
category: porting
|
||||
severity: important
|
||||
---
|
||||
|
||||
# Conversion `--cast-bf16` Needs an FP32-Keep Suffix Allowlist
|
||||
|
||||
## What Happened
|
||||
|
||||
`scripts/checkpoint_conversion/convert_magi_human_to_diffusers.py --cast-bf16`
|
||||
produced a converted distill DiT checkpoint that loaded cleanly, ran end-to-
|
||||
end, and emitted reasonable output — but `test_magi_human_distill_parity`
|
||||
showed `diff_mean=0.114` against the upstream reference. The base DiT was
|
||||
bit-exact with the same conversion script. Only the distill variant
|
||||
regressed.
|
||||
|
||||
The error was small enough that visual quality looked normal, but large
|
||||
enough to fail bit-exact parity. The MagiHuman base + distill DiTs share
|
||||
most of their architecture, so a difference that affected only distill was
|
||||
counterintuitive.
|
||||
|
||||
## Root Cause
|
||||
|
||||
`--cast-bf16` was downcasting **all** fp32 tensors to bf16 indiscriminately.
|
||||
The base checkpoint and the FastVideo `final_linear` / adapter modules
|
||||
require eight specific tensors to remain in fp32:
|
||||
|
||||
- LayerNorm `gamma` / `beta` weights for the final residual exit
|
||||
- Adapter projection biases
|
||||
- A handful of scale parameters in the output projection chain
|
||||
|
||||
These tensors participate in chains where bf16 precision causes accumulation
|
||||
error large enough to drift the parity check. The base DiT happened to not
|
||||
hit those specific chains in the path the test exercised (different
|
||||
attention mask shape, different audio interleave); the distill variant did.
|
||||
|
||||
## Fix / Workaround
|
||||
|
||||
Added `_FP32_KEEP_SUFFIXES` allowlist to
|
||||
`convert_magi_human_to_diffusers.py` (commit `829f70d3`) and gated `--cast-
|
||||
bf16` on it. Tensors whose state-dict key ends with any allowlisted suffix
|
||||
keep their original fp32 dtype regardless of the flag.
|
||||
|
||||
Distill DiT parity went from `diff_mean=0.114` (silently wrong) to bit-exact
|
||||
in one commit.
|
||||
|
||||
## Prevention
|
||||
|
||||
1. **Treat `--cast-bf16` as opinionated, not blanket.** Any conversion
|
||||
script that supports a global dtype downcast flag MUST own an explicit
|
||||
allowlist of fp32-keep tensors, documented at the top of the file.
|
||||
|
||||
2. **The `add-model-conversion` skill** should enforce two checks for any
|
||||
converter that ships a `--cast-bf16`-style flag:
|
||||
- Run the parity test for **every** variant of the model (base, distill,
|
||||
SR, etc.), not just the headline variant. Different variants exercise
|
||||
different code paths.
|
||||
- Diff the converted checkpoint's dtype map against the upstream
|
||||
reference and assert the allowlist covers every fp32 tensor in the
|
||||
reference.
|
||||
|
||||
3. **For MagiHuman specifically**: if you add or rename DiT modules that
|
||||
touch `final_linear`, the adapter, or any LayerNorm in the residual exit
|
||||
path, **check that any fp32-required tensors are covered by
|
||||
`_FP32_KEEP_SUFFIXES`** in the conversion script and re-run
|
||||
`test_magi_human_distill_parity` (it's the canary).
|
||||
|
||||
4. The lesson generalizes beyond MagiHuman: any DiT that uses bf16 mixed
|
||||
precision but keeps specific tensors in fp32 (a common pattern with
|
||||
flash-attn-style backends) needs this allowlist for any conversion that
|
||||
downcasts.
|
||||
@@ -0,0 +1,77 @@
|
||||
---
|
||||
date: 2026-05-07
|
||||
experiment: PR #1280 (daVinci-MagiHuman port), DiT parity bring-up
|
||||
category: porting
|
||||
severity: important
|
||||
---
|
||||
|
||||
# DiT Dtype Boundary Alignment with Flash-Attn-Style Backends
|
||||
|
||||
## What Happened
|
||||
|
||||
DiT bit-exact parity for daVinci-MagiHuman against the upstream reference
|
||||
sat at `diff_max=0.5` after the architecture port was complete and weight
|
||||
loading was correct. The error grew with depth (later layers diverged more
|
||||
than earlier ones), suggesting an accumulating numerical drift rather than
|
||||
a structural mismatch. None of the obvious culprits (RoPE, GQA expansion,
|
||||
attention mask handling) accounted for the pattern.
|
||||
|
||||
## Root Cause
|
||||
|
||||
Four cumulative dtype-boundary mismatches, each individually small but
|
||||
together pushing parity from `diff_max=0.5` to bit-exact (`diff_max=0.0`):
|
||||
|
||||
1. **SDPA inputs were not cast to bf16.** Upstream's `flash_attn_with_cp`
|
||||
internally casts Q/K/V to bf16 at `dit_module.py:508` before the kernel.
|
||||
FastVideo was passing fp32 tensors through, getting numerically different
|
||||
intermediates even though the kernel accepts both.
|
||||
|
||||
2. **Post-attention output was kept in bf16 across the per-head gating
|
||||
multiply.** Upstream upcasts to fp32 before the gating, FastVideo did the
|
||||
gate in bf16 then upcast.
|
||||
|
||||
3. **A residual-stream cast at the block boundary.** FastVideo had a
|
||||
`.to(bf16)` then `.to(fp32)` at the start of each block. Upstream keeps
|
||||
the residual stream **continuously in fp32** across all 40 layers; only
|
||||
the inputs to specific kernels are temporarily downcast.
|
||||
|
||||
4. **Parity test scheduler used a double-shift.** A separate per-block fix
|
||||
(Wave 11 production migration) — single-shift schedule is what upstream
|
||||
uses; the parity test was double-shifting.
|
||||
|
||||
## Fix / Workaround
|
||||
|
||||
Four cumulative changes in `fastvideo/models/dits/magi_human.py` (commit
|
||||
`3a4816cb`), each with a comment at the call site explaining the upstream
|
||||
parity rationale:
|
||||
|
||||
- Cast SDPA inputs to bf16 right before the attention call.
|
||||
- Upcast attention output to fp32 before the per-head gate multiply.
|
||||
- Drop the residual-stream `.to(bf16)`/`.to(fp32)` wrapper at the block
|
||||
boundary; let the residual stay fp32 throughout.
|
||||
- Single-shift schedule in the parity test fixture (matches upstream Wave 11).
|
||||
|
||||
## Prevention
|
||||
|
||||
1. **For any DiT port with a flash-attn-style backend**, treat the dtype of
|
||||
the residual stream as a load-bearing invariant, not a performance knob.
|
||||
Document it in the model's per-pipeline AGENTS.md. MagiHuman's invariant:
|
||||
*residual stream stays fp32 across all blocks; only kernel inputs are
|
||||
temporarily bf16*.
|
||||
|
||||
2. **Use layer-by-layer activation hooks** when DiT parity is close-but-not-
|
||||
bit-exact and the gap grows with depth. The
|
||||
`fastvideo/hooks/activation_trace.py` infra exists exactly for this case
|
||||
(`add-model-trace` skill). In MagiHuman's case it would have localized the
|
||||
first divergence point in one pass.
|
||||
|
||||
3. **The `add-model-port-dit` skill** should explicitly call out:
|
||||
- SDPA input dtype must match the upstream kernel's internal cast.
|
||||
- Post-attention upcast happens **before** any per-head gate, not after.
|
||||
- Residual stream dtype across block boundaries is a parity invariant.
|
||||
These rules apply to any DiT port whose upstream uses a flash-attn-style
|
||||
backend (`flash_attn_with_cp`, `flex_flash_attn_func`, etc.).
|
||||
|
||||
4. **Add an "intermediate-layer parity" test** for new DiT ports — comparing
|
||||
activations at layer 5, 10, 20, 30 — not just the final output. A growing-
|
||||
with-depth pattern is otherwise indistinguishable from "almost right".
|
||||
@@ -0,0 +1,69 @@
|
||||
---
|
||||
date: 2026-05-07
|
||||
experiment: PR #1280 (daVinci-MagiHuman port), Wave 14
|
||||
category: porting
|
||||
severity: critical
|
||||
---
|
||||
|
||||
# Silent Channel-Major Token-Packing Bugs
|
||||
|
||||
## What Happened
|
||||
|
||||
While porting daVinci-MagiHuman (`fastvideo/pipelines/basic/magi_human/`),
|
||||
the pipeline-parity test passed bit-exactly but the E2E user-visible output
|
||||
was **pure static noise**. Latent tensors compared identically against the
|
||||
upstream reference at every checkpointed boundary, yet decoded videos showed
|
||||
no recognizable content. The discrepancy reproduced on every variant
|
||||
(base / distill / SR-540p / SR-1080p) with the same noise profile.
|
||||
|
||||
## Root Cause
|
||||
|
||||
Video tokens were being packed **spatial-major** instead of **channel-major**:
|
||||
|
||||
```python
|
||||
# What we had (spatial-major, WRONG)
|
||||
einops.rearrange(x, "b c (T pT) (H pH) (W pW) -> b (T H W) (pT pH pW C)", ...)
|
||||
|
||||
# What upstream's UnfoldNd produces (channel-major, CORRECT)
|
||||
einops.rearrange(x, "b c (T pT) (H pH) (W pW) -> b (T H W) (C pT pH pW)", ...)
|
||||
```
|
||||
|
||||
A single-character einops reorder. The pipeline-parity test used FastVideo's
|
||||
own packer on **both** sides of the comparison, so the bug was invisible there
|
||||
— both sides agreed on the wrong layout. The DiT consumed those tokens
|
||||
without complaint because the channel dimension only matters at decode time,
|
||||
when the VAE's first conv expects channel-major input. By that point the test
|
||||
boundary was already passed.
|
||||
|
||||
The bug was load-bearing for any token-packed format that downstream feeds
|
||||
into a `UnfoldNd`-shaped consumer. Wave 14 of the port took multiple bug-hunt
|
||||
iterations and an Oracle consultation to localize.
|
||||
|
||||
## Fix / Workaround
|
||||
|
||||
Single-character einops change in `stages/latent_preparation.py:_img2tokens`
|
||||
(commit `6d190693` of the original PR). After the fix, all four variants
|
||||
produced expected E2E output and the pipeline-parity tests still passed
|
||||
because both sides of the parity check are now correct.
|
||||
|
||||
## Prevention
|
||||
|
||||
1. **Never use the FastVideo-side packer on both sides of a parity test.**
|
||||
At least one parity boundary must compare against an upstream tensor
|
||||
produced by the upstream packer. For MagiHuman this means a separate
|
||||
`_img2tokens` parity test that feeds upstream `UnfoldNd` output as the
|
||||
reference, not FastVideo's reformatted equivalent.
|
||||
|
||||
2. **Add an E2E hash check** alongside latent-parity. The mp4 SHA was the
|
||||
first signal that something was wrong; if it had been part of the standard
|
||||
parity battery, the bug would have surfaced in Wave 1, not Wave 14. See
|
||||
`fastvideo/tests/ssim/test_magi_human_similarity.py` for the CI version.
|
||||
|
||||
3. **For any new model port that involves explicit tensor reshaping into
|
||||
tokens**, document the expected packing order (`(C pT pH pW)` vs
|
||||
`(pT pH pW C)`) at the call site and assert the layout matches the
|
||||
downstream consumer's expectation.
|
||||
|
||||
4. The `add-model-port-dit` skill's parity gate should require an E2E hash
|
||||
check for any DiT that does video token packing, not just latent
|
||||
bit-exactness.
|
||||
@@ -0,0 +1,41 @@
|
||||
---
|
||||
date: 2026-05-22
|
||||
experiment: PR #1386 DreamVerse app CI backend tests
|
||||
category: infrastructure
|
||||
severity: important
|
||||
---
|
||||
|
||||
# DreamVerse App CI Streaming Imports Need GPU
|
||||
|
||||
## What Happened
|
||||
|
||||
DreamVerse app CI backend pytest collection imports FastVideo streaming surfaces.
|
||||
When those tests run in a CPU-only Modal environment, collection can fail before
|
||||
any app assertions run with Triton reporting:
|
||||
|
||||
```text
|
||||
RuntimeError: 0 active drivers
|
||||
```
|
||||
|
||||
## Root Cause
|
||||
|
||||
Some streaming import paths can import `fastvideo_kernel` at module import time.
|
||||
Triton then probes for an active GPU driver during pytest collection. A CPU-only
|
||||
Modal container has no active driver, so the failure appears as an import-time
|
||||
collection error rather than a DreamVerse app behavior failure.
|
||||
|
||||
## Fix / Workaround
|
||||
|
||||
For PR #1386, use a surgical CI fix: allocate a GPU to
|
||||
`run_dreamverse_app_tests`. Do not refactor core streaming/kernel imports just to
|
||||
unstick this app CI path.
|
||||
|
||||
Keep `build_kernel=False` for this job. The DreamVerse app backend test imports
|
||||
streaming surfaces but does not need to rebuild or exercise custom kernels.
|
||||
|
||||
## Prevention
|
||||
|
||||
When adding or modifying DreamVerse app CI jobs that import FastVideo streaming
|
||||
modules, make the GPU requirement explicit if the import graph may touch
|
||||
`fastvideo_kernel`. Prefer small CI resource fixes for app test collection issues
|
||||
unless the product code genuinely requires lazy import cleanup.
|
||||
@@ -0,0 +1,48 @@
|
||||
# Lessons Learned Database
|
||||
|
||||
This directory stores documented mistakes, unexpected behaviors, and their fixes.
|
||||
Each lesson is a permanent record that helps agents and humans avoid repeating
|
||||
past errors.
|
||||
|
||||
## When to Create a Lesson
|
||||
|
||||
- An experiment failed for a non-obvious reason.
|
||||
- A configuration or hyperparameter choice led to wasted compute.
|
||||
- A porting, data, or infrastructure issue was discovered and resolved.
|
||||
- A workaround was needed for a known framework/library bug.
|
||||
|
||||
## File Naming
|
||||
|
||||
`<YYYY-MM-DD>_<short-slug>.md` — e.g., `2026-03-02_lr-too-high-for-lora.md`
|
||||
|
||||
## Template
|
||||
|
||||
```markdown
|
||||
---
|
||||
date: <ISO-8601>
|
||||
experiment: <reference to experiment_journal.md entry, if applicable>
|
||||
category: hyperparameter | data | infrastructure | evaluation | porting | other
|
||||
severity: critical | important | minor
|
||||
---
|
||||
|
||||
# <Short Descriptive Title>
|
||||
|
||||
## What Happened
|
||||
<Description of the problem and its symptoms.>
|
||||
|
||||
## Root Cause
|
||||
<Analysis of why it happened.>
|
||||
|
||||
## Fix / Workaround
|
||||
<What resolved the issue.>
|
||||
|
||||
## Prevention
|
||||
<How to avoid this in the future — updated skills, SOPs, or checks.>
|
||||
```
|
||||
|
||||
## Usage
|
||||
|
||||
- Before starting a task, **search this directory** for relevant lessons.
|
||||
- After completing or failing a task, **check if a new lesson should be created**.
|
||||
- Periodically review lessons for **patterns** — recurring themes may warrant
|
||||
a new skill, SOP, or codebase fix.
|
||||
Executable
+96
@@ -0,0 +1,96 @@
|
||||
#!/usr/bin/env bash
|
||||
# Sync .agents/skills/ into .claude/skills/ via per-skill symlinks.
|
||||
#
|
||||
# Why: Claude Code only scans .claude/skills/ and ~/.claude/skills/ for
|
||||
# user-invocable skills (no skillsPath config exists — see
|
||||
# https://code.claude.com/docs/en/skills.md). This repo's skills live
|
||||
# in .agents/skills/ so they travel with the repo and stay under git.
|
||||
# Run this once after cloning (or after adding/removing a skill) to
|
||||
# expose them to Claude Code without maintaining a parallel tree.
|
||||
#
|
||||
# Usage:
|
||||
# .agents/scripts/sync-skills.sh
|
||||
#
|
||||
# Idempotent and safe to re-run. Prunes stale symlinks whose source
|
||||
# has been removed from .agents/skills/. Leaves hand-written
|
||||
# .claude/skills/<name>/ directories untouched (only symlinks are
|
||||
# managed).
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
REPO_ROOT="$(git -C "$(dirname "$0")" rev-parse --show-toplevel)"
|
||||
SRC_DIR="$REPO_ROOT/.agents/skills"
|
||||
DST_DIR="$REPO_ROOT/.claude/skills"
|
||||
|
||||
if [[ ! -d "$SRC_DIR" ]]; then
|
||||
echo "Error: $SRC_DIR does not exist." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
mkdir -p "$DST_DIR"
|
||||
|
||||
linked=0
|
||||
unchanged=0
|
||||
skipped=0
|
||||
pruned=0
|
||||
|
||||
link_skill() {
|
||||
local name="$1"
|
||||
local src="$SRC_DIR/$name"
|
||||
local dst="$DST_DIR/$name"
|
||||
# Relative target keeps symlinks portable across clones.
|
||||
local rel="../../.agents/skills/$name"
|
||||
|
||||
if [[ -L "$dst" ]]; then
|
||||
if [[ "$(readlink "$dst")" == "$rel" ]]; then
|
||||
unchanged=$((unchanged + 1))
|
||||
return
|
||||
fi
|
||||
rm "$dst"
|
||||
elif [[ -e "$dst" ]]; then
|
||||
echo "Skipped (not a symlink): .claude/skills/$name" >&2
|
||||
skipped=$((skipped + 1))
|
||||
return
|
||||
fi
|
||||
|
||||
ln -s "$rel" "$dst"
|
||||
echo "Linked: .claude/skills/$name -> $rel"
|
||||
linked=$((linked + 1))
|
||||
}
|
||||
|
||||
prune_stale() {
|
||||
local link="$1"
|
||||
local target
|
||||
target="$(readlink "$link")"
|
||||
case "$target" in
|
||||
../../.agents/skills/*) ;;
|
||||
*) return ;;
|
||||
esac
|
||||
local name="${target##*/}"
|
||||
if [[ ! -d "$SRC_DIR/$name" ]]; then
|
||||
rm "$link"
|
||||
echo "Pruned stale: .claude/skills/$(basename "$link")"
|
||||
pruned=$((pruned + 1))
|
||||
fi
|
||||
}
|
||||
|
||||
for src in "$SRC_DIR"/*/; do
|
||||
[[ -d "$src" ]] || continue
|
||||
name="$(basename "$src")"
|
||||
# Only treat directories that actually contain a SKILL.md as skills.
|
||||
[[ -f "$src/SKILL.md" ]] || continue
|
||||
link_skill "$name"
|
||||
done
|
||||
|
||||
shopt -s nullglob
|
||||
for link in "$DST_DIR"/*; do
|
||||
[[ -L "$link" ]] || continue
|
||||
prune_stale "$link"
|
||||
done
|
||||
shopt -u nullglob
|
||||
|
||||
printf "\nSummary: %d linked, %d unchanged, %d pruned" "$linked" "$unchanged" "$pruned"
|
||||
if [[ "$skipped" -gt 0 ]]; then
|
||||
printf ", %d skipped (non-symlink collision)" "$skipped"
|
||||
fi
|
||||
printf "\n"
|
||||
@@ -0,0 +1,55 @@
|
||||
---
|
||||
name: <skill-name>
|
||||
description: <one-line description — Codex uses this for implicit invocation matching>
|
||||
---
|
||||
|
||||
# <Skill Name>
|
||||
|
||||
## Purpose
|
||||
<Why this skill exists and when to use it.>
|
||||
|
||||
## Prerequisites
|
||||
- <What must be true before using this skill>
|
||||
|
||||
## Inputs
|
||||
| Parameter | Required | Description |
|
||||
|-----------|----------|-------------|
|
||||
| `param1` | Yes | ... |
|
||||
|
||||
## Steps
|
||||
|
||||
1. **Step 1 title**
|
||||
- Detail...
|
||||
|
||||
2. **Step 2 title**
|
||||
- Detail...
|
||||
|
||||
## Outputs
|
||||
- <What this skill produces>
|
||||
|
||||
## Example Usage
|
||||
|
||||
```
|
||||
<Example invocation or prompt snippet>
|
||||
```
|
||||
|
||||
## References
|
||||
- <Links to relevant files in the codebase>
|
||||
|
||||
---
|
||||
|
||||
## Folder Structure
|
||||
|
||||
Each skill lives in its own directory under `.agents/skills/`:
|
||||
|
||||
```
|
||||
.agents/skills/<skill-name>/
|
||||
├── SKILL.md # Required: instructions + metadata (this file)
|
||||
├── scripts/ # Optional: executable helper scripts
|
||||
├── references/ # Optional: documentation, papers
|
||||
└── assets/ # Optional: templates, resources
|
||||
```
|
||||
|
||||
Skill discovery is directory-based; no hand-maintained registry entry is
|
||||
required. Run `.agents/scripts/sync-skills.sh` if a local Claude Code checkout
|
||||
needs refreshed `.claude/skills/` symlinks.
|
||||
@@ -0,0 +1,173 @@
|
||||
---
|
||||
name: add-model-01-prep
|
||||
description: Use at the start of a FastVideo model port to gather required inputs, inspect/download HF weights, clone and install the official reference repo in the current environment, create a local_tests README skeleton, and produce a handoff before conversion or implementation.
|
||||
---
|
||||
|
||||
# Add Model Prep
|
||||
|
||||
## Goal
|
||||
|
||||
Prepare external assets and the shared parity-test environment for a FastVideo
|
||||
model port. Stop before writing conversion scripts, model components, pipeline
|
||||
code, registry entries, or executable parity tests.
|
||||
|
||||
## Ask First
|
||||
|
||||
Ask once, then proceed if the HF token is already exported:
|
||||
|
||||
```text
|
||||
Before prep: (1) official reference repo or Diffusers pipeline URL, (2) HF repo
|
||||
id or local weights path and whether it has a root model_index.json, (3) target
|
||||
model_family, (4) workload types, (5) which token env var is exported:
|
||||
HF_TOKEN, HUGGINGFACE_HUB_TOKEN, or HF_API_KEY, (6) may I stage clone and
|
||||
weights under the FastVideo repo root, and (7) may I install official reference
|
||||
dependencies into the current FastVideo conda/env for parity tests?
|
||||
```
|
||||
|
||||
Useful optional inputs: `pipeline_class`, `reference_dir`, `hf_revision`,
|
||||
`official_revision`, `reuse_hints`, `download_scope`.
|
||||
|
||||
## Rules
|
||||
|
||||
- Follow `../add-model/shared/common_rules.md` for token/auth safety, state files,
|
||||
escape hatches, and skip/pass semantics.
|
||||
- Run from the FastVideo repo root.
|
||||
- Use repo-relative defaults: `<ReferenceDir>/`,
|
||||
`official_weights/<model_family>/`, `converted_weights/<model_family>/`.
|
||||
- Install official reference deps into the current FastVideo environment, not a
|
||||
new venv/conda env, so parity tests run both implementations with one shared
|
||||
numeric stack.
|
||||
- If the reference is a Diffusers class/package instead of a cloneable repo,
|
||||
record import path and version instead of cloning.
|
||||
- Prep may create only the local-test README and `PORT_STATUS.md` skeletons;
|
||||
executable `.py` parity tests belong to `../add-model-02-parity/SKILL.md`.
|
||||
|
||||
## Escape Hatches
|
||||
|
||||
Follow `../add-model/shared/common_rules.md`. Prep-specific ask cases include
|
||||
overwriting an existing clone or weight directory, installing untrusted/private
|
||||
deps, choosing between incompatible official references, large downloads outside
|
||||
the agreed scope, or missing gated-repo auth setup by env var name.
|
||||
|
||||
## Workflow
|
||||
|
||||
1. Verify the repo:
|
||||
|
||||
```bash
|
||||
git rev-parse --show-toplevel
|
||||
```
|
||||
|
||||
Expected markers: `fastvideo/`, `scripts/checkpoint_conversion/`,
|
||||
`scripts/huggingface/download_hf.py`, `fastvideo/registry.py`.
|
||||
|
||||
2. Inspect HF or local weight layout:
|
||||
|
||||
```bash
|
||||
python ".agents/skills/add-model-01-prep/scripts/inspect_hf_layout.py" \
|
||||
"Org/Model" \
|
||||
--revision "<revision>" \
|
||||
--json
|
||||
```
|
||||
|
||||
For a local path, replace `Org/Model` with `/path/to/weights`. Record
|
||||
`source_layout`, `needs_conversion`, `model_index_class`, and
|
||||
`components_seen`.
|
||||
|
||||
3. Download HF weights if needed:
|
||||
|
||||
```bash
|
||||
python ".agents/skills/add-model-01-prep/scripts/download_hf_weights.py" \
|
||||
"Org/Model" \
|
||||
"official_weights/<model_family>" \
|
||||
--revision "<revision>"
|
||||
```
|
||||
|
||||
For selected files, repeat `--file-name`. For partial snapshots, repeat
|
||||
`--allow-pattern` or `--ignore-pattern`. If the user provided a local path,
|
||||
record it instead of copying large weights by default.
|
||||
|
||||
4. Clone the official reference repo if applicable:
|
||||
|
||||
```bash
|
||||
python ".agents/skills/add-model-01-prep/scripts/clone_reference_repo.py" \
|
||||
"<official_repo_url>" \
|
||||
"<ReferenceDir>" \
|
||||
--branch "<tag-or-branch>" \
|
||||
--commit "<commit-sha>" \
|
||||
--update-gitignore
|
||||
```
|
||||
|
||||
Omit `--branch`, `--commit`, or `--update-gitignore` when not needed. The
|
||||
helper refuses to overwrite existing paths and prints remote/HEAD instead.
|
||||
|
||||
5. Keep prep assets ignored. Ensure `.gitignore` includes relevant entries:
|
||||
|
||||
```gitignore
|
||||
/<ReferenceDir>/
|
||||
/official_weights/
|
||||
/converted_weights/
|
||||
```
|
||||
|
||||
6. Follow the official repo's setup instructions in the current environment.
|
||||
Inspect dependency files and README install docs before installing anything:
|
||||
|
||||
- `README*`, install docs, or model-card instructions.
|
||||
- `requirements*.txt`, `pyproject.toml`, `setup.py`, `environment.yml`.
|
||||
|
||||
Use the current FastVideo conda/env. Do not create a new env even if upstream
|
||||
docs recommend one; translate the needed install commands into the active env.
|
||||
Prefer editable/no-deps first so the official source is importable without
|
||||
changing shared pins:
|
||||
|
||||
```bash
|
||||
uv pip install --no-deps -e ./<ReferenceDir>
|
||||
```
|
||||
|
||||
Then install only missing official deps needed for parity imports. Stop before
|
||||
installing requirements that would change FastVideo's core stack. If upstream
|
||||
requires private/non-PyPI deps, record that parity needs a local stub helper
|
||||
rather than pretending setup is complete.
|
||||
|
||||
7. Create the model-family local test skeleton and top-level port state file:
|
||||
|
||||
```bash
|
||||
mkdir -p tests/local_tests/<model_family>
|
||||
cp ".agents/skills/add-model-01-prep/templates/local_tests_readme.md" \
|
||||
tests/local_tests/<model_family>/README.md
|
||||
cp ".agents/skills/add-model-01-prep/templates/port_status.md" \
|
||||
tests/local_tests/<model_family>/PORT_STATUS.md
|
||||
```
|
||||
|
||||
Edit every placeholder in the README and `PORT_STATUS.md`. The README gives
|
||||
later review agents enough information to reproduce the shared environment and
|
||||
run/review parity work:
|
||||
|
||||
- official code URL or import path, local clone path, and commit/version;
|
||||
- HF URL or local weight path, revision, access notes, and token env var name
|
||||
only;
|
||||
- commands already run and any blocked official dependency installs;
|
||||
- shared-env install commands to re-run without changing core pins;
|
||||
- expected local parity test paths and pytest commands;
|
||||
- private-dependency stubs or known setup gaps;
|
||||
- PR/review notes explaining which parity tests are required before handoff.
|
||||
|
||||
Do not include raw tokens, absolute cache paths that are not repo-reproducible,
|
||||
or large generated outputs. If prep is blocked before imports work, still create
|
||||
the README with `official_env_status=blocked` and the exact blocker.
|
||||
|
||||
`PORT_STATUS.md` must follow `../add-model/contracts/port_state.md`. Record open
|
||||
questions and prep issues immediately, using stable IDs such as `Q001` and
|
||||
`I001`. Keep resolved questions/issues in the table with a resolution instead of
|
||||
deleting them.
|
||||
|
||||
## Handoff
|
||||
|
||||
End with the canonical prep handoff contract from
|
||||
`../add-model/contracts/prep_handoff.md` and update the shared state files before
|
||||
handoff.
|
||||
|
||||
## Helper Scripts
|
||||
|
||||
- `scripts/inspect_hf_layout.py`: classify HF/local layout.
|
||||
- `scripts/download_hf_weights.py`: download HF snapshot or selected files.
|
||||
- `scripts/clone_reference_repo.py`: clone reference repo safely.
|
||||
@@ -0,0 +1,119 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Clone an official reference repo without overwriting existing paths."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import subprocess
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(description="Clone a reference repo for FastVideo parity tests.")
|
||||
parser.add_argument("repo_url", help="Official reference repository URL")
|
||||
parser.add_argument("target_dir", help="Directory to clone into")
|
||||
parser.add_argument("--branch", help="Branch or tag to clone")
|
||||
parser.add_argument("--commit", help="Commit SHA to check out after clone")
|
||||
parser.add_argument(
|
||||
"--update-gitignore",
|
||||
action="store_true",
|
||||
help="Add the target directory to .gitignore if missing",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--gitignore",
|
||||
default=".gitignore",
|
||||
help="Path to gitignore file when --update-gitignore is used",
|
||||
)
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def run(command: list[str], check: bool = True) -> subprocess.CompletedProcess[str]:
|
||||
return subprocess.run(
|
||||
command,
|
||||
check=check,
|
||||
text=True,
|
||||
capture_output=True,
|
||||
)
|
||||
|
||||
|
||||
def print_existing_repo_info(target: Path) -> int:
|
||||
print(f"target_exists: {target}")
|
||||
if not (target / ".git").exists():
|
||||
print("error: target exists but is not a git repo", file=sys.stderr)
|
||||
return 1
|
||||
|
||||
remote = run(["git", "-C", str(target), "remote", "-v"], check=False)
|
||||
head = run(["git", "-C", str(target), "rev-parse", "HEAD"], check=False)
|
||||
if remote.stdout:
|
||||
print("remote_v:")
|
||||
print(remote.stdout.rstrip())
|
||||
if head.stdout:
|
||||
print(f"head: {head.stdout.strip()}")
|
||||
print("not_overwritten: true")
|
||||
return 0
|
||||
|
||||
|
||||
def gitignore_entry_for(target: Path) -> str:
|
||||
root = Path.cwd().resolve()
|
||||
resolved = target.resolve()
|
||||
try:
|
||||
relative = resolved.relative_to(root)
|
||||
except ValueError as exc:
|
||||
raise ValueError("--update-gitignore requires target_dir to be under the current directory") from exc
|
||||
|
||||
text = relative.as_posix().rstrip("/")
|
||||
return "/" + text + "/"
|
||||
|
||||
|
||||
def update_gitignore(path: Path, target: Path) -> bool:
|
||||
entry = gitignore_entry_for(target)
|
||||
existing = path.read_text().splitlines() if path.exists() else []
|
||||
if entry in existing:
|
||||
return False
|
||||
|
||||
new_text = "\n".join(existing).rstrip("\n")
|
||||
if new_text:
|
||||
new_text += "\n"
|
||||
new_text += entry + "\n"
|
||||
path.write_text(new_text)
|
||||
return True
|
||||
|
||||
|
||||
def main() -> int:
|
||||
args = parse_args()
|
||||
target = Path(args.target_dir)
|
||||
|
||||
if target.exists():
|
||||
return print_existing_repo_info(target)
|
||||
|
||||
command = ["git", "clone", "--depth", "1"]
|
||||
if args.branch:
|
||||
command.extend(["--branch", args.branch])
|
||||
command.extend([args.repo_url, str(target)])
|
||||
|
||||
try:
|
||||
run(command)
|
||||
if args.commit:
|
||||
run(["git", "-C", str(target), "fetch", "--depth", "1", "origin", args.commit])
|
||||
run(["git", "-C", str(target), "checkout", args.commit])
|
||||
except subprocess.CalledProcessError as exc:
|
||||
if exc.stdout:
|
||||
print(exc.stdout, end="")
|
||||
if exc.stderr:
|
||||
print(exc.stderr, end="", file=sys.stderr)
|
||||
return exc.returncode
|
||||
|
||||
head = run(["git", "-C", str(target), "rev-parse", "HEAD"])
|
||||
print(f"cloned: {target}")
|
||||
print(f"head: {head.stdout.strip()}")
|
||||
|
||||
if args.update_gitignore:
|
||||
changed = update_gitignore(Path(args.gitignore), target)
|
||||
print(f"gitignore_updated: {str(changed).lower()}")
|
||||
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -0,0 +1,103 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Download HF weights using the standard FastVideo token env vars."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import os
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
HF_TOKEN_ENV_KEYS = ("HF_TOKEN", "HUGGINGFACE_HUB_TOKEN", "HF_API_KEY")
|
||||
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Download a HF model snapshot or selected files into a local directory.")
|
||||
parser.add_argument("repo_id", help="HF repo id, for example Org/Model")
|
||||
parser.add_argument("local_dir", help="Destination directory")
|
||||
parser.add_argument("--repo-type", default="model", help="HF repo type (default: model)")
|
||||
parser.add_argument("--revision", help="HF branch, tag, or commit")
|
||||
parser.add_argument(
|
||||
"--file-name",
|
||||
action="append",
|
||||
default=[],
|
||||
help="Download one file; may be repeated. If omitted, download full snapshot.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--allow-pattern",
|
||||
action="append",
|
||||
default=[],
|
||||
help="Snapshot allow pattern; may be repeated. Ignored when --file-name is used.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--ignore-pattern",
|
||||
action="append",
|
||||
default=[],
|
||||
help="Snapshot ignore pattern; may be repeated. Ignored when --file-name is used.",
|
||||
)
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def resolve_token() -> tuple[str | None, str | None]:
|
||||
for key in HF_TOKEN_ENV_KEYS:
|
||||
value = os.environ.get(key)
|
||||
if value:
|
||||
return key, value
|
||||
return None, None
|
||||
|
||||
|
||||
def main() -> int:
|
||||
args = parse_args()
|
||||
token_env, token = resolve_token()
|
||||
local_dir = Path(args.local_dir).expanduser()
|
||||
|
||||
if token_env:
|
||||
print(f"token_env: {token_env}")
|
||||
else:
|
||||
print("token_env: none", file=sys.stderr)
|
||||
|
||||
try:
|
||||
if local_dir.exists() and not local_dir.is_dir():
|
||||
print(
|
||||
f"error: destination exists and is not a directory: {local_dir}",
|
||||
file=sys.stderr,
|
||||
)
|
||||
return 1
|
||||
local_dir.mkdir(parents=True, exist_ok=True)
|
||||
if args.file_name:
|
||||
from huggingface_hub import hf_hub_download
|
||||
|
||||
for file_name in args.file_name:
|
||||
path = hf_hub_download(
|
||||
repo_id=args.repo_id,
|
||||
filename=file_name,
|
||||
repo_type=args.repo_type,
|
||||
revision=args.revision,
|
||||
local_dir=str(local_dir),
|
||||
token=token,
|
||||
)
|
||||
print(f"downloaded_file: {path}")
|
||||
else:
|
||||
from huggingface_hub import snapshot_download
|
||||
|
||||
path = snapshot_download(
|
||||
repo_id=args.repo_id,
|
||||
repo_type=args.repo_type,
|
||||
revision=args.revision,
|
||||
local_dir=str(local_dir),
|
||||
token=token,
|
||||
allow_patterns=args.allow_pattern or None,
|
||||
ignore_patterns=args.ignore_pattern or None,
|
||||
)
|
||||
print(f"downloaded_snapshot: {path}")
|
||||
except Exception as exc: # noqa: BLE001 - CLI should print concise failures.
|
||||
print(f"error: {exc}", file=sys.stderr)
|
||||
return 1
|
||||
|
||||
print(f"local_dir: {local_dir.resolve()}")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -0,0 +1,260 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Inspect a Hugging Face repo or local weight directory layout."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
HF_TOKEN_ENV_KEYS = ("HF_TOKEN", "HUGGINGFACE_HUB_TOKEN", "HF_API_KEY")
|
||||
RAW_WEIGHT_SUFFIXES = (".safetensors", ".pt", ".pth", ".ckpt", ".bin")
|
||||
KNOWN_COMPONENTS = {
|
||||
"audio_vae",
|
||||
"conditioner",
|
||||
"feature_extractor",
|
||||
"image_encoder",
|
||||
"scheduler",
|
||||
"text_encoder",
|
||||
"text_encoder_2",
|
||||
"tokenizer",
|
||||
"tokenizer_2",
|
||||
"transformer",
|
||||
"transformer_2",
|
||||
"unet",
|
||||
"upsampler",
|
||||
"vae",
|
||||
"vocoder",
|
||||
}
|
||||
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Classify a HF repo or local directory as Diffusers, raw, custom, or unknown.")
|
||||
parser.add_argument("source", help="HF repo id or local weights directory")
|
||||
parser.add_argument("--repo-type", default="model", help="HF repo type (default: model)")
|
||||
parser.add_argument("--revision", help="HF revision to inspect")
|
||||
parser.add_argument(
|
||||
"--max-local-files",
|
||||
type=int,
|
||||
default=20000,
|
||||
help="Maximum local files to scan recursively (default: 20000)",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--sample-limit",
|
||||
type=int,
|
||||
default=80,
|
||||
help="Number of file paths to print in human output (default: 80)",
|
||||
)
|
||||
parser.add_argument("--json", action="store_true", help="Emit JSON only")
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def resolve_token() -> tuple[str | None, str | None]:
|
||||
for key in HF_TOKEN_ENV_KEYS:
|
||||
value = os.environ.get(key)
|
||||
if value:
|
||||
return key, value
|
||||
return None, None
|
||||
|
||||
|
||||
def load_local_files(root: Path, max_files: int) -> tuple[list[str], bool]:
|
||||
files: list[str] = []
|
||||
truncated = False
|
||||
for path in root.rglob("*"):
|
||||
if not path.is_file():
|
||||
continue
|
||||
files.append(path.relative_to(root).as_posix())
|
||||
if len(files) >= max_files:
|
||||
truncated = True
|
||||
break
|
||||
return sorted(files), truncated
|
||||
|
||||
|
||||
def load_local_model_index(root: Path) -> tuple[dict[str, Any] | None, str | None]:
|
||||
index_path = root / "model_index.json"
|
||||
if not index_path.is_file():
|
||||
return None, None
|
||||
try:
|
||||
return json.loads(index_path.read_text()), None
|
||||
except Exception as exc: # noqa: BLE001 - surface malformed JSON clearly.
|
||||
return None, f"failed to parse local model_index.json: {exc}"
|
||||
|
||||
|
||||
def load_remote_files(
|
||||
repo_id: str,
|
||||
repo_type: str,
|
||||
revision: str | None,
|
||||
token: str | None,
|
||||
) -> list[str]:
|
||||
from huggingface_hub import list_repo_files
|
||||
|
||||
return sorted(list_repo_files(
|
||||
repo_id,
|
||||
repo_type=repo_type,
|
||||
revision=revision,
|
||||
token=token,
|
||||
))
|
||||
|
||||
|
||||
def load_remote_model_index(
|
||||
repo_id: str,
|
||||
repo_type: str,
|
||||
revision: str | None,
|
||||
token: str | None,
|
||||
) -> tuple[dict[str, Any] | None, str | None]:
|
||||
from huggingface_hub import hf_hub_download
|
||||
|
||||
try:
|
||||
path = hf_hub_download(
|
||||
repo_id=repo_id,
|
||||
filename="model_index.json",
|
||||
repo_type=repo_type,
|
||||
revision=revision,
|
||||
token=token,
|
||||
)
|
||||
except Exception as exc: # noqa: BLE001 - missing/inaccessible file is data.
|
||||
return None, f"failed to download model_index.json: {exc}"
|
||||
|
||||
try:
|
||||
return json.loads(Path(path).read_text()), None
|
||||
except Exception as exc: # noqa: BLE001 - surface malformed JSON clearly.
|
||||
return None, f"failed to parse remote model_index.json: {exc}"
|
||||
|
||||
|
||||
def root_file_names(files: list[str]) -> set[str]:
|
||||
return {name for name in files if "/" not in name}
|
||||
|
||||
|
||||
def component_names(files: list[str], model_index: dict[str, Any] | None) -> list[str]:
|
||||
components: set[str] = set()
|
||||
for name in files:
|
||||
parts = name.split("/", 1)
|
||||
if len(parts) != 2:
|
||||
continue
|
||||
top, rest = parts
|
||||
if top in KNOWN_COMPONENTS or rest == "config.json":
|
||||
components.add(top)
|
||||
|
||||
if model_index:
|
||||
for key, value in model_index.items():
|
||||
if key.startswith("_"):
|
||||
continue
|
||||
if isinstance(value, list) and len(value) == 2:
|
||||
components.add(key)
|
||||
|
||||
return sorted(components)
|
||||
|
||||
|
||||
def classify_layout(
|
||||
files: list[str],
|
||||
model_index: dict[str, Any] | None,
|
||||
components: list[str],
|
||||
) -> tuple[str, str]:
|
||||
roots = root_file_names(files)
|
||||
raw_weight_files = [name for name in roots if name.endswith(RAW_WEIGHT_SUFFIXES)]
|
||||
has_model_index = "model_index.json" in roots or model_index is not None
|
||||
|
||||
if has_model_index and components:
|
||||
return "diffusers", "no"
|
||||
if has_model_index:
|
||||
return "custom", "unknown"
|
||||
if raw_weight_files:
|
||||
return "raw_official", "yes"
|
||||
if any(name.endswith(RAW_WEIGHT_SUFFIXES) for name in files):
|
||||
return "custom", "yes"
|
||||
return "unknown", "unknown"
|
||||
|
||||
|
||||
def build_result(args: argparse.Namespace) -> dict[str, Any]:
|
||||
token_env, token = resolve_token()
|
||||
source_path = Path(args.source).expanduser()
|
||||
is_local = source_path.exists()
|
||||
|
||||
if is_local:
|
||||
root = source_path.resolve()
|
||||
if not root.is_dir():
|
||||
raise ValueError(f"local source is not a directory: {root}")
|
||||
files, truncated = load_local_files(root, args.max_local_files)
|
||||
model_index, model_index_error = load_local_model_index(root)
|
||||
source_kind = "local"
|
||||
source = str(root)
|
||||
else:
|
||||
files = load_remote_files(args.source, args.repo_type, args.revision, token)
|
||||
truncated = False
|
||||
model_index, model_index_error = load_remote_model_index(
|
||||
args.source,
|
||||
args.repo_type,
|
||||
args.revision,
|
||||
token,
|
||||
)
|
||||
source_kind = "hf"
|
||||
source = args.source
|
||||
|
||||
components = component_names(files, model_index)
|
||||
source_layout, needs_conversion = classify_layout(files, model_index, components)
|
||||
|
||||
return {
|
||||
"source": source,
|
||||
"source_kind": source_kind,
|
||||
"repo_type": None if is_local else args.repo_type,
|
||||
"revision": args.revision,
|
||||
"token_env": token_env,
|
||||
"source_layout": source_layout,
|
||||
"needs_conversion": needs_conversion,
|
||||
"model_index_class": (model_index or {}).get("_class_name"),
|
||||
"model_index_diffusers_version": (model_index or {}).get("_diffusers_version"),
|
||||
"model_index_error": model_index_error,
|
||||
"components_seen": components,
|
||||
"file_count": len(files),
|
||||
"file_scan_truncated": truncated,
|
||||
"files_sample": files[:args.sample_limit],
|
||||
}
|
||||
|
||||
|
||||
def print_human(result: dict[str, Any]) -> None:
|
||||
for key in (
|
||||
"source",
|
||||
"source_kind",
|
||||
"repo_type",
|
||||
"revision",
|
||||
"token_env",
|
||||
"source_layout",
|
||||
"needs_conversion",
|
||||
"model_index_class",
|
||||
"model_index_diffusers_version",
|
||||
"model_index_error",
|
||||
"file_count",
|
||||
"file_scan_truncated",
|
||||
):
|
||||
value = result.get(key)
|
||||
if value is not None:
|
||||
print(f"{key}: {value}")
|
||||
|
||||
components = result["components_seen"]
|
||||
print("components_seen: " + (", ".join(components) if components else "none"))
|
||||
print("files_sample:")
|
||||
for name in result["files_sample"]:
|
||||
print(f" {name}")
|
||||
|
||||
|
||||
def main() -> int:
|
||||
args = parse_args()
|
||||
try:
|
||||
result = build_result(args)
|
||||
except Exception as exc: # noqa: BLE001 - CLI should print concise failures.
|
||||
print(f"error: {exc}", file=sys.stderr)
|
||||
return 1
|
||||
|
||||
if args.json:
|
||||
print(json.dumps(result, indent=2, sort_keys=True))
|
||||
else:
|
||||
print_human(result)
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -0,0 +1,122 @@
|
||||
# <Model Family> Local Tests
|
||||
|
||||
Local-only parity and smoke tests for the `<model_family>` FastVideo port. These
|
||||
tests compare FastVideo against the official reference implementation and are
|
||||
not expected to run in CI unless explicitly promoted later.
|
||||
|
||||
Port progress, open questions, issues, and handoff notes live in
|
||||
`tests/local_tests/<model_family>/PORT_STATUS.md`.
|
||||
|
||||
## Reference Assets
|
||||
|
||||
| Field | Value |
|
||||
|---|---|
|
||||
| Model family | `<model_family>` |
|
||||
| Workload types | `<T2V/I2V/V2V/T2I/or compatibility shim with rationale>` |
|
||||
| Official reference | `<url or import path>` |
|
||||
| Local reference dir | `<ReferenceDir or none>` |
|
||||
| Official commit/version | `<sha, tag, package version, or unknown>` |
|
||||
| HF weights | `<HF repo id/url or local path>` |
|
||||
| HF revision | `<revision or default>` |
|
||||
| Local weights dir | `<official_weights/model_family or local path>` |
|
||||
| Source layout | `<diffusers/raw_official/monolithic/separate_components/mixed/custom/unknown>` |
|
||||
| Needs conversion | `<yes/no/unknown>` |
|
||||
|
||||
Do not write token values in this file. Use only the token env var name:
|
||||
`<HF_TOKEN or HUGGINGFACE_HUB_TOKEN or HF_API_KEY>`.
|
||||
|
||||
## Shared Environment Setup
|
||||
|
||||
Run from the FastVideo repo root in the same conda/env used for FastVideo.
|
||||
Do not create a separate upstream environment for parity tests.
|
||||
|
||||
```bash
|
||||
# Official reference source, if cloneable.
|
||||
python ".agents/skills/add-model-01-prep/scripts/clone_reference_repo.py" \
|
||||
"<official_repo_url>" \
|
||||
"<ReferenceDir>" \
|
||||
--commit "<commit-sha>" \
|
||||
--update-gitignore
|
||||
|
||||
# Editable install without changing shared core pins.
|
||||
uv pip install --no-deps -e ./<ReferenceDir>
|
||||
|
||||
# Additional official deps installed or required for imports:
|
||||
# <package list or none>
|
||||
```
|
||||
|
||||
Do not change core dependency versions (`torch`, `diffusers`, `transformers`,
|
||||
`flash-attn`, `triton`, CUDA packages) without explicit approval.
|
||||
|
||||
## Official Environment Status
|
||||
|
||||
```text
|
||||
dependency_changes: <none | installed no-deps editable | installed official deps in current env | blocked on user>
|
||||
official_env_status: <imports_ok | private_deps_need_stubs | blocked>
|
||||
private_dep_stubs: <none or tests/local_tests/helpers/<model_family>_upstream.py>
|
||||
blocked_on: <none or exact blocker>
|
||||
```
|
||||
|
||||
## Weight Setup
|
||||
|
||||
```bash
|
||||
python ".agents/skills/add-model-01-prep/scripts/download_hf_weights.py" \
|
||||
"<Org/Model>" \
|
||||
"official_weights/<model_family>" \
|
||||
--revision "<revision>"
|
||||
```
|
||||
|
||||
If weights are local-only, record the local path and do not copy large files into
|
||||
the repository.
|
||||
|
||||
## Prototype And Conversion Artifacts
|
||||
|
||||
State-dict key/shape dumps are generated after FastVideo native prototypes exist
|
||||
and are used to build the conversion mapping.
|
||||
|
||||
```text
|
||||
official_key_dumps:
|
||||
<component>: converted_weights/<model_family>/_mapping/<component>_official_keys.json
|
||||
fastvideo_key_dumps:
|
||||
<component>: converted_weights/<model_family>/_mapping/<component>_fastvideo_keys.json
|
||||
conversion_script: scripts/checkpoint_conversion/<model_family>_to_diffusers.py
|
||||
conversion_source_layout: <diffusers | separate_components | monolithic | mixed | custom>
|
||||
converted_weights_dir: converted_weights/<model_family>
|
||||
strict_load_status: <not_run | pass | pass_with_documented_exclusions | blocked>
|
||||
```
|
||||
|
||||
For monolithic official checkpoints, record the component prefix split here. For
|
||||
example, a single checkpoint may contain transformer, VAE/pretransform,
|
||||
conditioner, and scheduler/vocoder keys that the conversion script writes into
|
||||
separate FastVideo component subfolders.
|
||||
|
||||
## Expected Parity Tests
|
||||
|
||||
Planned local tests for this family:
|
||||
|
||||
| Component | Official files / args | Test | Concerns | Status |
|
||||
|---|---|---|---|---|
|
||||
| `<component>` | `<definition path; instantiation path + args>` | `tests/local_tests/<bucket>/test_<model_family>_<component>_parity.py` | `<prototype or setup concerns>` | `<planned/scaffold_skip/debug_red/non_skip_pass/blocked>` |
|
||||
| `pipeline` | `<official pipeline call>` | `tests/local_tests/pipelines/test_<model_family>_pipeline_parity.py` | `<pipeline concerns>` | `<planned/scaffold_skip/debug_red/non_skip_pass/blocked>` |
|
||||
|
||||
Include reused components in this table. Reuse is accepted only after the
|
||||
FastVideo component definition and official instantiation arguments have both
|
||||
been checked and the component parity test passes non-skip.
|
||||
|
||||
Run the relevant tests with:
|
||||
|
||||
```bash
|
||||
pytest tests/local_tests/<bucket>/test_<model_family>_<component>_parity.py -v -s
|
||||
pytest tests/local_tests/pipelines/test_<model_family>_pipeline_parity.py -v -s
|
||||
```
|
||||
|
||||
## Review Notes
|
||||
|
||||
- Required before handoff: non-skip PASS for each required component parity
|
||||
test, including reused components that own weights or numerical behavior.
|
||||
- Pipeline parity may start as a scaffold, but final handoff requires non-skip
|
||||
PASS or an explicit blocker accepted through the escape-hatch process.
|
||||
- User decisions and pause points are tracked as `E###` rows in
|
||||
`PORT_STATUS.md`; do not rely on chat history for escape-hatch context.
|
||||
- Review agents should verify this README's setup commands still match the PR,
|
||||
then run the listed parity tests or report the exact blocker.
|
||||
@@ -0,0 +1,69 @@
|
||||
# <Model Family> Port Status
|
||||
|
||||
## Summary
|
||||
|
||||
- model_family: `<model_family>`
|
||||
- workload_types: `<T2V/I2V/V2V/T2I/or compatibility shim with rationale>`
|
||||
- official_ref: `<url or import path>`
|
||||
- official_ref_dir: `<ReferenceDir or none>`
|
||||
- hf_weights_path: `<HF repo id/url or local path>`
|
||||
- local_weights_dir: `<official_weights/model_family or local path>`
|
||||
- source_layout: `<diffusers/raw_official/monolithic/separate_components/mixed/custom/unknown>`
|
||||
- local_tests_readme: `tests/local_tests/<model_family>/README.md`
|
||||
|
||||
## Current Phase
|
||||
|
||||
- phase: `prep`
|
||||
- status: `in_progress`
|
||||
- owner: `prep`
|
||||
- last_updated: `<YYYY-MM-DD>`
|
||||
|
||||
## Component Matrix
|
||||
|
||||
| Component | Type | Reuse/Port | Official Definition | Official Instantiation | FastVideo Target | Prototype | Conversion | Parity | Open Issues |
|
||||
|---|---|---|---|---|---|---|---|---|---|
|
||||
| `<component>` | `<dit/vae/encoder/generic>` | `<unknown/reuse/port>` | `<path + symbols>` | `<path + args>` | `<target files>` | `<not_started/in_progress/pass/blocked>` | `<not_started/pass/blocked>` | `<not_started/scaffold_skip/debug_red/non_skip_pass/blocked>` | `<none or IDs>` |
|
||||
|
||||
## Conversion State
|
||||
|
||||
- conversion_script: `scripts/checkpoint_conversion/<model_family>_to_diffusers.py`
|
||||
- converted_weights_dir: `converted_weights/<model_family>`
|
||||
- source_layout: `<diffusers/separate_components/monolithic/mixed/custom/unknown>`
|
||||
- strict_load_status: `not_run`
|
||||
- passthrough_components: `<none or list>`
|
||||
- retry_history: `<none>`
|
||||
|
||||
## Parity Commands
|
||||
|
||||
| Scope | Command | Last Result | Notes |
|
||||
|---|---|---|---|
|
||||
| component | `pytest tests/local_tests/<bucket>/test_<model_family>_<component>_parity.py -v -s` | `not_run` | `<notes>` |
|
||||
| pipeline | `pytest tests/local_tests/pipelines/test_<model_family>_pipeline_parity.py -v -s` | `not_run` | `<notes>` |
|
||||
|
||||
## Open Questions
|
||||
|
||||
| ID | Question | Owner | Needed By Phase | Status | Resolution |
|
||||
|---|---|---|---|---|---|
|
||||
| Q001 | `<question>` | `<owner>` | `<phase>` | `<open/resolved>` | `<resolution or blank>` |
|
||||
|
||||
## Issues And Blockers
|
||||
|
||||
| ID | Phase | Component | Severity | Issue | Evidence | Owner | Status | Resolution |
|
||||
|---|---|---|---|---|---|---|---|---|
|
||||
| I001 | `<phase>` | `<component or all>` | `<low/medium/high/blocker>` | `<issue>` | `<logs/paths/commands>` | `<owner>` | `<open/resolved>` | `<resolution or blank>` |
|
||||
|
||||
## Escape Hatches
|
||||
|
||||
| ID | Phase | Decision Type | Question | Recommended Option | Status | Resolution |
|
||||
|---|---|---|---|---|---|---|
|
||||
| E001 | `<phase>` | `<scope/dependency/auth/cost/destructive/ambiguity/blocker>` | `<one precise question>` | `<safe recommended option>` | `<open/resolved>` | `<resolution or blank>` |
|
||||
|
||||
## Decisions
|
||||
|
||||
| Date | Decision | Rationale | Impact |
|
||||
|---|---|---|---|
|
||||
| `<YYYY-MM-DD>` | `<decision>` | `<why>` | `<affected components/phases>` |
|
||||
|
||||
## Handoff Notes
|
||||
|
||||
- `<short notes for the next agent>`
|
||||
@@ -0,0 +1,234 @@
|
||||
---
|
||||
name: add-model-02-parity
|
||||
description: Use during /add-model after reference/architecture study to scaffold and later activate local FastVideo component parity tests. Emphasizes early test creation, official-reference loading, standardized FastVideo loading, and non-skip handoff gates.
|
||||
---
|
||||
|
||||
# Add Model Parity
|
||||
|
||||
## Goal
|
||||
|
||||
Create parity tests as early as possible in a FastVideo port. The first pass can
|
||||
land before conversion or component implementation as an executable scaffold;
|
||||
handoff is blocked until the same tests become non-skip PASS with real weights.
|
||||
|
||||
## When To Run
|
||||
|
||||
Follow `../add-model/shared/common_rules.md` for token/auth safety, state files,
|
||||
escape hatches, and skip/pass semantics.
|
||||
|
||||
Run immediately after `/add-model` Phase 1 has identified:
|
||||
|
||||
- official component classes and call signatures;
|
||||
- FastVideo target component buckets/classes/configs;
|
||||
- local reference clone or import path from `add-model-01-prep`;
|
||||
- local raw or Diffusers weight path;
|
||||
- `official_env_status=imports_ok`, or private deps that will be stubbed
|
||||
locally in tests;
|
||||
- `local_tests_readme` documenting setup and planned review/test commands;
|
||||
- expected component inputs and output tensors.
|
||||
|
||||
Do not wait for all FastVideo components to be implemented. Write the tests
|
||||
first, then let component-porting subagents make them pass.
|
||||
|
||||
## Outputs
|
||||
|
||||
- One component parity test per required component, including reused components:
|
||||
`tests/local_tests/<bucket>/test_<family>_<component>_parity.py`.
|
||||
- Optional helper for upstream private deps:
|
||||
`tests/local_tests/helpers/<family>_upstream.py`.
|
||||
- Pipeline parity is owned later by `../add-model-09-pipeline/SKILL.md` after all
|
||||
component parity tests pass non-skip.
|
||||
- A parity status block for the `/add-model` parity verification phase.
|
||||
|
||||
## Early Scaffold Rules
|
||||
|
||||
- A scaffold may skip while the FastVideo class, converted weights, or official
|
||||
import is missing.
|
||||
- A scaffold must already encode the real official load path, FastVideo load
|
||||
path, deterministic inputs, expected output extraction, and tolerance target.
|
||||
- Each parity test must declare its coverage scope in the file docstring or a
|
||||
module constant: `production_loader`, `implementation_subcomponent`, or `both`.
|
||||
Implementation/subcomponent parity may bypass production loaders deliberately,
|
||||
but final handoff still needs production-loader coverage somewhere before the
|
||||
pipeline depends on that component.
|
||||
- Official reference imports must run in the current FastVideo environment; do
|
||||
not create or assume a separate upstream venv/conda env.
|
||||
- A scaffold is not evidence of correctness. It becomes evidence only after a
|
||||
local non-skip PASS.
|
||||
- Prefer env-var path overrides with repo-relative defaults.
|
||||
- Keep tests local-only under `tests/local_tests/`; package/CI quality tests are
|
||||
added later.
|
||||
- Update shared state files as described in
|
||||
`../add-model/shared/common_rules.md` whenever adding or activating parity
|
||||
tests.
|
||||
|
||||
## Component Template
|
||||
|
||||
Copy `templates/component_parity_test.py` and fill every `TODO` marker. The
|
||||
template is distilled from:
|
||||
|
||||
- `tests/local_tests/transformers/test_ltx2.py`
|
||||
- `tests/local_tests/transformers/test_gamecraft_parity.py`
|
||||
- `tests/local_tests/encoders/test_ltx2_gemma_parity.py`
|
||||
- `tests/local_tests/vaes/test_oobleck_vae_parity.py`
|
||||
- `tests/local_tests/sd35/test_sd35_component_parity.py`
|
||||
|
||||
The template supports three states:
|
||||
|
||||
| State | Meaning |
|
||||
|---|---|
|
||||
| Scaffold skip | Test is committed early, but official import, FastVideo class, or weights are not available yet. |
|
||||
| Debug red | Both sides load and the test fails numerically. This is useful: porting can chase the first drift. |
|
||||
| Non-skip pass | Required before `/add-model` handoff. |
|
||||
|
||||
## Subagent Dispatch Pattern
|
||||
|
||||
After Phase 1, dispatch one parity subagent per component before or alongside
|
||||
component implementation:
|
||||
|
||||
```text
|
||||
Create a local parity test scaffold for <family> <component>.
|
||||
|
||||
Use the prep handoff:
|
||||
- official_ref_dir/import: <...>
|
||||
- local_weights_dir: <...>
|
||||
- source_layout: <...>
|
||||
- needs_conversion: <yes/no>
|
||||
- official_env_status: <imports_ok | private_deps_need_stubs>
|
||||
- local_tests_readme: tests/local_tests/<model_family>/README.md
|
||||
- port_state_file: tests/local_tests/<model_family>/PORT_STATUS.md
|
||||
- official_definition_files: <paths + classes/functions>
|
||||
- official_instantiation_files: <paths + factory/pipeline/config call sites + args>
|
||||
- concerns_or_unknowns: <known ambiguous inputs, outputs, deps, or args>
|
||||
|
||||
The complete per-component packet must match
|
||||
`../add-model/contracts/component_context.md`.
|
||||
|
||||
Read the official component call path and the planned FastVideo component API.
|
||||
Add tests/local_tests/<bucket>/test_<family>_<component>_parity.py based on
|
||||
add-model-02-parity/templates/component_parity_test.py.
|
||||
|
||||
The scaffold must load the official model with real weights when available,
|
||||
load the FastVideo model through the standardized config/class/loader path when
|
||||
available, create deterministic inputs, compare concrete outputs, and skip only
|
||||
when a dependency is genuinely missing. Do not make an unconditional skip or a
|
||||
shape-only test.
|
||||
```
|
||||
|
||||
## FastVideo Load Patterns
|
||||
|
||||
Pick the narrowest load path that matches the component:
|
||||
|
||||
| Component | Preferred FastVideo load path |
|
||||
|---|---|
|
||||
| DiT / transformer | Bucket config + model class, or `TransformerLoader` when testing converted Diffusers component dirs. |
|
||||
| VAE | VAE class `from_pretrained(...)` when implemented, or bucket config + class for local converted dirs. |
|
||||
| Text/image encoder | Bucket config + model class; pass HF subpaths from `local_weights_dir` or converted component dirs. |
|
||||
| Scheduler/conditioner | Native class/config plus exact official kwargs. |
|
||||
|
||||
For early scaffolds, an import of the planned FastVideo class may be inside a
|
||||
helper that calls `pytest.skip` if the class does not exist yet. Replace that
|
||||
skip with a real import once the component PR adds the class.
|
||||
|
||||
Direct class/config construction is allowed for implementation or subcomponent
|
||||
parity, such as connector-only encoder checks or official monolithic-checkpoint
|
||||
mapping tests. Label that scope explicitly and add separate production-loader
|
||||
coverage when converted component dirs are available.
|
||||
|
||||
## Official Load Patterns
|
||||
|
||||
- Clone/reference repo path: add its source dir to `sys.path` before imports.
|
||||
- HF/Diffusers reference: import only inside the test, not production code.
|
||||
- Private deps: add a helper under `tests/local_tests/helpers/` to install
|
||||
stubs before importing upstream modules; do not rely on an external upstream
|
||||
environment.
|
||||
- Gated HF repos: resolve `HF_TOKEN`, `HUGGINGFACE_HUB_TOKEN`, or `HF_API_KEY`
|
||||
under the token rules in `../add-model/shared/common_rules.md`.
|
||||
|
||||
## Non-Skip Activation Checklist
|
||||
|
||||
Before `/add-model` handoff, each scaffolded test must be activated:
|
||||
|
||||
```text
|
||||
[ ] Official side imports and loads real weights.
|
||||
[ ] FastVideo side imports and loads the converted or original weights.
|
||||
[ ] Test executes at least one real forward call on both sides.
|
||||
[ ] Test compares output tensors, not only shapes or state-dict keys.
|
||||
[ ] Local pytest output contains PASSED, not SKIPPED or XFAIL.
|
||||
[ ] Tolerance is justified for the component scope and kernel alignment.
|
||||
```
|
||||
|
||||
## Component Parity Details
|
||||
|
||||
Reference imports:
|
||||
|
||||
- Import from `official_ref_dir` or the recorded package/import path.
|
||||
- If upstream has private deps, add a helper under
|
||||
`tests/local_tests/helpers/<family>_upstream.py` that installs minimal stubs
|
||||
before importing upstream modules.
|
||||
- Common stubs: identity compile/op-registration decorators, CP world size set to
|
||||
1, identity scatter/gather, and test-friendly custom-op kernels.
|
||||
- Stub decorators that register `torch.ops.<ns>.<op>` must preserve the
|
||||
`torch.library` registration side-effect. Identity decorators alone are not
|
||||
enough.
|
||||
- Delete stub helpers and every `install_stubs()` call as soon as the real deps
|
||||
become required installs. No-op shims are dead code.
|
||||
|
||||
Kernel and wrapper pitfalls:
|
||||
|
||||
- If parity routes flash-attn GQA through SDPA, expand KV heads manually on the
|
||||
SDPA side with `repeat_interleave` along the head axis.
|
||||
- If upstream VAE `decode()` denormalizes internally but FastVideo/Diffusers
|
||||
expects pre-denormalized latents, apply `z = z * std + mean` only on the
|
||||
FastVideo side in the parity test.
|
||||
- Per-channel VAE `latents_mean` / `latents_std` must be reshaped explicitly,
|
||||
e.g. `.view(1, z_dim, 1, 1, 1)` for 5D video latents.
|
||||
|
||||
Tolerance guide:
|
||||
|
||||
| Scope | Start `atol` / `rtol` | Notes |
|
||||
|---|---|---|
|
||||
| Single block, same kernel | `1e-4` / `1e-4` | Tight default. |
|
||||
| Full DiT, aligned kernels | `1e-2` / `1e-2` | Cross-layer accumulation. |
|
||||
| Full DiT, cross-kernel bf16 | `0.1` / `0.1` | Also require abs-mean drift below 5% and per-modality diagnostics. |
|
||||
| VAE decode fp32 | `5e-2` / `5e-2` | After normalization alignment. |
|
||||
| Encoder wrapper around same HF class | `1e-3` / `1e-3` | Should be near-zero. |
|
||||
|
||||
Element-wise `assert_close` alone is not enough for deep full-DiT parity. Also
|
||||
log global abs-mean drift and per-modality summaries.
|
||||
|
||||
When a non-skip component parity run is numerically red after weight/input
|
||||
checks, invoke `../add-model-08-trace/SKILL.md` before adding bespoke forward
|
||||
hooks. Use `docs/contributing/activation_trace.md` to keep
|
||||
`FASTVIDEO_TRACE_LAYERS`, `FASTVIDEO_TRACE_STATS`, and `FASTVIDEO_TRACE_STEPS`
|
||||
identical across FastVideo and upstream traces.
|
||||
|
||||
Useful local commands:
|
||||
|
||||
```bash
|
||||
pytest tests/local_tests/<bucket>/test_<family>_*parity*.py -v -s
|
||||
pytest tests/local_tests -k "<family> and parity" -v -s
|
||||
```
|
||||
|
||||
## Escape Hatches
|
||||
|
||||
Follow `../add-model/shared/common_rules.md`. Parity-specific ask cases include
|
||||
private dependency approval, choosing between incompatible official references,
|
||||
accepting a shape-only substitute, or loosening required tolerances.
|
||||
|
||||
## Pipeline Parity
|
||||
|
||||
Pipeline parity is later than component parity because it needs stages, presets,
|
||||
registry wiring, converted weights, and green component parity. Record official
|
||||
pipeline call notes in `local_tests_readme`, but do not treat pipeline parity as
|
||||
owned by this skill.
|
||||
|
||||
Use `../add-model-09-pipeline/SKILL.md` and its
|
||||
`templates/pipeline_parity_test.py` for pipeline parity scaffolding and
|
||||
debugging. Compare denoised latents or decoded media, not just successful
|
||||
generation.
|
||||
|
||||
## Handoff Status Block
|
||||
|
||||
Return `../add-model/contracts/parity_status.md` to `/add-model` and update the
|
||||
shared state files before handoff.
|
||||
@@ -0,0 +1,188 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
"""Component parity scaffold for <FAMILY> <COMPONENT>.
|
||||
|
||||
This file is intended to be created early in a port. It may skip until the
|
||||
official reference, FastVideo class, and real weights are available, but it must
|
||||
never become an unconditional skip or shape-only test.
|
||||
|
||||
Fill every TODO before considering this test active.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import importlib
|
||||
import os
|
||||
from pathlib import Path
|
||||
import sys
|
||||
|
||||
import pytest
|
||||
import torch
|
||||
from torch.testing import assert_close
|
||||
|
||||
os.environ.setdefault("MASTER_ADDR", "localhost")
|
||||
os.environ.setdefault("MASTER_PORT", "29519")
|
||||
os.environ.setdefault("DISABLE_SP", "1")
|
||||
os.environ.setdefault("FASTVIDEO_ATTENTION_BACKEND", "TORCH_SDPA")
|
||||
|
||||
REPO_ROOT = Path(__file__).resolve().parents[3]
|
||||
FAMILY = "<family>" # TODO: snake_case family name.
|
||||
COMPONENT = "<component>" # TODO: transformer | vae | encoder | conditioner | ...
|
||||
PARITY_SCOPE = "implementation_subcomponent" # TODO: production_loader | implementation_subcomponent | both
|
||||
OFFICIAL_MODULE = "<official.module>" # TODO: e.g. "ltx_core.model.transformer".
|
||||
OFFICIAL_CLASS = "<OfficialClass>" # TODO: official class/factory name.
|
||||
FASTVIDEO_CONFIG_MODULE = "fastvideo.configs.models.<bucket>" # TODO.
|
||||
FASTVIDEO_CONFIG_CLASS = "<FastVideoConfig>" # TODO.
|
||||
FASTVIDEO_MODEL_MODULE = "fastvideo.models.<bucket>.<module>" # TODO.
|
||||
FASTVIDEO_MODEL_CLASS = "<FastVideoModel>" # TODO.
|
||||
|
||||
OFFICIAL_REF_DIR = Path(os.getenv("<FAMILY_UPPER>_OFFICIAL_REF_DIR", REPO_ROOT / "<ReferenceDir>"))
|
||||
LOCAL_WEIGHTS_DIR = Path(os.getenv("<FAMILY_UPPER>_LOCAL_WEIGHTS_DIR", REPO_ROOT / "official_weights" / FAMILY))
|
||||
CONVERTED_WEIGHTS_DIR = Path(os.getenv("<FAMILY_UPPER>_CONVERTED_WEIGHTS_DIR",
|
||||
REPO_ROOT / "converted_weights" / FAMILY))
|
||||
|
||||
|
||||
def _resolve_hf_token() -> str | None:
|
||||
for key in ("HF_TOKEN", "HUGGINGFACE_HUB_TOKEN", "HF_API_KEY"):
|
||||
value = os.environ.get(key)
|
||||
if value:
|
||||
return value
|
||||
return None
|
||||
|
||||
|
||||
def _add_official_to_path() -> None:
|
||||
"""Add the official source path before importing upstream modules."""
|
||||
# TODO: adjust for the official repo layout. Common examples:
|
||||
# OFFICIAL_REF_DIR / "src"
|
||||
# OFFICIAL_REF_DIR / "packages" / "<pkg>" / "src"
|
||||
# OFFICIAL_REF_DIR
|
||||
official_src = OFFICIAL_REF_DIR / "src"
|
||||
if not official_src.exists():
|
||||
official_src = OFFICIAL_REF_DIR
|
||||
if official_src.exists() and str(official_src) not in sys.path:
|
||||
sys.path.insert(0, str(official_src))
|
||||
|
||||
|
||||
def _import_or_skip(module_name: str, attr_name: str | None = None):
|
||||
if "<" in module_name or (attr_name is not None and "<" in attr_name):
|
||||
pytest.skip(f"Template import placeholder not filled: {module_name}.{attr_name}")
|
||||
try:
|
||||
module = importlib.import_module(module_name)
|
||||
except Exception as exc: # noqa: BLE001 - local parity should skip missing refs.
|
||||
pytest.skip(f"Cannot import {module_name}: {exc}")
|
||||
if attr_name is None:
|
||||
return module
|
||||
try:
|
||||
return getattr(module, attr_name)
|
||||
except AttributeError:
|
||||
pytest.skip(f"{module_name} has no attribute {attr_name}")
|
||||
|
||||
|
||||
def _load_official_model(device: torch.device, dtype: torch.dtype) -> torch.nn.Module:
|
||||
"""Load the official component with real weights."""
|
||||
_add_official_to_path()
|
||||
if not OFFICIAL_REF_DIR.exists():
|
||||
pytest.skip(f"Official reference missing: {OFFICIAL_REF_DIR}")
|
||||
if not LOCAL_WEIGHTS_DIR.exists():
|
||||
pytest.skip(f"Local weights missing: {LOCAL_WEIGHTS_DIR}")
|
||||
|
||||
# TODO: import official class/factory and load real weights strictly.
|
||||
# Examples in-tree:
|
||||
# - LTX2: SingleGPUModelBuilder(...).build(device=device, dtype=dtype)
|
||||
# - GameCraft: torch.load(...)["module"] -> official_model.load_state_dict(...)
|
||||
# - Oobleck: create_model_from_config(config) + ckpt state_dict
|
||||
OfficialClass = _import_or_skip(OFFICIAL_MODULE, OFFICIAL_CLASS)
|
||||
model = OfficialClass() # TODO: pass official config kwargs.
|
||||
state_dict = {} # TODO: load official state dict from LOCAL_WEIGHTS_DIR.
|
||||
missing, unexpected = model.load_state_dict(state_dict, strict=True)
|
||||
assert not missing and not unexpected, (f"official load mismatch missing={missing[:5]} unexpected={unexpected[:5]}")
|
||||
return model.to(device=device, dtype=dtype).eval()
|
||||
|
||||
|
||||
def _load_fastvideo_model(device: torch.device, dtype: torch.dtype) -> torch.nn.Module:
|
||||
"""Load the FastVideo component with the same tensor content."""
|
||||
if not CONVERTED_WEIGHTS_DIR.exists() and not LOCAL_WEIGHTS_DIR.exists():
|
||||
pytest.skip(f"No FastVideo loadable weights: {CONVERTED_WEIGHTS_DIR} or {LOCAL_WEIGHTS_DIR}")
|
||||
|
||||
# TODO: replace with the bucket-specific FastVideo config/class/loader.
|
||||
# DiT examples:
|
||||
# from fastvideo.configs.models.dits import <Config>
|
||||
# from fastvideo.models.dits.<module> import <Model>
|
||||
# VAE examples:
|
||||
# from fastvideo.models.vaes.<module> import <VAE>
|
||||
# model = <VAE>.from_pretrained(...)
|
||||
FastVideoConfig = _import_or_skip(FASTVIDEO_CONFIG_MODULE, FASTVIDEO_CONFIG_CLASS)
|
||||
FastVideoModel = _import_or_skip(FASTVIDEO_MODEL_MODULE, FASTVIDEO_MODEL_CLASS)
|
||||
|
||||
config = FastVideoConfig()
|
||||
model = FastVideoModel(config=config)
|
||||
state_dict = {} # TODO: load converted or directly mapped state dict.
|
||||
missing, unexpected = model.load_state_dict(state_dict, strict=True)
|
||||
assert not missing and not unexpected, (
|
||||
f"FastVideo load mismatch missing={missing[:5]} unexpected={unexpected[:5]}")
|
||||
return model.to(device=device, dtype=dtype).eval()
|
||||
|
||||
|
||||
def _make_inputs(device: torch.device, dtype: torch.dtype) -> dict[str, torch.Tensor]:
|
||||
"""Create deterministic inputs matching the official component call."""
|
||||
torch.manual_seed(0)
|
||||
# TODO: replace with component-specific tensors and metadata.
|
||||
return {
|
||||
"hidden_states": torch.randn(1, 4, 16, device=device, dtype=dtype),
|
||||
"timestep": torch.tensor([10], device=device),
|
||||
}
|
||||
|
||||
|
||||
def _run_official(model: torch.nn.Module, inputs: dict[str, torch.Tensor]) -> torch.Tensor:
|
||||
"""Run official component and return the tensor to compare."""
|
||||
with torch.inference_mode():
|
||||
output = model(**inputs) # TODO: adapt official call signature.
|
||||
if isinstance(output, dict):
|
||||
sample = output.get("sample")
|
||||
output = sample if sample is not None else output.get("x")
|
||||
elif hasattr(output, "sample"):
|
||||
output = output.sample
|
||||
elif isinstance(output, tuple):
|
||||
output = output[0]
|
||||
assert torch.is_tensor(output), f"official output is not tensor: {type(output)}"
|
||||
return output.detach().float().cpu()
|
||||
|
||||
|
||||
def _run_fastvideo(model: torch.nn.Module, inputs: dict[str, torch.Tensor]) -> torch.Tensor:
|
||||
"""Run FastVideo component and return the tensor to compare."""
|
||||
with torch.inference_mode():
|
||||
output = model(**inputs) # TODO: adapt FastVideo call signature.
|
||||
if isinstance(output, dict):
|
||||
sample = output.get("sample")
|
||||
output = sample if sample is not None else output.get("x")
|
||||
elif hasattr(output, "sample"):
|
||||
output = output.sample
|
||||
elif isinstance(output, tuple):
|
||||
output = output[0]
|
||||
assert torch.is_tensor(output), f"FastVideo output is not tensor: {type(output)}"
|
||||
return output.detach().float().cpu()
|
||||
|
||||
|
||||
@pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA required for this parity test.")
|
||||
def test_component_parity():
|
||||
"""Compare official and FastVideo outputs on identical inputs."""
|
||||
device = torch.device("cuda:0")
|
||||
dtype = torch.bfloat16
|
||||
|
||||
official = _load_official_model(device, dtype)
|
||||
fastvideo = _load_fastvideo_model(device, dtype)
|
||||
inputs = _make_inputs(device, dtype)
|
||||
|
||||
official_out = _run_official(official, inputs)
|
||||
fastvideo_out = _run_fastvideo(fastvideo, inputs)
|
||||
|
||||
assert official_out.shape == fastvideo_out.shape
|
||||
diff = (official_out - fastvideo_out).abs()
|
||||
print(f"official abs_mean={official_out.abs().mean().item():.6f} "
|
||||
f"fastvideo abs_mean={fastvideo_out.abs().mean().item():.6f} "
|
||||
f"diff_max={diff.max().item():.6f} diff_mean={diff.mean().item():.6f}")
|
||||
|
||||
# TODO: pick tolerance by scope:
|
||||
# - single block / same kernel: 1e-4
|
||||
# - full DiT aligned kernels: 1e-2
|
||||
# - full DiT cross-kernel bf16: 1e-1 + abs_mean drift check
|
||||
# - VAE decode fp32: 5e-2 after normalization alignment
|
||||
assert_close(fastvideo_out, official_out, atol=1e-4, rtol=1e-4)
|
||||
@@ -0,0 +1,114 @@
|
||||
---
|
||||
name: add-model-03-port-dit
|
||||
description: Use during /add-model Phase 4 or Phase 6 to prototype or parity-debug one FastVideo-native DiT/transformer component.
|
||||
---
|
||||
|
||||
# Add Model Port DiT
|
||||
|
||||
## Goal
|
||||
|
||||
Prototype or parity-debug one diffusion transformer in FastVideo-native code.
|
||||
This skill is for one component only; do not work on the VAE, encoders,
|
||||
pipeline, or unrelated conversion code unless the current component cannot load
|
||||
without a minimal fix there.
|
||||
|
||||
## Inputs
|
||||
|
||||
Follow `../add-model/shared/component_skill_common.md` and require the complete
|
||||
packet from `../add-model/contracts/component_context.md`.
|
||||
|
||||
DiT-specific packet fields:
|
||||
|
||||
- `component`: transformer or DiT name.
|
||||
- `parity_test`: `tests/local_tests/<bucket>/test_<family>_<component>_parity.py`.
|
||||
- `weights`: converted transformer dir or local official path.
|
||||
- `target_files`: `fastvideo/models/dits/<family>.py` and
|
||||
`fastvideo/configs/models/dits/<family>.py`.
|
||||
|
||||
## Modes
|
||||
|
||||
Use the common prototype and parity-debug modes from
|
||||
`../add-model/shared/component_skill_common.md`.
|
||||
|
||||
DiT-specific prototype concerns include ambiguous official flags, shape
|
||||
mismatches, missing FastVideo layer equivalents, and dedicated output heads.
|
||||
|
||||
## Reuse Proof
|
||||
|
||||
Apply the shared reuse proof. DiT-specific comparison must include attention
|
||||
algorithm, positional embeddings, RoPE/patching, timestep/guidance embeddings,
|
||||
scaling constants, dtype casts, state-dict names, and every output head.
|
||||
|
||||
## Existing FastVideo Patterns
|
||||
|
||||
- Base class: `fastvideo/models/dits/base.py::BaseDiT`.
|
||||
- Config bases: `DiTConfig` and `DiTArchConfig` in
|
||||
`fastvideo/configs/models/dits/base.py`.
|
||||
- Use the matching DiT config bucket. Wrong bucket inheritance can typecheck but
|
||||
fail during pipeline wiring.
|
||||
- Config export: add the config to
|
||||
`fastvideo/configs/models/dits/__init__.py`.
|
||||
- Registry discovery: set `EntryClass = <ClassName>` in the model file.
|
||||
- Loader path: `TransformerLoader` reads `transformer/config.json`, calls
|
||||
`dit_config.update_model_arch(config)`, resolves `_class_name` through
|
||||
`ModelRegistry`, and constructs the class with `config` and `hf_config`.
|
||||
- Reference examples: `stable_audio.py`, `wanvideo.py`, `sd3.py`, `longcat.py`,
|
||||
and `ltx2.py`.
|
||||
- Layer guidance: `fastvideo/layers/AGENTS.md`.
|
||||
|
||||
## Implementation Rules
|
||||
|
||||
- Use FastVideo-native layers by default: `ReplicatedLinear` for DiT hot-path
|
||||
linears, `DistributedAttention` for standard full-sequence attention, and
|
||||
`LocalAttention` for local/window attention or simple single-GPU parity paths.
|
||||
- Raw SDPA is acceptable for cross-modality flat streams when no FastVideo
|
||||
distributed equivalent exists; document the SP gap in the module docstring.
|
||||
- Mirror official tensor contracts exactly: latent packing, patch ordering,
|
||||
timestep embedding scale, RoPE/positional embedding, guidance embedding,
|
||||
cross-attention context order, output head order, and dtype casts.
|
||||
- Preserve all output heads that the official DiT emits. Do not silently drop
|
||||
audio, depth, pose, mask, or auxiliary heads.
|
||||
- Put architecture fields on `DiTArchConfig`; keep inference steps, CFG scales,
|
||||
FPS, flow shift, and sampling defaults out of the arch config.
|
||||
- Define `_fsdp_shard_conditions`, `_compile_conditions`,
|
||||
`param_names_mapping`, and `reverse_param_names_mapping` where needed.
|
||||
- Follow the production import boundary in
|
||||
`../add-model/shared/common_rules.md`.
|
||||
|
||||
## Prototype Checks
|
||||
|
||||
Follow the shared prototype success criteria. A useful one-off check is:
|
||||
|
||||
```bash
|
||||
python - <<'PY'
|
||||
# Import the target config/class, instantiate with random weights, and print
|
||||
# state_dict names/shapes for the conversion mapping.
|
||||
PY
|
||||
```
|
||||
|
||||
## Parity-Debug Loop
|
||||
|
||||
Run the shared parity-debug loop. The component test command is:
|
||||
|
||||
```bash
|
||||
pytest <parity_test> -v -s
|
||||
```
|
||||
|
||||
For numerical drift, use `../add-model-08-trace/SKILL.md` before writing bespoke
|
||||
hooks. Start with FastVideo's activation trace (`fastvideo/hooks/activation_trace.py`;
|
||||
`docs/contributing/activation_trace.md`) and a block-level regex such as
|
||||
`FASTVIDEO_TRACE_LAYERS="^block\.layers\.[0-9]+$"`. Only fall back to custom
|
||||
per-block hooks if the needed boundary or statistic is not exposed by
|
||||
`FASTVIDEO_TRACE_STATS`.
|
||||
|
||||
## Escape Hatches
|
||||
|
||||
Follow `../add-model/shared/common_rules.md` and the component-specific guidance
|
||||
in `../add-model/shared/component_skill_common.md`. DiT-specific ask cases include
|
||||
dropping an output head/modality, accepting an unsupported kernel/private op, or
|
||||
choosing between incompatible official transformer definitions.
|
||||
|
||||
## Handoff
|
||||
|
||||
Return `../add-model/contracts/component_skill_handoff.md` following the common
|
||||
handoff rules in `../add-model/shared/component_skill_common.md`.
|
||||
@@ -0,0 +1,100 @@
|
||||
---
|
||||
name: add-model-04-port-vae
|
||||
description: Use during /add-model Phase 4 or Phase 6 to prototype or parity-debug one FastVideo-native VAE component.
|
||||
---
|
||||
|
||||
# Add Model Port VAE
|
||||
|
||||
## Goal
|
||||
|
||||
Prototype or parity-debug one VAE or autoencoder in FastVideo-native code. This
|
||||
skill covers video, image, and audio VAEs.
|
||||
|
||||
## Inputs
|
||||
|
||||
Follow `../add-model/shared/component_skill_common.md` and require the complete
|
||||
packet from `../add-model/contracts/component_context.md`.
|
||||
|
||||
VAE-specific packet fields:
|
||||
|
||||
- `component`: VAE or autoencoder name.
|
||||
- `parity_test`: `tests/local_tests/vaes/test_<family>_<component>_parity.py`.
|
||||
- `weights`: converted VAE dir, HF subfolder, or local official path.
|
||||
- `target_files`: `fastvideo/models/vaes/<arch_or_family>.py` and
|
||||
`fastvideo/configs/models/vaes/<arch_or_family>.py`.
|
||||
|
||||
## Modes
|
||||
|
||||
Use the common prototype and parity-debug modes from
|
||||
`../add-model/shared/component_skill_common.md`.
|
||||
|
||||
VAE-specific prototype concerns include latent normalization, stochastic
|
||||
posterior behavior, tiling incompatibility, temporal/spatial/audio layout, and
|
||||
decode output containers.
|
||||
|
||||
## Reuse Proof
|
||||
|
||||
Apply the shared reuse proof. VAE-specific comparison must include latent layout,
|
||||
temporal/spatial/audio compression, scaling factor, mean/std normalization,
|
||||
posterior behavior, encode/decode output objects, tiling flags, and cropping.
|
||||
|
||||
## Existing FastVideo Patterns
|
||||
|
||||
- Shared tiling wrapper: `fastvideo/models/vaes/common.py::ParallelTiledVAE`.
|
||||
- Config bases: `VAEConfig` and `VAEArchConfig` in
|
||||
`fastvideo/configs/models/vaes/base.py`.
|
||||
- Use the matching VAE config bucket. Wrong bucket inheritance can typecheck but
|
||||
fail during pipeline wiring.
|
||||
- Config export: add the config to
|
||||
`fastvideo/configs/models/vaes/__init__.py`.
|
||||
- Registry discovery: set `EntryClass = <ClassName>` in the model file.
|
||||
- Loader path: VAE loaders resolve `_class_name` through `ModelRegistry` and
|
||||
load converted component weights from the VAE subdir.
|
||||
- Reference examples: `oobleck.py`, `autoencoder_kl.py`, `wanvae.py`,
|
||||
`ltx2vae.py`, and `gamecraftvae.py`.
|
||||
- Layer guidance: `fastvideo/layers/AGENTS.md`.
|
||||
|
||||
## Implementation Rules
|
||||
|
||||
- Name reusable VAE architectures by architecture (`oobleck.py`,
|
||||
`autoencoder_kl.py`); name family-specific VAEs by family.
|
||||
- Match official encode/decode contracts exactly: input layout, latent layout,
|
||||
temporal/spatial/audio compression, scaling factor, mean/std normalization,
|
||||
posterior sampling behavior, decode output object, and frame/sample cropping.
|
||||
- Compare deterministic outputs in parity: decode outputs, encode mean/mode, or
|
||||
round-trip tensors. Do not compare stochastic samples unless the RNG path is
|
||||
explicitly controlled.
|
||||
- Use FastVideo tiling only when it preserves official numerics for the tested
|
||||
shape; disable it in config for audio or unsupported dimensions.
|
||||
- Put architecture constants on `VAEArchConfig`; put `load_encoder`,
|
||||
`load_decoder`, tiling, dtype, and pretrained path fields on `VAEConfig`.
|
||||
- Follow the production import boundary in
|
||||
`../add-model/shared/common_rules.md`.
|
||||
|
||||
## Prototype Checks
|
||||
|
||||
Follow the shared prototype success criteria.
|
||||
|
||||
## Parity-Debug Loop
|
||||
|
||||
Run the shared parity-debug loop. The component test command is:
|
||||
|
||||
```bash
|
||||
pytest <parity_test> -v -s
|
||||
```
|
||||
|
||||
For numerical drift, check normalization, latent scaling, posterior mode vs
|
||||
sample, channel order, and temporal/spatial/audio cropping before changing
|
||||
layers.
|
||||
|
||||
## Escape Hatches
|
||||
|
||||
Follow `../add-model/shared/common_rules.md` and the component-specific guidance
|
||||
in `../add-model/shared/component_skill_common.md`. VAE-specific ask cases include
|
||||
dropping an encode/decode path, accepting an unsupported private op, or choosing
|
||||
between incompatible official VAE definitions.
|
||||
|
||||
## Handoff
|
||||
|
||||
Return `../add-model/contracts/component_skill_handoff.md` following the common
|
||||
handoff rules in `../add-model/shared/component_skill_common.md`.
|
||||
@@ -0,0 +1,119 @@
|
||||
---
|
||||
name: add-model-05-port-encoder
|
||||
description: Use during /add-model Phase 4 or Phase 6 to prototype or parity-debug one FastVideo-native text, image, audio, or compound encoder component.
|
||||
---
|
||||
|
||||
# Add Model Port Encoder
|
||||
|
||||
## Goal
|
||||
|
||||
Prototype or parity-debug one encoder or encoder-like conditioner in
|
||||
FastVideo-native code. Use this for text encoders, image encoders, audio
|
||||
encoders, and compound conditioners that fit the encoder config/loader bucket.
|
||||
|
||||
## Inputs
|
||||
|
||||
Follow `../add-model/shared/component_skill_common.md` and require the complete
|
||||
packet from `../add-model/contracts/component_context.md`.
|
||||
|
||||
Encoder-specific packet fields:
|
||||
|
||||
- `component`: encoder or encoder-like conditioner name.
|
||||
- `parity_test`: `tests/local_tests/encoders/test_<family>_<component>_parity.py`.
|
||||
- `weights`: converted encoder dir, HF subfolder, or external HF id.
|
||||
- `target_files`: `fastvideo/models/encoders/<arch_or_family>.py` and
|
||||
`fastvideo/configs/models/encoders/<arch_or_family>.py`.
|
||||
|
||||
## Modes
|
||||
|
||||
Use the common prototype and parity-debug modes from
|
||||
`../add-model/shared/component_skill_common.md`.
|
||||
|
||||
Encoder-specific prototype concerns include tokenizer kwargs, hidden-state
|
||||
extraction, output packing, connector order, and external/passthrough weight
|
||||
needs.
|
||||
|
||||
## Reuse Proof
|
||||
|
||||
Apply the shared reuse proof. Encoder-specific comparison must include tokenizer
|
||||
contracts, hidden-state extraction, masks, positional IDs, output packing,
|
||||
connector/projection ordering, passthrough paths, and returned dataclass shape.
|
||||
|
||||
## Existing FastVideo Patterns
|
||||
|
||||
- Base classes: `TextEncoder` and `ImageEncoder` in
|
||||
`fastvideo/models/encoders/base.py`.
|
||||
- Output type: `BaseEncoderOutput`.
|
||||
- Config bases: `TextEncoderConfig`, `ImageEncoderConfig`,
|
||||
`TextEncoderArchConfig`, and `ImageEncoderArchConfig` in
|
||||
`fastvideo/configs/models/encoders/base.py`.
|
||||
- Use the matching encoder config bucket. Wrong bucket inheritance can typecheck
|
||||
but fail during pipeline wiring.
|
||||
- Config export: add the config to
|
||||
`fastvideo/configs/models/encoders/__init__.py`.
|
||||
- Registry discovery: set `EntryClass = <ClassName>` or a list of class names in
|
||||
the model file.
|
||||
- Reference examples: native `t5.py`, `clip.py`, `siglip.py`, `llama.py`,
|
||||
`qwen2_5.py`, `gemma.py`, and compound `stable_audio_conditioner.py`.
|
||||
- Layer guidance: `fastvideo/layers/AGENTS.md`.
|
||||
|
||||
## Implementation Rules
|
||||
|
||||
- Reuse tokenizers and pure data utilities when needed, but do not add runtime
|
||||
third-party model-class imports as a placeholder for a component that owns
|
||||
weights or numerical behavior.
|
||||
- For LLM-style encoders, follow existing tensor-parallel patterns such as
|
||||
`QKVParallelLinear`, `MergedColumnParallelLinear`, `RowParallelLinear`,
|
||||
`VocabParallelEmbedding`, and `RMSNorm` when matching native examples.
|
||||
- Match official hidden-state extraction exactly: layer index, pooled output,
|
||||
attention mask dtype, padding side, truncation, special tokens, final norm,
|
||||
output_hidden_states, and returned tuple/dataclass shape.
|
||||
- For connector or conditioner modules, preserve sub-conditioner order and the
|
||||
exact packing of cross-attention tokens, masks, and global conditioning.
|
||||
- Put tokenizer kwargs and architecture constants on the arch config when they
|
||||
affect numerical behavior.
|
||||
- If an external HF encoder is explicitly accepted as a lazy wrapper, keep it
|
||||
isolated, document why it is not a native port, and still require parity for
|
||||
the wrapper's output contract.
|
||||
|
||||
Hybrid external-HF encoder checklist:
|
||||
|
||||
- Put external model folders in passthrough subfolders such as
|
||||
`text_encoder/<external_name>/`, or record a root `model_index.json` path field
|
||||
that the loader resolves to a local directory.
|
||||
- Keep external model parameters out of the FastVideo-owned state-dict surface
|
||||
when the external model is loaded lazily from its own HF files.
|
||||
- Convert and strict-check only the FastVideo-owned connector/projection weights;
|
||||
document external model weights as passthrough.
|
||||
- Add parity for the wrapper's final output contract and, when useful, a narrower
|
||||
connector-only parity test that labels its scope as
|
||||
`implementation_subcomponent`.
|
||||
- Verify the production loader resolves the same external path used by the
|
||||
pipeline, not just the direct class used in the parity test.
|
||||
|
||||
## Prototype Checks
|
||||
|
||||
Follow the shared prototype success criteria.
|
||||
|
||||
## Parity-Debug Loop
|
||||
|
||||
Run the shared parity-debug loop. The component test command is:
|
||||
|
||||
```bash
|
||||
pytest <parity_test> -v -s
|
||||
```
|
||||
|
||||
For numerical drift, check tokenization, masks, hidden-state selection,
|
||||
positional IDs, dtype/autocast, and output packing before changing layers.
|
||||
|
||||
## Escape Hatches
|
||||
|
||||
Follow `../add-model/shared/common_rules.md` and the component-specific guidance
|
||||
in `../add-model/shared/component_skill_common.md`. Encoder-specific ask cases
|
||||
include accepting private model-code execution, choosing between incompatible
|
||||
tokenizer/encoder references, or dropping a required conditioning stream.
|
||||
|
||||
## Handoff
|
||||
|
||||
Return `../add-model/contracts/component_skill_handoff.md` following the common
|
||||
handoff rules in `../add-model/shared/component_skill_common.md`.
|
||||
@@ -0,0 +1,111 @@
|
||||
---
|
||||
name: add-model-06-port-generic
|
||||
description: Use during /add-model Phase 4 or Phase 6 to prototype or parity-debug one non-DiT, non-VAE, non-encoder FastVideo component.
|
||||
---
|
||||
|
||||
# Add Model Port Generic
|
||||
|
||||
## Goal
|
||||
|
||||
Prototype or parity-debug one scheduler, conditioner, upsampler, vocoder,
|
||||
adapter, preprocessor, or unknown component in FastVideo-native code.
|
||||
|
||||
## Inputs
|
||||
|
||||
Follow `../add-model/shared/component_skill_common.md` and require the complete
|
||||
packet from `../add-model/contracts/component_context.md`.
|
||||
|
||||
Generic-component packet fields:
|
||||
|
||||
- `component`: component name.
|
||||
- `component_type`: scheduler, conditioner, upsampler, vocoder, adapter,
|
||||
preprocessor, or unknown.
|
||||
- `parity_test`: `tests/local_tests/<bucket>/test_<family>_<component>_parity.py`.
|
||||
- `weights`: converted component dir, HF subfolder, or none.
|
||||
- `target_files`: matching `fastvideo/models/` and `fastvideo/configs/models/`
|
||||
bucket files when applicable.
|
||||
|
||||
## Modes
|
||||
|
||||
Use the common prototype and parity-debug modes from
|
||||
`../add-model/shared/component_skill_common.md`.
|
||||
|
||||
Generic-component prototype concerns include stateless/stateful ambiguity,
|
||||
missing loader buckets, source prefixes, mutable scheduler state, and output
|
||||
container shape.
|
||||
|
||||
## Reuse Proof
|
||||
|
||||
Apply the shared reuse proof. Generic-component comparison must include mutable
|
||||
state, scaling constants, scheduler/conditioner semantics, output containers, and
|
||||
whether the component owns state or is stateless.
|
||||
|
||||
## Existing FastVideo Patterns
|
||||
|
||||
- Schedulers live under `fastvideo/models/schedulers/` and expose `EntryClass`.
|
||||
- Upsamplers use `fastvideo/models/upsamplers/` plus configs under
|
||||
`fastvideo/configs/models/upsamplers/`; see `hunyuan15.py`.
|
||||
- Vocoders and audio-specific modules can live under `fastvideo/models/audio/`
|
||||
with configs under `fastvideo/configs/models/audio/`; see `ltx2_audio_vae.py`.
|
||||
- Compound conditioners may fit the encoder bucket when the pipeline loader uses
|
||||
`ConditionerLoader`; see `stable_audio_conditioner.py`.
|
||||
- Registry discovery uses `EntryClass`; config bucket exports are required when
|
||||
pipeline configs import them by bucket.
|
||||
- Use the narrowest matching config bucket. Wrong bucket inheritance can typecheck
|
||||
but fail during pipeline wiring.
|
||||
- Layer guidance: `fastvideo/layers/AGENTS.md`.
|
||||
|
||||
## Bucket Decision
|
||||
|
||||
- If the component is a transformer/DiT, stop and use `add-model-03-port-dit`.
|
||||
- If the component is a VAE/autoencoder, stop and use `add-model-04-port-vae`.
|
||||
- If the component is a text/image/audio encoder or encoder-like conditioner,
|
||||
stop and use `add-model-05-port-encoder` unless the loader requires a different
|
||||
bucket.
|
||||
- Otherwise choose the narrowest existing bucket. Add a new bucket only when no
|
||||
existing loader/config shape can represent the component without misleading
|
||||
names or unsafe runtime behavior.
|
||||
|
||||
## Implementation Rules
|
||||
|
||||
- Match official behavior, not just shapes: constructor args, default values,
|
||||
runtime flags, RNG use, dtype/autocast, scaling constants, masks, and output
|
||||
containers all matter.
|
||||
- Keep the implementation minimal and native. Do not keep a runtime import of
|
||||
the official implementation as the production component.
|
||||
- For schedulers, compare timesteps, sigmas/noise levels, step outputs, shift
|
||||
handling, prediction type, and any mutable internal state.
|
||||
- For upsamplers, compare resize mode, align_corners, residual branches,
|
||||
causal padding, normalization, and exact target-shape behavior.
|
||||
- For vocoders/audio components, compare waveform shape, sample-rate contract,
|
||||
channel order, hop length, normalization, and dtype.
|
||||
- If private upstream deps are required only for tests, keep stubs under
|
||||
`tests/local_tests/helpers/` and do not import them from production code.
|
||||
|
||||
## Prototype Checks
|
||||
|
||||
Follow the shared prototype success criteria.
|
||||
|
||||
## Parity-Debug Loop
|
||||
|
||||
Run the shared parity-debug loop. The component test command is:
|
||||
|
||||
```bash
|
||||
pytest <parity_test> -v -s
|
||||
```
|
||||
|
||||
For numerical drift, add targeted intermediate comparisons in the test to
|
||||
identify the first divergent operation.
|
||||
|
||||
## Escape Hatches
|
||||
|
||||
Follow `../add-model/shared/common_rules.md` and the component-specific guidance
|
||||
in `../add-model/shared/component_skill_common.md`. Generic-component ask cases
|
||||
include creating a new loader bucket, accepting an unsupported private op,
|
||||
choosing between incompatible official definitions, or dropping a required
|
||||
component.
|
||||
|
||||
## Handoff
|
||||
|
||||
Return `../add-model/contracts/component_skill_handoff.md` following the common
|
||||
handoff rules in `../add-model/shared/component_skill_common.md`.
|
||||
@@ -0,0 +1,186 @@
|
||||
---
|
||||
name: add-model-07-conversion
|
||||
description: Use during /add-model Phase 5 to write and verify a FastVideo checkpoint conversion script after native component prototypes expose FastVideo state-dict keys/shapes.
|
||||
---
|
||||
|
||||
# Add Model Conversion
|
||||
|
||||
## Goal
|
||||
|
||||
Convert official weights into a FastVideo-loadable component layout after Phase 4
|
||||
native prototypes exist. The conversion script owns parameter mapping, component
|
||||
splitting, passthrough assets, config emission, and strict-load verification.
|
||||
|
||||
## Inputs
|
||||
|
||||
Follow `../add-model/shared/common_rules.md` for token/auth safety, state files,
|
||||
escape hatches, production boundaries, and skip/pass semantics.
|
||||
|
||||
Require the initial request from
|
||||
`../add-model/contracts/conversion_request.md`.
|
||||
|
||||
If the FastVideo key/shape dump is missing, return to `/add-model` Phase 4. Do
|
||||
not write a final mapping against an unimplemented component.
|
||||
|
||||
For Phase 6 retry requests from component skills, also require the retry shape
|
||||
from `../add-model/contracts/conversion_request.md`.
|
||||
|
||||
## Output
|
||||
|
||||
- `scripts/checkpoint_conversion/<family>_to_diffusers.py`.
|
||||
- `converted_weights/<family>/` with `model_index.json` and per-component
|
||||
subfolders.
|
||||
- Updated `tests/local_tests/<model_family>/README.md` with conversion command,
|
||||
source layout, output path, and strict-load status.
|
||||
- Updated `tests/local_tests/<model_family>/PORT_STATUS.md` with conversion
|
||||
state, retry history, open questions, and issues/blockers.
|
||||
|
||||
## Reference Scripts
|
||||
|
||||
- `scripts/checkpoint_conversion/convert_ltx2_weights.py`: component prefix
|
||||
splitting, metadata config extraction, passthrough Gemma/tokenizer assets, and
|
||||
optional component-only output.
|
||||
- `scripts/checkpoint_conversion/stable_audio_to_diffusers.py`: monolithic
|
||||
`model.safetensors` split into transformer/VAE/conditioner, plus copied
|
||||
passthrough subfolders. Use this shape for single-checkpoint official repos.
|
||||
- `scripts/checkpoint_conversion/convert_gamecraft_full.py`: separate official
|
||||
sources for transformer, VAE, text encoders, tokenizers, scheduler, and root
|
||||
`model_index.json`.
|
||||
- `scripts/checkpoint_conversion/longcat_to_fastvideo.py`: fused QKV/KV split,
|
||||
renamed native transformer weights, and copied existing Diffusers components.
|
||||
- `scripts/checkpoint_conversion/pt_to_safetensors.py`: simple `.pt` extraction
|
||||
helper for nested checkpoint dictionaries.
|
||||
|
||||
## Source Layout Decision
|
||||
|
||||
Choose exactly one primary layout:
|
||||
|
||||
| Layout | Conversion behavior |
|
||||
|---|---|
|
||||
| `diffusers` | Usually no tensor remap; verify configs/classes and copy or update `_class_name` only when needed. |
|
||||
| `raw_official` | Convert a raw official checkpoint file or directory. Choose explicit component ownership before writing output. |
|
||||
| `separate_components` | Convert/copy each component from its own file or directory. |
|
||||
| `monolithic` | Load one model checkpoint and split state dict by authoritative prefixes into component buckets. |
|
||||
| `mixed` | Convert some components and copy passthrough components such as tokenizers, text encoders, schedulers, or already-Diffusers VAE dirs. |
|
||||
| `custom` | Document why none of the above fits before writing conversion code. |
|
||||
|
||||
Monolithic checkpoints need explicit prefix ownership. For example, Stable Audio
|
||||
uses one `model.safetensors` with DiT, pretransform/VAE, and conditioner keys;
|
||||
the converter splits those keys into FastVideo component subfolders and writes
|
||||
per-component configs.
|
||||
|
||||
## Script Shape
|
||||
|
||||
Start from `templates/family_to_diffusers.py` or the closest reference script.
|
||||
Keep the script explicit and reviewable:
|
||||
|
||||
- `COMPONENT_SPECS` or `COMPONENT_PREFIXES` declares component ownership.
|
||||
- `PARAM_NAME_MAP` declares key renames.
|
||||
- `SKIP_PATTERNS` declares intentionally dropped training-only keys.
|
||||
- tensor split/fuse helpers are named by operation, e.g. `split_qkv`.
|
||||
- `build_component_configs(...)` writes loader-compatible config files. Most
|
||||
model components use `config.json`; schedulers use `scheduler_config.json`.
|
||||
- `build_model_index(...)` writes a root `model_index.json` matching the target
|
||||
FastVideo pipeline and component classes.
|
||||
- verification reports missing, unexpected, skipped, unchanged, renamed, and
|
||||
shape-mismatched keys.
|
||||
|
||||
`model_index.json` library tokens must match FastVideo loaders:
|
||||
|
||||
- standard native DiT/VAE/audio/vocoder/upsampler components loaded by existing
|
||||
Diffusers-style loaders usually use `"diffusers"` with a FastVideo
|
||||
`_class_name` in the component `config.json`;
|
||||
- text encoders, tokenizers, image encoders, processors, and feature extractors
|
||||
usually use `"transformers"`;
|
||||
- `conditioner` currently expects `"fastvideo"`;
|
||||
- use fully qualified `"fastvideo.<module>"` only when intentionally relying on
|
||||
the custom fastvideo-library escape path;
|
||||
- do not write bare `"fastvideo"` for transformer, VAE, or other loaders that
|
||||
expect `"diffusers"` unless the loader explicitly expects it.
|
||||
|
||||
## Mapping Rules
|
||||
|
||||
- Use Phase 4 key/shape dumps to derive mappings. Do not guess from official key
|
||||
names alone.
|
||||
- Preserve each component's official file paths, parity test path, and prototype
|
||||
concerns in comments or structured constants near the mapping that uses them.
|
||||
- Every official inference parameter should be mapped, copied through, or listed
|
||||
as intentionally skipped with a reason.
|
||||
- Every FastVideo prototype parameter should receive a tensor or be listed as an
|
||||
intentional external/passthrough parameter.
|
||||
- Shape matches are necessary but not sufficient; check semantic pairing for
|
||||
Q/K/V, gate/up/down, norm scale/bias, LoRA/base, and modality-specific heads.
|
||||
- If official and FastVideo fuse or split tensors differently, convert tensors in
|
||||
the script rather than changing production code to match checkpoint quirks.
|
||||
|
||||
## Verification
|
||||
|
||||
Run conversion locally, then verify before returning to Phase 6. For retry
|
||||
requests, update the mapping, rerun conversion, and refresh only the implicated
|
||||
converted component when safe; otherwise rerun the full conversion.
|
||||
|
||||
```bash
|
||||
python scripts/checkpoint_conversion/<family>_to_diffusers.py \
|
||||
--src <official_weights> \
|
||||
--revision <hf_revision> \
|
||||
--dst converted_weights/<model_family>
|
||||
```
|
||||
|
||||
Omit `--revision` for local sources or when prep recorded `default` / `none`.
|
||||
|
||||
Minimum output layout:
|
||||
|
||||
```text
|
||||
converted_weights/<family>/
|
||||
model_index.json
|
||||
transformer/config.json
|
||||
transformer/*.safetensors
|
||||
vae/config.json
|
||||
vae/*.safetensors
|
||||
scheduler/scheduler_config.json as needed
|
||||
text_encoder/... as needed
|
||||
```
|
||||
|
||||
Required checks:
|
||||
|
||||
- `model_index.json` exists and lists every required component.
|
||||
- Each converted component has the config filename its loader expects and
|
||||
safetensors weights when it owns weights. Scheduler dirs require
|
||||
`scheduler_config.json`; most other native model dirs use `config.json`.
|
||||
- Weight filenames may vary by loader: transformer and VAE loaders glob all
|
||||
`*.safetensors`; text encoders may load `*.safetensors`, `*.bin`, and
|
||||
sometimes `*.pt`; `conditioner` currently expects
|
||||
`diffusion_pytorch_model.safetensors`. Use the loader's actual accepted layout
|
||||
rather than assuming one global filename.
|
||||
- Passthrough components are copied or referenced deliberately.
|
||||
- Each emitted component config validates through the same path production
|
||||
loaders use. Instantiate the relevant config and call `update_model_arch(...)`
|
||||
or `update_model_config(...)` with the emitted JSON so unknown keys fail during
|
||||
conversion, not at pipeline load time.
|
||||
- Record production loader strictness for every stateful component. If the loader
|
||||
intentionally uses non-strict loading, add explicit missing/unexpected-key
|
||||
assertions in the parity test and document exactly which keys are allowed.
|
||||
- Each new FastVideo component strict-loads converted weights where its production
|
||||
loader is strict. If strict loading is impossible, record the exact allowed
|
||||
missing/unexpected keys and why they are not inference weights.
|
||||
- Retry fixes include the original component parity evidence and the new
|
||||
strict-load result in `local_tests_readme` so the component subagent can resume
|
||||
without rediscovering context.
|
||||
- `local_tests_readme` records the command, output directory, and strict-load
|
||||
result.
|
||||
|
||||
Do not chase numerical parity in this skill except to identify a conversion
|
||||
mapping bug. Long parity-debug loops belong to `/add-model` Phase 6.
|
||||
|
||||
## Escape Hatches
|
||||
|
||||
Follow `../add-model/shared/common_rules.md`. Conversion-specific ask cases
|
||||
include selecting between incompatible official checkpoints, publishing/uploading
|
||||
weights, overwriting an existing converted repo not created by this run,
|
||||
accepting non-strict missing inference weights, or dropping a component/output
|
||||
from scope.
|
||||
|
||||
## Handoff
|
||||
|
||||
Return `../add-model/contracts/conversion_handoff.md` and update the shared state
|
||||
files before handoff.
|
||||
@@ -0,0 +1,293 @@
|
||||
#!/usr/bin/env python3
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
"""Convert <model_family> official weights to a FastVideo Diffusers-style tree.
|
||||
|
||||
This template supports both separate component sources and a monolithic pipeline
|
||||
checkpoint that must be split by component prefix. Replace every TODO before
|
||||
using it for a real port.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import shutil
|
||||
from collections import OrderedDict
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
import torch
|
||||
from safetensors import safe_open
|
||||
from safetensors.torch import load_file, save_file
|
||||
|
||||
try:
|
||||
from huggingface_hub import snapshot_download
|
||||
except ImportError: # pragma: no cover - optional local conversion dependency
|
||||
snapshot_download = None
|
||||
|
||||
# TODO: fill with authoritative component prefixes for monolithic checkpoints.
|
||||
# Example: {"model.model.": "transformer", "pretransform.model.": "vae"}
|
||||
COMPONENT_PREFIXES: dict[str, str] = {}
|
||||
|
||||
# TODO: fill with component-specific source paths for separate-component repos.
|
||||
# Example: {"transformer": "transformer/model.safetensors", "vae": "vae/"}
|
||||
SEPARATE_COMPONENT_PATHS: dict[str, str] = {}
|
||||
|
||||
# TODO: copy passthrough dirs that are already loadable by FastVideo/Diffusers.
|
||||
PASSTHROUGH_SUBFOLDERS: tuple[str, ...] = ("tokenizer", "scheduler")
|
||||
|
||||
# TODO: add regex renames derived from Phase 4 key/shape dumps.
|
||||
PARAM_NAME_MAP: dict[str, str] = {}
|
||||
|
||||
# TODO: include training-only or dynamically-computed keys that must not load.
|
||||
SKIP_PATTERNS: tuple[str, ...] = ()
|
||||
|
||||
|
||||
def _hf_token() -> str | None:
|
||||
return (os.environ.get("HF_TOKEN") or os.environ.get("HUGGINGFACE_HUB_TOKEN") or os.environ.get("HF_API_KEY"))
|
||||
|
||||
|
||||
def resolve_src(src: str, revision: str | None) -> Path:
|
||||
if os.path.exists(src):
|
||||
return Path(src)
|
||||
if snapshot_download is None:
|
||||
raise RuntimeError("huggingface_hub is required when --src is a repo id")
|
||||
return Path(snapshot_download(repo_id=src, revision=revision, token=_hf_token()))
|
||||
|
||||
|
||||
def load_checkpoint(path: Path) -> dict[str, torch.Tensor]:
|
||||
if path.is_dir():
|
||||
weights: dict[str, torch.Tensor] = {}
|
||||
for shard in sorted(path.glob("*.safetensors")):
|
||||
weights.update(load_file(str(shard)))
|
||||
if weights:
|
||||
return weights
|
||||
raise FileNotFoundError(f"No safetensors found in {path}")
|
||||
|
||||
if path.suffix == ".safetensors":
|
||||
return load_file(str(path))
|
||||
|
||||
checkpoint = torch.load(path, map_location="cpu", weights_only=True)
|
||||
if isinstance(checkpoint, dict):
|
||||
for key in ("state_dict", "model_state_dict", "model", "module", "ema"):
|
||||
if key in checkpoint and isinstance(checkpoint[key], dict):
|
||||
return checkpoint[key]
|
||||
return checkpoint
|
||||
raise TypeError(f"Unsupported checkpoint type: {type(checkpoint)!r}")
|
||||
|
||||
|
||||
def should_skip_key(key: str) -> bool:
|
||||
return any(re.search(pattern, key) for pattern in SKIP_PATTERNS)
|
||||
|
||||
|
||||
def apply_mapping(key: str) -> str | None:
|
||||
if should_skip_key(key):
|
||||
return None
|
||||
for pattern, replacement in PARAM_NAME_MAP.items():
|
||||
if re.match(pattern, key):
|
||||
return re.sub(pattern, replacement, key)
|
||||
return key
|
||||
|
||||
|
||||
def split_monolithic(state: dict[str, torch.Tensor], ) -> dict[str, OrderedDict[str, torch.Tensor]]:
|
||||
components: dict[str, OrderedDict[str, torch.Tensor]] = {
|
||||
name: OrderedDict()
|
||||
for name in set(COMPONENT_PREFIXES.values())
|
||||
}
|
||||
intentionally_skipped: list[str] = []
|
||||
unowned: list[str] = []
|
||||
for key, value in state.items():
|
||||
if should_skip_key(key):
|
||||
intentionally_skipped.append(key)
|
||||
continue
|
||||
for prefix, component in COMPONENT_PREFIXES.items():
|
||||
if key.startswith(prefix):
|
||||
mapped = apply_mapping(key[len(prefix):])
|
||||
if mapped is not None:
|
||||
components[component][mapped] = value
|
||||
break
|
||||
else:
|
||||
unowned.append(key)
|
||||
if unowned:
|
||||
sample = ", ".join(unowned[:10])
|
||||
raise ValueError(f"Unowned monolithic keys: {len(unowned)}. "
|
||||
f"Add COMPONENT_PREFIXES or SKIP_PATTERNS entries. Sample: {sample}")
|
||||
if intentionally_skipped:
|
||||
print(f"Intentionally skipped {len(intentionally_skipped)} keys")
|
||||
return {name: weights for name, weights in components.items() if weights}
|
||||
|
||||
|
||||
def load_separate_components(src_dir: Path) -> dict[str, OrderedDict[str, torch.Tensor]]:
|
||||
components: dict[str, OrderedDict[str, torch.Tensor]] = {}
|
||||
for component, rel_path in SEPARATE_COMPONENT_PATHS.items():
|
||||
state = load_checkpoint(src_dir / rel_path)
|
||||
converted: OrderedDict[str, torch.Tensor] = OrderedDict()
|
||||
for key, value in state.items():
|
||||
mapped = apply_mapping(key)
|
||||
if mapped is not None:
|
||||
converted[mapped] = value
|
||||
components[component] = converted
|
||||
return components
|
||||
|
||||
|
||||
def build_component_configs(_src_dir: Path) -> dict[str, dict[str, Any]]:
|
||||
# TODO: emit config content accepted by FastVideo loaders. Most components use
|
||||
# config.json; schedulers use scheduler_config.json.
|
||||
return {
|
||||
"transformer": {
|
||||
"_class_name": "<FastVideoTransformerClass>"
|
||||
},
|
||||
"vae": {
|
||||
"_class_name": "<FastVideoVAEClass>"
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def config_filename(component: str) -> str:
|
||||
if component == "scheduler":
|
||||
return "scheduler_config.json"
|
||||
return "config.json"
|
||||
|
||||
|
||||
def source_label(src: str) -> str:
|
||||
if os.path.exists(src):
|
||||
return Path(src).name
|
||||
return src
|
||||
|
||||
|
||||
def build_model_index(
|
||||
src: str,
|
||||
revision: str | None,
|
||||
available_components: set[str],
|
||||
) -> dict[str, Any]:
|
||||
# TODO: match the target pipeline and every required component.
|
||||
index: dict[str, Any] = {
|
||||
"_class_name": "<FastVideoPipelineClass>",
|
||||
"_diffusers_version": "0.30.0",
|
||||
"_fastvideo_converted_from": source_label(src),
|
||||
# Existing transformer/VAE loaders expect "diffusers" even when
|
||||
# _class_name names a FastVideo-native class registered in FastVideo.
|
||||
"transformer": ["diffusers", "<FastVideoTransformerClass>"],
|
||||
"vae": ["diffusers", "<FastVideoVAEClass>"],
|
||||
}
|
||||
if revision:
|
||||
index["_fastvideo_converted_revision"] = revision
|
||||
return {key: value for key, value in index.items() if key.startswith("_") or key in available_components}
|
||||
|
||||
|
||||
def validate_component_configs(configs: dict[str, dict[str, Any]]) -> None:
|
||||
# TODO: instantiate each FastVideo config and call update_model_arch(...) or
|
||||
# update_model_config(...) with this JSON so unknown emitted keys fail here.
|
||||
placeholder_configs = [name for name, config in configs.items() if "<" in json.dumps(config)]
|
||||
if placeholder_configs:
|
||||
raise ValueError(f"Replace config placeholders for: {placeholder_configs}")
|
||||
|
||||
|
||||
def verify_conversion(
|
||||
dst_dir: Path,
|
||||
components: dict[str, OrderedDict[str, torch.Tensor]],
|
||||
) -> None:
|
||||
del dst_dir, components
|
||||
# TODO: load each emitted stateful component through its production loader and
|
||||
# assert strict load, or document exact allowed missing/unexpected keys.
|
||||
raise NotImplementedError("Implement production config validation and strict-load checks")
|
||||
|
||||
|
||||
def write_component(
|
||||
dst_dir: Path,
|
||||
name: str,
|
||||
state: dict[str, torch.Tensor],
|
||||
config: dict[str, Any] | None,
|
||||
) -> None:
|
||||
component_dir = dst_dir / name
|
||||
if component_dir.exists() and any(component_dir.iterdir()):
|
||||
shutil.rmtree(component_dir)
|
||||
component_dir.mkdir(parents=True, exist_ok=True)
|
||||
save_file(dict(state), str(component_dir / "diffusion_pytorch_model.safetensors"))
|
||||
if config is not None:
|
||||
config_path = component_dir / config_filename(name)
|
||||
with config_path.open("w", encoding="utf-8") as f:
|
||||
json.dump(config, f, indent=2)
|
||||
f.write("\n")
|
||||
print(f"Wrote {name}: {len(state)} tensors")
|
||||
|
||||
|
||||
def copy_passthrough(src_dir: Path, dst_dir: Path) -> list[str]:
|
||||
copied: list[str] = []
|
||||
for subfolder in PASSTHROUGH_SUBFOLDERS:
|
||||
src = src_dir / subfolder
|
||||
if not src.is_dir():
|
||||
continue
|
||||
dst = dst_dir / subfolder
|
||||
if dst.exists():
|
||||
shutil.rmtree(dst)
|
||||
shutil.copytree(src, dst)
|
||||
copied.append(subfolder)
|
||||
print(f"Copied {subfolder}/")
|
||||
return copied
|
||||
|
||||
|
||||
def default_monolithic_checkpoint(src_path: Path) -> Path:
|
||||
if src_path.is_file():
|
||||
return src_path
|
||||
return src_path / "model.safetensors"
|
||||
|
||||
|
||||
def convert(
|
||||
src: str,
|
||||
dst: str,
|
||||
layout: str,
|
||||
revision: str | None,
|
||||
) -> None:
|
||||
src_path = resolve_src(src, revision)
|
||||
dst_dir = Path(dst)
|
||||
dst_dir.mkdir(parents=True, exist_ok=True)
|
||||
model_index_path = dst_dir / "model_index.json"
|
||||
|
||||
if layout in {"monolithic", "raw_official"}:
|
||||
# TODO: replace model.safetensors with the official monolithic file name.
|
||||
components = split_monolithic(load_checkpoint(default_monolithic_checkpoint(src_path)))
|
||||
elif layout in {"separate_components", "mixed"}:
|
||||
if not src_path.is_dir():
|
||||
raise ValueError(f"{layout} layout requires a source directory: {src_path}")
|
||||
components = load_separate_components(src_path)
|
||||
else:
|
||||
raise ValueError(f"Unsupported template layout: {layout}")
|
||||
|
||||
copied = (copy_passthrough(src_path, dst_dir) if src_path.is_dir() else [])
|
||||
configs = build_component_configs(src_path if src_path.is_dir() else src_path.parent)
|
||||
validate_component_configs(configs)
|
||||
for name, state in components.items():
|
||||
write_component(dst_dir, name, state, configs.get(name))
|
||||
|
||||
available = set(components) | set(copied)
|
||||
with model_index_path.open("w", encoding="utf-8") as f:
|
||||
json.dump(build_model_index(src, revision, available), f, indent=2)
|
||||
f.write("\n")
|
||||
print(f"Wrote {dst_dir / 'model_index.json'}")
|
||||
verify_conversion(dst_dir, components)
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("--src", required=True, help="HF repo id, local dir, or checkpoint path")
|
||||
parser.add_argument("--revision", help="HF branch, tag, or commit for repo sources")
|
||||
parser.add_argument(
|
||||
"--dst",
|
||||
required=True,
|
||||
help="Output converted_weights/<model_family> directory",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--layout",
|
||||
choices=("raw_official", "monolithic", "separate_components", "mixed"),
|
||||
required=True,
|
||||
help="Official source layout",
|
||||
)
|
||||
args = parser.parse_args()
|
||||
convert(args.src, args.dst, args.layout, args.revision)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,214 @@
|
||||
---
|
||||
name: add-model-08-trace
|
||||
description: Use during /add-model Phase 6 when component parity has failed and root cause requires layer-by-layer divergence analysis. Uses FastVideo activation trace first, falling back to custom hooks only for boundaries or stats the utility cannot observe.
|
||||
---
|
||||
|
||||
# Add-Model Trace
|
||||
|
||||
## Manual Invocation
|
||||
|
||||
Load this skill when `/add-model` Phase 6 component parity has failed and the
|
||||
root cause requires layer-by-layer divergence analysis. This skill is not
|
||||
auto-fired. The calling subagent (DiT, VAE, encoder, or generic port skill)
|
||||
loads it when its standard parity-debug loop hits a wall and cannot isolate
|
||||
the divergence from end-to-end tensor comparisons alone.
|
||||
|
||||
Do not load this skill for first-pass parity failures. Try weight-diff and
|
||||
end-to-end tensor comparison first. Load this skill only when those do not
|
||||
isolate the cause.
|
||||
|
||||
## Goal
|
||||
|
||||
Find the first numerical divergence point between FastVideo's port and the
|
||||
official reference, layer by layer, by instrumenting both sides at matching
|
||||
tensor boundaries. The investigation must leave zero source residue in
|
||||
production code when it closes.
|
||||
|
||||
## When To Run
|
||||
|
||||
After a component parity test FAILS at a bf16-noise-realistic tolerance AND
|
||||
the calling subagent's first-pass debug (weight-diff, end-to-end tensor
|
||||
compare) does not isolate the cause.
|
||||
|
||||
Required inputs before starting:
|
||||
|
||||
- A working FastVideo loader for the component under investigation.
|
||||
- A working official loader, typically via
|
||||
`tests/local_tests/helpers/<family>_upstream.py::load_upstream_<component>`.
|
||||
- Shared deterministic test inputs (same tensors on both sides).
|
||||
- The component parity test file path and its current failure output.
|
||||
|
||||
## Primary Path: FastVideo Activation Trace
|
||||
|
||||
Use FastVideo's first-class activation trace before writing custom hooks:
|
||||
`fastvideo/hooks/activation_trace.py`, documented in
|
||||
`docs/contributing/activation_trace.md`.
|
||||
|
||||
Pipeline runs attach trace to the transformer during pipeline initialization.
|
||||
Component-only parity harnesses may call `attach_activation_trace(model)` from
|
||||
local test/debug code; do not add trace calls to production model code.
|
||||
|
||||
Prefix the failing parity command with a tight layer regex:
|
||||
|
||||
```bash
|
||||
FASTVIDEO_TRACE_ACTIVATIONS=1 \
|
||||
FASTVIDEO_TRACE_LAYERS="^block\.layers\.[0-9]+$" \
|
||||
FASTVIDEO_TRACE_STATS="abs_mean,sum,max,shape" \
|
||||
FASTVIDEO_TRACE_STEPS="0" \
|
||||
FASTVIDEO_TRACE_OUTPUT="/tmp/opencode/fv_trace.jsonl" \
|
||||
pytest tests/local_tests -k "parity" -v -s
|
||||
```
|
||||
|
||||
Match the layer regex to the actual `model.named_modules()` names. Empty or
|
||||
broad regexes are expensive; prefer block-level names first, then narrow to
|
||||
submodules after the first divergent block is known.
|
||||
|
||||
## Trace Compare Contract
|
||||
|
||||
One JSONL file per side. FastVideo output should use `FASTVIDEO_TRACE_OUTPUT`;
|
||||
the upstream harness should emit the same JSONL shape:
|
||||
|
||||
```json
|
||||
{"module":"block.layers.0","tensor":"out","step":0,"abs_mean":0.0123,"sum":1.0,"max":0.5,"shape":[1,16,32]}
|
||||
```
|
||||
|
||||
Compare rows by `(module, step, tensor)`. The first row whose `shape`,
|
||||
`abs_mean`, or `max` diverges beyond the component tolerance is the first broken
|
||||
boundary. Keep `FASTVIDEO_TRACE_LAYERS`, `FASTVIDEO_TRACE_STATS`, and
|
||||
`FASTVIDEO_TRACE_STEPS` identical between sides; if row order differs, sort or
|
||||
normalize before diffing.
|
||||
|
||||
## Drill-Down Loop
|
||||
|
||||
**Initial run:** trace every top-level block (`^block\.layers\.[0-9]+$` or the
|
||||
family's equivalent). Identify the first block index where `abs_mean` or `max`
|
||||
drifts beyond tolerance while earlier blocks match.
|
||||
|
||||
**Drill run:** tighten `FASTVIDEO_TRACE_LAYERS` to submodules inside the first
|
||||
divergent block: attention output, MLP projections, norm outputs, modality
|
||||
adapters, or other named boundaries exposed by `named_modules()`.
|
||||
|
||||
**Iterate:** if the first divergent operation is a free function or tensor op not
|
||||
visible as an `nn.Module`, use the fallback instrumentation hierarchy below.
|
||||
|
||||
The loop ends when the first divergent submodule or operation is identified with
|
||||
a file:line citation in the official source.
|
||||
|
||||
## Fallback Instrumentation Hierarchy
|
||||
|
||||
Use these only when activation trace cannot observe the needed boundary or
|
||||
statistic.
|
||||
|
||||
### (1) Custom forward hooks
|
||||
|
||||
`module.register_forward_hook(...)` and `register_forward_pre_hook(...)`.
|
||||
Always within `try/finally` with `handle.remove()`. Zero source residue.
|
||||
|
||||
### (2) Runtime monkey-patch
|
||||
|
||||
`module.attr = wrapped_func` or `cls.method = wrapped_method`, restored via
|
||||
`try/finally` (save original first). Use for free functions and non-Module sites
|
||||
such as activation functions (`swiglu`, `apply_rotary_emb`).
|
||||
|
||||
### (3) Source edits in FastVideo's own code
|
||||
|
||||
Only when (1) and (2) are insufficient. Track all edits within a single named
|
||||
`git stash` boundary OR a temporary branch. Run `git diff` before closing the
|
||||
investigation to confirm cleanup.
|
||||
|
||||
### (4) Source edits in official repo source
|
||||
|
||||
Allowed only when hook and monkey-patch approaches cannot capture the site.
|
||||
For git-tracked or editable official clones, use `git diff` in the clone path to
|
||||
verify cleanup. For non-editable site-packages, back up the target file before
|
||||
editing and restore it before handoff.
|
||||
|
||||
## Hypothesis Toggles
|
||||
|
||||
Use env-var-gated monkey-patches to A/B test suspect implementations without
|
||||
source edits. Pattern: `<FAMILY>_DEBUG_PATCH_<HYPOTHESIS>=1`.
|
||||
|
||||
Example from the magi-human investigation:
|
||||
|
||||
```
|
||||
MAGI_DEBUG_PATCH_LINEAR=1
|
||||
```
|
||||
|
||||
This patched `PackedExpertLinear.forward` to mirror upstream's
|
||||
`_BF16ComputeLinear` explicit-cast pattern, isolating a dtype-cast difference
|
||||
as the root cause.
|
||||
|
||||
Document all toggles in the script docstring. Each toggle must:
|
||||
|
||||
- save the original before patching;
|
||||
- restore the original in a `try/finally` block;
|
||||
- print a `[debug] Patched <ClassName>.<method>` line to stdout when active.
|
||||
|
||||
## Cleanup Gate
|
||||
|
||||
The calling agent MUST report `[cleanup-gate] PASS` on all five items before
|
||||
handoff. Do not hand off with any item unresolved.
|
||||
|
||||
1. `git diff` in the FastVideo repo: empty. No stray prints, hooks, or
|
||||
monkey-patches in production code.
|
||||
2. `git diff` in the official-repo clone (if used): empty. For non-editable
|
||||
site-packages installs: `diff original.py original.py.trace-backup` is
|
||||
empty OR `pip install --force-reinstall <pkg>` succeeded and the installed
|
||||
file matches the original.
|
||||
3. `git stash list`: only the named investigation stash (or empty). No
|
||||
unnamed stashes left from this session.
|
||||
4. No new untracked files outside `/tmp/opencode/` (logs) and the existing
|
||||
debug script directory (`tests/local_tests/transformers/` or equivalent).
|
||||
5. `mypy` clean on any production files touched during the investigation.
|
||||
|
||||
## Escape Hatches
|
||||
|
||||
Escalate to the calling bucket skill when:
|
||||
|
||||
- A forward hook on an official module raises because of a custom `forward`
|
||||
signature or varlen handler args that the hook closure cannot satisfy. The
|
||||
bucket skill has component-specific knowledge to work around this.
|
||||
- The first divergent layer is `block[0]`, meaning the divergence is in the
|
||||
adapter, modality dispatcher, coordinate embedding, or packing step before
|
||||
any block runs. Check those sites first; the bug is not in attention or MLP.
|
||||
- Per-block drift is never zero anywhere across all blocks. This usually means
|
||||
the inputs are not bit-identical between sides. Verify with a state-dict
|
||||
compare (weight-diff script) AND confirm the input tensors are the same
|
||||
object or have identical values before the forward call.
|
||||
|
||||
## Handoff
|
||||
|
||||
Return to the calling subagent with:
|
||||
|
||||
- FastVideo trace JSONL path and upstream trace JSONL path.
|
||||
- Trace settings used: `FASTVIDEO_TRACE_LAYERS`, `FASTVIDEO_TRACE_STATS`, and
|
||||
`FASTVIDEO_TRACE_STEPS`.
|
||||
- The first divergent `(module, step, tensor)` row and observed drift.
|
||||
- The upstream file:line citation where the divergence originates.
|
||||
- Fallback hook/patch verdict if activation trace could not observe the boundary.
|
||||
- Hypothesis verdict if an A/B toggle was used, for example `PATCH_LINEAR=1`.
|
||||
- Cleanup-gate status: `[cleanup-gate] PASS` or a list of unresolved items.
|
||||
|
||||
The calling agent uses this to scope the production fix in the FastVideo
|
||||
component file.
|
||||
|
||||
## References
|
||||
|
||||
- `docs/contributing/activation_trace.md` for canonical activation-trace env vars,
|
||||
JSONL output, cost model, and troubleshooting.
|
||||
- `fastvideo/hooks/activation_trace.py` for the implementation and
|
||||
`attach_activation_trace(model)` entry point.
|
||||
- `templates/block_trace_debug.py` in this skill directory: fallback custom-hook
|
||||
template when activation trace cannot observe the needed boundary or stat.
|
||||
- `tests/local_tests/transformers/_debug_magi_human_block_parity.py` in the
|
||||
FastVideo3 repo: historical worked example for custom hook/patch debugging.
|
||||
- `add-model/SKILL.md` Phase 6: the calling context for this skill.
|
||||
- `add-model-03-port-dit/SKILL.md`, `add-model-04-port-vae/SKILL.md`,
|
||||
`add-model-05-port-encoder/SKILL.md`, `add-model-06-port-generic/SKILL.md`:
|
||||
bucket-specific debug language and component-specific escape-hatch knowledge.
|
||||
|
||||
## Changelog
|
||||
|
||||
| Date | Change |
|
||||
|---|---|
|
||||
| 2026-05-01 | Initial skill extracted from `_debug_magi_human_block_parity.py` pattern. |
|
||||
@@ -0,0 +1,334 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
"""Per-block divergence debugger template for FastVideo model ports.
|
||||
|
||||
Run directly (not a pytest test):
|
||||
python tests/local_tests/transformers/_debug_<family>_<component>_parity.py
|
||||
|
||||
Generalizes: tests/local_tests/transformers/_debug_magi_human_block_parity.py
|
||||
|
||||
Fill FAMILY, COMPONENT, and the two loader functions. Run once for the initial
|
||||
drift table, then set <FAMILY>_DEBUG_DRILL_LAYER=NN to drill into submodules.
|
||||
Add <FAMILY>_DEBUG_PATCH_<HYPOTHESIS>=1 to A/B test a suspect implementation.
|
||||
|
||||
CLEANUP: all hooks removed in try/finally; monkey-patches restored in
|
||||
try/finally; source edits tracked in a named git stash. Zero source residue.
|
||||
See add-model-08-trace/SKILL.md for the full cleanup gate checklist.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import gc
|
||||
import os
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
import torch
|
||||
|
||||
FAMILY: str = "<family>" # e.g. "magi_human", "ltx2", "wan"
|
||||
COMPONENT: str = "<component>" # e.g. "dit", "vae", "encoder"
|
||||
DRILL_LAYER_ENV: str = "<FAMILY>_DEBUG_DRILL_LAYER"
|
||||
HYPOTHESIS_ENV: str = "<FAMILY>_DEBUG_PATCH_<HYPOTHESIS>"
|
||||
REL_THRESHOLD: float = 0.005 # 0.5% abs_mean drift flags a block as divergent
|
||||
LOG_DIR: Path = Path("/tmp/opencode")
|
||||
|
||||
REPO_ROOT = Path(__file__).resolve().parents[3]
|
||||
sys.path.insert(0, str(REPO_ROOT))
|
||||
|
||||
|
||||
def load_official(device: torch.device) -> torch.nn.Module:
|
||||
"""Load the official upstream model. TODO: implement for your family.
|
||||
|
||||
Example (magi-human):
|
||||
from tests.local_tests.helpers.magi_human_upstream import install_stubs, load_upstream_dit
|
||||
install_stubs()
|
||||
return load_upstream_dit(base_shard_dir, device=device, dtype=None)
|
||||
"""
|
||||
raise NotImplementedError(f"Fill load_official() for {FAMILY}/{COMPONENT}.")
|
||||
|
||||
|
||||
def load_fastvideo(device: torch.device) -> torch.nn.Module:
|
||||
"""Load the FastVideo-native model. TODO: implement for your family.
|
||||
|
||||
Example (magi-human):
|
||||
from fastvideo.configs.models.dits.magi_human import MagiHumanVideoConfig
|
||||
from fastvideo.models.dits.magi_human import MagiHumanDiT
|
||||
from safetensors.torch import load_file; import glob
|
||||
fv = MagiHumanDiT(MagiHumanVideoConfig())
|
||||
state = {}
|
||||
for shard in sorted(glob.glob(str(transformer_dir / "*.safetensors"))): state.update(load_file(shard))
|
||||
fv.load_state_dict(state, strict=False); return fv.to(device).eval()
|
||||
"""
|
||||
raise NotImplementedError(f"Fill load_fastvideo() for {FAMILY}/{COMPONENT}.")
|
||||
|
||||
|
||||
def build_inputs(device: torch.device) -> dict[str, Any]:
|
||||
"""Return deterministic inputs shared by both sides. TODO: replace.
|
||||
|
||||
Both sides must receive the SAME tensors (clone before each forward call).
|
||||
Non-identical inputs cause non-zero drift everywhere.
|
||||
"""
|
||||
torch.manual_seed(0)
|
||||
return {"x": torch.randn(64, 1024, dtype=torch.bfloat16, device=device)}
|
||||
|
||||
|
||||
def _stat(name: str, t: torch.Tensor) -> dict:
|
||||
f = t.detach().float()
|
||||
return {
|
||||
"name": name,
|
||||
"shape": tuple(t.shape),
|
||||
"abs_mean": f.abs().mean().item(),
|
||||
"sum": f.sum().item(),
|
||||
"min": f.min().item(),
|
||||
"max": f.max().item(),
|
||||
}
|
||||
|
||||
|
||||
def _attach_block_hooks(
|
||||
model: torch.nn.Module,
|
||||
label: str,
|
||||
log: list[dict],
|
||||
tensors: dict[str, torch.Tensor] | None = None,
|
||||
drill_layer: int | None = None,
|
||||
) -> list[Any]:
|
||||
"""Return hook handles. Caller MUST remove them in try/finally."""
|
||||
handles: list[Any] = []
|
||||
|
||||
def _hook(name: str):
|
||||
|
||||
def fn(_module, _inputs, outputs):
|
||||
t = outputs[0] if isinstance(outputs, tuple) else outputs
|
||||
if not torch.is_tensor(t):
|
||||
return
|
||||
log.append({"side": label, **_stat(name, t)})
|
||||
if tensors is not None:
|
||||
tensors[name] = t.detach().float().cpu()
|
||||
|
||||
return fn
|
||||
|
||||
def _pre_hook(name: str):
|
||||
# Pre-hooks observe a free function's output by intercepting the next
|
||||
# module's input (useful when the activation is not an nn.Module).
|
||||
def fn(_module, inputs):
|
||||
t = inputs[0] if isinstance(inputs, tuple) else inputs
|
||||
if not torch.is_tensor(t):
|
||||
return
|
||||
key = f"{name}<in>"
|
||||
log.append({"side": label, **_stat(key, t)})
|
||||
if tensors is not None:
|
||||
tensors[key] = t.detach().float().cpu()
|
||||
|
||||
return fn
|
||||
|
||||
# TODO: adapt attribute paths to your model. Remove adapter block if absent.
|
||||
if hasattr(model, "adapter"):
|
||||
handles.append(model.adapter.register_forward_hook(_hook("adapter")))
|
||||
|
||||
# TODO: adapt model.block.layers to your block container.
|
||||
# Alternatives: model.transformer.layers, model.blocks, model.layers
|
||||
block_layers = model.block.layers # type: ignore[attr-defined]
|
||||
for i, layer in enumerate(block_layers):
|
||||
handles.append(layer.register_forward_hook(_hook(f"block[{i:02d}]")))
|
||||
if drill_layer is not None and i == drill_layer:
|
||||
tag = f"L{i:02d}"
|
||||
# TODO: adapt submodule names to your layer's attributes.
|
||||
# magi-human uses: attention, mlp.pre_norm, mlp.up_gate_proj,
|
||||
# mlp.down_proj (pre+post), mlp, attn_post_norm, mlp_post_norm.
|
||||
if hasattr(layer, "attention"):
|
||||
handles.append(layer.attention.register_forward_hook(_hook(f"{tag}.attention")))
|
||||
if hasattr(layer, "mlp"):
|
||||
mlp = layer.mlp
|
||||
if hasattr(mlp, "pre_norm"):
|
||||
handles.append(mlp.pre_norm.register_forward_hook(_hook(f"{tag}.mlp.pre_norm")))
|
||||
if hasattr(mlp, "up_gate_proj"):
|
||||
handles.append(mlp.up_gate_proj.register_forward_hook(_hook(f"{tag}.mlp.up_gate_proj")))
|
||||
if hasattr(mlp, "down_proj"):
|
||||
handles.append(mlp.down_proj.register_forward_pre_hook(_pre_hook(f"{tag}.mlp.down_proj")))
|
||||
handles.append(mlp.down_proj.register_forward_hook(_hook(f"{tag}.mlp.down_proj")))
|
||||
handles.append(mlp.register_forward_hook(_hook(f"{tag}.mlp")))
|
||||
if hasattr(layer, "attn_post_norm"):
|
||||
handles.append(layer.attn_post_norm.register_forward_hook(_hook(f"{tag}.attn_post_norm")))
|
||||
if hasattr(layer, "mlp_post_norm"):
|
||||
handles.append(layer.mlp_post_norm.register_forward_hook(_hook(f"{tag}.mlp_post_norm")))
|
||||
return handles
|
||||
|
||||
|
||||
def _apply_hypothesis_patch() -> bool:
|
||||
"""Apply an optional monkey-patch gated by HYPOTHESIS_ENV. TODO: implement.
|
||||
|
||||
Pattern: save original on the class, patch, restore in _restore_hypothesis_patch().
|
||||
"""
|
||||
if os.getenv(HYPOTHESIS_ENV) != "1":
|
||||
return False
|
||||
# TODO: import FastVideo class, save original, apply patch.
|
||||
print(f"[debug] Hypothesis patch {HYPOTHESIS_ENV}=1 applied.")
|
||||
return True
|
||||
|
||||
|
||||
def _restore_hypothesis_patch() -> None:
|
||||
if os.getenv(HYPOTHESIS_ENV) != "1":
|
||||
return
|
||||
# TODO: restore original, e.g.: _mod.TargetClass.method = _mod._ORIGINAL_METHOD
|
||||
|
||||
|
||||
def _write_log(entries: list[dict], path: Path) -> None:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
with open(path, "w") as f:
|
||||
for e in entries:
|
||||
f.write(f"{e['name']} {e['shape']} "
|
||||
f"{e['abs_mean']:.8f} {e['sum']:.4f} "
|
||||
f"{e['min']:.6f} {e['max']:.6f}\n")
|
||||
|
||||
|
||||
def _sort_key(name: str, drill_layer: int) -> tuple:
|
||||
if name == "adapter":
|
||||
return (0, "")
|
||||
if name.startswith(f"L{drill_layer:02d}."):
|
||||
sub_order = {
|
||||
"attention": 0,
|
||||
"attn_post_norm": 1,
|
||||
"mlp.pre_norm": 2,
|
||||
"mlp.up_gate_proj": 3,
|
||||
"mlp.down_proj<in>": 4,
|
||||
"mlp.down_proj": 5,
|
||||
"mlp": 6,
|
||||
"mlp_post_norm": 7,
|
||||
}.get(name.split(".", 1)[1], 9)
|
||||
return (1, f"block[{drill_layer:02d}]", sub_order)
|
||||
if name.startswith("block["):
|
||||
return (1, name, 99)
|
||||
return (2, name, 0)
|
||||
|
||||
|
||||
def _print_table(by_name: dict[str, dict], drill_layer: int) -> int | None:
|
||||
hdr = (f"{'name':<18} {'up_shape':<22} {'up_absmean':>12} {'fv_absmean':>12} "
|
||||
f"{'absmean_diff':>14} {'rel%':>8} {'up_sum':>14} {'fv_sum':>14} {'sum_diff':>12}")
|
||||
print(f"\n{hdr}\n{'-' * len(hdr)}")
|
||||
first_div: int | None = None
|
||||
for name in sorted(by_name.keys(), key=lambda n: _sort_key(n, drill_layer)):
|
||||
d = by_name[name]
|
||||
up, fv = d.get("up"), d.get("fv")
|
||||
if up is None or fv is None:
|
||||
continue
|
||||
am_diff = abs(up["abs_mean"] - fv["abs_mean"])
|
||||
am_rel = am_diff / max(up["abs_mean"], 1e-9)
|
||||
sum_diff = abs(up["sum"] - fv["sum"])
|
||||
flag = ""
|
||||
if name.startswith("block[") and am_rel > REL_THRESHOLD:
|
||||
flag = " <<< DIVERGE"
|
||||
if first_div is None:
|
||||
first_div = int(name[len("block["):-1])
|
||||
print(f"{name:<18} {str(up['shape']):<22} {up['abs_mean']:>12.6f} "
|
||||
f"{fv['abs_mean']:>12.6f} {am_diff:>14.6f} {am_rel * 100:>7.3f}% "
|
||||
f"{up['sum']:>14.4f} {fv['sum']:>14.4f} {sum_diff:>12.4f}{flag}")
|
||||
return first_div
|
||||
|
||||
|
||||
def _print_elementwise(up_t: dict[str, torch.Tensor], fv_t: dict[str, torch.Tensor], drill_layer: int) -> None:
|
||||
common = set(up_t.keys()) & set(fv_t.keys())
|
||||
if not common:
|
||||
return
|
||||
hdr = f"{'name':<30} {'shape':<22} {'diff_max':>12} {'diff_mean':>12} {'diff_rel%':>10}"
|
||||
print(f"\nElement-wise diffs for drilled L{drill_layer:02d} submodules:\n{hdr}\n{'-' * len(hdr)}")
|
||||
for name in sorted(common):
|
||||
a, b = up_t[name], fv_t[name]
|
||||
if a.shape != b.shape:
|
||||
continue
|
||||
diff = (a - b).abs()
|
||||
rel = (diff.mean().item() / max(a.abs().mean().item(), 1e-9)) * 100
|
||||
print(f"{name:<30} {str(tuple(a.shape)):<22} "
|
||||
f"{diff.max().item():>12.6f} {diff.mean().item():>12.6f} {rel:>9.4f}%")
|
||||
|
||||
|
||||
def main() -> None:
|
||||
if not torch.cuda.is_available():
|
||||
print("Need CUDA. Skipping.")
|
||||
return
|
||||
|
||||
# TODO: add precondition checks (official clone present, weights available).
|
||||
|
||||
drill_layer = int(os.getenv(DRILL_LAYER_ENV, "0"))
|
||||
device = torch.device("cuda:0")
|
||||
patched = _apply_hypothesis_patch()
|
||||
try:
|
||||
inputs = build_inputs(device)
|
||||
|
||||
print("Loading official model...")
|
||||
official = load_official(device)
|
||||
up_log: list[dict] = []
|
||||
up_t: dict[str, torch.Tensor] = {}
|
||||
up_handles = _attach_block_hooks(official, "up", up_log, up_t, drill_layer)
|
||||
print("Running official forward (with hooks)...")
|
||||
try:
|
||||
with torch.inference_mode():
|
||||
# TODO: adapt forward call signature to your component.
|
||||
ref_out = official(**{k: v.clone() for k, v in inputs.items()})
|
||||
if isinstance(ref_out, dict):
|
||||
sample = ref_out.get("sample")
|
||||
ref_out = sample if sample is not None else ref_out.get("x")
|
||||
elif hasattr(ref_out, "sample"):
|
||||
ref_out = ref_out.sample
|
||||
elif isinstance(ref_out, tuple):
|
||||
ref_out = ref_out[0]
|
||||
assert torch.is_tensor(ref_out), f"official output is not tensor: {type(ref_out)}"
|
||||
ref_out = ref_out.detach().float().cpu()
|
||||
finally:
|
||||
for h in up_handles:
|
||||
h.remove()
|
||||
del official
|
||||
gc.collect()
|
||||
torch.cuda.empty_cache()
|
||||
|
||||
print("Loading FastVideo model...")
|
||||
fv = load_fastvideo(device)
|
||||
fv_log: list[dict] = []
|
||||
fv_t: dict[str, torch.Tensor] = {}
|
||||
fv_handles = _attach_block_hooks(fv, "fv", fv_log, fv_t, drill_layer)
|
||||
print("Running FastVideo forward (with hooks)...")
|
||||
try:
|
||||
with torch.inference_mode():
|
||||
# TODO: adapt forward call signature to your component.
|
||||
fv_out = fv(**{k: v.clone() for k, v in inputs.items()})
|
||||
if isinstance(fv_out, dict):
|
||||
sample = fv_out.get("sample")
|
||||
fv_out = sample if sample is not None else fv_out.get("x")
|
||||
elif hasattr(fv_out, "sample"):
|
||||
fv_out = fv_out.sample
|
||||
elif isinstance(fv_out, tuple):
|
||||
fv_out = fv_out[0]
|
||||
assert torch.is_tensor(fv_out), f"FastVideo output is not tensor: {type(fv_out)}"
|
||||
fv_out = fv_out.detach().float().cpu()
|
||||
finally:
|
||||
for h in fv_handles:
|
||||
h.remove()
|
||||
finally:
|
||||
_restore_hypothesis_patch()
|
||||
|
||||
LOG_DIR.mkdir(parents=True, exist_ok=True)
|
||||
up_path = LOG_DIR / f"{FAMILY}_{COMPONENT}_up_layers.log"
|
||||
fv_path = LOG_DIR / f"{FAMILY}_{COMPONENT}_fv_layers.log"
|
||||
_write_log(up_log, up_path)
|
||||
_write_log(fv_log, fv_path)
|
||||
print(f"\nLogs: {up_path} {fv_path}\nDiff: diff {up_path} {fv_path}")
|
||||
|
||||
by_name: dict[str, dict] = {}
|
||||
for entry in up_log + fv_log:
|
||||
by_name.setdefault(entry["name"], {})[entry["side"]] = entry
|
||||
first_div = _print_table(by_name, drill_layer)
|
||||
|
||||
print()
|
||||
if first_div is not None:
|
||||
print(f"First block exceeding {REL_THRESHOLD * 100:.2f}% drift: block[{first_div:02d}]")
|
||||
print(f"Re-run with {DRILL_LAYER_ENV}={first_div} to drill submodules.")
|
||||
else:
|
||||
print(f"No block exceeded {REL_THRESHOLD * 100:.2f}% -- divergence is amortized or pre-block.")
|
||||
|
||||
diff = (ref_out - fv_out).abs()
|
||||
print(f"\nFinal ref_abs={ref_out.abs().mean():.6f} fv_abs={fv_out.abs().mean():.6f} "
|
||||
f"diff_max={diff.max():.6f} diff_mean={diff.mean():.6f}")
|
||||
_print_elementwise(up_t, fv_t, drill_layer)
|
||||
if patched:
|
||||
print(f"\n[debug] Hypothesis {HYPOTHESIS_ENV}=1 was active this run.")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,255 @@
|
||||
---
|
||||
name: add-model-09-pipeline
|
||||
description: Use during /add-model Phase 7 after all required component parity tests pass to define FastVideo pipeline wiring, configs, presets, registry entries, examples, smoke tests, and pipeline parity tests.
|
||||
---
|
||||
|
||||
# Add Model Pipeline
|
||||
|
||||
## Goal
|
||||
|
||||
Implement and verify the end-to-end FastVideo pipeline after the native
|
||||
components and converted weights have passed non-skip component parity. This
|
||||
skill owns pipeline class/stage wiring, pipeline configs, presets, registry
|
||||
entries, examples, smoke tests, and pipeline parity-debug.
|
||||
|
||||
FastVideo has one pipeline architecture: stage-based composition through
|
||||
`ComposedPipelineBase`. Add or specialize stages only when existing stages cannot
|
||||
represent the official behavior safely.
|
||||
|
||||
## Hard Gate
|
||||
|
||||
Do not start pipeline work until every required component, including reused
|
||||
components, has a non-skip local parity PASS.
|
||||
|
||||
If any component row is missing, skipped, red, or blocked, return to `/add-model`
|
||||
Phase 6. Pipeline parity cannot distinguish stage wiring mistakes from broken
|
||||
component numerics when component parity is still unresolved.
|
||||
|
||||
## Inputs
|
||||
|
||||
Follow `../add-model/shared/common_rules.md` for token/auth safety, state files,
|
||||
escape hatches, production boundaries, and skip/pass semantics.
|
||||
|
||||
Require a complete packet matching
|
||||
`../add-model/contracts/pipeline_context.md`.
|
||||
|
||||
The packet must include:
|
||||
|
||||
- official pipeline files and official call/default sources;
|
||||
- workload types, input/output modalities, and output contract;
|
||||
- converted or source `model_index.json` path;
|
||||
- component parity rows, all `non_skip_pass`;
|
||||
- target FastVideo pipeline/config/preset/registry/example/test paths;
|
||||
- `local_tests_readme` and `port_state_file` paths.
|
||||
|
||||
## Outputs
|
||||
|
||||
- Pipeline package under `fastvideo/pipelines/basic/<family>/`.
|
||||
- Pipeline config under `fastvideo/configs/pipelines/<family>.py` or a documented
|
||||
family-local config file when that matches existing project style.
|
||||
- Presets under `fastvideo/pipelines/basic/<family>/presets.py`.
|
||||
- Registry updates in `fastvideo/registry.py`.
|
||||
- Basic example under `examples/inference/basic/basic_<family>*.py`.
|
||||
- Local smoke and parity tests under `tests/local_tests/pipelines/`.
|
||||
- Updated `tests/local_tests/<model_family>/README.md`.
|
||||
- Updated `tests/local_tests/<model_family>/PORT_STATUS.md`.
|
||||
- Handoff matching `../add-model/contracts/pipeline_handoff.md`.
|
||||
|
||||
## Mode: Pipeline Definition
|
||||
|
||||
Use this mode first.
|
||||
|
||||
1. Read the official pipeline call path before editing FastVideo code.
|
||||
2. Compare official defaults against the planned FastVideo config and presets:
|
||||
steps, CFG scales, secondary CFG, flow shift, schedulers, sigmas, seed/RNG,
|
||||
resolution, frames, FPS, duration, VAE scaling, decode slicing, negative
|
||||
prompt defaults, and output heads.
|
||||
3. Create or update the pipeline class with `_required_config_modules` matching
|
||||
the emitted `model_index.json` and `ComposedPipelineBase.load_modules`.
|
||||
Runtime pipeline resolution is exact: `model_index.json["_class_name"]` must
|
||||
match a registered `EntryClass.__name__`, or a wrapper/alias class in
|
||||
`EntryClass`. Registry detectors do not select the executable pipeline class.
|
||||
4. Add new public generation kwargs to `fastvideo/api/sampling_param.py` before
|
||||
examples or presets use them. `SamplingParam.update()` ignores unknown keys
|
||||
except for logging, and preset defaults apply only to declared fields. Add CLI
|
||||
args when the option should be available from command-line entrypoints.
|
||||
5. Put loader-time changes in `load_modules()` or earlier, not
|
||||
`initialize_pipeline()`. `ComposedPipelineBase.__init__` loads modules before
|
||||
`post_init()` calls `initialize_pipeline()`, so process-global flags, loader
|
||||
path rewrites, dtype overrides, and tokenizer path changes needed for loading
|
||||
cannot be introduced there.
|
||||
6. Use `self.get_module("transformer_2", None)` and similar optional accessors
|
||||
for truly optional modules. Do not hard-require optional modules by accident.
|
||||
7. Avoid mutating class-level `_required_config_modules` in custom code. If a
|
||||
pipeline needs dynamic modules, copy the list to an instance-owned value or
|
||||
pass `required_config_modules` explicitly so one pipeline instance cannot leak
|
||||
module requirements into another.
|
||||
8. Create the stage chain in official execution order. Prefer existing shared
|
||||
stages for standard text encoding, timestep preparation, latent preparation,
|
||||
denoising, and decoding.
|
||||
9. Add model-specific stages only for family-specific behavior that does not fit
|
||||
the shared stage contracts.
|
||||
10. Add pipeline config classes for wiring and runtime defaults. Do not duplicate
|
||||
component architecture fields unless a loader requires them in the subconfig.
|
||||
Family-local config files such as
|
||||
`fastvideo/pipelines/basic/<family>/pipeline_configs.py` are valid only when
|
||||
`fastvideo/registry.py` imports and registers the classes explicitly.
|
||||
11. Add `InferencePreset` objects with `model_family`, `name`, `version`,
|
||||
`defaults`, optional validation-only `stage_schemas`, and an `ALL_PRESETS`
|
||||
tuple. `stage_schemas` validates user-facing `stage_overrides` names; it does
|
||||
not drive `create_pipeline_stages()` execution.
|
||||
12. Register config classes and presets in `fastvideo/registry.py`: add
|
||||
`register_configs(...)`, import the family's `ALL_PRESETS`, and append it to
|
||||
`_register_presets()`. Detectors should cover HF paths and `_class_name`
|
||||
strings for config/preset lookup, but not as a replacement for exact pipeline
|
||||
class-name resolution.
|
||||
13. Add a basic example with a user-story docstring and normal file-path inputs
|
||||
for image, audio, or video references. Keep orchestration glue in the
|
||||
pipeline or a helper, not in the example.
|
||||
14. Add a separate smoke test
|
||||
`tests/local_tests/pipelines/test_<family>_pipeline_smoke.py` that proves
|
||||
imports, `EntryClass`, registry, presets, config defaults, and at least one
|
||||
real load/generate path when weights are local. Older local tests sometimes
|
||||
colocate smoke checks in parity files; new ports should use the separate file
|
||||
convention.
|
||||
15. Add or update pipeline parity test scaffolding with
|
||||
`templates/pipeline_parity_test.py`.
|
||||
16. Update `local_tests_readme` and `port_state_file` with commands, statuses,
|
||||
default sources, decisions, and blockers.
|
||||
|
||||
Production import boundaries are defined in
|
||||
`../add-model/shared/common_rules.md`.
|
||||
|
||||
## Mode: Pipeline Parity Debug
|
||||
|
||||
Run after pipeline definition and after smoke can execute far enough to load the
|
||||
pipeline. Loop until pipeline parity is a non-skip PASS or a precise blocker is
|
||||
returned.
|
||||
|
||||
Mandatory order:
|
||||
|
||||
```bash
|
||||
pytest tests/local_tests/pipelines/test_<family>_pipeline_smoke.py -v -s
|
||||
DISABLE_SP=1 pytest tests/local_tests/pipelines/test_<family>_pipeline_parity.py -v -s
|
||||
python examples/inference/basic/basic_<family>.py
|
||||
```
|
||||
|
||||
Pipeline parity must compare real outputs, not only successful generation:
|
||||
|
||||
- denoised latents when decode parity is expensive or nondeterministic;
|
||||
- decoded videos/images when visual output should be deterministic enough;
|
||||
- decoded waveform or audio features for audio pipelines;
|
||||
- separate video and audio targets for joint AV pipelines unless a validated
|
||||
joint metric exists.
|
||||
|
||||
Debug pipeline drift in this order:
|
||||
|
||||
1. Confirm both sides use the same component weights and component parity PASS
|
||||
results are still valid.
|
||||
2. Align official and FastVideo call arguments, presets, and default values.
|
||||
3. Align scheduler timesteps, sigmas/noise levels, prediction type, flow shift,
|
||||
guidance math, and secondary-guidance branches.
|
||||
4. Align RNG: initial latents/noise, generator device, seed, per-step noise, VAE
|
||||
sampling, and any official `+1 frame` or crop/slice behavior.
|
||||
5. Align conditioning: prompt templates, negative prompts, masks, image/audio
|
||||
preprocessing, modality packing, text truncation, and dtype/autocast.
|
||||
6. Align decode: latent scaling, per-channel mean/std, tiling flags, output
|
||||
channel order, sample rate, FPS, and final slicing.
|
||||
7. Add targeted stage-level diagnostics to identify the first divergent stage.
|
||||
|
||||
If stage diagnostics show the first bad stage is transformer/denoising or a
|
||||
mid-DiT block, enable activation trace before adding ad hoc pipeline prints; see
|
||||
`docs/contributing/activation_trace.md` and `../add-model-08-trace/SKILL.md`.
|
||||
Keep `FASTVIDEO_TRACE_LAYERS`, `FASTVIDEO_TRACE_STATS`, and
|
||||
`FASTVIDEO_TRACE_STEPS` identical across reruns so pipeline parity traces diff
|
||||
one-to-one.
|
||||
|
||||
If the first divergence belongs to component implementation, strict loading, or
|
||||
conversion mapping, stop pipeline edits and return `next_step=return_to_phase_6`
|
||||
with the exact failing evidence. Do not patch conversion from this skill.
|
||||
|
||||
## Stage And Variant Rules
|
||||
|
||||
- Canonical video T2V order: `InputValidationStage`, `TextEncodingStage`,
|
||||
`ConditioningStage`, `TimestepPreparationStage`, `LatentPreparationStage`,
|
||||
`DenoisingStage`, `DecodingStage`.
|
||||
- Canonical video I2V delta adds image loading/encoding and image VAE encoding in
|
||||
the official order, commonly: `TextEncodingStage`, `ImageEncodingStage`,
|
||||
`ConditioningStage`, `TimestepPreparationStage`, `LatentPreparationStage`,
|
||||
`ImageVAEEncodingStage`, `DenoisingStage`, `DecodingStage`.
|
||||
- Treat `ConditioningStage` as default-present for Wan-style pipelines, but still
|
||||
follow the reference if another family truly skips or replaces it.
|
||||
- T2V video pipelines usually use validation, text encoding, conditioning,
|
||||
timestep preparation, latent preparation, denoising, and decoding.
|
||||
- I2V adds image loading/encoding and image-latent preparation according to the
|
||||
official pipeline, not by assuming CLIP or Wan-specific branches.
|
||||
- Pick image, audio, and video encoders from the reference. Do not assume CLIP or
|
||||
any other common encoder unless the reference uses it.
|
||||
- Cross-attention class names are not prescribed; match the family style and
|
||||
preserve the official tensor contract.
|
||||
- `WorkloadType` currently has no `T2A`, `A2A`, or `AV` values. Until that enum
|
||||
is extended, audio-only pipelines may register with `WorkloadType.T2V` and
|
||||
preset `workload_type="t2v"` as a compatibility shim, but must document the
|
||||
rationale in code and `PORT_STATUS.md`.
|
||||
- Audio-only pipelines should not force real video semantics into presets. Use
|
||||
minimal video-shaped placeholders such as small `height`/`width` and
|
||||
`num_frames=1` only when shared `VideoGenerator`/validation paths require them,
|
||||
and document that the real output is audio.
|
||||
- Record modality-specific shape knobs and output contract in the pipeline
|
||||
handoff: video uses `height`, `width`, `num_frames`, and `fps`; audio uses
|
||||
`audio_seconds` and `sampling_rate`; joint AV records both plus whether output
|
||||
is muxed or paired files.
|
||||
- Use sibling pipeline classes/configs when required modules, HF repo layout,
|
||||
stage chains, or inputs differ materially.
|
||||
- Use one kwargs-driven pipeline class only when variants share weights,
|
||||
modules, stage chain, and safe call semantics.
|
||||
- Split later if components diverge, workload tags require separate discovery,
|
||||
signatures become unsafe, or stage branches become substantial.
|
||||
- If the DiT branches on `added_kv_proj_dim`, document the T2V/I2V split.
|
||||
- If the reference uses `transformer_2`, `boundary_ratio`, `guidance_scale_2`, or
|
||||
DMD step lists, keep those on config, presets, or stages deliberately.
|
||||
- Support every official output head in scope. If a head is out of scope, record
|
||||
explicit user approval in `PORT_STATUS.md`.
|
||||
|
||||
Pipeline verification order:
|
||||
|
||||
```bash
|
||||
pytest tests/local_tests/pipelines/test_<family>_pipeline_smoke.py -v -s
|
||||
DISABLE_SP=1 pytest -v -s tests/local_tests/pipelines/test_<family>_pipeline_parity.py
|
||||
python examples/inference/basic/basic_<family>.py
|
||||
```
|
||||
|
||||
Smoke tests prove loadability only. They are not a substitute for numerical
|
||||
component or pipeline parity.
|
||||
|
||||
## Escape Hatches
|
||||
|
||||
Follow `../add-model/shared/common_rules.md`. Pipeline-specific ask cases include
|
||||
dropping a public mode, modality, or output head; adding a new workload enum;
|
||||
changing official defaults for user-facing behavior; accepting a known pipeline
|
||||
parity blocker; running GPU-heavy quality work outside the agreed scope; or
|
||||
publishing/uploading generated references or converted weights.
|
||||
|
||||
## Handoff
|
||||
|
||||
Return `../add-model/contracts/pipeline_handoff.md` and update the shared state
|
||||
files before handoff.
|
||||
|
||||
Do not hand back a green pipeline if smoke or parity skipped locally. A skip is a
|
||||
setup gap, not a pass.
|
||||
|
||||
## References
|
||||
|
||||
- `fastvideo/pipelines/composed_pipeline_base.py` for module loading and stage
|
||||
execution.
|
||||
- `fastvideo/pipelines/basic/wan/` for standard video T2V/I2V/DMD variants.
|
||||
- `fastvideo/pipelines/basic/stable_audio/` for audio-specific stage composition.
|
||||
- `fastvideo/configs/pipelines/stable_audio.py` and
|
||||
`fastvideo/pipelines/basic/stable_audio/presets.py` for config/preset shape.
|
||||
- `fastvideo/registry.py` for `register_configs(...)` and preset registration.
|
||||
- `tests/local_tests/pipelines/test_gamecraft_pipeline_parity.py` for latent
|
||||
parity structure.
|
||||
- `tests/local_tests/pipelines/test_stable_audio_pipeline_parity.py` for audio
|
||||
parity structure.
|
||||
- `tests/local_tests/pipelines/test_stable_audio_pipeline_smoke.py` for no-GPU
|
||||
import/registry/preset preflight shape.
|
||||
@@ -0,0 +1,147 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
"""Pipeline parity scaffold for TODO_MODEL_FAMILY.
|
||||
|
||||
Copy this file to
|
||||
`tests/local_tests/pipelines/test_<family>_pipeline_parity.py` and replace every
|
||||
TODO before treating it as an executable scaffold.
|
||||
|
||||
The filled test should compare denoised latents, decoded media, audio waveform,
|
||||
or another concrete output from the official pipeline against FastVideo. A
|
||||
successful generation without tensor/media comparison is not parity.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
import pytest
|
||||
import torch
|
||||
from torch.testing import assert_close
|
||||
|
||||
_REPO_ROOT = Path(__file__).resolve().parents[3]
|
||||
_MODEL_FAMILY = "TODO_MODEL_FAMILY"
|
||||
_OFFICIAL_REF_ENV = "TODO_OFFICIAL_REF_PATH"
|
||||
_OFFICIAL_REF_DEFAULT = _REPO_ROOT / "TODO_OFFICIAL_REF_DIR"
|
||||
_FASTVIDEO_MODEL_ENV = "TODO_FASTVIDEO_MODEL_PATH"
|
||||
_FASTVIDEO_MODEL_DEFAULT = _REPO_ROOT / "converted_weights" / _MODEL_FAMILY
|
||||
|
||||
|
||||
def _path_from_env(env_name: str, default: Path) -> Path:
|
||||
return Path(os.getenv(env_name, str(default))).expanduser()
|
||||
|
||||
|
||||
def _add_official_to_path() -> Path:
|
||||
official_path = _path_from_env(_OFFICIAL_REF_ENV, _OFFICIAL_REF_DEFAULT)
|
||||
if not official_path.exists():
|
||||
pytest.skip(f"Official reference not found at {official_path}")
|
||||
if str(official_path) not in sys.path:
|
||||
sys.path.insert(0, str(official_path))
|
||||
return official_path
|
||||
|
||||
|
||||
def _log_tensor_stats(label: str, tensor: torch.Tensor) -> None:
|
||||
value = tensor.detach().float()
|
||||
print(f"[{_MODEL_FAMILY} PIPELINE] {label}: shape={tuple(tensor.shape)} "
|
||||
f"dtype={tensor.dtype} device={tensor.device} "
|
||||
f"min={value.min().item():.6f} max={value.max().item():.6f} "
|
||||
f"mean={value.mean().item():.6f} std={value.std().item():.6f}")
|
||||
|
||||
|
||||
def _extract_tensor(output: Any, key: str) -> torch.Tensor:
|
||||
if isinstance(output, dict):
|
||||
value = output.get(key)
|
||||
else:
|
||||
value = getattr(output, key, None)
|
||||
if value is None:
|
||||
raise AssertionError(f"Pipeline output did not contain {key!r}")
|
||||
if not torch.is_tensor(value):
|
||||
try:
|
||||
import numpy as np
|
||||
value = torch.from_numpy(np.asarray(value))
|
||||
except Exception as exc: # pragma: no cover - scaffold guard
|
||||
raise AssertionError(f"Could not convert {key!r} to tensor") from exc
|
||||
return value.detach().float().cpu()
|
||||
|
||||
|
||||
def _run_official_pipeline(
|
||||
official_path: Path,
|
||||
params: dict[str, Any],
|
||||
device: torch.device,
|
||||
) -> Any:
|
||||
del official_path, params, device
|
||||
pytest.skip("TODO: import the official pipeline/factory, load official weights, "
|
||||
"run with params, and return the comparison target.")
|
||||
|
||||
|
||||
def _run_fastvideo_pipeline(model_path: Path, params: dict[str, Any]) -> Any:
|
||||
from fastvideo import VideoGenerator
|
||||
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
str(model_path),
|
||||
num_gpus=1,
|
||||
use_fsdp_inference=False,
|
||||
dit_cpu_offload=False,
|
||||
vae_cpu_offload=False,
|
||||
text_encoder_cpu_offload=False,
|
||||
)
|
||||
try:
|
||||
return generator.generate_video(
|
||||
prompt=params["prompt"],
|
||||
negative_prompt=params.get("negative_prompt"),
|
||||
output_path=f"outputs_{_MODEL_FAMILY}/pipeline_parity",
|
||||
save_video=False,
|
||||
height=params.get("height"),
|
||||
width=params.get("width"),
|
||||
num_frames=params.get("num_frames"),
|
||||
fps=params.get("fps"),
|
||||
num_inference_steps=params["num_inference_steps"],
|
||||
guidance_scale=params.get("guidance_scale"),
|
||||
seed=params["seed"],
|
||||
)
|
||||
finally:
|
||||
generator.shutdown()
|
||||
|
||||
|
||||
@pytest.mark.skipif(
|
||||
not torch.cuda.is_available(),
|
||||
reason="TODO_MODEL_FAMILY pipeline parity requires CUDA.",
|
||||
)
|
||||
def test_todo_model_family_pipeline_official_parity() -> None:
|
||||
official_path = _add_official_to_path()
|
||||
fastvideo_model_path = _path_from_env(
|
||||
_FASTVIDEO_MODEL_ENV,
|
||||
_FASTVIDEO_MODEL_DEFAULT,
|
||||
)
|
||||
if not fastvideo_model_path.exists():
|
||||
pytest.skip(f"FastVideo model path not found at {fastvideo_model_path}")
|
||||
|
||||
device = torch.device("cuda:0")
|
||||
params = {
|
||||
"prompt": "TODO: stable parity prompt",
|
||||
"negative_prompt": "",
|
||||
"height": 64,
|
||||
"width": 64,
|
||||
"num_frames": 9,
|
||||
"fps": 8,
|
||||
"num_inference_steps": 4,
|
||||
"guidance_scale": 1.0,
|
||||
"seed": 0,
|
||||
}
|
||||
|
||||
official_output = _run_official_pipeline(official_path, params, device)
|
||||
fastvideo_output = _run_fastvideo_pipeline(fastvideo_model_path, params)
|
||||
|
||||
comparison_key = "TODO_COMPARISON_KEY"
|
||||
official_tensor = _extract_tensor(official_output, comparison_key)
|
||||
fastvideo_tensor = _extract_tensor(fastvideo_output, comparison_key)
|
||||
|
||||
_log_tensor_stats("official", official_tensor)
|
||||
_log_tensor_stats("fastvideo", fastvideo_tensor)
|
||||
assert official_tensor.shape == fastvideo_tensor.shape
|
||||
|
||||
diff = (official_tensor - fastvideo_tensor).abs()
|
||||
print(f"diff max={diff.max().item():.6f} "
|
||||
f"mean={diff.mean().item():.6f} median={diff.median().item():.6f}")
|
||||
assert_close(fastvideo_tensor, official_tensor, atol=1e-2, rtol=1e-2)
|
||||
@@ -0,0 +1,141 @@
|
||||
---
|
||||
name: add-model-10-pr-review
|
||||
description: Review rubric for FastVideo PRs that add or modify model families, variants, first-class components, checkpoint conversion, pipelines, parity coverage, or generated-media quality baselines. Use when reviewing a PR whose diff touches fastvideo/models/, fastvideo/pipelines/basic/, fastvideo/registry.py, scripts/checkpoint_conversion/, fastvideo/tests/ssim/, or related model-port surfaces. Pairs with review-pr-link as a project-scoped review pass; produces findings, not fixes.
|
||||
---
|
||||
|
||||
# Add-Model PR Review
|
||||
|
||||
Use this skill when a reviewed PR appears to add, port, or substantially modify
|
||||
a FastVideo model family, model variant, first-class model component,
|
||||
checkpoint conversion, model pipeline, or local parity coverage.
|
||||
|
||||
This is a review skill, not an implementation workflow. Do not run `/add-model`
|
||||
or start writing missing port code during review. Use the add-model skill stack
|
||||
as a rubric for findings.
|
||||
|
||||
## Trigger Paths
|
||||
|
||||
Trigger this skill if `git diff --name-only <base>...HEAD` includes any of:
|
||||
|
||||
- `fastvideo/models/dits/`, `fastvideo/configs/models/dits/`
|
||||
- `fastvideo/models/vaes/`, `fastvideo/configs/models/vaes/`
|
||||
- `fastvideo/models/encoders/`, `fastvideo/configs/models/encoders/`
|
||||
- `fastvideo/models/schedulers/`, `fastvideo/configs/models/schedulers/`
|
||||
- `fastvideo/models/upsamplers/`, `fastvideo/configs/models/upsamplers/`
|
||||
- `fastvideo/models/audio/`, `fastvideo/configs/models/audio/`
|
||||
- `fastvideo/pipelines/basic/`, `fastvideo/configs/pipelines/`
|
||||
- `fastvideo/registry.py`, `fastvideo/api/sampling_param.py`
|
||||
- `scripts/checkpoint_conversion/`
|
||||
- `examples/inference/basic/`
|
||||
- `tests/local_tests/`, especially component or pipeline parity tests
|
||||
- `fastvideo/tests/ssim/` or other quality-regression tests for generated media
|
||||
|
||||
Also trigger when the PR title/body claims a new model, model variant, VAE,
|
||||
encoder, scheduler, conditioner, pipeline, conversion script, or generated-media
|
||||
quality baseline even if the path list is incomplete.
|
||||
|
||||
## Review Inputs
|
||||
|
||||
Read these add-model references as review checklists:
|
||||
|
||||
- `../add-model/SKILL.md`: phase gates and final handoff requirements.
|
||||
- `../add-model/shared/common_rules.md`: token/auth safety, production import
|
||||
boundaries, state files, and skip/pass semantics.
|
||||
- `../add-model/contracts/final_handoff.md`: final evidence expected from a
|
||||
complete port.
|
||||
- `../add-model/contracts/component_context.md` and
|
||||
`../add-model/contracts/component_skill_handoff.md`: component evidence and
|
||||
parity-debug expectations.
|
||||
- `../add-model/contracts/conversion_request.md` and
|
||||
`../add-model/contracts/conversion_handoff.md`: conversion evidence,
|
||||
strict-load status, config validation, and retry context.
|
||||
- `../add-model/contracts/pipeline_context.md` and
|
||||
`../add-model/contracts/pipeline_handoff.md`: pipeline class/stage/config/
|
||||
preset/registry/example evidence.
|
||||
|
||||
Then read only the satellite skill(s) that match touched areas:
|
||||
|
||||
- DiT/transformer changes: `../add-model-03-port-dit/SKILL.md`.
|
||||
- VAE changes: `../add-model-04-port-vae/SKILL.md`.
|
||||
- Encoder/conditioner changes: `../add-model-05-port-encoder/SKILL.md`.
|
||||
- Scheduler/upsampler/vocoder/other components:
|
||||
`../add-model-06-port-generic/SKILL.md`.
|
||||
- Component parity tests: `../add-model-02-parity/SKILL.md`.
|
||||
- Checkpoint conversion: `../add-model-07-conversion/SKILL.md`.
|
||||
- Pipeline/config/presets/registry/examples:
|
||||
`../add-model-09-pipeline/SKILL.md`.
|
||||
- Prep/state docs: `../add-model-01-prep/SKILL.md`.
|
||||
|
||||
## Required Review Lanes
|
||||
|
||||
For a full model-family or model-variant PR, cover all lanes. For a
|
||||
component-only PR, cover the component, conversion/parity as applicable, and the
|
||||
documented downstream consumer.
|
||||
|
||||
1. Scope and source-of-truth lane:
|
||||
Verify the PR clearly identifies the official reference, weights/revision,
|
||||
supported variants, modalities, output heads, and any approved scope cuts.
|
||||
|
||||
2. Component lane:
|
||||
Verify each required component is FastVideo-native or has a documented and
|
||||
accepted lazy-wrapper exception. Check bucket/config inheritance, `EntryClass`,
|
||||
state-dict surface, reused-component evidence, and output heads.
|
||||
|
||||
3. Conversion lane:
|
||||
Verify mappings are derived from prototype key/shape dumps, source layout is
|
||||
supported, skipped keys are intentional, emitted configs validate through
|
||||
production paths, component strict-load status is recorded, `model_index.json`
|
||||
library tokens match loaders, and revisions are pinned when converting from
|
||||
HF.
|
||||
|
||||
4. Component parity lane:
|
||||
Verify local parity tests exist for every required component, including reused
|
||||
components. Scaffolds may skip in CI, but the PR must provide local non-skip
|
||||
PASS evidence or an explicit accepted blocker.
|
||||
|
||||
5. Pipeline lane:
|
||||
Verify stage order, required modules, `_class_name` / `EntryClass.__name__`
|
||||
resolution, config defaults, presets, `SamplingParam` fields, registry
|
||||
registration, examples, smoke tests, and pipeline parity.
|
||||
|
||||
6. Quality and evidence lane:
|
||||
Verify media quality regression is added or explicitly deferred, examples run,
|
||||
generated outputs are non-corrupt, `tests/local_tests/<family>/README.md` and
|
||||
`PORT_STATUS.md` are current, and final blockers are surfaced in the review.
|
||||
|
||||
## Findings To Prioritize
|
||||
|
||||
Prioritize review findings in this order:
|
||||
|
||||
- Missing or skipped required component parity without accepted blocker.
|
||||
- Pipeline parity/smoke/example missing or skipped for a pipeline PR.
|
||||
- Conversion emits unloadable or unvalidated configs/weights.
|
||||
- Wrong `model_index.json` `_class_name`, component library token, or registry
|
||||
class resolution.
|
||||
- Runtime diffusers/transformers model-class imports for components that own
|
||||
weights or numerical behavior.
|
||||
- Dropped modalities, output heads, variants, or conditioning streams without
|
||||
explicit approval.
|
||||
- Reused FastVideo component lacks exact definition/instantiation proof or
|
||||
non-skip parity.
|
||||
- Public generation kwargs/preset defaults missing from `SamplingParam`.
|
||||
- Tests only check shapes, importability, or successful generation without
|
||||
numerical/media comparison.
|
||||
- Tokens, credentials, reference clones, staged weights, or generated bulk assets
|
||||
committed to the PR.
|
||||
|
||||
## Output Format
|
||||
|
||||
Write normal code-review findings first, ordered by severity. Include file and
|
||||
line references from the PR diff when possible.
|
||||
|
||||
Use this phrasing for missing add-model evidence:
|
||||
|
||||
```text
|
||||
This PR does not satisfy the add-model <component|conversion|pipeline|final>
|
||||
gate because <specific required evidence> is missing. The risk is <runtime load,
|
||||
numerical parity, dropped output, registry resolution, etc.>.
|
||||
```
|
||||
|
||||
Keep the summary short. Mention which lanes were reviewed and which could not be
|
||||
verified because assets, GPU time, or external credentials were unavailable.
|
||||
@@ -0,0 +1,438 @@
|
||||
---
|
||||
name: add-model
|
||||
description: Manual /add-model workflow for implementing a FastVideo model or first-class component port after add-model-01-prep has staged reference code and weights. Organizes the port into numbered phases with conversion rules, component policies, parity gates, and handoff checks.
|
||||
---
|
||||
|
||||
# Add Model
|
||||
|
||||
## Manual Invocation
|
||||
|
||||
This skill is for explicit `/add-model` use only. Do not auto-start it from a
|
||||
casual model-port mention. The setup-only workflow is
|
||||
`../add-model-01-prep/SKILL.md`.
|
||||
|
||||
## Goal
|
||||
|
||||
Port a new FastVideo model family, model variant, or first-class reusable
|
||||
component so it can be loaded through FastVideo's native model, config, stage,
|
||||
registry, preset, and test infrastructure.
|
||||
|
||||
FastVideo has one pipeline architecture: stage-based composition via
|
||||
`ComposedPipelineBase`. Vary the stages and modules, not the architecture.
|
||||
|
||||
## Scope Shapes
|
||||
|
||||
Use this skill for either shape:
|
||||
|
||||
| Shape | Required output |
|
||||
|---|---|
|
||||
| Full model family or variant | Native components, conversion if needed, pipeline config/class, presets, registry, smoke test, local parity tests, example, quality regression. |
|
||||
| First-class component contribution | Native component class/config, bucket export, component parity test, and a documented downstream pipeline that will consume it. Skip pipeline/preset/registry rows only when the contribution is intentionally component-only. |
|
||||
|
||||
If upstream ships many variants, lock scope before coding. "Base model" means
|
||||
checkpoint variant, not a modality subset. If the base checkpoint produces
|
||||
audio, pose, depth, masks, or other output heads, either support those outputs
|
||||
or get explicit user agreement to drop them.
|
||||
|
||||
## Required Input
|
||||
|
||||
Start from an `add-model-01-prep` handoff, or equivalent fields matching
|
||||
`contracts/prep_handoff.md`.
|
||||
|
||||
Before Phase 0, read the shared rules and all relevant schemas:
|
||||
|
||||
- `shared/common_rules.md`
|
||||
- `contracts/prep_handoff.md`
|
||||
- `contracts/port_state.md`
|
||||
- `contracts/escape_hatch.md`
|
||||
- `contracts/component_context.md`
|
||||
- `contracts/parity_status.md`
|
||||
- `contracts/conversion_request.md`
|
||||
- `contracts/conversion_handoff.md`
|
||||
- `contracts/component_skill_handoff.md`
|
||||
- `contracts/pipeline_context.md`
|
||||
- `contracts/pipeline_handoff.md`
|
||||
- `contracts/final_handoff.md`
|
||||
|
||||
## Hard Rules
|
||||
|
||||
- Follow `shared/common_rules.md` for token/auth safety, state files, escape
|
||||
hatches, production import boundaries, and skip/pass semantics.
|
||||
- If the prep handoff is missing or ambiguous, stop and run
|
||||
`../add-model-01-prep/SKILL.md`.
|
||||
- If a needed component is not ported, do not ship the pipeline that needs it.
|
||||
- Wan is grandfathered for missing local parity; do not copy its missing-test
|
||||
precedent for new work.
|
||||
|
||||
## Escape Hatches
|
||||
|
||||
Follow `shared/common_rules.md` and `contracts/escape_hatch.md`. The main
|
||||
orchestrator should ask only when no phase skill can safely continue under the
|
||||
shared rules.
|
||||
|
||||
## Files Map
|
||||
|
||||
| Area | Paths |
|
||||
|---|---|
|
||||
| DiT | `fastvideo/models/dits/<family>.py`, `fastvideo/configs/models/dits/<family>.py`, bucket `__init__.py`. |
|
||||
| VAE | `fastvideo/models/vaes/<arch_or_family>.py`, `fastvideo/configs/models/vaes/<arch_or_family>.py`, bucket `__init__.py`. Name by shared arch when reusable (`oobleck.py`, `autoencoder_kl.py`), otherwise by family (`wanvae.py`). |
|
||||
| Encoder / conditioner / scheduler / upsampler | Native class/config in the matching `fastvideo/models/<bucket>/` and `fastvideo/configs/models/<bucket>/` bucket. |
|
||||
| Lazy loader wrapper | Optional `fastvideo/models/<bucket>/<family>_loader.py` or similar thin `nn.Module` wrapper when a component is fetched from an external HF repo and should be hidden from host-pipeline state-dict matching. |
|
||||
| Conversion | `scripts/checkpoint_conversion/<family>_to_diffusers.py` only when `needs_conversion=yes`. |
|
||||
| Pipeline | `fastvideo/pipelines/basic/<family>/<family>_pipeline.py` plus sibling files for variants whose components or required modules differ. |
|
||||
| Pipeline config | `fastvideo/configs/pipelines/<family>.py` or `fastvideo/pipelines/basic/<family>/pipeline_configs.py`. |
|
||||
| Stages | `fastvideo/pipelines/basic/<family>/stages/` only for model-specific stage subclasses. |
|
||||
| Presets / registry | `fastvideo/pipelines/basic/<family>/presets.py`, `fastvideo/registry.py`. |
|
||||
| Tests | Component parity under `tests/local_tests/<bucket>/`; pipeline smoke/parity under `tests/local_tests/pipelines/`; CI-backed quality tests under `fastvideo/tests/`. |
|
||||
| Example | `examples/inference/basic/basic_<family>*.py`, one per public mode/variant. |
|
||||
|
||||
## Phase 0: Scope And Handoff Gate
|
||||
|
||||
1. Validate every required handoff field.
|
||||
2. Resolve `needs_conversion=unknown` before component work:
|
||||
|
||||
```bash
|
||||
python ".agents/skills/add-model-01-prep/scripts/inspect_hf_layout.py" \
|
||||
"<hf-or-local-path>" \
|
||||
--json
|
||||
```
|
||||
|
||||
3. List first-PR scope across both axes:
|
||||
- Variant axis: base, distill, SR/refine, causal, DMD, I2V, V2V, etc.
|
||||
- Modality axis: video, image, audio, pose, depth, masks, text, etc.
|
||||
4. For component-only work, explicitly name the downstream full-pipeline PR or
|
||||
planned consumer.
|
||||
5. Confirm `official_env_status` is `imports_ok` or
|
||||
`private_deps_need_stubs`. If it is `blocked`, return to
|
||||
`../add-model-01-prep/SKILL.md` before parity scaffolding.
|
||||
6. Confirm `local_tests_readme` exists and records official setup, HF weights,
|
||||
dependency changes, and planned parity commands for reviewers.
|
||||
7. Confirm `port_state_file` exists, follows `contracts/port_state.md`, and has
|
||||
rows for open questions/issues found during prep.
|
||||
8. If there are multiple official implementations, choose the one whose
|
||||
architecture matches the published weights. A blessed library port can be a
|
||||
better parity reference than a highly configurable research repo; document
|
||||
the choice in tests.
|
||||
|
||||
## Phase 1: Reference And Architecture Study
|
||||
|
||||
Read the official pipeline call path before writing code.
|
||||
|
||||
Record:
|
||||
|
||||
- Required modules from `model_index.json` or equivalent: transformer, VAE,
|
||||
text encoders, tokenizers, scheduler, image encoders, audio VAE, vocoder,
|
||||
conditioners, upsamplers.
|
||||
- Input/output modalities and every dedicated DiT output head.
|
||||
- Text/image/audio encoding flow, latent shape, dtype, scaling, packing,
|
||||
scheduler/timestep math, guidance math, VAE normalization, and decode flow.
|
||||
- Whether the official code relies on private deps, custom ops, or special
|
||||
kernels that parity tests must stub.
|
||||
|
||||
Arch config rule:
|
||||
|
||||
- `ArchConfig` fields must match the emitted per-component config, especially
|
||||
`transformer/config.json`, one-to-one.
|
||||
- Pipeline knobs do not belong on the DiT arch config: inference steps, CFG
|
||||
scales, flow shift, FPS, VAE stride, text target length, data-proxy knobs,
|
||||
eval defaults, and sampling defaults go on `PipelineConfig`, presets, or
|
||||
stages.
|
||||
- If the HF repo is raw or has empty configs, synthesize
|
||||
`transformer/config.json` from the official Python model-config class, not
|
||||
from data/eval config classes.
|
||||
|
||||
## Phase 2: Early Parity Scaffolding
|
||||
|
||||
Create component parity tests before or alongside implementation. Use
|
||||
`../add-model-02-parity/SKILL.md` and its `templates/component_parity_test.py`.
|
||||
The official reference must import in the current FastVideo environment, or the
|
||||
prep handoff must identify private deps that will be stubbed locally for tests.
|
||||
Use `local_tests_readme` as the reviewer-facing source for setup commands and
|
||||
update its planned test table as parity scaffolds are added.
|
||||
|
||||
This phase is early by design:
|
||||
|
||||
- Official loading can be implemented from the reference study.
|
||||
- FastVideo loading can target planned standardized class/config/loader paths.
|
||||
- Tests may initially skip because the FastVideo class or converted weights do
|
||||
not exist yet.
|
||||
- The scaffold must still contain real official loading, deterministic inputs,
|
||||
output extraction, and concrete tensor comparisons. No unconditional skips,
|
||||
no shape-only tests.
|
||||
|
||||
Use subagents here: dispatch one parity-test subagent per required component,
|
||||
including components that may be reused. Their output becomes the red/skip
|
||||
target that porting or reuse-verification subagents make pass later.
|
||||
|
||||
## Phase 3: Reuse Gate And Component Dispatch
|
||||
|
||||
Build a component inventory before implementation:
|
||||
|
||||
| Field | Meaning |
|
||||
|---|---|
|
||||
| Component | transformer, VAE, text encoder, image encoder, scheduler, conditioner, upsampler, vocoder, etc. |
|
||||
| Official definition | Repo-relative source file, class/function name, and relevant line/range if known. |
|
||||
| Official instantiation | Repo-relative pipeline/config/factory call site plus constructor args and runtime flags. |
|
||||
| FastVideo target | Existing class to reuse or new bucket/file/config to add. |
|
||||
| Parity test | Required local test path, including reused components. |
|
||||
| Status | `reuse_pending`, `reuse_proven`, `port_pending`, `non_skip_pass`, or `blocked`. |
|
||||
|
||||
Reuse is allowed only from the checked-out FastVideo tree. Do not wait for or
|
||||
depend on an open PR adding a native class; add the native port directly in this
|
||||
PR if the current tree cannot be reused.
|
||||
|
||||
Reuse decision:
|
||||
|
||||
1. Record exact official definition and instantiation evidence for every
|
||||
component.
|
||||
2. If an existing FastVideo class and config match both definition and
|
||||
instantiation, pass that reused target to the bucket-specific skill in
|
||||
`mode=prototype` and require reuse evidence plus key/shape dumps.
|
||||
3. If either definition or instantiation differs, port the component directly as
|
||||
FastVideo-native code through the bucket-specific skill.
|
||||
4. Reused components still require non-skip component parity against the exact
|
||||
official instantiation used by the target pipeline.
|
||||
|
||||
Porting subagent dispatch:
|
||||
|
||||
- Dispatch one subagent per component after Phase 2 parity scaffolds exist.
|
||||
- Use `../add-model-03-port-dit/SKILL.md` for DiTs/transformers.
|
||||
- Use `../add-model-04-port-vae/SKILL.md` for VAEs.
|
||||
- Use `../add-model-05-port-encoder/SKILL.md` for text, image, audio, or compound
|
||||
encoders/conditioners that fit the encoder config bucket.
|
||||
- Use `../add-model-06-port-generic/SKILL.md` for schedulers, upsamplers,
|
||||
vocoders, adapters, preprocessors, or unknown components.
|
||||
- Each subagent owns one component only and must loop on that component's local
|
||||
parity test until it produces a non-skip PASS or returns a precise blocker.
|
||||
|
||||
Every component subagent must receive a complete packet matching
|
||||
`contracts/component_context.md`. If any required path is unknown, pass `unknown`
|
||||
plus the exact search already performed. Do not silently omit ambiguous official
|
||||
files or prototype concerns.
|
||||
|
||||
Bucket, layer, and attention rules live in the bucket-specific skills and
|
||||
`fastvideo/layers/AGENTS.md`.
|
||||
|
||||
## Phase 4: Native Component Prototype
|
||||
|
||||
Conversion needs a FastVideo state-dict surface. Use the Phase 3
|
||||
bucket-specific skill in `mode=prototype` for every required component, including
|
||||
reused components.
|
||||
|
||||
Prototype success criteria:
|
||||
|
||||
- the FastVideo-native or reused class/config can import and instantiate with the
|
||||
exact official architecture args;
|
||||
- official and FastVideo key/shape dumps exist for every stateful component;
|
||||
- `local_tests_readme` and `port_state_file` record prototype status and concerns;
|
||||
- the returned handoff matches `contracts/component_skill_handoff.md`.
|
||||
|
||||
Do not chase numerical parity in Phase 4. Prototype mode ends when conversion has
|
||||
the key/shape surface it needs, or when the component skill returns a precise
|
||||
blocker or escape hatch.
|
||||
|
||||
## Phase 5: Param Mapping And Weight Conversion
|
||||
|
||||
Use `../add-model-07-conversion/SKILL.md` after Phase 4 prototypes exist.
|
||||
Send a request matching `contracts/conversion_request.md`; consume the returned
|
||||
`contracts/conversion_handoff.md` update before Phase 6.
|
||||
|
||||
Use the prep handoff's `needs_conversion` value:
|
||||
|
||||
- `no`: verify the source already has the component layout FastVideo loaders can
|
||||
consume, then record any passthrough components.
|
||||
- `yes`: write `scripts/checkpoint_conversion/<family>_to_diffusers.py` and
|
||||
output `converted_weights/<family>/`.
|
||||
- `unknown`: return to Phase 0.
|
||||
|
||||
The conversion skill owns source-layout handling, mapping derivation, config and
|
||||
`model_index.json` emission, passthrough assets, strict-load verification, and
|
||||
Phase 6 retry requests. Component skills must not patch conversion scripts or
|
||||
converted weights ad hoc.
|
||||
|
||||
## Phase 6: Component Parity Debug
|
||||
|
||||
This is the expected expensive loop. Dispatch one subagent per required
|
||||
component, including reused components, using the bucket-specific skill in
|
||||
`mode=parity-debug`.
|
||||
|
||||
Each subagent gets:
|
||||
|
||||
- the complete component context packet from Phase 3/4;
|
||||
- updated conversion mapping notes and strict-load result from Phase 5;
|
||||
- any prototype concerns or unknowns that were not resolved before conversion.
|
||||
|
||||
The bucket-specific skills own parity-debug tactics. If a failure belongs to
|
||||
conversion, route it through `../add-model-07-conversion/SKILL.md` with a retry
|
||||
request matching `contracts/conversion_request.md`, then resume the component
|
||||
skill with the updated conversion handoff.
|
||||
|
||||
When a component failure narrows to layer-by-layer numerical drift, load
|
||||
`../add-model-08-trace/SKILL.md` before writing custom hooks. It uses
|
||||
`fastvideo/hooks/activation_trace.py`; canonical env vars and JSONL format are
|
||||
documented in `docs/contributing/activation_trace.md`.
|
||||
|
||||
Phase 6 ends only when every required component handoff reports
|
||||
`parity_status=non_skip_pass`, or when a precise blocker or escape hatch is
|
||||
recorded in `port_state_file`.
|
||||
|
||||
## Phase 7: Pipeline, Stages, And Variants
|
||||
|
||||
Do not start Phase 7 until every required component, reused or ported, has a
|
||||
non-skip local parity PASS from Phase 6. If any component parity test is still
|
||||
`scaffold_skip`, `debug_red`, `blocked`, or missing, resume Phase 6 first.
|
||||
|
||||
Use `../add-model-09-pipeline/SKILL.md` for pipeline definition and parity-debug.
|
||||
Send a complete packet matching `contracts/pipeline_context.md`; consume the
|
||||
returned `contracts/pipeline_handoff.md` before moving to quality regression or
|
||||
final handoff.
|
||||
|
||||
The pipeline skill owns:
|
||||
|
||||
- pipeline class, stage chain, and optional model-specific stages;
|
||||
- pipeline config, presets, registry updates, and examples;
|
||||
- official args/defaults/presets comparison before setting FastVideo defaults;
|
||||
- pipeline smoke and parity tests;
|
||||
- continuous pipeline parity-debug until non-skip PASS or precise blocker;
|
||||
- updates to `local_tests_readme` and `port_state_file`.
|
||||
|
||||
The pipeline handoff must explicitly cover stage order, variants, modality and
|
||||
output-head handling, config/preset/registry/example status, smoke/parity tests,
|
||||
and any return-to-Phase-6 evidence.
|
||||
|
||||
## Phase 8: PipelineConfig, Presets, Registry, Examples
|
||||
|
||||
This phase is implemented through `../add-model-09-pipeline/SKILL.md` after the
|
||||
Phase 7 component-parity gate passes. Accept the pipeline handoff only if it
|
||||
covers configs, presets, registry detection/exact class resolution, examples,
|
||||
new `SamplingParam` fields for public kwargs/defaults, and local smoke/parity
|
||||
status. Detailed rules live in `../add-model-09-pipeline/SKILL.md`.
|
||||
|
||||
## Phase 9: Parity Activation And Local Verification
|
||||
|
||||
Local parity is author-run, not CI-enforced. CI may only run package-level
|
||||
quality tests later. Before handoff, Phase 2 scaffolds must be activated into
|
||||
non-skip PASS results.
|
||||
|
||||
Order is mandatory:
|
||||
|
||||
1. Run conversion if needed.
|
||||
2. Run component parity for every required component, including reused ones.
|
||||
3. Run pipeline smoke.
|
||||
4. Run pipeline parity.
|
||||
5. Run the basic example.
|
||||
|
||||
If pipeline smoke or parity points back to component implementation,
|
||||
strict-load, or conversion mapping, return to Phase 6 or Phase 5 rather than
|
||||
patching around the issue in the pipeline.
|
||||
|
||||
Skip policy:
|
||||
|
||||
- Follow `shared/common_rules.md`: a committed local test may skip for absent
|
||||
clones/weights, but a local skip is not a verified pass.
|
||||
|
||||
Use the commands and tolerance guidance from `../add-model-02-parity/SKILL.md` for
|
||||
component checks and from `../add-model-09-pipeline/SKILL.md` for pipeline smoke,
|
||||
pipeline parity, and examples. Record exact commands, status, and blockers in
|
||||
`local_tests_readme` and `port_state_file`.
|
||||
|
||||
## Phase 10: Quality Regression
|
||||
|
||||
Video outputs:
|
||||
|
||||
- Add `fastvideo/tests/ssim/test_<family>_similarity.py` when output video
|
||||
quality must be preserved.
|
||||
- Seed references through `seed-ssim-references` after the test exists.
|
||||
|
||||
Audio outputs:
|
||||
|
||||
- SSIM does not apply. Use an audio-specific regression metric such as
|
||||
mel-spectrogram L1, multi-resolution STFT, CLAP cosine, or a project-approved
|
||||
learned metric.
|
||||
- Document the metric and hardware/runtime assumptions in the test.
|
||||
|
||||
Joint AV outputs:
|
||||
|
||||
- Keep video and audio regression checks separate unless there is a validated
|
||||
joint metric.
|
||||
|
||||
## Phase 11: Post-Parity Review And Handoff
|
||||
|
||||
After parity is green, run a hot-path review before handoff:
|
||||
|
||||
- Hoist constant tensor allocations out of sampler/denoising loops.
|
||||
- Replace per-step `randn_like` churn with preallocated buffers plus
|
||||
`.normal_()` when safe.
|
||||
- Move `torch.backends.*` flag changes to one-shot setup/load paths.
|
||||
- Delete `batch.extra` writes that nothing reads.
|
||||
- Derive magic constants from configs when possible.
|
||||
|
||||
Pre-handoff checklist:
|
||||
|
||||
```text
|
||||
[ ] Prep handoff is complete and committed nowhere with token values.
|
||||
[ ] Conversion was run if needed and output loads with real weights.
|
||||
[ ] Every required component, reused or newly ported, has a non-skip local parity PASS.
|
||||
[ ] `local_tests_readme` lists every component parity test, command, status, and blocker if any.
|
||||
[ ] `port_state_file` has every open question/issue either resolved or listed as an explicit blocker.
|
||||
[ ] Any `next_step=ask_user` has a matching `escape_hatch` block and `E###` row.
|
||||
[ ] Pipeline smoke has a non-skip local PASS.
|
||||
[ ] Pipeline parity has a non-skip local PASS against the official reference.
|
||||
[ ] Basic example runs and writes a non-corrupt output.
|
||||
[ ] Video SSIM or audio-specific quality regression is added or explicitly deferred.
|
||||
[ ] Runtime production code has no diffusers/transformers model-class imports.
|
||||
[ ] Production comments are WHY-focused; examples have user-story docstrings.
|
||||
[ ] Post-parity hot-path pass is complete.
|
||||
```
|
||||
|
||||
Ask before deleting any reference clone or staged weights created by
|
||||
`add-model-01-prep`. Leave `.gitignore` entries so future parity assets stay
|
||||
untracked. Never commit the clone, weights, `.env`, credentials, or anything
|
||||
matching `*secret*`.
|
||||
|
||||
## References
|
||||
|
||||
- `../add-model-01-prep/SKILL.md` for user-input collection, HF inspection,
|
||||
weight staging, reference cloning, and setup handoff.
|
||||
- `contracts/` for canonical handoff schemas used by prep, parity, conversion,
|
||||
component porting, escape hatches, and final handoff.
|
||||
- `../add-model-02-parity/SKILL.md` for early component parity scaffolds and
|
||||
activation templates.
|
||||
- `../add-model-07-conversion/SKILL.md` for Phase 5 mapping, conversion scripts,
|
||||
monolithic checkpoint splitting, and strict-load checks.
|
||||
- `../add-model-03-port-dit/SKILL.md`, `../add-model-04-port-vae/SKILL.md`,
|
||||
`../add-model-05-port-encoder/SKILL.md`, and
|
||||
`../add-model-06-port-generic/SKILL.md` for component subagent implementation
|
||||
and parity-debug loops.
|
||||
- `../add-model-09-pipeline/SKILL.md` for pipeline definition, config/preset/
|
||||
registry/example wiring, smoke tests, and pipeline parity-debug.
|
||||
- `fastvideo/layers/AGENTS.md` for native layer selection and state-dict surface
|
||||
guidance.
|
||||
- `docs/contributing/coding_agents.md` for narrative context.
|
||||
- `docs/design/overview.md` for pipeline/config/registry architecture.
|
||||
- `fastvideo/pipelines/basic/wan/` for standard T2V/I2V/DMD/Causal variants.
|
||||
- `fastvideo/pipelines/basic/ltx2/` for non-standard stages and audio/video
|
||||
patterns.
|
||||
- `tests/local_tests/pipelines/test_gamecraft_pipeline_parity.py` for pipeline
|
||||
parity shape.
|
||||
- `tests/local_tests/transformers/test_ltx2.py`,
|
||||
`tests/local_tests/vaes/test_ltx2_vae.py`, and
|
||||
`tests/local_tests/encoders/test_ltx2_gemma_parity.py` for component parity.
|
||||
- `scripts/checkpoint_conversion/convert_ltx2_weights.py` for modern conversion
|
||||
script shape.
|
||||
- `scripts/checkpoint_conversion/wan_to_diffusers.py` for legacy regex mapping
|
||||
reference only.
|
||||
|
||||
## Changelog
|
||||
|
||||
| Date | Change |
|
||||
|---|---|
|
||||
| 2026-04-24 | Initial FastVideo add-model workflow. |
|
||||
| 2026-04-30 | Split external setup into `add-model-01-prep`. |
|
||||
| 2026-04-30 | Rewrote as manual `/add-model` phase workflow and incorporated prior review decisions. |
|
||||
| 2026-04-30 | Extracted early parity scaffolding into `add-model-02-parity` and moved it before conversion/component implementation. |
|
||||
| 2026-04-30 | Added component reuse proof gate, bucket-specific porting skills, and parity PASS requirement for reused components. |
|
||||
| 2026-04-30 | Split prototype, conversion, and parity-debug phases; added conversion skill for monolithic and separate checkpoint layouts. |
|
||||
| 2026-04-30 | Extracted handoff schemas into `contracts/` for shared use across skills. |
|
||||
| 2026-04-30 | Added pipeline skill contract and Phase 7 component-parity gate. |
|
||||
| 2026-04-30 | Added escape-hatch contract for user decisions and `ask_user` handoffs. |
|
||||
@@ -0,0 +1,29 @@
|
||||
# Add Model Contracts
|
||||
|
||||
Canonical handoff schemas for the `/add-model` workflow. When a skill needs to
|
||||
send or receive structured context, use these files instead of inventing a local
|
||||
schema.
|
||||
|
||||
| Contract | Use |
|
||||
|---|---|
|
||||
| `prep_handoff.md` | `add-model-01-prep` output and `/add-model` Phase 0 input. |
|
||||
| `port_state.md` | Per-port `PORT_STATUS.md` file tracking progress, open questions, and issues. |
|
||||
| `escape_hatch.md` | Shared pause-and-ask schema for user decisions the workflow cannot safely choose. |
|
||||
| `component_context.md` | Per-component packet passed to parity, prototype, conversion, and parity-debug subagents. |
|
||||
| `parity_status.md` | `add-model-02-parity` scaffold/activation status returned to `/add-model`. |
|
||||
| `conversion_request.md` | Phase 5 conversion input and Phase 6 conversion retry request. |
|
||||
| `conversion_handoff.md` | `add-model-07-conversion` output back to `/add-model` and component subagents. |
|
||||
| `component_skill_handoff.md` | Component porting skill output in prototype or parity-debug mode. |
|
||||
| `pipeline_context.md` | Phase 7 packet passed to `add-model-09-pipeline` after component parity is green. |
|
||||
| `pipeline_handoff.md` | `add-model-09-pipeline` output back to `/add-model` after pipeline definition or parity-debug. |
|
||||
| `final_handoff.md` | Final `/add-model` pre-handoff checklist summary. |
|
||||
|
||||
Rules:
|
||||
|
||||
- Do not omit required fields. Use `unknown` plus the search already performed
|
||||
when the value is not known yet.
|
||||
- Do not include raw token values. Use env var names only.
|
||||
- Keep model-specific mapping details in conversion scripts and the local tests
|
||||
README/status notes, not in generic skill docs.
|
||||
- Use `next_step=ask_user` only with an `escape_hatch` block matching
|
||||
`escape_hatch.md`.
|
||||
@@ -0,0 +1,55 @@
|
||||
# Component Context Contract
|
||||
|
||||
Canonical per-component packet passed from `/add-model` to parity, prototype,
|
||||
conversion, and parity-debug subagents.
|
||||
|
||||
```text
|
||||
component_context:
|
||||
model_family: <snake_case>
|
||||
component: <name>
|
||||
component_type: <dit|vae|encoder|scheduler|conditioner|upsampler|vocoder|generic>
|
||||
mode: parity-scaffold | prototype | parity-debug
|
||||
official_ref_dir: <path or import path>
|
||||
official_definition_files:
|
||||
- path: <repo-relative or absolute path in official repo>
|
||||
symbols: <class/function names>
|
||||
notes: <layer graph, output contract, state-dict owner>
|
||||
official_instantiation_files:
|
||||
- path: <repo-relative or absolute path in official repo>
|
||||
symbols: <factory/pipeline/config names>
|
||||
args: <constructor args, config values, runtime flags>
|
||||
official_weight_source: <checkpoint file, subfolder, prefix, or passthrough source>
|
||||
fastvideo_target_files:
|
||||
- fastvideo/models/<bucket>/<file>.py
|
||||
- fastvideo/configs/models/<bucket>/<file>.py
|
||||
local_tests_readme: tests/local_tests/<model_family>/README.md
|
||||
port_state_file: tests/local_tests/<model_family>/PORT_STATUS.md
|
||||
parity_test: tests/local_tests/<bucket>/test_<family>_<component>_parity.py
|
||||
prototype_key_dumps:
|
||||
official: converted_weights/<family>/_mapping/<component>_official_keys.json | planned | unknown
|
||||
fastvideo: converted_weights/<family>/_mapping/<component>_fastvideo_keys.json | planned | unknown
|
||||
conversion:
|
||||
script: scripts/checkpoint_conversion/<family>_to_diffusers.py | not_created | not_needed | unknown
|
||||
converted_component_dir: converted_weights/<family>/<component> | not_created | not_needed | unknown
|
||||
model_index_library: <diffusers|transformers|fastvideo|fastvideo.*|unknown|none>
|
||||
config_file: <config.json|scheduler_config.json|none|unknown>
|
||||
mapping_notes: <key prefixes, split/fuse concerns, skipped keys, not_created, not_needed, or unknown>
|
||||
production_loader_strictness: <strict|non_strict_with_allowed_keys|stateless|unknown>
|
||||
strict_load: <not_run | pass | pass_with_documented_exclusions | blocked>
|
||||
concerns_or_unknowns:
|
||||
- <prototype mismatch, ambiguous arg, missing op, dtype concern, output head, etc.>
|
||||
```
|
||||
|
||||
Rules:
|
||||
|
||||
- If any required path is unknown, pass `unknown` plus the exact search already
|
||||
performed.
|
||||
- Do not silently omit ambiguous official files, instantiation args, or prototype
|
||||
concerns.
|
||||
- For reused components, still fill every field and set `fastvideo_target_files`
|
||||
to the reused class/config.
|
||||
- In `mode=parity-scaffold`, prototype and conversion fields may be `planned`,
|
||||
`not_created`, `not_needed`, or `unknown`; do not invent paths or statuses that
|
||||
do not exist yet.
|
||||
- Update `port_state_file` when concerns, issues, conversion status, or parity
|
||||
status change.
|
||||
@@ -0,0 +1,41 @@
|
||||
# Component Skill Handoff Contract
|
||||
|
||||
Returned by `add-model-03-port-dit`, `add-model-04-port-vae`,
|
||||
`add-model-05-port-encoder`, and `add-model-06-port-generic`.
|
||||
|
||||
```text
|
||||
component: <name>
|
||||
mode: prototype | parity-debug
|
||||
files_changed: <model/config/export/test/readme paths>
|
||||
official_files_used: <definition files, instantiation files>
|
||||
prototype_key_dumps: <official path, fastvideo path, or none>
|
||||
port_state_file: tests/local_tests/<model_family>/PORT_STATUS.md
|
||||
concerns_or_unknowns: <remaining or newly discovered concerns>
|
||||
parity_test: <path>
|
||||
parity_status: scaffold_skip | debug_red | non_skip_pass | blocked
|
||||
production_loader_strictness: strict | non_strict_with_allowed_keys | stateless
|
||||
strict_load: pass | pass_with_documented_exclusions | blocked | not_run
|
||||
pytest_output: <command + short result>
|
||||
blocker: <none or exact missing dependency/weights/numeric mismatch>
|
||||
conversion_retry_request: <none or failing keys/shapes/prefixes/evidence for add-model-07-conversion>
|
||||
readme_updated: yes | no
|
||||
next_step: phase_5_conversion | phase_5_conversion_retry | phase_6_continue | ask_user | blocked
|
||||
escape_hatch: <none or block matching contracts/escape_hatch.md>
|
||||
```
|
||||
|
||||
Rules:
|
||||
|
||||
- In `mode=prototype`, parity may be `scaffold_skip` or `blocked`; key dumps are
|
||||
the required artifact. Successful prototype handoff should use
|
||||
`next_step=phase_5_conversion`.
|
||||
- In `mode=parity-debug`, final success requires `parity_status=non_skip_pass`.
|
||||
- If conversion is implicated, return `conversion_retry_request` and do not edit
|
||||
conversion scripts or converted weights directly.
|
||||
- If production loading is non-strict, list allowed missing/unexpected keys in
|
||||
the parity test or handoff and mark `strict_load=pass_with_documented_exclusions`.
|
||||
- Update `port_state_file` before returning: component row, open questions,
|
||||
issues/blockers, decisions, and handoff notes.
|
||||
- Return an `escape_hatch` only for user decisions, not for normal component
|
||||
implementation or parity-debug failures.
|
||||
- Use `next_step=ask_user` only with an `escape_hatch` block and a matching
|
||||
`PORT_STATUS.md` row.
|
||||
@@ -0,0 +1,41 @@
|
||||
# Conversion Handoff Contract
|
||||
|
||||
Returned by `../add-model-07-conversion/SKILL.md` to `/add-model` and component
|
||||
parity-debug subagents.
|
||||
|
||||
```text
|
||||
conversion_script: scripts/checkpoint_conversion/<family>_to_diffusers.py
|
||||
source_layout: <diffusers|raw_official|separate_components|monolithic|mixed|custom>
|
||||
converted_weights_dir: converted_weights/<model_family>
|
||||
port_state_file: tests/local_tests/<model_family>/PORT_STATUS.md
|
||||
components_written: <list>
|
||||
passthrough_components: <list>
|
||||
strict_load: pass | pass_with_documented_exclusions | blocked
|
||||
component_context_updates:
|
||||
- component: <name>
|
||||
converted_component_dir: <path>
|
||||
model_index_library: <diffusers|transformers|fastvideo|fastvideo.*>
|
||||
config_file: <path or none>
|
||||
config_validation: pass | blocked | not_applicable
|
||||
mapping_notes: <prefixes, split/fuse ops, skipped keys>
|
||||
production_loader_strictness: strict | non_strict_with_allowed_keys | stateless
|
||||
strict_load: pass | pass_with_documented_exclusions | blocked | not_run
|
||||
retry_resolved: <yes | no | not_a_retry>
|
||||
concerns_or_unknowns: <remaining list>
|
||||
blocked_on: <none or exact blocker>
|
||||
next_step: phase_6_component_parity_debug | ask_user
|
||||
escape_hatch: <none or block matching contracts/escape_hatch.md>
|
||||
```
|
||||
|
||||
Rules:
|
||||
|
||||
- Include strict-load evidence for every stateful converted component.
|
||||
- If a component intentionally loads non-strictly, list the exact missing or
|
||||
unexpected keys and why they are safe.
|
||||
- Include the actual `model_index.json` library token, config filename, and config
|
||||
validation result for every emitted component.
|
||||
- Preserve retry evidence so the requesting component subagent can resume with
|
||||
updated context.
|
||||
- Keep `port_state_file` synchronized with `component_context_updates`.
|
||||
- Use `next_step=ask_user` only with an `escape_hatch` block and a matching
|
||||
`PORT_STATUS.md` row.
|
||||
@@ -0,0 +1,54 @@
|
||||
# Conversion Request Contract
|
||||
|
||||
Consumed by `../add-model-07-conversion/SKILL.md` in Phase 5 and during Phase 6
|
||||
conversion retries.
|
||||
|
||||
Initial conversion request:
|
||||
|
||||
```text
|
||||
model_family: <snake_case>
|
||||
source_layout: diffusers | raw_official | monolithic | separate_components | mixed | custom
|
||||
official_weights: <HF repo, local dir, or checkpoint file>
|
||||
hf_revision: <revision | default | none>
|
||||
converted_weights_dir: converted_weights/<model_family>
|
||||
local_tests_readme: tests/local_tests/<model_family>/README.md
|
||||
port_state_file: tests/local_tests/<model_family>/PORT_STATUS.md
|
||||
components:
|
||||
- name: <transformer|vae|text_encoder|conditioner|scheduler|...>
|
||||
component_type: <dit|vae|encoder|scheduler|conditioner|upsampler|vocoder|generic>
|
||||
official_definition_files: <paths + symbols>
|
||||
official_instantiation_files: <paths + call sites + args>
|
||||
official_weight_source: <checkpoint file, prefix, subfolder, or passthrough source>
|
||||
official_keys: <path to official key/shape dump>
|
||||
fastvideo_keys: <path to FastVideo prototype key/shape dump>
|
||||
fastvideo_class: <class name>
|
||||
model_index_library: <diffusers|transformers|fastvideo|fastvideo.*>
|
||||
config_filename: <config.json|scheduler_config.json|none>
|
||||
production_loader_strictness: <strict|non_strict_with_allowed_keys|stateless>
|
||||
source_prefix_or_path: <prefix or path>
|
||||
parity_test: <component parity test path>
|
||||
prototype_concerns_or_unknowns: <short list>
|
||||
```
|
||||
|
||||
Retry request from a component skill:
|
||||
|
||||
```text
|
||||
conversion_retry_request:
|
||||
component: <name>
|
||||
parity_test: <path>
|
||||
failing_keys: <official and FastVideo keys, if known>
|
||||
expected_actual_shapes: <expected vs actual shapes, if known>
|
||||
source_prefix_or_path: <prefix/path implicated by the failure>
|
||||
evidence: <strict-load error, first divergent tensor, parity log excerpt>
|
||||
suspected_fix: <rename | split | fuse | skip | component bucket | config | unknown>
|
||||
```
|
||||
|
||||
Rules:
|
||||
|
||||
- Phase 5 conversion requires Phase 4 official/FastVideo key dumps.
|
||||
- Component skills must use the retry request instead of editing conversion
|
||||
scripts or converted weights directly.
|
||||
- Conversion must update `port_state_file` with conversion status, retry history,
|
||||
strict-load status, new issues, and resolved issues.
|
||||
- Conversion must validate emitted config keys through the production config
|
||||
update path and record the config filename expected by each loader.
|
||||
@@ -0,0 +1,51 @@
|
||||
# Escape Hatch Contract
|
||||
|
||||
Canonical pause-and-ask schema for `/add-model` skills. Use this when the next
|
||||
action requires user input instead of autonomous debugging.
|
||||
|
||||
```text
|
||||
escape_hatch:
|
||||
needs_user_input: yes | no
|
||||
decision_type: scope | dependency | auth | cost | destructive | ambiguity | blocker
|
||||
question: <one precise question>
|
||||
recommended_option: <safe recommended choice>
|
||||
options:
|
||||
- <option + consequence>
|
||||
safe_default: <what the agent will do after approval, or none>
|
||||
blocked_until_answered: yes | no
|
||||
state_snapshot:
|
||||
phase: <phase or skill mode>
|
||||
files_changed:
|
||||
- <paths>
|
||||
command_or_test: <last relevant command, or not_run>
|
||||
evidence: <short logs, paths, error text, or blocker ID>
|
||||
```
|
||||
|
||||
Use `needs_user_input=no` when the handoff is green or the next step is already
|
||||
specified by the workflow.
|
||||
|
||||
Ask the user only for decisions the workflow cannot safely choose:
|
||||
|
||||
- product or PR scope changes, including dropping a modality, output head, or
|
||||
variant;
|
||||
- core dependency changes, version pin changes, or installing untrusted/private
|
||||
dependencies;
|
||||
- auth setup for gated repos, using env var names only and never token values;
|
||||
- large downloads, publishing weights, SSIM/reference uploads, or GPU-heavy work
|
||||
where cost/runtime approval is needed;
|
||||
- destructive file/git operations, overwriting existing clones/weights, or
|
||||
deleting staged assets;
|
||||
- ambiguous official sources of truth with incompatible behavior;
|
||||
- accepting a known blocker, loosening parity/quality tolerances, or shipping
|
||||
without required non-skip parity.
|
||||
|
||||
Do not ask for normal recoverable failures:
|
||||
|
||||
- missing imports, missing local paths, skipped tests, failing parity, conversion
|
||||
mapping errors, strict-load failures, format/lint failures, or implementation
|
||||
bugs covered by the skill workflow.
|
||||
|
||||
Before returning `next_step=ask_user`, update
|
||||
`tests/local_tests/<model_family>/PORT_STATUS.md` with the blocker/question ID,
|
||||
include the exact evidence, and provide one recommended option plus at most three
|
||||
alternatives.
|
||||
@@ -0,0 +1,38 @@
|
||||
# Final Handoff Contract
|
||||
|
||||
Completed by `/add-model` before handing work back to the user or opening a PR.
|
||||
|
||||
```text
|
||||
final_handoff:
|
||||
prep_handoff_complete: yes | no
|
||||
conversion_status: not_needed | pass | blocked
|
||||
components:
|
||||
- name: <component>
|
||||
reuse_or_port: reused | ported
|
||||
parity_test: <path>
|
||||
parity_status: non_skip_pass | blocked
|
||||
concerns_or_unknowns: <none or list>
|
||||
pipeline_smoke: pass | blocked | not_run
|
||||
pipeline_parity: pass | blocked | not_run
|
||||
example_status: pass | blocked | not_run
|
||||
quality_regression: added | deferred_with_reason | not_applicable
|
||||
local_tests_readme: tests/local_tests/<model_family>/README.md
|
||||
port_state_file: tests/local_tests/<model_family>/PORT_STATUS.md
|
||||
token_values_committed: no
|
||||
runtime_third_party_model_imports: none | listed_with_rationale
|
||||
blockers: <none or list>
|
||||
escape_hatch: <none or block matching contracts/escape_hatch.md>
|
||||
```
|
||||
|
||||
Required before handoff:
|
||||
|
||||
- Every required component, reused or ported, has non-skip local parity PASS.
|
||||
- Pipeline smoke and pipeline parity are non-skip PASS, or a blocker is explicit.
|
||||
- Basic example runs and writes a non-corrupt output.
|
||||
- `local_tests_readme` lists every component parity command/status/blocker.
|
||||
- `port_state_file` has no unresolved blocker that is omitted from the final
|
||||
response or PR notes.
|
||||
- No raw HF token values, credentials, `.env`, reference clone, or staged weight
|
||||
blobs are committed.
|
||||
- If final handoff is blocked on user input, include an `escape_hatch` block and
|
||||
matching `PORT_STATUS.md` row.
|
||||
@@ -0,0 +1,44 @@
|
||||
# Parity Status Contract
|
||||
|
||||
Returned by `../add-model-02-parity/SKILL.md` to `/add-model` and later updated by
|
||||
component parity-debug subagents.
|
||||
|
||||
```text
|
||||
component_parity:
|
||||
- component: <name>
|
||||
test: tests/local_tests/<bucket>/test_<family>_<component>_parity.py
|
||||
status: scaffold_skip | debug_red | non_skip_pass | blocked
|
||||
missing: <none | fastvideo_class | converted_weights | official_import | ...>
|
||||
coverage_scope: production_loader | implementation_subcomponent | both
|
||||
official_definition_files: <paths>
|
||||
official_instantiation_files: <paths>
|
||||
concerns_or_unknowns: <short list>
|
||||
pipeline_parity:
|
||||
test: <path or not-created>
|
||||
status: not_started | scaffold_skip | debug_red | non_skip_pass | blocked
|
||||
local_tests_readme: tests/local_tests/<model_family>/README.md
|
||||
port_state_file: tests/local_tests/<model_family>/PORT_STATUS.md
|
||||
notes: <short list>
|
||||
escape_hatch: <none or block matching contracts/escape_hatch.md>
|
||||
```
|
||||
|
||||
Status meanings:
|
||||
|
||||
- `scaffold_skip`: test is present but skips for a specific missing dependency,
|
||||
FastVideo class, or weights.
|
||||
- `debug_red`: both sides load and the test fails numerically.
|
||||
- `non_skip_pass`: required before final handoff for every required component,
|
||||
including reused components.
|
||||
- `blocked`: a precise missing dependency, weight, official call path, or
|
||||
component/conversion regression prevents local activation.
|
||||
|
||||
Coverage meanings:
|
||||
|
||||
- `production_loader`: FastVideo side loads through the same loader/path used by
|
||||
a pipeline.
|
||||
- `implementation_subcomponent`: FastVideo side constructs classes or remaps
|
||||
tensors directly to isolate implementation behavior.
|
||||
- `both`: the test covers both production loading and implementation behavior.
|
||||
|
||||
Use `escape_hatch` only when blocked status requires a user decision. Normal
|
||||
skips or red parity should be debugged by the workflow without asking.
|
||||
@@ -0,0 +1,82 @@
|
||||
# Pipeline Context Contract
|
||||
|
||||
Canonical packet passed from `/add-model` to `add-model-09-pipeline` for pipeline
|
||||
definition and pipeline parity-debug work.
|
||||
|
||||
```text
|
||||
pipeline_context:
|
||||
model_family: <snake_case>
|
||||
mode: pipeline-definition | pipeline-parity-debug
|
||||
workload_types:
|
||||
- <T2V|I2V|V2V|T2I|compatibility-shim-with-rationale>
|
||||
modalities:
|
||||
inputs: <text/image/video/audio/pose/depth/mask/etc.>
|
||||
outputs: <video/image/audio/joint-av/latents/etc.>
|
||||
official_ref_dir: <path or import path>
|
||||
official_pipeline_files:
|
||||
- path: <repo-relative or absolute path in official repo>
|
||||
symbols: <pipeline/factory/sample functions>
|
||||
notes: <stage order, mutable state, output contract>
|
||||
official_call:
|
||||
command_or_api: <official CLI, Python call, or package entrypoint>
|
||||
args_and_defaults: <height, width, frames, fps, duration, steps, CFG, scheduler, seeds, etc.>
|
||||
preset_source: <model card, config file, official script, or unknown>
|
||||
scheduler_and_rng: <timestep/sigma/noise/generator behavior>
|
||||
output_contract: <decoded media, denoised latents, waveform, dict keys, etc.>
|
||||
model_index:
|
||||
class_name: <FastVideo pipeline class name to emit in model_index.json>
|
||||
entry_class_names: <registered EntryClass.__name__ values that must include class_name>
|
||||
required_modules: <text_encoder, tokenizer, vae, transformer, scheduler, etc.>
|
||||
passthrough_modules: <tokenizer, scheduler, processor, external HF dirs, or none>
|
||||
sampling_param:
|
||||
new_fields: <none or list of public kwargs/preset defaults to add to SamplingParam>
|
||||
cli_fields: <none or list of fields that need CLI args>
|
||||
placeholder_fields: <none or video-shaped compatibility placeholders with rationale>
|
||||
components:
|
||||
- name: <component>
|
||||
component_type: <dit|vae|encoder|scheduler|conditioner|upsampler|vocoder|generic>
|
||||
parity_test: tests/local_tests/<bucket>/test_<family>_<component>_parity.py
|
||||
parity_status: non_skip_pass
|
||||
fastvideo_target_files: <model/config/export files>
|
||||
converted_component_dir: converted_weights/<family>/<component>
|
||||
conversion:
|
||||
converted_weights_dir: converted_weights/<family>
|
||||
source_layout: <diffusers|raw_official|monolithic|separate_components|mixed|custom>
|
||||
model_index_path: converted_weights/<family>/model_index.json
|
||||
fastvideo_targets:
|
||||
pipeline_files:
|
||||
- fastvideo/pipelines/basic/<family>/<family>_pipeline.py
|
||||
stage_files:
|
||||
- fastvideo/pipelines/basic/<family>/stages/<stage>.py
|
||||
pipeline_config_files:
|
||||
- fastvideo/configs/pipelines/<family>.py
|
||||
preset_file: fastvideo/pipelines/basic/<family>/presets.py
|
||||
registry_file: fastvideo/registry.py
|
||||
example_files:
|
||||
- examples/inference/basic/basic_<family>.py
|
||||
smoke_test: tests/local_tests/pipelines/test_<family>_pipeline_smoke.py
|
||||
parity_test: tests/local_tests/pipelines/test_<family>_pipeline_parity.py
|
||||
local_tests_readme: tests/local_tests/<model_family>/README.md
|
||||
port_state_file: tests/local_tests/<model_family>/PORT_STATUS.md
|
||||
concerns_or_unknowns:
|
||||
- <pipeline branch, unsupported workload, output head, preset ambiguity, etc.>
|
||||
```
|
||||
|
||||
Rules:
|
||||
|
||||
- Start only after every required component, reused or ported, has
|
||||
`parity_status=non_skip_pass`. If any row is missing or skipped, return to
|
||||
`/add-model` Phase 6.
|
||||
- Record official call arguments and default sources before writing FastVideo
|
||||
presets. Do not invent inference defaults from memory.
|
||||
- `model_index.class_name` must match a registered pipeline `EntryClass.__name__`;
|
||||
registry detectors are not sufficient for executable pipeline resolution.
|
||||
- Every public generation kwarg or preset default must be represented in
|
||||
`SamplingParam`, or documented as an intentional internal-only field.
|
||||
- `T2A`, `A2A`, and `AV` may be used only after `WorkloadType` supports them;
|
||||
otherwise record the compatibility shim and rationale explicitly.
|
||||
- Keep token values out of the packet. Use only token environment variable names.
|
||||
- If a target path is unknown, use `unknown` plus the exact search already
|
||||
performed.
|
||||
- Update `local_tests_readme` and `port_state_file` whenever pipeline smoke,
|
||||
parity, presets, registry, examples, or blockers change.
|
||||
@@ -0,0 +1,71 @@
|
||||
# Pipeline Handoff Contract
|
||||
|
||||
Returned by `add-model-09-pipeline` to `/add-model` after pipeline definition or
|
||||
pipeline parity-debug work.
|
||||
|
||||
```text
|
||||
pipeline_handoff:
|
||||
model_family: <snake_case>
|
||||
mode: pipeline-definition | pipeline-parity-debug
|
||||
files_changed:
|
||||
- <pipeline/config/preset/registry/stage/example/test/readme/status paths>
|
||||
official_files_used:
|
||||
- <definition/call/default source paths>
|
||||
required_config_modules:
|
||||
emitted: <list from pipeline class>
|
||||
model_index: <list from converted or source model_index.json>
|
||||
status: match | mismatch | blocked
|
||||
pipeline_class_resolution:
|
||||
model_index_class_name: <_class_name>
|
||||
entry_class_names: <registered EntryClass.__name__ values>
|
||||
status: exact_match | alias_added | blocked
|
||||
sampling_param:
|
||||
fields_added: <none or list>
|
||||
cli_fields_added: <none or list>
|
||||
unknown_kwargs_checked: yes | no | blocked
|
||||
stage_chain:
|
||||
- <stage names in execution order>
|
||||
pipeline_config:
|
||||
file: <path>
|
||||
classes: <class names>
|
||||
official_defaults_checked: yes | no | blocked
|
||||
presets:
|
||||
file: <path>
|
||||
names: <preset names>
|
||||
status: pass | blocked | not_run
|
||||
registry:
|
||||
status: pass | blocked | not_run
|
||||
detectors: <HF paths and model_index _class_name strings covered>
|
||||
smoke_test:
|
||||
path: tests/local_tests/pipelines/test_<family>_pipeline_smoke.py
|
||||
status: non_skip_pass | blocked | not_run
|
||||
pytest_output: <command + short result>
|
||||
pipeline_parity:
|
||||
path: tests/local_tests/pipelines/test_<family>_pipeline_parity.py
|
||||
status: scaffold_skip | debug_red | non_skip_pass | blocked
|
||||
pytest_output: <command + short result>
|
||||
comparison_target: <latents|decoded video|audio|joint outputs>
|
||||
example:
|
||||
path: examples/inference/basic/basic_<family>.py
|
||||
status: pass | blocked | not_run
|
||||
output: <path or none>
|
||||
readme_updated: yes | no
|
||||
port_state_updated: yes | no
|
||||
blockers: <none or exact blocker list>
|
||||
next_step: phase_10_quality_regression | return_to_phase_6 | ask_user
|
||||
escape_hatch: <none or block matching contracts/escape_hatch.md>
|
||||
```
|
||||
|
||||
Rules:
|
||||
|
||||
- `pipeline-definition` may return with parity still `scaffold_skip` only if the
|
||||
exact missing dependency, weight, or call-path blocker is recorded.
|
||||
- Final `/add-model` handoff requires `smoke_test.status=non_skip_pass` and
|
||||
`pipeline_parity.status=non_skip_pass`, unless the user explicitly accepts a
|
||||
documented blocker.
|
||||
- If parity failure traces to a component, conversion, or strict-load issue,
|
||||
return `next_step=return_to_phase_6` and include the exact failing evidence.
|
||||
- Keep `local_tests_readme` and `port_state_file` synchronized with this
|
||||
handoff before returning.
|
||||
- Use `next_step=ask_user` only with an `escape_hatch` block and a matching
|
||||
`PORT_STATUS.md` row.
|
||||
@@ -0,0 +1,89 @@
|
||||
# Port State Contract
|
||||
|
||||
Canonical per-port state file created during prep and updated by every
|
||||
`/add-model` phase.
|
||||
|
||||
Path:
|
||||
|
||||
```text
|
||||
tests/local_tests/<model_family>/PORT_STATUS.md
|
||||
```
|
||||
|
||||
Purpose:
|
||||
|
||||
- Single source of truth for resumable port progress.
|
||||
- Tracks component status, conversion status, parity status, open questions,
|
||||
blockers, escape hatches, and issue history.
|
||||
- Lets review agents run the same setup/tests without reconstructing handoffs
|
||||
from conversation history.
|
||||
|
||||
Required sections:
|
||||
|
||||
```text
|
||||
# <Model Family> Port Status
|
||||
|
||||
## Summary
|
||||
- model_family:
|
||||
- workload_types:
|
||||
- official_ref:
|
||||
- official_ref_dir:
|
||||
- hf_weights_path:
|
||||
- local_weights_dir:
|
||||
- source_layout:
|
||||
- local_tests_readme:
|
||||
|
||||
## Current Phase
|
||||
- phase:
|
||||
- status: not_started | in_progress | blocked | complete
|
||||
- owner: orchestrator | prep | parity | conversion | component:<name> | pipeline
|
||||
- last_updated:
|
||||
|
||||
## Component Matrix
|
||||
| Component | Type | Reuse/Port | Official Definition | Official Instantiation | FastVideo Target | Prototype | Conversion | Parity | Open Issues |
|
||||
|---|---|---|---|---|---|---|---|---|---|
|
||||
|
||||
## Conversion State
|
||||
- conversion_script:
|
||||
- converted_weights_dir:
|
||||
- source_layout:
|
||||
- strict_load_status:
|
||||
- passthrough_components:
|
||||
- retry_history:
|
||||
|
||||
## Parity Commands
|
||||
| Scope | Command | Last Result | Notes |
|
||||
|---|---|---|---|
|
||||
|
||||
## Open Questions
|
||||
| ID | Question | Owner | Needed By Phase | Status | Resolution |
|
||||
|---|---|---|---|---|---|
|
||||
|
||||
## Issues And Blockers
|
||||
| ID | Phase | Component | Severity | Issue | Evidence | Owner | Status | Resolution |
|
||||
|---|---|---|---|---|---|---|---|---|
|
||||
|
||||
## Escape Hatches
|
||||
| ID | Phase | Decision Type | Question | Recommended Option | Status | Resolution |
|
||||
|---|---|---|---|---|---|---|
|
||||
|
||||
## Decisions
|
||||
| Date | Decision | Rationale | Impact |
|
||||
|---|---|---|---|
|
||||
|
||||
## Handoff Notes
|
||||
- <short notes for the next agent>
|
||||
```
|
||||
|
||||
Rules:
|
||||
|
||||
- Update this file whenever a phase starts, blocks, resolves an issue, or hands
|
||||
off to another skill.
|
||||
- Record open questions and issues immediately. Do not leave blockers only in
|
||||
chat history or subagent responses.
|
||||
- Use stable IDs: `Q001`, `Q002`, `I001`, `I002`, etc.
|
||||
- Use stable escape-hatch IDs: `E001`, `E002`, etc. Link them from handoff
|
||||
`escape_hatch.state_snapshot.evidence` when returning `next_step=ask_user`.
|
||||
- Do not include raw token values, machine-local cache internals, or large output
|
||||
dumps. Use repo-relative paths when possible.
|
||||
- If a question or issue is resolved, keep the row and fill `Resolution` instead
|
||||
of deleting it.
|
||||
@@ -0,0 +1,42 @@
|
||||
# Prep Handoff Contract
|
||||
|
||||
Produced by `../add-model-01-prep/SKILL.md` and consumed by `/add-model` Phase 0.
|
||||
|
||||
```text
|
||||
model_family: <snake_case>
|
||||
workload_types: <T2V/I2V/V2V/T2I/or compatibility shim with rationale>
|
||||
official_ref: <url or import path>
|
||||
official_ref_dir: <ReferenceDir or none>
|
||||
official_ref_commit: <sha or unknown>
|
||||
hf_weights_path: <HF id or local path>
|
||||
hf_revision: <revision or default>
|
||||
local_weights_dir: official_weights/<model_family> or <local path>
|
||||
source_layout: diffusers | raw_official | monolithic | separate_components | mixed | custom | unknown
|
||||
model_index_class: <_class_name or none>
|
||||
components_seen: <components>
|
||||
needs_conversion: yes | no | unknown
|
||||
hf_token_env: <env var name only>
|
||||
dependency_changes: none | installed no-deps editable | installed official deps in current env | blocked on user
|
||||
official_env_status: imports_ok | private_deps_need_stubs | blocked
|
||||
local_tests_readme: tests/local_tests/<model_family>/README.md
|
||||
port_state_file: tests/local_tests/<model_family>/PORT_STATUS.md
|
||||
gitignore_entries_added: <list>
|
||||
next_step: add-model | ask_user
|
||||
open_questions: <short list>
|
||||
escape_hatch: <none or block matching contracts/escape_hatch.md>
|
||||
```
|
||||
|
||||
Validation:
|
||||
|
||||
- `official_env_status` must be `imports_ok` or `private_deps_need_stubs` before
|
||||
component parity scaffolding.
|
||||
- `local_tests_readme` must exist and describe official setup, HF weights,
|
||||
dependency changes, planned parity commands, and review notes.
|
||||
- `port_state_file` must exist and follow `contracts/port_state.md`.
|
||||
- Prep does not go directly to conversion; `/add-model` must run component
|
||||
prototype/key-dump Phase 4 before Phase 5 conversion.
|
||||
- `T2A`, `A2A`, and `AV` may be used only after `WorkloadType` supports them;
|
||||
otherwise record the compatibility shim and rationale explicitly.
|
||||
- Never include HF token values.
|
||||
- Use `next_step=ask_user` only with an `escape_hatch` block and a matching
|
||||
`PORT_STATUS.md` row.
|
||||
@@ -0,0 +1,96 @@
|
||||
# Shared Add-Model Rules
|
||||
|
||||
These rules apply to every `add-model` related skill: prep, parity, conversion,
|
||||
component porting, pipeline, and the main `/add-model` orchestrator.
|
||||
|
||||
## Token And Auth Safety
|
||||
|
||||
- Never accept, print, echo, log, hard-code, or commit raw HF token values.
|
||||
- Refer only to token environment variable names: `HF_TOKEN`,
|
||||
`HUGGINGFACE_HUB_TOKEN`, or `HF_API_KEY`.
|
||||
- Scripts may read those environment variables but must not print their values.
|
||||
- Ask for auth setup only by env var name. Do not ask the user to paste a token.
|
||||
- Read scope is needed for gated repos during conversion/load. Write scope is
|
||||
needed for publishing converted weights or seeding generated references.
|
||||
|
||||
## Shared State Files
|
||||
|
||||
- `tests/local_tests/<model_family>/README.md` is the reviewer-facing setup and
|
||||
verification log. Keep it current with setup commands, dependency blockers,
|
||||
parity commands, conversion commands, and pass/blocker status.
|
||||
- `tests/local_tests/<model_family>/PORT_STATUS.md` is the per-port state file.
|
||||
It must follow `../contracts/port_state.md` and keep stable `Q###`, `I###`, and
|
||||
`E###` IDs.
|
||||
- Keep resolved questions/issues in `PORT_STATUS.md` with the resolution instead
|
||||
of deleting them.
|
||||
- Before returning a handoff, update both state files when the skill changed
|
||||
setup, tests, conversion, parity status, blockers, or decisions.
|
||||
- Do not include raw tokens, non-reproducible absolute cache paths, large
|
||||
generated outputs, `.env`, credentials, or anything matching `*secret*`.
|
||||
|
||||
## Escape Hatches
|
||||
|
||||
Continue autonomously for recoverable setup, implementation, conversion,
|
||||
strict-load, smoke, parity-debug, lint, or test failures. Stop and ask the user
|
||||
only when the next action requires a product, cost, safety, auth, dependency, or
|
||||
scope decision the workflow cannot safely choose.
|
||||
|
||||
Use `../contracts/escape_hatch.md` whenever returning `next_step=ask_user`.
|
||||
|
||||
Ask for user input only for:
|
||||
|
||||
- scope changes, such as dropping a modality, output head, variant, component, or
|
||||
public mode;
|
||||
- core dependency changes, untrusted/private dependency installs, or version pin
|
||||
changes;
|
||||
- auth setup for gated repos, using env var names only;
|
||||
- large downloads, publishing weights, SSIM/reference uploads, or GPU-heavy work
|
||||
where cost/runtime approval is needed;
|
||||
- destructive file/git operations, overwriting existing clones/weights, or
|
||||
deleting staged assets;
|
||||
- incompatible official sources of truth where no reference can be chosen from
|
||||
published weights and docs;
|
||||
- accepting a blocker, loosening tolerances, using shape-only substitutes, or
|
||||
shipping without required non-skip parity.
|
||||
|
||||
Do not ask for normal recoverable failures: missing imports, missing local paths,
|
||||
skipped tests, failing parity, conversion mapping bugs, strict-load errors,
|
||||
format/lint failures, smoke failures, registry import issues, example failures,
|
||||
or implementation bugs covered by the phase workflow.
|
||||
|
||||
Before asking:
|
||||
|
||||
- update `PORT_STATUS.md` with an `E###` escape-hatch row plus any linked `Q###`
|
||||
or `I###` row;
|
||||
- include exact evidence: command, path, short error text, parity or strict-load
|
||||
excerpt, or blocker ID;
|
||||
- provide one recommended option and at most three alternatives;
|
||||
- set the relevant handoff `next_step=ask_user` and include the `escape_hatch`
|
||||
block.
|
||||
|
||||
Skill-specific escape-hatch sections may add extra examples, but they must not
|
||||
weaken these shared rules.
|
||||
|
||||
## Production Boundary
|
||||
|
||||
- No runtime `from diffusers import <model class>` or
|
||||
`from transformers import <model class>` in `fastvideo/` production code.
|
||||
- Components that own weights or numerical behavior must be FastVideo-native
|
||||
unless the user explicitly accepts a documented lazy-wrapper exception.
|
||||
- Allowed third-party runtime exceptions are tokenizers and pure data utilities
|
||||
when they match existing project patterns.
|
||||
- Tests may import diffusers/transformers as parity references.
|
||||
- Production comments explain why, not what or provenance. Avoid narrative
|
||||
comments like `vendored from`, `matches upstream`, `REVIEW`, or session-history
|
||||
commentary.
|
||||
|
||||
## Verification Semantics
|
||||
|
||||
- A committed local test may skip when clones, weights, or private deps are absent
|
||||
so CI and other contributors are not blocked.
|
||||
- On the porter's machine, a skip is not a pass. Fix the missing import, weights,
|
||||
or path before claiming verification.
|
||||
- New ports require local non-skip parity for required components and pipeline
|
||||
parity when a pipeline is in scope.
|
||||
- Smoke tests prove loadability only. They are not a substitute for numerical
|
||||
component or pipeline parity.
|
||||
@@ -0,0 +1,117 @@
|
||||
# Component Skill Common Instructions
|
||||
|
||||
These instructions apply to `add-model-03-port-dit`, `add-model-04-port-vae`,
|
||||
`add-model-05-port-encoder`, and `add-model-06-port-generic`. Bucket-specific skills
|
||||
add target paths, implementation patterns, drift checks, and scope questions.
|
||||
|
||||
## Required Context
|
||||
|
||||
Require the complete packet from `../contracts/component_context.md`.
|
||||
|
||||
Do not start if the official definition files, official instantiation files, or
|
||||
parity test path are missing. Ask the `/add-model` orchestrator for the complete
|
||||
component context packet instead of rediscovering broad scope silently.
|
||||
|
||||
If the parity scaffold is missing, create it first with
|
||||
`../../add-model-02-parity/templates/component_parity_test.py`.
|
||||
|
||||
## Prototype Mode
|
||||
|
||||
Prototype mode runs before conversion:
|
||||
|
||||
- implement or prove reuse for the minimal native component, config, export, and
|
||||
`EntryClass` surface needed by the relevant loader;
|
||||
- instantiate with random weights using the exact official architecture args, or
|
||||
instantiate/document stateless components with no weights;
|
||||
- dump official and FastVideo `state_dict()` names/shapes for every stateful
|
||||
component so conversion can derive mappings from real surfaces;
|
||||
- return concerns discovered during prototype work, such as ambiguous official
|
||||
flags, shape mismatches, private ops, missing loader buckets, passthrough
|
||||
weights, or output heads;
|
||||
- update `local_tests_readme` with prototype status and key-dump paths;
|
||||
- update `port_state_file` with prototype status, open questions, issues, and
|
||||
handoff notes;
|
||||
- do not chase numerical parity and do not block on converted weights.
|
||||
|
||||
Prototype mode succeeds when the component imports, instantiates with official
|
||||
args, and required key/shape dumps exist. Converted weights and parity PASS are
|
||||
not required yet.
|
||||
|
||||
## Parity-Debug Mode
|
||||
|
||||
Parity-debug mode runs after conversion:
|
||||
|
||||
- strict-load converted weights through the same path the pipeline will use, or
|
||||
document that the component is stateless or an approved passthrough;
|
||||
- use `conversion_context` and `concerns_or_unknowns` to decide whether a failure
|
||||
belongs to mapping, loading, implementation, tokenization, normalization,
|
||||
scheduler semantics, or the parity test;
|
||||
- run only the component parity test first with `pytest <parity_test> -v -s`;
|
||||
- if it skips, fix the missing official import, FastVideo class, tokenizer,
|
||||
converted weights, or path;
|
||||
- if it fails numerically, add targeted intermediate comparisons to identify the
|
||||
first divergent operation or tensor;
|
||||
- update component implementation only when the failure is a component
|
||||
layer/config/forward/contract bug;
|
||||
- update `local_tests_readme` with the command, result, and blocker or PASS;
|
||||
- update `port_state_file` with parity status, resolved/new issues, open
|
||||
questions, and handoff notes;
|
||||
- keep iterating until the test is a non-skip PASS or return a precise blocker.
|
||||
|
||||
## Conversion Boundary
|
||||
|
||||
Component skills must not patch conversion scripts or converted weights ad hoc.
|
||||
If the first drift or strict-load failure points to wrong keys, missing tensors,
|
||||
shape mismatches, component prefixes, split/fuse logic, skipped-key policy, or
|
||||
config emission, return a conversion retry request for
|
||||
`../../add-model-07-conversion/SKILL.md` matching
|
||||
`../contracts/conversion_request.md`.
|
||||
Resume parity-debug only after conversion returns an updated handoff.
|
||||
|
||||
## Reuse Proof
|
||||
|
||||
When `fastvideo_target_files` point to existing FastVideo code instead of a new
|
||||
port:
|
||||
|
||||
- compare the official definition against the FastVideo target: graph/operation
|
||||
structure, parameter or state shapes, normalization, activation, positional or
|
||||
temporal behavior, scaling constants, dtype behavior, state-dict names, output
|
||||
containers, and output tensors;
|
||||
- compare the official instantiation against the FastVideo config and loader
|
||||
args: constructor args, config values, defaults, variant flags, optional
|
||||
submodules, checkpoint metadata, tokenizer/media paths, and loader path;
|
||||
- treat a matching class instantiated with different args as not reusable;
|
||||
- record reuse evidence in `local_tests_readme` and keep the reused component in
|
||||
`prototype_key_dumps` when it owns state so conversion and parity-debug use the
|
||||
same surface;
|
||||
- still run parity-debug to a non-skip PASS. If mismatch is found, return the
|
||||
concern so `/add-model` can switch the component to a native port.
|
||||
|
||||
## Handoff
|
||||
|
||||
Return `../contracts/component_skill_handoff.md`.
|
||||
|
||||
Mode-specific expectations:
|
||||
|
||||
- In `mode=prototype`, `parity_status` may be `scaffold_skip` or `blocked`; key
|
||||
dumps are the required artifact and successful prototype handoff should use
|
||||
`next_step=phase_5_conversion`.
|
||||
- In `mode=parity-debug`, final success requires
|
||||
`parity_status=non_skip_pass`.
|
||||
- If conversion is implicated, return `conversion_retry_request` and leave
|
||||
conversion edits to `add-model-07-conversion`.
|
||||
- If production loading is non-strict, list allowed missing/unexpected keys in
|
||||
the parity test or handoff and mark
|
||||
`strict_load=pass_with_documented_exclusions`.
|
||||
|
||||
## Escape Hatches
|
||||
|
||||
Follow `common_rules.md`. Do not ask for normal prototype or parity-debug
|
||||
failures such as missing imports, tokenizer/path issues, red parity,
|
||||
strict-load failures, key mismatches, shape mismatches, or implementation bugs.
|
||||
Return conversion retry requests or precise blockers as directed by the workflow.
|
||||
|
||||
Ask only when component work requires a scope or safety decision, such as
|
||||
dropping a required stream/output/path, changing core dependencies, accepting
|
||||
private model code or unsupported private ops, choosing between incompatible
|
||||
official definitions, creating a new loader bucket, or loosening required parity.
|
||||
@@ -0,0 +1,99 @@
|
||||
---
|
||||
name: ci-runner
|
||||
description: Work on FastVideo's Slurm-only, change-aware GPU CI lanes, static Buildkite graph, trusted ci-runner policy, lane scripts, and GB200 validation.
|
||||
---
|
||||
|
||||
# Slinky Slurm CI lanes
|
||||
|
||||
FastVideo's `ci-runner` Buildkite queue is the control plane for all active
|
||||
GPU CI. A private host-owned dispatcher leases GPUs from the Slinky Slurm tray
|
||||
and runs the immutable PR SHA inside an isolated Enroot container. Buildkite
|
||||
pipeline upload and Slurm submission occur on the login plane; every test
|
||||
payload executes on Slurm compute.
|
||||
|
||||
The files under `fastvideo/tests/modal/` and `.buildkite/scripts/pr_test.sh`
|
||||
are dormant rollback code. Never add an active Buildkite or slash-command
|
||||
route to them. `pr_test.sh` must continue to reject Buildkite invocations.
|
||||
|
||||
The private operator bundle is deliberately outside this repository because
|
||||
it contains site paths and credentials. See
|
||||
`docs/contributing/ci_architecture.md`; this skill covers the repository half
|
||||
and the coordination contract with that bundle.
|
||||
|
||||
## Invariants
|
||||
|
||||
- `.buildkite/pipeline.yml` contains exactly one static step for every active
|
||||
GPU lane. Each step pins a unique key and label, a 90-minute timeout, the
|
||||
trusted `/opt/fastvideo-ci-runner/run-ci` command (`run-unit` is the one
|
||||
compatibility wrapper), step-level internal `TEST_TYPE`, and
|
||||
`queue: "ci-runner"`.
|
||||
- Active CI contains no `pr_test.sh` command, Modal invocation, default queue,
|
||||
Buildkite plugin, `soft_fail`, or job-controlled artifact glob.
|
||||
- The six Fastcheck lanes use `:microscope:` labels. Full-Suite-only lanes use
|
||||
`:test_tube:` or `:bar_chart:` so direct reruns update the right aggregate.
|
||||
- SSIM and vanilla training request all four GPUs. Keep both in the
|
||||
`fastvideo/slinky/whole-tray` Buildkite concurrency group with a limit of one
|
||||
so the second job does not consume an agent or command timeout while waiting
|
||||
for the same tray.
|
||||
- `/test full` schedules all twenty lanes. `/merge`, `ready`, and new pushes to
|
||||
ready PRs use the trusted base-branch planner in
|
||||
`.github/scripts/plan_merge_ci.py`: automatic Fastcheck remains the universal
|
||||
six-lane baseline, and the merge build adds only path-relevant integration
|
||||
lanes. Unknown source/build paths fail closed to all fourteen additive lanes.
|
||||
The trusted uploader still normalizes and validates the complete static graph
|
||||
before Buildkite evaluates its plan conditions.
|
||||
- Focused merge builds may pass allowlisted golden-gate and SSIM test basenames.
|
||||
The private host validates the lane plan and basenames before staging them,
|
||||
and the in-container scripts validate them again. Direct `/test ssim`,
|
||||
explicit `/test full`, and the weekly main-branch schedule run the complete
|
||||
SSIM matrix.
|
||||
- The trusted uploader serves exactly three entry pipelines:
|
||||
`pr-fastcheck` for automatic PR builds, `ci` for slash-command/ready-label
|
||||
API builds, and `fastvideo-performance-lane` for the weekly schedule. Keep
|
||||
incoming GitHub webhook processing disabled on `ci` so it cannot duplicate
|
||||
`pr-fastcheck` on every PR update.
|
||||
- Test payloads live in `.buildkite/scripts/unit_test.sh` or executable
|
||||
`.buildkite/scripts/lanes/<lane>.sh`. Backend policy (GPU count, extras,
|
||||
secrets, kernel build, artifacts) stays in the agent-owned lane table.
|
||||
- Tests must preserve an inherited `MASTER_PORT`. Packed containers share the
|
||||
tray network namespace, so the private runner assigns a distinct port range
|
||||
per GPU lease and the SSIM scheduler assigns task offsets within its range.
|
||||
- The ARM64 runner image includes the pinned FA4 CuTe overlay validated on
|
||||
GB200. Keep SSIM at `FASTVIDEO_FA4=1` because its references were seeded with
|
||||
FA4; keep lanes with FA2 baselines at `FASTVIDEO_FA4=0`. A runner image change
|
||||
must revalidate both the FA4 import and an actual GB200 forward kernel.
|
||||
- `fastvideo/tests/ssim/ci_runner.py` is the active four-GPU SSIM scheduler.
|
||||
New SSIM files are discovered through `REQUIRED_GPUS` and
|
||||
`*_MODEL_TO_PARAMS`; do not wire them through the dormant Modal scheduler.
|
||||
- The host policy fail-closes unknown tuples. A repository-side lane change is
|
||||
inert until the operator updates the private lane table and uploader policy
|
||||
in the same rollout.
|
||||
|
||||
## Adding or changing a lane
|
||||
|
||||
1. Read the closest `AGENTS.md` and the domain-specific testing guide.
|
||||
2. Add or update the executable lane payload under `.buildkite/scripts/`.
|
||||
Keep it deterministic and free of host-specific paths or credential fetches.
|
||||
3. Add the static pipeline step and canonical `/test <name>` mapping. Keep the
|
||||
`<name>-ci` alias only when compatibility requires it.
|
||||
4. Add its source/test path ownership to `.github/scripts/plan_merge_ci.py`.
|
||||
Prefer the narrowest correctness-preserving lane set; leave unknown paths
|
||||
fail-closed. Extend `fastvideo/tests/contract/test_ci_test_collection.py`,
|
||||
`test_merge_ci_plan.py`, and focused CPU-only scheduler/policy tests.
|
||||
5. Coordinate the private lane row: GPU count (1-4), wall time, script, scope
|
||||
pairs, step key, command, HF cache/token, tracking mode, extras, attention
|
||||
backend policy, kernel policy, and artifact relay. Active training lanes
|
||||
keep W&B offline and do not stage a W&B credential.
|
||||
6. Update the trusted pipeline-uploader schema. A mismatch must reject the
|
||||
pipeline rather than silently skip a lane.
|
||||
7. Run `pre-commit run --files <changed paths>`, the planner's representative
|
||||
diff matrix, contract tests, private driver tests, and a real GB200 canary.
|
||||
Multi-GPU, hardware-reference, training, performance, and SSIM changes need
|
||||
their own target-hardware evidence.
|
||||
|
||||
## Rollback
|
||||
|
||||
Rollback the Slurm routing/configuration change or pause the `ci-runner` queue.
|
||||
Do not silently reactivate Modal. A manual Modal experiment requires the
|
||||
explicit local opt-in documented in `ci_architecture.md`; returning it to
|
||||
production CI needs a separate reviewed decision.
|
||||
@@ -0,0 +1,338 @@
|
||||
---
|
||||
name: decompose-pipeline-pr
|
||||
description: Decompose an oversized FastVideo pipeline PR into a stack of independently-reviewable PRs. Tiers the diff by blast radius (invisible / dead code / cross-cutting infra / activation), produces a branch graph and worktree bootstrap, drafts the AGENTS.md manifest, flags missing tests on cross-cutting infra changes, and extracts lessons from the PR body.
|
||||
---
|
||||
|
||||
# Decompose Pipeline PR
|
||||
|
||||
## Purpose
|
||||
|
||||
When a PR adds a new pipeline (or first-class component port) and crosses
|
||||
~3,000 LOC, single-shot review converges to rubber-stamping. This skill
|
||||
decomposes such a PR into a stack of independently-reviewable PRs without
|
||||
disturbing `main`.
|
||||
|
||||
It is the inverse of `add-model`: where `add-model` walks adding a new
|
||||
pipeline as a fresh PR, this skill walks decomposing an existing oversized
|
||||
pipeline PR.
|
||||
|
||||
**Worked example:** PR #1280 (daVinci-MagiHuman, 9,812 LOC, 56 files) →
|
||||
2 prerequisite PRs off main + 8-PR stack:
|
||||
- #1293 `will/activation-trace` (prerequisite)
|
||||
- #1294 `will/loader-infra` (prerequisite)
|
||||
- #1295 (1/8) housekeeping
|
||||
- #1296 (2/8) t5gemma encoder
|
||||
- #1297 (3/8) DiT
|
||||
- #1298 (4/8) pipeline stages
|
||||
- #1299 (5/8) pipeline orchestrator
|
||||
- #1300 (6/8) provenance (AGENTS.md, JOURNAL.md, lessons)
|
||||
- #1301 (7/8) conversion scripts
|
||||
- #1302 (8/8) registry activation
|
||||
|
||||
## Prerequisites
|
||||
|
||||
- Open PR number on `hao-ai-lab/FastVideo` (or any FastVideo fork)
|
||||
- `gh` CLI authenticated against the target remote
|
||||
- Local git worktree support (`git worktree`)
|
||||
- Git config `user.name` / `user.email` set
|
||||
- Pre-commit installed (`pre-commit install --hook-type pre-commit --hook-type commit-msg`)
|
||||
- The target PR's branch fetched locally as `origin/<feature-branch>`
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Required | Description |
|
||||
|-----------|----------|-------------|
|
||||
| PR number or URL | Yes | E.g. `1280` or `https://github.com/hao-ai-lab/FastVideo/pull/1280` |
|
||||
| Max desired PR size | No | Defaults to ~2,500 LOC of code per stack PR (excluding generated/journal files) |
|
||||
| Output directory | No | Defaults to `.agents/tmp/decompose-<pr-number>/` (gitignored) |
|
||||
|
||||
## Steps
|
||||
|
||||
### 1. Verify ground truth (do not trust `gh pr diff --name-only`)
|
||||
|
||||
`gh pr diff <N> --name-only` has been observed to emit phantom file entries.
|
||||
Always cross-check against the authoritative `git diff`:
|
||||
|
||||
```bash
|
||||
mkdir -p .agents/tmp/decompose-<N>
|
||||
git fetch origin pull/<N>/head:<feature-branch>
|
||||
git diff origin/main..origin/<feature-branch> --name-status \
|
||||
> .agents/tmp/decompose-<N>/files.txt
|
||||
git diff origin/main..origin/<feature-branch> --stat
|
||||
```
|
||||
|
||||
Use the `--name-status` output as the authoritative file list. If it
|
||||
disagrees with `gh pr diff --name-only`, trust the git diff.
|
||||
|
||||
### 2. Tier the diff by blast radius
|
||||
|
||||
Classify every changed file into one of four tiers:
|
||||
|
||||
| Tier | Description | Examples |
|
||||
|---|---|---|
|
||||
| **Tier 0 — Invisible** | Lint/style/CI configs that don't affect runtime | `.gitignore`, `pyproject.toml` (codespell only), agent documentation |
|
||||
| **Tier 1 — Dead code** | New files in their own dirs; aggregator one-liners | `fastvideo/models/dits/<new>/`, `fastvideo/pipelines/basic/<new>/`, `examples/inference/basic/basic_<new>*.py`, `tests/local_tests/<new>/`, `__init__.py` exports |
|
||||
| **Tier 2 — Cross-cutting infra** | Modifications to files used by every pipeline | See protected-paths list below |
|
||||
| **Tier 3 — Activation switch** | `register_configs(...)` calls + the example scripts that demo them | `fastvideo/registry.py` |
|
||||
|
||||
**FastVideo Tier 2 protected paths:**
|
||||
```
|
||||
fastvideo/utils.py
|
||||
fastvideo/pipelines/composed_pipeline_base.py
|
||||
fastvideo/models/loader/component_loader.py
|
||||
fastvideo/configs/models/dits/__init__.py
|
||||
fastvideo/configs/models/encoders/__init__.py
|
||||
fastvideo/configs/models/vaes/__init__.py
|
||||
fastvideo/envs.py
|
||||
fastvideo/fastvideo_args.py
|
||||
fastvideo/distributed/**
|
||||
fastvideo/layers/**
|
||||
fastvideo/attention/**
|
||||
fastvideo/registry.py # treat as Tier 3 if change is the activation
|
||||
```
|
||||
|
||||
Tier 3 detection (mechanical):
|
||||
```bash
|
||||
git diff origin/main..origin/<feature-branch> -- fastvideo/registry.py | \
|
||||
grep -E "^\+.*register_configs\("
|
||||
```
|
||||
|
||||
If `registry.py` only contains `register_configs` additions, treat it as
|
||||
Tier 3. If it modifies existing behavior, treat it as Tier 2 (rare).
|
||||
|
||||
### 3. Identify reusable Tier-1 components
|
||||
|
||||
Within Tier 1, look for sub-trees that are **not** model-specific and could
|
||||
land separately:
|
||||
|
||||
- Encoders matching a known multi-model base (T5/T5-Gemma/Llama/Gemma/CLIP variants)
|
||||
- New stage classes that subclass shared bases without referencing the new model
|
||||
- Hook/profiler/debug infra under `fastvideo/hooks/`
|
||||
- New helpers that have no model-specific dependencies
|
||||
|
||||
These get split into their own PRs (e.g. PR 4 `t5gemma-encoder` in the
|
||||
MagiHuman example).
|
||||
|
||||
### 4. Hunt for missing test coverage on Tier 2 changes
|
||||
|
||||
For every Tier-2 file modified, check whether the original PR added unit
|
||||
tests for the new behavior:
|
||||
|
||||
```bash
|
||||
for f in <list-of-tier-2-files>; do
|
||||
echo "=== Tests for $f ==="
|
||||
git diff origin/main..origin/<feature-branch> -- \
|
||||
"$(echo $f | sed 's|fastvideo/|fastvideo/tests/|; s|\.py|*|')"
|
||||
done
|
||||
```
|
||||
|
||||
If a Tier-2 PR has no accompanying tests, **emit a "must-add tests" list**
|
||||
with a sketch of the case grid. Tier-2 PRs do not ship without those tests.
|
||||
|
||||
The MagiHuman example required this for PR-B (`utils.py`): the original PR
|
||||
shipped no `test_utils_loader.py`, so the decomposition added 9 unit-test
|
||||
cases covering the umbrella-detector boundary, the optional-component-dirs
|
||||
relaxation, and regression coverage on every existing 2-segment HF id.
|
||||
|
||||
### 5. Build the dependency DAG and topo-sort
|
||||
|
||||
Edges:
|
||||
- Tier 2 infra → Tier 1 code that imports it
|
||||
- Reusable Tier 1 components → model-specific Tier 1 code that uses them
|
||||
(encoder before DiT before pipeline)
|
||||
- Tier 1 → Tier 3 (activation always last)
|
||||
- Tier 0 has no dependents (lands first as a freebie)
|
||||
|
||||
Topo-sort produces the stack ordering. Pull Tier-2 PRs **out of the stack**
|
||||
when they have no model-specific dependency — they should land off main
|
||||
with their own focused review, not buried in a model port.
|
||||
|
||||
Render as a tree (markdown):
|
||||
|
||||
```
|
||||
main
|
||||
├─ <prereq-A>
|
||||
│ └─ <prereq-B>
|
||||
│ ├─ <stack-01-housekeeping>
|
||||
│ │ └─ <stack-02-encoder>
|
||||
│ │ └─ <stack-03-dit>
|
||||
│ │ └─ ...
|
||||
│ │ └─ <stack-N-activate>
|
||||
│ └─ (parallel) <skill-pr> off main
|
||||
```
|
||||
|
||||
### 6. Detect mis-shelved docs and debug scratch
|
||||
|
||||
Two categories to flag:
|
||||
|
||||
- **Mis-shelved docs**: Markdown files under `tests/local_tests/` are
|
||||
journals, not tests. Flag for relocation to the package dir as
|
||||
`JOURNAL.md`.
|
||||
- **Debug scratch**: files starting with `_debug_`, `_scratch_`, or
|
||||
`_explore_`. Flag for drop (do not carry into any output PR).
|
||||
|
||||
For MagiHuman: `tests/local_tests/magi-human.md` → relocate. Two
|
||||
`_debug_magi_human_*.py` files → drop.
|
||||
|
||||
### 7. Author the AGENTS.md manifest skeleton
|
||||
|
||||
For the new pipeline package, generate a 6-section `AGENTS.md` scaffold
|
||||
with the file table pre-populated from the diff:
|
||||
|
||||
1. **Manifest** — file table by role
|
||||
2. **Parity invariants** — load-bearing rules with one-paragraph each + lesson refs
|
||||
3. **Cross-refs** — "If you change X, re-run Y" matrix
|
||||
4. **Run book** — single pytest command + prereqs (HF tokens, GPU, wall-time)
|
||||
5. **Open questions** — known issues (e.g. tolerance carve-outs)
|
||||
6. **Provenance** — PR table with branch names and source SHA
|
||||
|
||||
The provenance section is filled incrementally during stack execution and
|
||||
finalized in the activation PR.
|
||||
|
||||
### 8. Extract lessons from the PR body
|
||||
|
||||
Scan the PR body for sections titled "Key implementation work", "Bug hunt",
|
||||
"Lessons", or sentences with patterns like "took N waves to localize",
|
||||
"silent regression", "investigation revealed". Each becomes a candidate
|
||||
`.agents/lessons/<YYYY-MM-DD>_<slug>.md` draft.
|
||||
|
||||
Lessons MUST follow the existing template in
|
||||
`.agents/lessons/README.md`:
|
||||
- YAML frontmatter: `date`, `experiment`, `category`, `severity`
|
||||
- Sections: What Happened, Root Cause, Fix / Workaround, Prevention
|
||||
- Filename: `<YYYY-MM-DD>_<short-slug>.md`
|
||||
|
||||
Lessons co-locate with the code they concern: a conversion-script lesson
|
||||
lands in the same PR as the conversion script, not in the docs PR.
|
||||
|
||||
### 9. Emit the commit-footer convention
|
||||
|
||||
Every commit in the stack ends with:
|
||||
|
||||
```
|
||||
<Feature>-Stack: N/M
|
||||
```
|
||||
|
||||
E.g. `Magi-Stack: 5/8`. Use the package directory name as the feature key.
|
||||
After all PRs squash-merge, `git log --grep='^<Feature>-Stack:'` reconstructs
|
||||
the lineage even if PR numbers later get renumbered.
|
||||
|
||||
### 10. Produce the worktree bootstrap
|
||||
|
||||
Generate a runnable bash script:
|
||||
|
||||
```bash
|
||||
#!/bin/bash
|
||||
set -euo pipefail
|
||||
|
||||
REPO=/home/<user>/FastVideo
|
||||
WORKTREE=/home/<user>/FastVideoMagi # NB: directory name must be a valid
|
||||
# Python identifier (no hyphens) so
|
||||
# mypy doesn't choke
|
||||
SOURCE_PR=<N>
|
||||
SOURCE_BRANCH=will/<feature>
|
||||
SOURCE_SHA=$(git -C "$REPO" rev-parse "origin/$SOURCE_BRANCH")
|
||||
|
||||
git -C "$REPO" fetch origin main:main
|
||||
git -C "$REPO" fetch "origin/$SOURCE_BRANCH"
|
||||
git -C "$REPO" worktree add "$WORKTREE" origin/main
|
||||
|
||||
# Capture baseline for provenance. Everything under .agents/tmp is transient
|
||||
# and ignored by git.
|
||||
OUTPUT_DIR="$REPO/.agents/tmp/decompose-$SOURCE_PR"
|
||||
mkdir -p "$OUTPUT_DIR"
|
||||
cat > "$OUTPUT_DIR/<feature>-baseline-${SOURCE_SHA:0:8}.txt" <<EOF
|
||||
Source PR: <repo>#$SOURCE_PR
|
||||
Source SHA: $SOURCE_SHA
|
||||
Authoritative file count: $(git -C "$REPO" diff origin/main..origin/$SOURCE_BRANCH --name-only | wc -l)
|
||||
Date captured: $(date -u +%Y-%m-%dT%H:%M:%SZ)
|
||||
EOF
|
||||
```
|
||||
|
||||
### 11. Author preserve via `git checkout`, not `cherry-pick`
|
||||
|
||||
For each stack PR:
|
||||
|
||||
```bash
|
||||
git -C "$WORKTREE" switch -c <new-branch> <base-branch>
|
||||
git -C "$WORKTREE" checkout origin/<source-branch> -- <file1> <file2> ...
|
||||
git -C "$WORKTREE" commit -m "[<scope>]: <subject>
|
||||
|
||||
<body>
|
||||
|
||||
<Feature>-Stack: N/M"
|
||||
git -C "$WORKTREE" push -u origin <new-branch>
|
||||
gh pr create --base <base-branch> --head <new-branch> --title "..." --body "$(cat <<EOF ... EOF)"
|
||||
```
|
||||
|
||||
Notes:
|
||||
- `git checkout origin/<source> -- <files>` extracts only the named files,
|
||||
preserving the diff. The original PR's author is **not** preserved on the
|
||||
new commit (it's authored by whoever runs the script). Reference the
|
||||
original PR + source SHA in every commit body and PR description for
|
||||
authorship attribution.
|
||||
- **Never use `git cherry-pick`** for this workflow — cherry-pick applies
|
||||
whole commits, which mixes concerns across PR boundaries.
|
||||
|
||||
## Outputs
|
||||
|
||||
The skill produces all transient planning artifacts under
|
||||
`.agents/tmp/decompose-<pr>/`:
|
||||
|
||||
1. A markdown decomposition plan (`plan.md`)
|
||||
2. A proposed branch graph
|
||||
3. A worktree-bootstrap script (`bootstrap.sh`)
|
||||
4. Per-PR file allocation lists (under `stack/`)
|
||||
5. AGENTS.md scaffolds for any new pipeline packages
|
||||
6. Draft lesson files (placed alongside the PR that owns the code they concern)
|
||||
7. A finalized provenance table for the package AGENTS.md
|
||||
|
||||
## Anti-Patterns
|
||||
|
||||
The skill should warn against:
|
||||
|
||||
- **"Just rebase the megaPR into smaller commits."** Doesn't help review;
|
||||
reviewer still sees one PR.
|
||||
- **Co-locating tests under the new package.** FastVideo's convention is
|
||||
by-kind under `fastvideo/tests/` and `tests/local_tests/<family>/`. Don't
|
||||
invent a new layout per pipeline.
|
||||
- **Splitting Tier 2 changes into "one file per PR."** Tier 2 PRs are
|
||||
about semantic units (e.g., "loader umbrella + optional component dirs"
|
||||
together because they jointly define the new diffusers-format contract),
|
||||
not file-count.
|
||||
- **Landing the activation switch first** ("just register, the code can
|
||||
be empty"). The skill enforces activation-last so every intermediate
|
||||
state is dead code, not broken code.
|
||||
- **Trusting `gh pr diff --name-only`.** Cross-check against
|
||||
`git diff origin/main..origin/<feature-branch> --name-status` —
|
||||
`gh`'s output has been observed to include phantom entries.
|
||||
- **Worktree dir names with hyphens.** mypy interprets them as invalid
|
||||
Python package names and refuses to run. Use CamelCase or underscores.
|
||||
- **Skipping the lesson-extraction step.** PR bodies contain the most
|
||||
expensive learnings of the original implementation. Losing them to a
|
||||
squash-merge is the silent decay of institutional knowledge.
|
||||
|
||||
## Example Usage
|
||||
|
||||
```
|
||||
User: split PR 1280
|
||||
Agent: [invokes decompose-pipeline-pr]
|
||||
→ produces .agents/tmp/decompose-1280/plan.md with:
|
||||
- tiered file table (56 files: 3 tier-0, 35 tier-1, 9 tier-2,
|
||||
9 tier-3)
|
||||
- branch graph (PR-A + PR-B + 8-PR stack)
|
||||
- worktree bootstrap script
|
||||
- per-PR file lists
|
||||
- AGENTS.md scaffold for fastvideo/pipelines/basic/magi_human/
|
||||
- 3 draft lessons extracted from the PR body
|
||||
→ asks user to confirm before opening branches
|
||||
```
|
||||
|
||||
## References
|
||||
|
||||
- The MagiHuman decomposition (worked example):
|
||||
`fastvideo/pipelines/basic/magi_human/AGENTS.md` (after PR #1302 merges)
|
||||
- Existing skill: `.agents/skills/add-model/SKILL.md` (the inverse — adding
|
||||
a new pipeline as a fresh PR)
|
||||
- Lesson template: `.agents/lessons/README.md`
|
||||
- Skill template: `.agents/skills/SKILL_TEMPLATE.md`
|
||||
@@ -0,0 +1,196 @@
|
||||
---
|
||||
name: dreamverse-deploy
|
||||
description: Use when redeploying the migrated Dreamverse app backend and frontend on a chosen local GPU; tears down existing ports, launches services, and waits for readiness checks.
|
||||
---
|
||||
|
||||
# dreamverse-deploy — redeploy migrated Dreamverse on a chosen GPU
|
||||
|
||||
**Scope:** project (lives in this repo at `.agents/skills/dreamverse-deploy/`)
|
||||
|
||||
**When to use:** you want to (re)launch the migrated `apps/dreamverse/` backend
|
||||
and frontend on this dev node, pinned to a specific physical GPU. Tears down
|
||||
any existing deploy on the same ports first, then boots fresh and waits for
|
||||
both `/readyz` and the FE root to return 200.
|
||||
|
||||
## Prerequisites
|
||||
|
||||
- Working tree containing `apps/dreamverse/`
|
||||
- `dreamverse-server` installed from this checkout; if missing, run
|
||||
`uv pip install -e ".[dreamverse]"`
|
||||
- Local conda env at `~/miniconda3/envs/fv-main/` with `flashinfer-python`,
|
||||
`cerebras-cloud-sdk`, `openai` installed (override the default path with
|
||||
`DREAMVERSE_PYTHON=/path/to/python`)
|
||||
- `~/.env` exporting `CEREBRAS_API_KEY`, `GROQ_API_KEY`, etc.
|
||||
- npm available in `$PATH` (or set `NPM=/path/to/npm`)
|
||||
- `gcc-13` + `g++-13` at `/usr/bin/` (workaround for nvcc gcc-15 rejection)
|
||||
- **Recommended:** native ffmpeg at `$HOME/opt/ffmpeg-native/bin/ffmpeg`, built
|
||||
via `bash apps/dreamverse/scripts/install_native_ffmpeg.sh`. The deploy
|
||||
detects that binary directly and exports it for the backend. The installer's
|
||||
generated `apps/dreamverse/scripts/ffmpeg-env.sh` is for manual launches.
|
||||
When the binary is missing, the deploy falls back to system ffmpeg with a
|
||||
warning. Set
|
||||
`DREAMVERSE_REQUIRE_NATIVE_FFMPEG=true` to make the missing binary a hard
|
||||
failure.
|
||||
|
||||
If any required prereq is missing, the script fails fast with a clear message.
|
||||
|
||||
## Usage
|
||||
|
||||
```bash
|
||||
# Deploy on GPU 4 with the current web port. The legacy helper default remains
|
||||
# 5274, so pass 5299 explicitly. Torch compile and warmup are both off.
|
||||
./.agents/skills/dreamverse-deploy/scripts/dreamverse-deploy.sh 4 8009 5299
|
||||
|
||||
# Deploy on GPU 6 with custom ports
|
||||
./.agents/skills/dreamverse-deploy/scripts/dreamverse-deploy.sh 6 8089 5275
|
||||
|
||||
# Deploy on GPU 0 with warmup enabled
|
||||
./.agents/skills/dreamverse-deploy/scripts/dreamverse-deploy.sh --warmup 0 8009 5299
|
||||
|
||||
# Deploy with torch.compile enabled (max-autotune; first segment ~3-4min,
|
||||
# subsequent segments save ~3s — only worth it for benchmarking)
|
||||
./.agents/skills/dreamverse-deploy/scripts/dreamverse-deploy.sh --torch-compile 4 8009 5299
|
||||
|
||||
# Deploy with both warmup AND torch.compile enabled
|
||||
./.agents/skills/dreamverse-deploy/scripts/dreamverse-deploy.sh --warmup --torch-compile 4 8009 5299
|
||||
|
||||
# Flags can appear before, between, or after positional args
|
||||
./.agents/skills/dreamverse-deploy/scripts/dreamverse-deploy.sh 4 8089 5275 --warmup
|
||||
```
|
||||
|
||||
### Arguments
|
||||
|
||||
| Position | Name | Default | Notes |
|
||||
|---|---|---|---|
|
||||
| 1 | `GPU` | (required) | Physical GPU index, e.g. `4` |
|
||||
| 2 | `BACKEND_PORT` | `8009` | TCP port for the FastAPI server |
|
||||
| 3 | `FRONTEND_PORT` | `5274` | TCP port for the Next.js dev server |
|
||||
|
||||
### Flags
|
||||
|
||||
| Flag | Default | Notes |
|
||||
|---|---|---|
|
||||
| `--warmup` / `--no-warmup` | off | Run GPU warmup at boot (~minutes). Overrides `DREAMVERSE_WARMUP` |
|
||||
| `--torch-compile` / `--no-torch-compile` | off | Enable max-autotune `torch.compile`. First segment ~3-4min when on, ~45s when off. Overrides `DREAMVERSE_TORCH_COMPILE` |
|
||||
| `--nvenc` / `--no-nvenc` | off | Use `h264_nvenc` hardware encoder instead of `libx264` software. Eliminates ~1100ms/segment of CPU encoding cost (raises realtime ratio from ~0.78x → ≥1.0x, eliminating inter-segment buffer-drain stutter). Requires native ffmpeg built with `--enable-nvenc` (the install script's default since the NVENC update). Hard-fails up-front if the binary is missing or lacks NVENC. Overrides `DREAMVERSE_NVENC` |
|
||||
| `-h` / `--help` | — | Show usage |
|
||||
|
||||
Flags can appear in any position relative to the positional args. Explicit flag values always win over env-var defaults.
|
||||
|
||||
### Environment variables (used when no flag is given)
|
||||
|
||||
| Var | Default | Purpose |
|
||||
|---|---|---|
|
||||
| `DREAMVERSE_WARMUP` | `false` | Same as `--warmup`/`--no-warmup`. Flag takes precedence |
|
||||
| `DREAMVERSE_TORCH_COMPILE` | `false` | Same as `--torch-compile`/`--no-torch-compile`. Flag takes precedence |
|
||||
| `DREAMVERSE_NVENC` | `false` | Same as `--nvenc`/`--no-nvenc`. Flag takes precedence |
|
||||
| `DREAMVERSE_PYTHON` | `~/miniconda3/envs/fv-main/bin/python` | Conda environment used for the flashinfer prerequisite probe; `dreamverse-server` itself is resolved from `PATH` |
|
||||
| `DREAMVERSE_REPO_ROOT` | git rev-parse | Repo root override |
|
||||
| `DREAMVERSE_LOG_DIR` | `/tmp/opencode/dreamverse-deploy` | Directory for the per-GPU backend and per-port frontend logs |
|
||||
| `DREAMVERSE_REQUIRE_NATIVE_FFMPEG` | `false` | If `true`, fail when `$HOME/opt/ffmpeg-native/bin/ffmpeg` is absent |
|
||||
|
||||
## What it does
|
||||
|
||||
1. Validates prereqs.
|
||||
2. Kills any process on the target backend/frontend ports + waits for the
|
||||
target GPU to release memory (allows up to 30s for cleanup).
|
||||
3. Sources `~/.env`.
|
||||
4. Exports the env recipe required for boot:
|
||||
- `CUDA_VISIBLE_DEVICES=<gpu>`
|
||||
- `FASTVIDEO_ENABLE_DEVTOOLS=1`
|
||||
- `FASTVIDEO_ENABLE_STARTUP_WARMUP=<DREAMVERSE_WARMUP>`
|
||||
- `FASTVIDEO_GPU_COUNT=1`
|
||||
- `ENABLE_TORCH_COMPILE=<0|1 derived from DREAMVERSE_TORCH_COMPILE>`
|
||||
- `CC=/usr/bin/gcc-13 CXX=/usr/bin/g++-13 CUDAHOSTCXX=/usr/bin/g++-13`
|
||||
- `NVCC_PREPEND_FLAGS="-ccbin /usr/bin/gcc-13 -allow-unsupported-compiler"`
|
||||
- `FASTVIDEO_FFMPEG_BIN=$HOME/opt/ffmpeg-native/bin/ffmpeg` +
|
||||
`FASTVIDEO_VIDEO_CODEC=<libx264|h264_nvenc>` (when the native binary exists)
|
||||
5. Launches the installed `dreamverse-server` console command in a detached
|
||||
`setsid` session and captures its PID.
|
||||
6. Polls `/readyz` until 200. The budget is 5 minutes by default, 8 minutes
|
||||
with one startup optimization enabled, and 15 minutes with both warmup and
|
||||
`torch.compile` enabled.
|
||||
7. Launches the devtools frontend through npm in a detached session and
|
||||
captures its PID.
|
||||
8. Polls FE `/` until 200 (max 60s).
|
||||
9. Prints URLs, PIDs, and log paths.
|
||||
|
||||
## What it does NOT do
|
||||
|
||||
- Does not modify `~/.env` or the FastVideo `.venv`.
|
||||
- Does not push code or commit anything.
|
||||
- Does not run Playwright. Use the e2e wrapper separately:
|
||||
```bash
|
||||
cd apps/dreamverse/web
|
||||
PLAYWRIGHT_SKIP_WEBSERVER=1 BACKEND_HOST=127.0.0.1 BACKEND_PORT=8009 \
|
||||
PLAYWRIGHT_BASE_URL=http://127.0.0.1:5299 \
|
||||
NEXT_PUBLIC_INCLUDE_DEVTOOLS=1 \
|
||||
npm exec -- playwright test
|
||||
```
|
||||
The standard suite runs by default; the long-running two-segment
|
||||
audio-continuation spec is gated behind
|
||||
`PLAYWRIGHT_LONG_RUNNING=1` (see below).
|
||||
|
||||
## Long-running e2e (paired with `--warmup --torch-compile`)
|
||||
|
||||
[`apps/dreamverse/web/e2e/long-running-segments.spec.ts`](../../../apps/dreamverse/web/e2e/long-running-segments.spec.ts)
|
||||
drives a real two-segment session through the FE, captures every WS
|
||||
frame, and asserts segments 1 AND 2 both reach `media_segment_complete`
|
||||
with at least one binary fMP4 chunk per segment. It guards against the
|
||||
BrokenPipe regression previously caused by dropped LTX-2 audio continuation
|
||||
kwargs.
|
||||
Skipped by default. Enable with:
|
||||
|
||||
```bash
|
||||
./.agents/skills/dreamverse-deploy/scripts/dreamverse-deploy.sh \
|
||||
--warmup --torch-compile 4 8009 5299
|
||||
|
||||
cd apps/dreamverse/web
|
||||
PLAYWRIGHT_SKIP_WEBSERVER=1 \
|
||||
BACKEND_HOST=127.0.0.1 \
|
||||
BACKEND_PORT=8009 \
|
||||
PLAYWRIGHT_BASE_URL=http://127.0.0.1:5299 \
|
||||
NEXT_PUBLIC_INCLUDE_DEVTOOLS=1 \
|
||||
PLAYWRIGHT_LONG_RUNNING=1 \
|
||||
npm exec -- playwright test e2e/long-running-segments.spec.ts
|
||||
```
|
||||
|
||||
Expected runtime: ~7-9 minutes on a B200 (torch.compile max-autotune
|
||||
warm-up dominates the cold start; per-test timeout is 900s). The spec
|
||||
hard-fails on any WS `error`/`step_error` frame so the BrokenPipe
|
||||
regression surfaces with the actual ffmpeg/audio diagnostics rather
|
||||
than an opaque "test timed out".
|
||||
|
||||
## Teardown
|
||||
|
||||
Stop both services without redeploying:
|
||||
|
||||
```bash
|
||||
# Stop services on default ports (port-pattern based)
|
||||
./.agents/skills/dreamverse-deploy/scripts/dreamverse-deploy.sh --stop
|
||||
|
||||
# Stop AND nuke any process holding GPU N
|
||||
./.agents/skills/dreamverse-deploy/scripts/dreamverse-deploy.sh --stop 4
|
||||
```
|
||||
|
||||
The redeploy path (`<GPU>` mode) automatically nukes any process holding the
|
||||
target GPU before launching — including orphan `multiproc_executor` worker
|
||||
subprocesses left over from a parent backend that was killed without grace.
|
||||
This was the failure mode of an earlier naive port-only kill: parent dies,
|
||||
children survive, GPU stays full, next deploy OOMs.
|
||||
|
||||
## Notes
|
||||
|
||||
- The installed `dreamverse-server` console command enters
|
||||
`apps/dreamverse/dreamverse/server_entry.py`, which loads the current
|
||||
Dreamverse runtime from `apps/dreamverse/dreamverse/`.
|
||||
- The B200 / sm_100a NVCC flags are mandatory on this dev node because the
|
||||
conda toolchain ships gcc-15, which nvcc rejects. The script requires the
|
||||
configured gcc-13 and g++-13 binaries during preflight.
|
||||
|
||||
## Deployment boundary
|
||||
|
||||
This skill is for a local checkout on a directly attached GPU. For a container
|
||||
image, use `apps/dreamverse/docker/README.md`. For Modal, follow
|
||||
`apps/dreamverse/scripts/modal/README.md`; do not adapt this process-killing
|
||||
workflow to a remote deployment.
|
||||
@@ -0,0 +1,460 @@
|
||||
#!/usr/bin/env bash
|
||||
# See ../SKILL.md for full usage.
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
is_pid_alive() {
|
||||
kill -0 "$1" 2>/dev/null
|
||||
}
|
||||
|
||||
terminate_pid() {
|
||||
local pid="$1"
|
||||
local label="${2:-pid=${pid}}"
|
||||
|
||||
[[ -n "${pid}" ]] && [[ "${pid}" != "$$" ]] || return 0
|
||||
is_pid_alive "${pid}" || return 0
|
||||
|
||||
kill "${pid}" 2>/dev/null || true
|
||||
for _ in $(seq 1 10); do
|
||||
is_pid_alive "${pid}" || return 0
|
||||
sleep 0.5
|
||||
done
|
||||
|
||||
if is_pid_alive "${pid}"; then
|
||||
kill -9 "${pid}" 2>/dev/null && echo " force-killed ${label}" || true
|
||||
fi
|
||||
}
|
||||
|
||||
terminate_pattern() {
|
||||
local pattern="$1"
|
||||
local pid
|
||||
|
||||
if ! command -v pgrep >/dev/null 2>&1; then
|
||||
pkill -TERM -f "${pattern}" 2>/dev/null || true
|
||||
sleep 2
|
||||
pkill -KILL -f "${pattern}" 2>/dev/null || true
|
||||
return 0
|
||||
fi
|
||||
|
||||
for pid in $(pgrep -f -- "${pattern}" 2>/dev/null || true); do
|
||||
terminate_pid "${pid}" "pattern='${pattern}' pid=${pid}"
|
||||
done
|
||||
}
|
||||
|
||||
list_port_pids() {
|
||||
local port="$1"
|
||||
|
||||
if command -v lsof >/dev/null 2>&1; then
|
||||
lsof -t -iTCP:"${port}" -sTCP:LISTEN 2>/dev/null || true
|
||||
return 0
|
||||
fi
|
||||
|
||||
ss -tlnp 2>/dev/null | awk -v port=":${port}" '
|
||||
$0 ~ port {
|
||||
while (match($0, /pid=[0-9]+/)) {
|
||||
print substr($0, RSTART + 4, RLENGTH - 4)
|
||||
$0 = substr($0, RSTART + RLENGTH)
|
||||
}
|
||||
}
|
||||
' || true
|
||||
}
|
||||
|
||||
if [[ "${1:-}" == "--stop" ]]; then
|
||||
for pat in 'apps/dreamverse/dreamverse/main.py' 'dreamverse-server --host 0.0.0.0 --port' 'next dev --port' 'next-server (v'; do
|
||||
terminate_pattern "${pat}"
|
||||
done
|
||||
if [[ -n "${2:-}" ]] && [[ "${2}" =~ ^[0-9]+$ ]]; then
|
||||
gpu_uuid="$(nvidia-smi --query-gpu=index,uuid --format=csv,noheader 2>/dev/null | awk -F', ' -v g="${2}" '$1==g {print $2}')"
|
||||
if [[ -n "${gpu_uuid}" ]]; then
|
||||
for pid in $(nvidia-smi --query-compute-apps=pid,gpu_uuid --format=csv,noheader 2>/dev/null \
|
||||
| awk -F', ' -v u="${gpu_uuid}" '$2==u {print $1}'); do
|
||||
terminate_pid "${pid}" "GPU${2} pid=${pid}"
|
||||
done
|
||||
fi
|
||||
fi
|
||||
sleep 2
|
||||
echo "stopped: ports may take a few seconds to free"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Args
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
usage() {
|
||||
cat <<USAGE
|
||||
Usage: $(basename "$0") [FLAGS] <GPU> [BACKEND_PORT] [FRONTEND_PORT]
|
||||
$(basename "$0") --stop [GPU]
|
||||
|
||||
Positional:
|
||||
GPU Physical GPU index (required), e.g. 4
|
||||
BACKEND_PORT default 8009
|
||||
FRONTEND_PORT default 5274
|
||||
|
||||
Flags (override env vars when both set):
|
||||
--warmup / --no-warmup run GPU warmup at boot (default off)
|
||||
--torch-compile / --no-torch-compile
|
||||
enable max-autotune torch.compile
|
||||
(default off — first segment ~3-4min
|
||||
when on, ~45s when off)
|
||||
--nvenc / --no-nvenc use h264_nvenc hardware encoder (default
|
||||
off — uses libx264 software encoder).
|
||||
Requires native ffmpeg built with NVENC.
|
||||
-h, --help show this help
|
||||
|
||||
Env overrides:
|
||||
DREAMVERSE_WARMUP 'true'|'false' (default false)
|
||||
DREAMVERSE_TORCH_COMPILE 'true'|'false' (default false)
|
||||
DREAMVERSE_NVENC 'true'|'false' (default false)
|
||||
DREAMVERSE_REPO_ROOT default: \$(git rev-parse --show-toplevel)
|
||||
DREAMVERSE_LOG_DIR default: /tmp/opencode/dreamverse-deploy
|
||||
DREAMVERSE_REQUIRE_NATIVE_FFMPEG 'true'|'false' (default false)
|
||||
USAGE
|
||||
}
|
||||
|
||||
WARMUP_OVERRIDE=""
|
||||
TORCH_COMPILE_OVERRIDE=""
|
||||
NVENC_OVERRIDE=""
|
||||
POSITIONAL=()
|
||||
while [[ $# -gt 0 ]]; do
|
||||
case "$1" in
|
||||
-h|--help) usage; exit 0 ;;
|
||||
--warmup) WARMUP_OVERRIDE=true; shift ;;
|
||||
--no-warmup) WARMUP_OVERRIDE=false; shift ;;
|
||||
--torch-compile) TORCH_COMPILE_OVERRIDE=true; shift ;;
|
||||
--no-torch-compile) TORCH_COMPILE_OVERRIDE=false; shift ;;
|
||||
--nvenc) NVENC_OVERRIDE=true; shift ;;
|
||||
--no-nvenc) NVENC_OVERRIDE=false; shift ;;
|
||||
--) shift; while [[ $# -gt 0 ]]; do POSITIONAL+=("$1"); shift; done ;;
|
||||
-*) echo "error: unknown flag '$1'" >&2; usage >&2; exit 2 ;;
|
||||
*) POSITIONAL+=("$1"); shift ;;
|
||||
esac
|
||||
done
|
||||
set -- "${POSITIONAL[@]+"${POSITIONAL[@]}"}"
|
||||
|
||||
if [[ $# -lt 1 ]]; then
|
||||
usage >&2
|
||||
exit 2
|
||||
fi
|
||||
|
||||
GPU="${1}"
|
||||
BACKEND_PORT="${2:-8009}"
|
||||
FRONTEND_PORT="${3:-5274}"
|
||||
|
||||
if ! [[ "${GPU}" =~ ^[0-9]+$ ]]; then
|
||||
echo "error: GPU must be a non-negative integer (got '${GPU}')" >&2
|
||||
exit 2
|
||||
fi
|
||||
|
||||
WARMUP="${WARMUP_OVERRIDE:-${DREAMVERSE_WARMUP:-false}}"
|
||||
case "${WARMUP}" in
|
||||
true|false) ;;
|
||||
*) echo "error: warmup must be 'true' or 'false' (got '${WARMUP}')" >&2; exit 2 ;;
|
||||
esac
|
||||
|
||||
TORCH_COMPILE="${TORCH_COMPILE_OVERRIDE:-${DREAMVERSE_TORCH_COMPILE:-false}}"
|
||||
case "${TORCH_COMPILE}" in
|
||||
true|false) ;;
|
||||
*) echo "error: torch-compile must be 'true' or 'false' (got '${TORCH_COMPILE}')" >&2; exit 2 ;;
|
||||
esac
|
||||
TORCH_COMPILE_FLAG=$([[ "${TORCH_COMPILE}" == "true" ]] && echo 1 || echo 0)
|
||||
|
||||
NVENC="${NVENC_OVERRIDE:-${DREAMVERSE_NVENC:-false}}"
|
||||
case "${NVENC}" in
|
||||
true|false) ;;
|
||||
*) echo "error: nvenc must be 'true' or 'false' (got '${NVENC}')" >&2; exit 2 ;;
|
||||
esac
|
||||
|
||||
REPO_ROOT="${DREAMVERSE_REPO_ROOT:-$(git rev-parse --show-toplevel 2>/dev/null || pwd)}"
|
||||
LOG_DIR="${DREAMVERSE_LOG_DIR:-/tmp/opencode/dreamverse-deploy}"
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Prereq checks
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
bail() { echo "error: $*" >&2; exit 3; }
|
||||
|
||||
[[ -d "${REPO_ROOT}/apps/dreamverse" ]] \
|
||||
|| bail "REPO_ROOT '${REPO_ROOT}' does not contain apps/dreamverse/. Are you on a migration branch?"
|
||||
DREAMVERSE_SERVER="$(command -v dreamverse-server 2>/dev/null || true)"
|
||||
[[ -n "${DREAMVERSE_SERVER}" ]] && [[ -x "${DREAMVERSE_SERVER}" ]] \
|
||||
|| bail "dreamverse-server not executable or not in PATH (run: uv pip install -e \".[dreamverse]\")"
|
||||
|
||||
CONDA_ENV_PYTHON="${DREAMVERSE_PYTHON:-${HOME}/miniconda3/envs/fv-main/bin/python}"
|
||||
[[ -x "${CONDA_ENV_PYTHON}" ]] \
|
||||
|| bail "conda env python missing at ${CONDA_ENV_PYTHON} (set DREAMVERSE_PYTHON to override)"
|
||||
"${CONDA_ENV_PYTHON}" -c 'import flashinfer' 2>/dev/null \
|
||||
|| bail "flashinfer-python not installed in ${CONDA_ENV_PYTHON} (run: ${CONDA_ENV_PYTHON} -m pip install flashinfer-python --no-build-isolation)"
|
||||
|
||||
NPM="${NPM:-npm}"
|
||||
NPM_REQUESTED="${NPM}"
|
||||
NPM="$(command -v "${NPM}" 2>/dev/null || true)"
|
||||
[[ -n "${NPM}" ]] && [[ -x "${NPM}" ]] || bail "npm not executable or not in PATH: ${NPM_REQUESTED} (set NPM to override)"
|
||||
|
||||
GCC13="$(command -v "${GCC13:-gcc-13}" 2>/dev/null || true)"
|
||||
GPP13="$(command -v "${GPP13:-g++-13}" 2>/dev/null || true)"
|
||||
[[ -n "${GCC13}" ]] && command -v "${GCC13}" >/dev/null 2>&1 \
|
||||
|| bail "gcc-13 not found or not executable (needed for nvcc workaround). Set GCC13 or install gcc-13 in PATH"
|
||||
[[ -n "${GPP13}" ]] && command -v "${GPP13}" >/dev/null 2>&1 \
|
||||
|| bail "g++-13 not found or not executable (needed for nvcc workaround). Set GPP13 or install g++-13 in PATH"
|
||||
|
||||
[[ -f "${HOME}/.env" ]] || echo "warn: ${HOME}/.env missing — provider API keys may be unset" >&2
|
||||
|
||||
NATIVE_FFMPEG_BIN="${HOME}/opt/ffmpeg-native/bin/ffmpeg"
|
||||
if [[ "${NVENC}" == "true" ]]; then
|
||||
NATIVE_VIDEO_CODEC=h264_nvenc
|
||||
else
|
||||
NATIVE_VIDEO_CODEC=libx264
|
||||
fi
|
||||
REQUIRE_NATIVE_FFMPEG="${DREAMVERSE_REQUIRE_NATIVE_FFMPEG:-false}"
|
||||
case "${REQUIRE_NATIVE_FFMPEG}" in
|
||||
true|false) ;;
|
||||
*) bail "DREAMVERSE_REQUIRE_NATIVE_FFMPEG must be 'true' or 'false' (got '${REQUIRE_NATIVE_FFMPEG}')" ;;
|
||||
esac
|
||||
if [[ -x "${NATIVE_FFMPEG_BIN}" ]]; then
|
||||
if [[ "${NVENC}" == "true" ]]; then
|
||||
encoder_list="$("${NATIVE_FFMPEG_BIN}" -hide_banner -encoders 2>/dev/null || true)"
|
||||
if [[ "${encoder_list}" != *h264_nvenc* ]]; then
|
||||
bail "--nvenc requested but ${NATIVE_FFMPEG_BIN} was not built with NVENC. Rebuild: bash apps/dreamverse/scripts/install_native_ffmpeg.sh (with ENABLE_NVENC=1, the default)"
|
||||
fi
|
||||
if ! "${NATIVE_FFMPEG_BIN}" -hide_banner -loglevel error -y \
|
||||
-f lavfi -i 'color=red:size=64x64:rate=24:duration=0.2' \
|
||||
-c:v h264_nvenc -f null - >/dev/null 2>&1; then
|
||||
bail "--nvenc requested but the GPU on this host has no NVENC silicon (probe failed: 'OpenEncodeSessionEx unsupported device'). Datacenter Blackwell (B200) and some H100 SKUs ship without NVENC; --nvenc only works on hosts with NVENC-capable GPUs (RTX 50-series, T4, A10, etc.)."
|
||||
fi
|
||||
fi
|
||||
echo " native ffmpeg: ${NATIVE_FFMPEG_BIN} (codec=${NATIVE_VIDEO_CODEC})"
|
||||
elif [[ "${REQUIRE_NATIVE_FFMPEG}" == "true" ]] || [[ "${NVENC}" == "true" ]]; then
|
||||
bail "${NATIVE_FFMPEG_BIN} missing (required by --nvenc or DREAMVERSE_REQUIRE_NATIVE_FFMPEG=true). Run: bash apps/dreamverse/scripts/install_native_ffmpeg.sh"
|
||||
else
|
||||
echo "warn: ${NATIVE_FFMPEG_BIN} missing — backend will fall back to system ffmpeg (\$(command -v ffmpeg))." >&2
|
||||
echo " Build native ffmpeg with: bash apps/dreamverse/scripts/install_native_ffmpeg.sh" >&2
|
||||
fi
|
||||
echo " python: ${CONDA_ENV_PYTHON}"
|
||||
|
||||
mkdir -p "${LOG_DIR}"
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Teardown anything on target ports
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
echo "[1/8] killing any existing deploy on ports ${BACKEND_PORT}/${FRONTEND_PORT} and GPU ${GPU}..."
|
||||
|
||||
kill_port_pid() {
|
||||
local port="$1"
|
||||
local pid
|
||||
|
||||
for pid in $(list_port_pids "${port}"); do
|
||||
terminate_pid "${pid}" "port=${port} pid=${pid}"
|
||||
done
|
||||
}
|
||||
|
||||
for pat in "dreamverse-server --host 0.0.0.0 --port ${BACKEND_PORT}" "next dev --port ${FRONTEND_PORT}" "NEXT_PUBLIC_INCLUDE_DEVTOOLS=1 next dev --port ${FRONTEND_PORT}"; do
|
||||
terminate_pattern "${pat}"
|
||||
done
|
||||
kill_port_pid "${BACKEND_PORT}"
|
||||
kill_port_pid "${FRONTEND_PORT}"
|
||||
|
||||
gpu_uuid="$(nvidia-smi --query-gpu=index,uuid --format=csv,noheader 2>/dev/null | awk -F', ' -v g="${GPU}" '$1==g {print $2}')"
|
||||
if [[ -n "${gpu_uuid}" ]]; then
|
||||
for pid in $(nvidia-smi --query-compute-apps=pid,gpu_uuid --format=csv,noheader 2>/dev/null \
|
||||
| awk -F', ' -v u="${gpu_uuid}" '$2==u {print $1}'); do
|
||||
if [[ -n "${pid}" ]] && [[ "${pid}" != "$$" ]]; then
|
||||
cmd="$(ps -p "${pid}" -o comm= 2>/dev/null || true)"
|
||||
terminate_pid "${pid}" "GPU${GPU} pid=${pid} (${cmd:-?})"
|
||||
fi
|
||||
done
|
||||
fi
|
||||
|
||||
for i in $(seq 1 30); do
|
||||
free_be=true
|
||||
free_fe=true
|
||||
ss -tln 2>/dev/null | grep -qE ":${BACKEND_PORT}\b" && free_be=false
|
||||
ss -tln 2>/dev/null | grep -qE ":${FRONTEND_PORT}\b" && free_fe=false
|
||||
gpu_mem="$(nvidia-smi --query-gpu=memory.used --format=csv,noheader,nounits 2>/dev/null | sed -n "$((GPU + 1))p" || echo 99999)"
|
||||
if "${free_be}" && "${free_fe}" && [[ "${gpu_mem}" -lt 1000 ]]; then
|
||||
break
|
||||
fi
|
||||
sleep 1
|
||||
done
|
||||
|
||||
gpu_mem="$(nvidia-smi --query-gpu=memory.used --format=csv,noheader,nounits 2>/dev/null | sed -n "$((GPU + 1))p" || echo 0)"
|
||||
echo " ports cleared; GPU${GPU} at ${gpu_mem} MiB"
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Launch backend
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
echo "[2/8] launching backend on GPU ${GPU} port ${BACKEND_PORT} (warmup=${WARMUP} torch_compile=${TORCH_COMPILE} nvenc=${NVENC})..."
|
||||
|
||||
backend_log="${LOG_DIR}/backend-gpu${GPU}.log"
|
||||
: > "${backend_log}"
|
||||
|
||||
setsid bash -c "
|
||||
set -a
|
||||
if [[ -f \"${HOME}/.env\" ]]; then
|
||||
source \"${HOME}/.env\"
|
||||
fi
|
||||
set +a
|
||||
if [[ -x \"${NATIVE_FFMPEG_BIN}\" ]]; then
|
||||
export FASTVIDEO_FFMPEG_BIN=\"${NATIVE_FFMPEG_BIN}\"
|
||||
export FASTVIDEO_VIDEO_CODEC=\"${NATIVE_VIDEO_CODEC}\"
|
||||
fi
|
||||
export DREAMVERSE_PYTHON=\"${CONDA_ENV_PYTHON}\"
|
||||
export CUDA_VISIBLE_DEVICES=${GPU}
|
||||
export FASTVIDEO_ENABLE_DEVTOOLS=1
|
||||
export FASTVIDEO_ENABLE_STARTUP_WARMUP=${WARMUP}
|
||||
export FASTVIDEO_GPU_COUNT=1
|
||||
export ENABLE_TORCH_COMPILE=${TORCH_COMPILE_FLAG}
|
||||
export CC=${GCC13}
|
||||
export CXX=${GPP13}
|
||||
export CUDAHOSTCXX=${GPP13}
|
||||
export NVCC_PREPEND_FLAGS=\"-ccbin ${GCC13} -allow-unsupported-compiler\"
|
||||
cd \"${REPO_ROOT}\"
|
||||
exec \"${DREAMVERSE_SERVER}\" --host 0.0.0.0 --port ${BACKEND_PORT}
|
||||
" > "${backend_log}" 2>&1 < /dev/null &
|
||||
disown
|
||||
|
||||
# Wait briefly, then resolve actual python PID (the inner process, not the
|
||||
# wrapper bash).
|
||||
sleep 4
|
||||
backend_pid="$(pgrep -f "dreamverse-server --host 0.0.0.0 --port ${BACKEND_PORT}" | head -1 || true)"
|
||||
|
||||
if [[ -z "${backend_pid}" ]]; then
|
||||
echo "error: backend failed to spawn. Last 30 lines of log:" >&2
|
||||
tail -30 "${backend_log}" >&2
|
||||
exit 4
|
||||
fi
|
||||
|
||||
echo " backend pid=${backend_pid} log=${backend_log}"
|
||||
|
||||
# Poll /readyz. Deadline scales with warmup + torch.compile flags
|
||||
# because warmup runs two synthetic segments before /readyz=200, and
|
||||
# torch.compile max-autotune adds ~3-4min cold start to the first
|
||||
# segment. Empirical worst case (warmup=true, torch_compile=true):
|
||||
# ~7 min on B200; we budget 15 min for safety.
|
||||
if [[ "${WARMUP}" == "true" ]] && [[ "${TORCH_COMPILE}" == "true" ]]; then
|
||||
READYZ_BUDGET_SECONDS=900
|
||||
elif [[ "${WARMUP}" == "true" ]] || [[ "${TORCH_COMPILE}" == "true" ]]; then
|
||||
READYZ_BUDGET_SECONDS=480
|
||||
else
|
||||
READYZ_BUDGET_SECONDS=300
|
||||
fi
|
||||
READYZ_POLL_INTERVAL=6
|
||||
READYZ_MAX_ITERS=$(( READYZ_BUDGET_SECONDS / READYZ_POLL_INTERVAL ))
|
||||
|
||||
echo "[3/8] polling http://127.0.0.1:${BACKEND_PORT}/readyz (budget=${READYZ_BUDGET_SECONDS}s) ..."
|
||||
ready=0
|
||||
for i in $(seq 1 ${READYZ_MAX_ITERS}); do
|
||||
code="$(curl -s -o /dev/null -w '%{http_code}' --max-time 2 "http://127.0.0.1:${BACKEND_PORT}/readyz" 2>/dev/null || echo 000)"
|
||||
if [[ "${code}" == "200" ]]; then
|
||||
ready=1
|
||||
break
|
||||
fi
|
||||
if ! kill -0 "${backend_pid}" 2>/dev/null; then
|
||||
echo "error: backend pid ${backend_pid} died. Last 50 lines:" >&2
|
||||
tail -50 "${backend_log}" >&2
|
||||
exit 5
|
||||
fi
|
||||
sleep ${READYZ_POLL_INTERVAL}
|
||||
done
|
||||
|
||||
if [[ "${ready}" != "1" ]]; then
|
||||
echo "error: backend did not become /readyz=200 within ${READYZ_BUDGET_SECONDS}s. Last 50 lines:" >&2
|
||||
tail -50 "${backend_log}" >&2
|
||||
exit 5
|
||||
fi
|
||||
|
||||
echo "[4/8] backend /readyz OK"
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Launch frontend
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
echo "[5/8] launching frontend on port ${FRONTEND_PORT}..."
|
||||
|
||||
frontend_log="${LOG_DIR}/frontend-port${FRONTEND_PORT}.log"
|
||||
: > "${frontend_log}"
|
||||
|
||||
# Resolve dev script: dev:devtools forces port 5274 + devtools env. If the
|
||||
# requested port differs, run `next dev --port` directly with devtools env.
|
||||
fe_cmd="run dev:devtools"
|
||||
if [[ "${FRONTEND_PORT}" != "5274" ]]; then
|
||||
fe_cmd="exec -- next dev --port ${FRONTEND_PORT}"
|
||||
fi
|
||||
|
||||
setsid bash -c "
|
||||
cd \"${REPO_ROOT}/apps/dreamverse/web\"
|
||||
export NEXT_PUBLIC_INCLUDE_DEVTOOLS=1
|
||||
export BACKEND_URL=http://127.0.0.1:${BACKEND_PORT}
|
||||
export BACKEND_HOST=127.0.0.1
|
||||
export BACKEND_PORT=${BACKEND_PORT}
|
||||
exec '${NPM}' ${fe_cmd}
|
||||
" > "${frontend_log}" 2>&1 < /dev/null &
|
||||
disown
|
||||
|
||||
sleep 4
|
||||
frontend_pid="$(pgrep -f "next dev --port ${FRONTEND_PORT}" | head -1 || true)"
|
||||
if [[ -z "${frontend_pid}" ]]; then
|
||||
echo "error: frontend failed to spawn. Last 30 lines:" >&2
|
||||
tail -30 "${frontend_log}" >&2
|
||||
exit 6
|
||||
fi
|
||||
|
||||
echo " frontend pid=${frontend_pid} log=${frontend_log}"
|
||||
|
||||
# Poll FE root
|
||||
echo "[6/8] polling http://127.0.0.1:${FRONTEND_PORT}/ ..."
|
||||
fe_ready=0
|
||||
for i in $(seq 1 30); do
|
||||
code="$(curl -s -o /dev/null -w '%{http_code}' --max-time 2 "http://127.0.0.1:${FRONTEND_PORT}/" 2>/dev/null || echo 000)"
|
||||
if [[ "${code}" == "200" ]]; then
|
||||
fe_ready=1
|
||||
break
|
||||
fi
|
||||
if ! kill -0 "${frontend_pid}" 2>/dev/null; then
|
||||
echo "error: frontend pid ${frontend_pid} died. Last 30 lines:" >&2
|
||||
tail -30 "${frontend_log}" >&2
|
||||
exit 7
|
||||
fi
|
||||
sleep 2
|
||||
done
|
||||
|
||||
if [[ "${fe_ready}" != "1" ]]; then
|
||||
echo "error: frontend did not respond 200 within 60s. Last 30 lines:" >&2
|
||||
tail -30 "${frontend_log}" >&2
|
||||
exit 7
|
||||
fi
|
||||
|
||||
echo "[7/8] frontend / OK"
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Print summary
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
cwd="$(readlink "/proc/${backend_pid}/cwd" 2>/dev/null || echo unknown)"
|
||||
gpu_mem_now="$(nvidia-smi --query-gpu=memory.used --format=csv,noheader,nounits 2>/dev/null | sed -n "$((GPU + 1))p" || echo 0)"
|
||||
ffmpeg_in_use="$(tr '\0' '\n' < "/proc/${backend_pid}/environ" 2>/dev/null | sed -n 's/^FASTVIDEO_FFMPEG_BIN=//p' | head -1)"
|
||||
[[ -z "${ffmpeg_in_use}" ]] && ffmpeg_in_use="$(command -v ffmpeg 2>/dev/null || echo '<not found>') (system fallback)"
|
||||
|
||||
cat <<SUMMARY
|
||||
[8/8] redeploy OK
|
||||
|
||||
Frontend : http://localhost:${FRONTEND_PORT} (PID ${frontend_pid})
|
||||
Backend : http://localhost:${BACKEND_PORT} (PID ${backend_pid})
|
||||
cwd=${cwd}
|
||||
gpu=${GPU} mem=${gpu_mem_now} MiB
|
||||
ffmpeg=${ffmpeg_in_use}
|
||||
|
||||
Logs : ${backend_log}
|
||||
${frontend_log}
|
||||
|
||||
Stop : ./.agents/skills/dreamverse-deploy/scripts/dreamverse-deploy.sh --stop
|
||||
|
||||
E2E : cd apps/dreamverse/web && \\
|
||||
PLAYWRIGHT_SKIP_WEBSERVER=1 \\
|
||||
BACKEND_URL=http://127.0.0.1:${BACKEND_PORT} \\
|
||||
PLAYWRIGHT_BASE_URL=http://127.0.0.1:${FRONTEND_PORT} \\
|
||||
NEXT_PUBLIC_INCLUDE_DEVTOOLS=1 \\
|
||||
npm exec -- playwright test
|
||||
SUMMARY
|
||||
@@ -0,0 +1,79 @@
|
||||
---
|
||||
name: env-var-conventions
|
||||
description: Add, read, rename, or remove an environment variable in FastVideo, or change the environment-variable policy. Use before touching fastvideo/envs.py, os.environ, os.getenv, or monkeypatch.setenv in fastvideo/, and when fastvideo/tests/contract/test_env_policy.py fails.
|
||||
---
|
||||
|
||||
# Environment Variable Conventions
|
||||
|
||||
## Purpose
|
||||
|
||||
FastVideo registers its environment variables as typed fields in
|
||||
`fastvideo/envs.py`. The policy that governs them is
|
||||
`docs/contributing/env_vars.md`, and the contract test
|
||||
`fastvideo/tests/contract/test_env_policy.py` enforces the policy in the unit
|
||||
CI lane. This skill routes an environment-variable change through that policy.
|
||||
The policy doc is the single source of the rules; read it instead of relying
|
||||
on a summary here.
|
||||
|
||||
## Prerequisites
|
||||
|
||||
- Read `docs/contributing/env_vars.md` in full.
|
||||
- Decide whether the setting belongs in an environment variable or an argument
|
||||
(rule 5 in the policy doc). Settings that users change per deployment are
|
||||
arguments; add them through `fastvideo/fastvideo_args.py` instead.
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Required | Description |
|
||||
| ---------- | -------- | -------------------------------------------------------------- |
|
||||
| `change` | Yes | Add, read, rename, or remove a variable, or change the policy. |
|
||||
| `variable` | Yes | The variable name, with the `FASTVIDEO_` prefix. |
|
||||
|
||||
## Steps
|
||||
|
||||
1. **Declare or edit the variable in `fastvideo/envs.py`.**
|
||||
- Pick the field type and category that the policy doc lists.
|
||||
- Write a description that states what the variable does and its units.
|
||||
- To rename, keep the old name in `deprecated_names`. To remove, add the
|
||||
name to `DEPRECATED_VARIABLES`. Update the uses in `examples/`,
|
||||
`scripts/`, `docs/`, `apps/`, and the tests.
|
||||
2. **Read the variable with `envs.NAME.get()` inside a function.**
|
||||
- In tests, change the value with `envs.NAME.override(value)`, and a variable
|
||||
outside the registry with `envs.override_external(name, value)`; the
|
||||
`env_overrides` fixture keeps either until the end of the test.
|
||||
- Name a variable that only tests read `FASTVIDEO_TEST_*`.
|
||||
- Do not call `os.environ`, `os.getenv`, or `monkeypatch.setenv` for a
|
||||
FastVideo variable.
|
||||
- To set a variable that another tool reads, call `envs.set_external`,
|
||||
`envs.setdefault_external`, or `envs.unset_external`.
|
||||
3. **Regenerate the table in the policy doc.**
|
||||
- Run `python fastvideo/tests/contract/test_env_policy.py`.
|
||||
4. **Run the contract test.**
|
||||
- Run `pytest fastvideo/tests/contract/test_env_policy.py`.
|
||||
- When the test reports a fixed known violation, delete or lower its entry
|
||||
in `KNOWN_VIOLATIONS`. Never add an entry to `KNOWN_VIOLATIONS`.
|
||||
5. **When the policy itself changes, update the policy doc and the contract
|
||||
test in the same pull request.**
|
||||
- The rules in `docs/contributing/env_vars.md`, the checks and allowlist in
|
||||
`fastvideo/tests/contract/test_env_policy.py`, and this skill must agree.
|
||||
|
||||
## Outputs
|
||||
|
||||
- A registry entry in `fastvideo/envs.py` and call sites that use
|
||||
`envs.NAME.get()`.
|
||||
- A regenerated table in `docs/contributing/env_vars.md`.
|
||||
- A passing `fastvideo/tests/contract/test_env_policy.py`.
|
||||
|
||||
## Example Usage
|
||||
|
||||
```
|
||||
Add a FASTVIDEO_DEBUG_MY_STAGE switch that logs MyStage inputs.
|
||||
```
|
||||
|
||||
## References
|
||||
|
||||
- `docs/contributing/env_vars.md`: the policy, the field types, and the
|
||||
violation kinds that the contract test reports.
|
||||
- `fastvideo/envs.py`: the registry.
|
||||
- `fastvideo/tests/contract/test_env_policy.py`: the contract test,
|
||||
`EXTERNAL_ALLOWLIST`, and `KNOWN_VIOLATIONS`.
|
||||
@@ -0,0 +1,642 @@
|
||||
---
|
||||
name: reseed-performance-baseline
|
||||
description: Re-seed the HF performance-tracking baseline for an intentional runtime, dependency, environment-caused benchmark shift, or reviewed v2 calibration using one or more reviewed normalized performance JSONs. Use when performance CI fails because metrics such as latency, throughput, component time, or peak memory changed for an accepted reason and the rolling median baseline in FastVideo/performance-tracking must be advanced, or when a new v2 exact comparable identity needs its first approved baseline. The workflow backs up existing history under /tmp, validates all source JSONs for the same legacy (model_id, gpu_type) target or the same v2 exact identity, rejects internally inconsistent source batches, uploads one success=true baseline record per accepted source JSON, and offers to clean local temp state after a successful upload.
|
||||
---
|
||||
|
||||
# Re-seed Performance Baseline
|
||||
|
||||
## Purpose
|
||||
|
||||
Replace or advance the rolling performance baseline in the HF dataset
|
||||
`FastVideo/performance-tracking`. Legacy targets are scoped by
|
||||
`(model_id, gpu_type)`. V2 targets are scoped by exact comparable identity:
|
||||
`workload_id`, `variant_id`, `benchmark_version`, `hardware_profile_id`,
|
||||
`software_profile_id`, and `recipe_fingerprint`.
|
||||
|
||||
Performance comparison uses the median of up to the last 5 successful,
|
||||
baseline-eligible records for the same target. Failed or calibration-only
|
||||
records are useful audit history, but they do not move the future baseline
|
||||
because `compare_baseline.py` loads records with `successful_only=True` and
|
||||
`baseline_eligible_only=True`.
|
||||
|
||||
This skill now reseeds from a reviewed batch of one or more source performance
|
||||
JSONs. It uploads one new `success=true` record per accepted source JSON; it
|
||||
does not blindly replicate one measurement into 3 or 5 records. The effective
|
||||
reseed size is therefore dynamic and equals the number of provided, validated,
|
||||
internally consistent source JSONs.
|
||||
|
||||
For baseline shifts with existing history, if the operator provides fewer than
|
||||
3 records, call out that the last-5 rolling median may not move immediately. If
|
||||
the operator provides 3 consistent shifted records, the rolling median usually
|
||||
moves immediately. If the operator provides 5 consistent shifted records, the
|
||||
last-5 window is effectively reset to the new runtime profile. For the first
|
||||
approved v2 baseline of a new exact identity, one reviewed calibration seed is
|
||||
enough for the next comparable run to leave `CALIBRATION_NEEDED`.
|
||||
|
||||
These records are intentional operator-approved baseline resets, not ordinary
|
||||
independent main-branch persistence. Mark them clearly with provenance fields
|
||||
so the HF history remains auditable.
|
||||
|
||||
Use this skill when a performance test fails for an intentional and reviewed
|
||||
reason, such as a torch/runtime/container upgrade that legitimately increases
|
||||
peak memory or changes timings. This is the performance equivalent of
|
||||
`reseed-ssim-references`: backup first, scope tightly, require explicit human
|
||||
approval, then upload reviewed accepted baseline records.
|
||||
|
||||
## When to use
|
||||
|
||||
- A PR or main run failed the rolling performance comparison by more than the
|
||||
allowed regression threshold, and maintainers agree the shift is caused by
|
||||
an intentional runtime, dependency, hardware image, or benchmark environment
|
||||
change rather than a FastVideo logic regression.
|
||||
- One or more shifted source result JSONs have been reviewed and accepted, and
|
||||
the operator wants to use those exact reviewed results to advance the rolling
|
||||
baseline.
|
||||
- The source batch is internally consistent: no provided source JSON regresses
|
||||
against the batch median by more than the configured tolerance.
|
||||
|
||||
## When not to use
|
||||
|
||||
- The benchmark failure might be a real code regression. Fix or investigate
|
||||
the code path first.
|
||||
- The fixed benchmark thresholds in
|
||||
`.buildkite/performance-benchmarks/tests/*.json` are too low. Those are a
|
||||
separate gate from the rolling HF baseline and may need a code review change.
|
||||
- There is no clear source run, commit, and rationale. Baseline history is a
|
||||
production signal; do not edit it without provenance.
|
||||
- The provided source JSONs disagree materially with each other. Rerun or
|
||||
investigate instead of uploading a noisy reseed batch.
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Required | Description |
|
||||
|-----------|----------|-------------|
|
||||
| `model_id` | Legacy required; v2 inferred | Benchmark id, e.g. `wan-t2v-1.3b-2gpu`. This maps to the HF subdirectory after `sanitize(model_id)`. For v2 records, use the `model_id` from each source artifact only as the upload directory; comparison is by exact identity. |
|
||||
| `gpu_type` | Legacy required; v2 inferred | Exact GPU device string from the performance record, e.g. the L40S device name emitted by CI. V2 hardware matching uses `hardware_profile_id`; preserve `gpu_type` as display metadata. |
|
||||
| `source_results` | Yes | One or more local paths or Buildkite artifact URLs for accepted shifted performance JSONs. Prefer normalized `normalized_perf_*.json` artifacts emitted by `compare_baseline.py`. Accept `source_result` as an alias only for a single JSON. |
|
||||
| `max_intra_batch_regression` | No | Maximum allowed regression of any source JSON against the source batch median. Default: `0.05` (5%). |
|
||||
| `intent_rationale` | Yes | One-line explanation for why the baseline shift is legitimate. This is written into provenance and should be reused in the PR. |
|
||||
|
||||
Hardcoded defaults:
|
||||
|
||||
- HF repo: `FastVideo/performance-tracking` (`HF_REPO_ID` override is
|
||||
supported by the code, but use the default unless the user explicitly asks).
|
||||
- Local sync root: `/tmp/perf-tracking` (`PERFORMANCE_TRACKING_ROOT` override
|
||||
is supported).
|
||||
- Prepared-record staging root: `/tmp/performance_reseed_prepared`
|
||||
(`PERFORMANCE_RESEED_STAGING_ROOT` override is supported). Keep it separate
|
||||
and non-nested from the sync root.
|
||||
- Backup root: `/tmp/performance_reseed_backup`.
|
||||
- Download scratch root for source artifact URLs: `/tmp/performance_reseed_source`.
|
||||
- Baseline window: last 5 `success=true`, `baseline_eligible=true` records
|
||||
for the same legacy `(model_id, gpu_type)` target or the same v2 exact
|
||||
comparable identity.
|
||||
- Reseed count: dynamic. Upload exactly one accepted seed record per validated
|
||||
source JSON.
|
||||
|
||||
## Steps
|
||||
|
||||
### 1. Validate the target and source results
|
||||
|
||||
Normalize `source_results` to a list. If the user passes a single
|
||||
`source_result`, treat it as a one-element `source_results` list and report
|
||||
that a single record may not move the last-5 median immediately.
|
||||
|
||||
If any source result is a Buildkite artifact URL, download it first into a
|
||||
local scratch directory under `/tmp/performance_reseed_source/` and use the
|
||||
downloaded JSON path for the rest of the workflow. If the agent cannot access
|
||||
the artifact because Buildkite authentication is missing, ask the user to
|
||||
download the artifact manually and provide the local path.
|
||||
|
||||
Prefer the normalized Buildkite artifact emitted by `compare_baseline.py`:
|
||||
|
||||
```text
|
||||
perf_reports/results/normalized_perf_*.json
|
||||
```
|
||||
|
||||
That file is already in the HF tracking schema. Load each normalized JSON
|
||||
directly:
|
||||
|
||||
```python
|
||||
import json
|
||||
|
||||
with open(source_result, encoding="utf-8") as f:
|
||||
record = json.load(f)
|
||||
```
|
||||
|
||||
Classify the source batch before continuing:
|
||||
|
||||
- **Legacy source records** have no v2 exact identity fields. Stop if any
|
||||
normalized record's `model_id` or `gpu_type` does not match the requested
|
||||
`model_id` and `gpu_type`.
|
||||
- **V2 source records** have exact identity fields. Stop unless every source
|
||||
record has all six comparable identity fields and they are identical across
|
||||
the batch: `workload_id`, `variant_id`, `benchmark_version`,
|
||||
`hardware_profile_id`, `software_profile_id`, and `recipe_fingerprint`.
|
||||
Do not fall back to legacy `(model_id, gpu_type)` matching for v2 records.
|
||||
|
||||
The source records may have `success: false` when they came from failed
|
||||
rolling baseline comparisons. That is expected; only the reviewed reseed
|
||||
records become new `success: true` baseline records after explicit approval.
|
||||
For a first v2 baseline seed, the source records must instead be successful
|
||||
scheduled-main full-suite `CALIBRATION_NEEDED` normalized artifacts. Reject PR,
|
||||
local, direct-run, non-main-branch, or non-full-suite calibration artifacts as
|
||||
seed sources.
|
||||
|
||||
Sort validated source records by their original `timestamp` ascending before
|
||||
preparing the seed records. If a source timestamp is missing or unparsable,
|
||||
preserve input order for those records and print a warning. This makes the
|
||||
fresh reseed timestamps deterministic and makes it clear which records enter
|
||||
the last-5 window when more than 5 source JSONs are provided.
|
||||
|
||||
Check that `HF_API_KEY` is exported. The sync path may be public, but the
|
||||
upload path requires write access.
|
||||
|
||||
### 1a. Check source batch consistency
|
||||
|
||||
Before syncing or preparing uploads, reject source batches that are internally
|
||||
inconsistent. Use the same metric direction as `compare_baseline.py`:
|
||||
|
||||
- Lower is better: `latency`, `memory`, `text_encoder_time_s`, `dit_time_s`,
|
||||
`vae_decode_time_s`.
|
||||
- Higher is better: `throughput`.
|
||||
|
||||
For each metric with at least two non-null source values:
|
||||
|
||||
1. Compute the source batch median.
|
||||
2. For lower-is-better metrics, compute `(source_value - batch_median) / batch_median`.
|
||||
3. For `throughput`, compute `(batch_median - source_value) / batch_median`.
|
||||
4. Stop if any source record regresses against the batch median by more than
|
||||
`max_intra_batch_regression`.
|
||||
|
||||
Default `max_intra_batch_regression` to `0.05`. Print a table with per-source values, batch median, and
|
||||
worst intra-batch regression.
|
||||
|
||||
This check prevents uploading a mixed batch where one JSON is materially
|
||||
slower or faster than the others. If the batch fails this check, ask the user
|
||||
to provide a cleaner batch or explicitly investigate the variance. Do not
|
||||
silently drop outliers unless the user gives a concrete reviewed reason and a
|
||||
new source list.
|
||||
|
||||
### 1b. How to obtain source results from CI
|
||||
|
||||
The performance CI exports normalized source results for failed rolling
|
||||
baseline comparisons when `compare_baseline.py` ran. The preferred artifacts
|
||||
come from:
|
||||
|
||||
```text
|
||||
perf_reports/results/normalized_perf_*.json
|
||||
```
|
||||
|
||||
The normal operator flow is:
|
||||
|
||||
1. Open the failed Buildkite performance job or several reruns of the same
|
||||
benchmark after the accepted environment shift.
|
||||
2. Download the `normalized_perf_*.json` artifacts for the target benchmark.
|
||||
3. Pass all reviewed local paths or artifact URLs as `source_results`.
|
||||
|
||||
Do not scrape the Markdown performance summary to reconstruct JSON. The
|
||||
normalized JSON artifacts are the only supported source of truth for reseed
|
||||
metrics and provenance. Raw `fastvideo/tests/performance/results/perf_*.json`
|
||||
artifacts are not accepted by this skill. If no normalized JSON artifact is
|
||||
present, that run is not a valid source for baseline reseeding.
|
||||
|
||||
### 2. Sync and back up existing HF records under /tmp
|
||||
|
||||
Use `fastvideo/performance/hf_store.py` helpers directly. Do **not** use
|
||||
`compare_baseline.py` as a sync shortcut; on full main runs it can persist
|
||||
records, while this step must only fetch and back up existing history.
|
||||
|
||||
The sync command pattern is:
|
||||
|
||||
```bash
|
||||
export PERFORMANCE_TRACKING_ROOT="${PERFORMANCE_TRACKING_ROOT:-/tmp/perf-tracking}"
|
||||
export HF_REPO_ID="${HF_REPO_ID:-FastVideo/performance-tracking}"
|
||||
python -c 'from fastvideo.performance.hf_store import sync_from_hf; import os; sync_from_hf(os.environ["PERFORMANCE_TRACKING_ROOT"], strict=True)'
|
||||
```
|
||||
|
||||
For legacy records, back up the sanitized model directory under `/tmp`:
|
||||
|
||||
```bash
|
||||
SHORT_COMMIT=$(git rev-parse --short=12 HEAD)
|
||||
TIMESTAMP=$(date -u +%Y%m%d_%H%M%S)
|
||||
MODEL_SAFE=$(python - <<'PY'
|
||||
from fastvideo.performance.hf_store import sanitize
|
||||
print(sanitize("<model_id>"))
|
||||
PY
|
||||
)
|
||||
BACKUP_DIR="/tmp/performance_reseed_backup/${TIMESTAMP}_${SHORT_COMMIT}_${MODEL_SAFE}"
|
||||
mkdir -p "$BACKUP_DIR"
|
||||
cp -R "${PERFORMANCE_TRACKING_ROOT}/${MODEL_SAFE}" "$BACKUP_DIR/" 2>/dev/null || true
|
||||
```
|
||||
|
||||
For v2 records, back up the full local tracking root after sync. Exact identity
|
||||
lookup scans across model directories, so a benchmark rename may have relevant
|
||||
history outside the current source artifact's `model_id` directory:
|
||||
|
||||
```bash
|
||||
BACKUP_DIR="/tmp/performance_reseed_backup/${TIMESTAMP}_${SHORT_COMMIT}_v2_exact_identity"
|
||||
mkdir -p "$BACKUP_DIR"
|
||||
cp -R "${PERFORMANCE_TRACKING_ROOT}" "$BACKUP_DIR/tracking-root"
|
||||
```
|
||||
|
||||
Write provenance next to the backup:
|
||||
|
||||
```bash
|
||||
cat > "$BACKUP_DIR/PROVENANCE.txt" <<EOF
|
||||
model_id: <model_id>
|
||||
gpu_type: <gpu_type>
|
||||
source_results:
|
||||
- <source_result_1>
|
||||
- <source_result_2>
|
||||
reseed_record_count: <len(source_results)>
|
||||
max_intra_batch_regression: <threshold>
|
||||
head_commit: $(git rev-parse HEAD)
|
||||
timestamp_utc: $(date -u +%FT%TZ)
|
||||
reason: <intent_rationale>
|
||||
EOF
|
||||
```
|
||||
|
||||
If the backup has no prior records, this is not a destructive reseed; it is a
|
||||
first baseline seed. Continue, but report that baseline history was empty.
|
||||
|
||||
### 3. Compute old baseline and candidate shift
|
||||
|
||||
Load the last 5 successful baseline records for the target.
|
||||
|
||||
For legacy targets:
|
||||
|
||||
```python
|
||||
from fastvideo.performance.hf_store import load_records_for_model
|
||||
|
||||
records = load_records_for_model(
|
||||
"/tmp/perf-tracking",
|
||||
"<model_id>",
|
||||
"<gpu_type>",
|
||||
last_n=5,
|
||||
successful_only=True,
|
||||
baseline_eligible_only=True,
|
||||
)
|
||||
```
|
||||
|
||||
For v2 exact-identity targets:
|
||||
|
||||
```python
|
||||
from fastvideo.performance.hf_store import load_records_for_identity
|
||||
|
||||
records = load_records_for_identity(
|
||||
"/tmp/perf-tracking",
|
||||
{
|
||||
"workload_id": "<workload_id>",
|
||||
"variant_id": "<variant_id>",
|
||||
"benchmark_version": "<benchmark_version>",
|
||||
"hardware_profile_id": "<hardware_profile_id>",
|
||||
"software_profile_id": "<software_profile_id>",
|
||||
"recipe_fingerprint": "<recipe_fingerprint>",
|
||||
},
|
||||
last_n=5,
|
||||
successful_only=True,
|
||||
baseline_eligible_only=True,
|
||||
)
|
||||
```
|
||||
|
||||
Print a small table showing old medians, source batch medians, candidate
|
||||
medians after appending the proposed seed records, and source batch spread for:
|
||||
|
||||
- `latency`
|
||||
- `throughput`
|
||||
- `memory`
|
||||
- `text_encoder_time_s`
|
||||
- `dit_time_s`
|
||||
- `vae_decode_time_s`
|
||||
|
||||
Also print how many successful old records exist. Make clear:
|
||||
|
||||
- 1 seed record usually does not move an existing last-5 median by itself, but
|
||||
it is enough to establish the first v2 baseline for a new exact identity.
|
||||
- 3 consistent seed records usually move the last-5 median immediately.
|
||||
- 5 consistent seed records effectively reset the last-5 window.
|
||||
- The records are intentional approved baseline resets and must be labeled
|
||||
that way.
|
||||
|
||||
### 4. Confirm intent
|
||||
|
||||
Require an explicit confirmation phrase before preparing the upload:
|
||||
|
||||
> About to RE-SEED performance baseline for `<target description>`.
|
||||
> This will upload `<N>` new `success=true` records to
|
||||
> `FastVideo/performance-tracking/<sanitize(model_id)>/` or the source
|
||||
> artifact's v2 model directory, one per accepted source JSON.
|
||||
>
|
||||
> Reason: `<intent_rationale>`
|
||||
> Source results: `<source_results>`
|
||||
> Reseed record count: `<N>`
|
||||
> Max intra-batch regression: `<threshold>`
|
||||
> Note: these records come from a reviewed source batch and are intended to
|
||||
> move the rolling median to the accepted runtime profile. They are not
|
||||
> ordinary main-branch persistence.
|
||||
> HEAD: `<git rev-parse --short=12 HEAD>`
|
||||
> Backup: `<BACKUP_DIR>`
|
||||
>
|
||||
> Reply `confirm performance reseed` to proceed, anything else to abort.
|
||||
|
||||
Do not continue unless the user types exactly `confirm performance reseed`.
|
||||
|
||||
### 5. Create the accepted seed records
|
||||
|
||||
Create one seed record from each normalized source result.
|
||||
|
||||
For first v2 baseline seeds, use the scoped utility. It validates exact
|
||||
identity, requires successful scheduled-main full-suite `CALIBRATION_NEEDED`
|
||||
source artifacts, preserves the normalized v2 identity and metadata fields,
|
||||
and writes seed records with `success=true`, `baseline_eligible=true`, and
|
||||
`comparison_status=PASS`:
|
||||
|
||||
```bash
|
||||
python fastvideo/tests/performance/seed_baseline.py \
|
||||
--source-result <normalized_perf_1.json> \
|
||||
--source-result <normalized_perf_2.json> \
|
||||
--intent-rationale "<intent_rationale>" \
|
||||
--max-intra-batch-regression 0.05 \
|
||||
--tracking-root "${PERFORMANCE_TRACKING_ROOT}" \
|
||||
--staging-root "${PERFORMANCE_RESEED_STAGING_ROOT:-/tmp/performance_reseed_prepared}"
|
||||
```
|
||||
|
||||
The utility is prepare-only and intentionally has no upload option. Upload the
|
||||
scoped records only after the separate confirmation in step 6.
|
||||
|
||||
The utility validates against an isolated fresh HF snapshot and leaves
|
||||
`PERFORMANCE_TRACKING_ROOT` untouched; that argument only proves the staging
|
||||
root is separate from the operator's tracking mirror. Before writing, it stops
|
||||
if the exact identity already has a successful baseline-eligible record or if
|
||||
the workload/variant/version already trusts another recipe. It atomically
|
||||
reserves the exact identity and writes a digest-protected upload manifest bound
|
||||
to the current HF endpoint, repository id, and repository type. Keep the
|
||||
prepared records, manifest, source files, and reservation unchanged until the
|
||||
operation is uploaded or explicitly cleaned up.
|
||||
|
||||
If the prepared seed records look correct, upload only those scoped records in
|
||||
step 7. Do not rerun the utility with a different source list after approval.
|
||||
|
||||
For legacy reseeds or accepted v2 baseline shifts from regression artifacts,
|
||||
create one seed record from each normalized source result. Do not copy the
|
||||
source JSON wholesale.
|
||||
|
||||
Infer the baseline field allowlist from all existing HF records for the target
|
||||
after syncing, including both `success=true` and `success=false` records. For
|
||||
legacy targets the target is `(model_id, gpu_type)`. For v2 baseline-shift
|
||||
reseeds the target is the exact comparable identity. Use the union of
|
||||
non-provenance keys present in those target records, preserving only fields
|
||||
that also exist in the normalized source record or are explicitly set by the
|
||||
reseed workflow. Always include `model_id`, `timestamp`, `success`,
|
||||
`baseline_eligible`, and `comparison_status` because the upload path and
|
||||
baseline loader depend on them. Always set `timestamp` to a fresh reseed
|
||||
timestamp, `success` to `true`, `baseline_eligible` to `true`, and
|
||||
`comparison_status` to `PASS`. Do not include unrelated source-only fields
|
||||
that are absent from existing HF records.
|
||||
|
||||
Exclude existing provenance or operator metadata from the inferred baseline
|
||||
field allowlist. At minimum, exclude keys prefixed with `baseline_reseed` and
|
||||
any fields known to be local-only audit metadata.
|
||||
|
||||
If there are no previous HF records for the target, fall back to this default
|
||||
baseline field list:
|
||||
|
||||
- `model_id`
|
||||
- `timestamp`
|
||||
- `commit_sha`
|
||||
- `gpu_type`
|
||||
- `latency`
|
||||
- `throughput`
|
||||
- `memory`
|
||||
- `text_encoder_time_s`
|
||||
- `dit_time_s`
|
||||
- `vae_decode_time_s`
|
||||
- `success`
|
||||
- `baseline_eligible`
|
||||
- `comparison_status`
|
||||
|
||||
For v2 baseline-shift reseeds with no previous HF records for the exact
|
||||
identity, also preserve:
|
||||
|
||||
- `workload_id`
|
||||
- `variant_id`
|
||||
- `benchmark_version`
|
||||
- `recipe_fingerprint`
|
||||
- `hardware_profile_id`
|
||||
- `software_profile_id`
|
||||
- `recipe`
|
||||
- `hardware_profile`
|
||||
- `software_profile`
|
||||
- `software_comparison_profile`
|
||||
|
||||
Do not upload extra fields from the source artifact.
|
||||
|
||||
Optional provenance fields are allowed and useful:
|
||||
|
||||
- `baseline_reseed: true`
|
||||
- `baseline_reseed_reason`
|
||||
- `baseline_reseed_source_result`
|
||||
- `baseline_reseed_source_timestamp`
|
||||
- `baseline_reseed_source_success`
|
||||
- `baseline_reseed_batch_size`
|
||||
- `baseline_reseed_batch_index`
|
||||
- `baseline_reseed_operator`
|
||||
- `baseline_reseed_max_intra_batch_regression`
|
||||
|
||||
The v2 calibration seed utility writes analogous first-seed provenance:
|
||||
|
||||
- `baseline_seed: true`
|
||||
- `baseline_seed_reason`
|
||||
- `baseline_seed_source_result`
|
||||
- `baseline_seed_source_status`
|
||||
- `baseline_seed_source_timestamp`
|
||||
- `baseline_seed_source_success`
|
||||
- `baseline_seed_source_run_source`
|
||||
- `baseline_seed_source_branch`
|
||||
- `baseline_seed_source_test_scope`
|
||||
- `baseline_seed_source_pr_number`
|
||||
- `baseline_seed_batch_size`
|
||||
- `baseline_seed_batch_index`
|
||||
- `baseline_seed_operator`
|
||||
|
||||
Use a fresh reseed timestamp for each seed record, not the original source
|
||||
result timestamp. This is required because
|
||||
`load_records_for_model(..., last_n=5)` keeps the last records after loading
|
||||
the model directory; stale filenames/timestamps may not enter the last-5
|
||||
window and therefore may not move the median. Preserve the original source
|
||||
timestamp in `baseline_reseed_source_timestamp`.
|
||||
|
||||
Use the existing filename convention from `_write_tracking_record()`:
|
||||
`<sanitize(timestamp)>_<sanitize(commit_sha)>.json` under the sanitized model
|
||||
directory, but include a deterministic suffix such as `_reseed_01`,
|
||||
`_reseed_02`, and so on before `.json` so multiple records from the same
|
||||
batch do not overwrite each other.
|
||||
|
||||
If a source record already exists on HF with `success=false`, do not edit it
|
||||
in place unless the user explicitly asked for an audit-preserving correction.
|
||||
Prefer uploading new accepted seed records so failed history remains visible.
|
||||
|
||||
### 6. Pause before upload
|
||||
|
||||
Print:
|
||||
|
||||
- Backup directory path under `/tmp`.
|
||||
- Prepared local record paths under `PERFORMANCE_RESEED_STAGING_ROOT`.
|
||||
- Prepared upload-manifest path under the identity reservation.
|
||||
- HF paths that will receive the new records.
|
||||
- Old rolling medians.
|
||||
- Source batch medians, source batch spread, reseed count, and candidate
|
||||
medians.
|
||||
- Rationale.
|
||||
|
||||
Ask the user to reply exactly `upload`. Anything else aborts and leaves the
|
||||
prepared records plus backup on disk.
|
||||
|
||||
### 7. Upload only the scoped records
|
||||
|
||||
For a first v2 calibration seed, use the manifest uploader after the user
|
||||
replies exactly `upload`:
|
||||
|
||||
```bash
|
||||
python -c 'from fastvideo.tests.performance.seed_baseline import upload_prepared_seed_manifest; print(upload_prepared_seed_manifest("<prepared_manifest>"))'
|
||||
```
|
||||
|
||||
The uploader verifies the source and prepared-record digests, pins and scans
|
||||
the current HF revision, rechecks exact-identity and recipe-cohort conflicts,
|
||||
and writes the entire batch in one commit whose `parent_commit` must still be
|
||||
current. A concurrent Hub update makes the commit fail. Do not retry
|
||||
automatically: preserve staging, refresh/review remote state, and request a new
|
||||
explicit `upload` after the conflict is understood. Each record goes to:
|
||||
|
||||
```text
|
||||
FastVideo/performance-tracking/<sanitize(model_id)>/<record_filename>.json
|
||||
```
|
||||
|
||||
Never call `upload_record()` once per first-seed record: that can partially
|
||||
land the batch and has no compare-and-swap guard.
|
||||
|
||||
For a legacy reseed or an accepted v2 baseline shift, the first-seed manifest
|
||||
validator does not apply because an eligible baseline already exists. Upload
|
||||
only the individually reviewed records prepared in step 5 with the shared
|
||||
`upload_record(local_path, record, strict=True)` helper. Stop on the first
|
||||
failure and report exactly which records reached HF; do not silently rerun or
|
||||
replicate the remainder.
|
||||
|
||||
Never bulk upload the tracking or staging root, and never modify another
|
||||
model's directory in the same operation.
|
||||
|
||||
### 8. Report outcome and offer cleanup
|
||||
|
||||
Report:
|
||||
|
||||
- Uploaded HF paths.
|
||||
- Backup directory under `/tmp`.
|
||||
- Local tracking root, usually `/tmp/perf-tracking`.
|
||||
- Old baseline window count and medians.
|
||||
- Source batch medians, source batch spread, reseed count, and candidate
|
||||
medians.
|
||||
- Expected effect based on reseed count.
|
||||
- Any separate threshold changes still needed in
|
||||
`.buildkite/performance-benchmarks/tests/*.json`.
|
||||
|
||||
Include the `intent_rationale` in the PR or follow-up comment so reviewers can
|
||||
distinguish an accepted baseline shift from a hidden regression.
|
||||
|
||||
After the upload is verified, ask whether the user wants to clear temporary
|
||||
local state. Explain what each directory is for:
|
||||
|
||||
- `PERFORMANCE_TRACKING_ROOT`, usually `/tmp/perf-tracking`: read-only local
|
||||
synced mirror used for operator review and reporting. First-v2 preparation
|
||||
independently proves remote state from a fresh temporary HF snapshot.
|
||||
- `PERFORMANCE_RESEED_STAGING_ROOT`, usually
|
||||
`/tmp/performance_reseed_prepared`: prepared local seed records used for the
|
||||
scoped upload, plus the identity reservation and digest manifest. Keeping
|
||||
this separate prevents aborted preparations from appearing in later
|
||||
baseline reads.
|
||||
- `/tmp/performance_reseed_backup/<...>`: local backup of the target model's
|
||||
pre-reseed HF history plus `PROVENANCE.txt`, kept so a bad reseed can be
|
||||
audited or corrected.
|
||||
- `/tmp/performance_reseed_source/<...>` when used: downloaded source JSON
|
||||
artifacts from Buildkite URLs.
|
||||
|
||||
Ask:
|
||||
|
||||
> Reseed succeeded. Do you want me to delete the local temp tracking mirror,
|
||||
> this reseed's prepared staging records, source downloads, and reseed backup
|
||||
> under `/tmp`? These files are local safety/audit artifacts only; HF already
|
||||
> has the uploaded records.
|
||||
>
|
||||
> Reply `cleanup reseed temp` to delete them, anything else to keep them.
|
||||
|
||||
Do not delete anything unless the user replies exactly
|
||||
`cleanup reseed temp`. If cleanup is requested, remove only the specific
|
||||
directories and prepared record paths created for this reseed. Do not remove
|
||||
the shared staging root when it contains other records. Remove this operation's
|
||||
identity reservation only with its prepared records and manifest, and never
|
||||
remove unrelated `/tmp` contents.
|
||||
|
||||
## Failure modes and handling
|
||||
|
||||
- **`HF_API_KEY` unset.** Stop before upload. Do not create an untracked
|
||||
process that appears to have reseeded but never reached HF.
|
||||
- **Source result does not match target.** Stop. The wrong benchmark or GPU
|
||||
would poison a separate baseline.
|
||||
- **Source batch is internally inconsistent.** Stop if any source regresses
|
||||
against the source batch median by more than `max_intra_batch_regression`.
|
||||
Ask for cleaner sources or a reviewed explanation before continuing.
|
||||
- **Too few source records to move the median.** Continue only after making
|
||||
clear that one or two records may not immediately move an existing last-5
|
||||
median. This warning does not block a first v2 calibration seed for an exact
|
||||
identity with no eligible baseline yet.
|
||||
- **The source results are noisy or suspicious.** Stop. Reseeding amplifies
|
||||
those measurements into the baseline, so they must be reviewed first.
|
||||
- **HF sync fails.** Stop for destructive reseeds. A stale or empty sync can
|
||||
make the old baseline look missing.
|
||||
- **The exact v2 identity already has an eligible baseline.** Stop. The
|
||||
`CALIBRATION_NEEDED` artifact is stale; use the reviewed baseline-shift path
|
||||
instead of the first-seed utility.
|
||||
- **The workload/variant/version trusts another recipe.** Stop. The source is
|
||||
stale relative to the current recipe cohort and must not bypass
|
||||
`RECIPE_MISMATCH` by creating a second trusted recipe.
|
||||
- **The staging root already has a prepared seed for the exact identity.**
|
||||
Stop and reuse, upload, or explicitly clean that preparation. Do not prepare
|
||||
another copy of the same measurement.
|
||||
- **The conditional Hub commit loses its parent race.** Stop without retrying.
|
||||
Keep the preparation, refresh and review the new remote state, then request
|
||||
a new explicit `upload` only if the seed is still valid.
|
||||
- **Candidate still violates fixed thresholds.** Report that this skill only
|
||||
handles the rolling HF baseline; update benchmark JSON thresholds in code
|
||||
review if maintainers accept the new absolute limit.
|
||||
- **The user aborts at either confirmation.** Leave the backup and prepared
|
||||
records on disk. Nothing should be uploaded.
|
||||
- **The user declines cleanup.** Keep `/tmp/perf-tracking`, the prepared seed
|
||||
records under `/tmp/performance_reseed_prepared`, the source download
|
||||
directory if any, and `/tmp/performance_reseed_backup/<...>` in place for
|
||||
audit/debugging.
|
||||
- **A bad seed was uploaded.** Use the backup and HF history to identify the
|
||||
uploaded file, then remove or supersede it with an explicitly reviewed
|
||||
corrective record. Do not silently rewrite unrelated history.
|
||||
|
||||
## References
|
||||
|
||||
- `.agents/skills/reseed-ssim-references/SKILL.md` — safety pattern for
|
||||
intentional baseline replacement.
|
||||
- `fastvideo/tests/performance/compare_baseline.py` — normalization, rolling
|
||||
median comparison, and persistence rules.
|
||||
- `fastvideo/performance/hf_store.py` — HF sync and record loading helpers.
|
||||
- `fastvideo/tests/performance/seed_baseline.py` — first-seed preparation,
|
||||
staging reservation, manifest validation, and conditional batch upload.
|
||||
- `fastvideo/tests/performance/test_inference_performance.py` — source result
|
||||
JSON schema.
|
||||
- `.buildkite/performance-benchmarks/tests/*.json` — fixed absolute benchmark
|
||||
thresholds, separate from rolling baseline comparisons.
|
||||
|
||||
## Changelog
|
||||
|
||||
| Date | Change |
|
||||
|------|--------|
|
||||
| 2026-05-03 | Initial version. Sister workflow to `reseed-ssim-references`, scoped to one performance `(model_id, gpu_type)` baseline seed with backup, confirmation, provenance, and `success=true` upload. |
|
||||
| 2026-05-03 | Previous policy: replicate one approved shifted source result into 3 success records by default, or 5 only when explicitly requested. Add provenance marker for replicated-source reseeds. Superseded by the 2026-05-08 dynamic multi-source policy. |
|
||||
| 2026-05-08 | Replace fixed 3/5 replication with dynamic multi-source reseeding: upload one seed record per reviewed source JSON, validate intra-batch consistency, move backup/source scratch under `/tmp`, and ask whether to clean temp state after successful upload. |
|
||||
| 2026-07-13 | Keep first-v2-seed preparation outside the canonical mirror, reserve staging identities atomically, reject stale or replayed calibration seeds, and upload reviewed manifests with a single parent-guarded Hub commit. |
|
||||
@@ -0,0 +1,344 @@
|
||||
---
|
||||
name: reseed-ssim-references
|
||||
description: Re-seed HF reference videos for a single existing SSIM test on Modal L40S. Always backs up current refs locally first, regenerates on Modal, pauses for the user to eyeball before-vs-after quality, then overwrites the targeted model subtree on `FastVideo/ssim-reference-videos` with `--force`. Use when an intentional code change (model port fix, attention backend swap, kernel upgrade, hyperparameter change) has invalidated existing refs and they need to be regenerated. Pairs with `seed-ssim-references`, which is for first-time seeding only.
|
||||
---
|
||||
|
||||
# Re-seed SSIM Reference Videos
|
||||
|
||||
## Purpose
|
||||
|
||||
Replace the existing SSIM reference videos for a single `(test_file, model_id)`
|
||||
pair on the HF dataset (`FastVideo/ssim-reference-videos`). This is **destructive**
|
||||
on HF — the old refs are overwritten — so the skill always:
|
||||
|
||||
1. Confirms intent with a one-liner the user has to type.
|
||||
2. Downloads the existing refs as a local, timestamped backup.
|
||||
3. Regenerates through the manual legacy Modal L40S maintenance path.
|
||||
4. Pauses for a side-by-side eyeball of backup vs new mp4s.
|
||||
5. Uploads with `--force`, scoped to the single `--model-id`.
|
||||
6. Reminds the user to keep the backup until the PR lands.
|
||||
|
||||
Pairs with `seed-ssim-references`, which is the inverse (first-time seeding
|
||||
only, refuses to overwrite). Re-seeding is intentionally a separate, more
|
||||
ceremonial operation because mistakenly clobbering production refs is much
|
||||
harder to recover from than failing closed.
|
||||
|
||||
## When to use
|
||||
|
||||
- An intentional code change (model port fix, kernel upgrade, attention
|
||||
backend swap, hyperparameter change in the test itself) has shifted the
|
||||
expected SSIM output and the existing refs no longer represent the new
|
||||
ground truth.
|
||||
- A test is failing in CI **for the right reason** (the new code is correct,
|
||||
the old refs are stale).
|
||||
|
||||
## When not to use
|
||||
|
||||
- A test is failing for the **wrong** reason (the port is buggy, not the
|
||||
refs). Fix the port; re-seeding hides the bug.
|
||||
- A brand-new test that has no refs on HF yet. Use `seed-ssim-references`.
|
||||
- "Just to clean up drift" without a concrete code change to point at. The
|
||||
PR description has to justify *why* refs changed; without a concrete
|
||||
change, there's nothing to write.
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Required | Description |
|
||||
|-----------|----------|-------------|
|
||||
| `test_file` | Yes | Path to the SSIM test, e.g. `fastvideo/tests/ssim/test_matrixgame_similarity.py`. Validated against `fastvideo/tests/ssim/test_*_similarity.py`. |
|
||||
| `model_id` | Yes | Single model id from the test's `*_MODEL_TO_PARAMS`, e.g. `Matrix-Game-2.0-Diffusers-Base`. Re-seed runs are **per model**. For multi-model tests, invoke the skill once per model. |
|
||||
| `intent_rationale` | Yes | One-line explanation of *why* refs are being regenerated (e.g. "Relax FA-2 head_size whitelist to include 80 — matrix_game now uses FLASH_ATTN instead of TORCH_SDPA"). Recorded in the backup directory and reused in the PR description. |
|
||||
|
||||
Hardcoded:
|
||||
|
||||
- Modal GPU: **L40S**. This is a manual reference-maintenance target, not the
|
||||
active Slurm CI compute path; changing the SKU also changes the historical
|
||||
`L40S_reference_videos` contract.
|
||||
- Quality tier: **`default`**. `full_quality` is a separate, deliberate
|
||||
operation.
|
||||
- HF repo: `FastVideo/ssim-reference-videos` (override via
|
||||
`FASTVIDEO_TEST_SSIM_REFERENCE_HF_REPO`).
|
||||
- Device folder: `L40S_reference_videos`.
|
||||
|
||||
## Prerequisites
|
||||
|
||||
The user has confirmed:
|
||||
|
||||
- `modal` CLI authenticated.
|
||||
- `hf` CLI authenticated, **and** `HF_API_KEY` (or `HUGGINGFACE_HUB_TOKEN` /
|
||||
`HF_TOKEN`) exported with **write** access to
|
||||
`FastVideo/ssim-reference-videos`.
|
||||
- The current branch's code is the change that motivated the re-seed (i.e.
|
||||
`git rev-parse HEAD` is the commit that intentionally invalidated refs).
|
||||
|
||||
Fail fast if any of these are missing.
|
||||
|
||||
## Steps
|
||||
|
||||
### 1. Validate inputs and confirm intent
|
||||
|
||||
- Verify `test_file` exists and matches `fastvideo/tests/ssim/test_*_similarity.py`.
|
||||
- Grep the file for `*_MODEL_TO_PARAMS` and assert `model_id` is one of its
|
||||
keys. If the file has only a single hardcoded model, accept that model id
|
||||
as the only valid value.
|
||||
- Print the rationale and ask the user to type **`confirm reseed`** (not just
|
||||
`y` — make it deliberate):
|
||||
|
||||
> About to RE-SEED references for model `<model_id>` from test `<test_file>`.
|
||||
> This will OVERWRITE existing refs on
|
||||
> `FastVideo/ssim-reference-videos/reference_videos/default/L40S_reference_videos/<model_id>/`
|
||||
> after backup + Modal regen + eyeball.
|
||||
>
|
||||
> Reason: `<intent_rationale>`
|
||||
> HEAD: `<git rev-parse --short=12 HEAD>`
|
||||
>
|
||||
> Reply `confirm reseed` to proceed, anything else to abort.
|
||||
|
||||
Stop until the user types exactly `confirm reseed`. Anything else aborts
|
||||
with no side effects.
|
||||
|
||||
### 2. Back up existing refs
|
||||
|
||||
Always required. The backup is the only graceful path back if anything goes
|
||||
wrong later.
|
||||
|
||||
```bash
|
||||
SHORT_COMMIT=$(git rev-parse --short=12 HEAD)
|
||||
TIMESTAMP=$(date -u +%Y%m%d_%H%M%S)
|
||||
MODEL_SAFE=$(echo "<model_id>" | tr '/' '_')
|
||||
BACKUP_DIR="ssim_reseed_backup/${TIMESTAMP}_${SHORT_COMMIT}_${MODEL_SAFE}"
|
||||
mkdir -p "$BACKUP_DIR"
|
||||
|
||||
hf download \
|
||||
--repo-type dataset FastVideo/ssim-reference-videos \
|
||||
--include "reference_videos/default/L40S_reference_videos/<model_id>/**" \
|
||||
--local-dir "$BACKUP_DIR"
|
||||
|
||||
mp4_count=$(find "$BACKUP_DIR" -name "*.mp4" | wc -l)
|
||||
echo "Backup mp4 count: $mp4_count"
|
||||
[ "$mp4_count" -gt 0 ] || {
|
||||
echo "ERROR: backup is empty for <model_id>. Either the model id is wrong"
|
||||
echo "or there are no existing refs (use seed-ssim-references instead)."
|
||||
exit 1
|
||||
}
|
||||
|
||||
# Provenance — used in the PR description
|
||||
cat > "$BACKUP_DIR/PROVENANCE.txt" <<EOF
|
||||
test_file: <test_file>
|
||||
model_id: <model_id>
|
||||
head_commit: $(git rev-parse HEAD)
|
||||
timestamp_utc: $(date -u +%FT%TZ)
|
||||
reason: <intent_rationale>
|
||||
EOF
|
||||
```
|
||||
|
||||
If the `hf download` produces zero mp4s, abort — the user has either picked a
|
||||
non-existent `model_id` or there are no refs yet (in which case
|
||||
`seed-ssim-references` is the right tool).
|
||||
|
||||
### 3. Regenerate on Modal L40S
|
||||
|
||||
Mirror CI's exact env recipe so the regenerated refs are byte-comparable to
|
||||
what CI will produce on the same commit. Two differences from CI:
|
||||
|
||||
1. **Pass the same env prefix CI uses** (`IMAGE_VERSION`, `BUILDKITE_*`) — see
|
||||
`.buildkite/pipeline.yml:1-3` and `.buildkite/scripts/pr_test.sh:62-83`.
|
||||
Without this, `ssim_test.py:17-18` resolves a different GHCR image tag
|
||||
(default is `latest`, CI is `py3.12-latest`), and `ssim_test.py:38-46`
|
||||
bakes different values into the image's frozen env block. **Mismatched
|
||||
image or env is the most common source of SSIM drift between reseed and
|
||||
CI runs.**
|
||||
2. **Do not pass `--skip-reference-download`**. Letting the test fetch the
|
||||
existing refs and run the full SSIM compare gives "before" SSIM numbers
|
||||
for the PR description, and the test still produces the new mp4s
|
||||
regardless of whether the comparison passes or fails.
|
||||
|
||||
```bash
|
||||
SUBDIR="${TIMESTAMP}_${SHORT_COMMIT}"
|
||||
|
||||
IMAGE_VERSION="py3.12-latest" \
|
||||
BUILDKITE_REPO="$(git config --get remote.origin.url)" \
|
||||
BUILDKITE_COMMIT="$(git rev-parse HEAD)" \
|
||||
BUILDKITE_PULL_REQUEST="${BUILDKITE_PULL_REQUEST:-false}" \
|
||||
modal run fastvideo/tests/modal/ssim_test.py \
|
||||
--git-repo="$(git config --get remote.origin.url)" \
|
||||
--git-commit="$(git rev-parse HEAD)" \
|
||||
--hf-api-key="$HF_API_KEY" \
|
||||
--test-files="<test_file>" \
|
||||
--sync-generated-to-volume \
|
||||
--generated-volume-subdir="$SUBDIR" \
|
||||
--no-fail-fast
|
||||
```
|
||||
|
||||
Capture the printed `modal volume get ...` hint — its `<SUBDIR>` matches
|
||||
`$SUBDIR` and is needed for step 4. Capture the SSIM numbers from the test
|
||||
output (or from the JSON next to the generated mp4) for the PR description.
|
||||
|
||||
### 4. Download generated videos
|
||||
|
||||
```bash
|
||||
modal volume get --force hf-model-weights \
|
||||
ssim_generated_videos/default/"$SUBDIR"/generated_videos \
|
||||
./generated_videos_modal/default
|
||||
```
|
||||
|
||||
After this, the new mp4s live at:
|
||||
|
||||
```
|
||||
./generated_videos_modal/default/generated_videos/L40S_reference_videos/<model_id>/<backend>/<prompt>.mp4
|
||||
```
|
||||
|
||||
`--force` is required when `./generated_videos_modal/default` already exists
|
||||
from a prior run; safe on the first run too.
|
||||
|
||||
### 5. PAUSE — user reviews quality side-by-side
|
||||
|
||||
Print the diff and the comparison:
|
||||
|
||||
```bash
|
||||
echo "=== File list diff (backup vs new) ==="
|
||||
diff -u \
|
||||
<(find "$BACKUP_DIR/reference_videos/default/L40S_reference_videos/<model_id>" -name "*.mp4" \
|
||||
| sed "s|$BACKUP_DIR/reference_videos/default/L40S_reference_videos/||" | sort) \
|
||||
<(find ./generated_videos_modal/default/generated_videos/L40S_reference_videos/<model_id> -name "*.mp4" \
|
||||
| sed "s|./generated_videos_modal/default/generated_videos/L40S_reference_videos/||" | sort) \
|
||||
|| true
|
||||
|
||||
echo
|
||||
echo "=== SSIM numbers from this run (paste into PR) ==="
|
||||
find ./generated_videos_modal/default/generated_videos/L40S_reference_videos/<model_id> -name "*_ssim.json" -exec cat {} \;
|
||||
```
|
||||
|
||||
Then stop and tell the user:
|
||||
|
||||
> Old refs backed up to `$BACKUP_DIR`.
|
||||
> New videos in `./generated_videos_modal/default/generated_videos/L40S_reference_videos/<model_id>/`.
|
||||
>
|
||||
> Open both in a video player. Confirm the new videos:
|
||||
> 1. Look correct (no obvious artifacts, no black/static frames).
|
||||
> 2. Are *intentionally* different from the backup in the way described
|
||||
> in `<intent_rationale>` (e.g. slight numerical drift only, not a
|
||||
> different scene / different motion / corrupted output).
|
||||
>
|
||||
> Reply **`upload`** to overwrite HF, anything else to abort.
|
||||
> Aborting leaves the backup and new videos on disk for inspection — nothing
|
||||
> on HF changes.
|
||||
|
||||
Do not proceed until the user types exactly `upload`. If they abort, leave
|
||||
everything on disk and stop here.
|
||||
|
||||
### 6. Copy into the local reference layout
|
||||
|
||||
Same as `seed-ssim-references` step 5:
|
||||
|
||||
```bash
|
||||
python fastvideo/tests/ssim/reference_videos_cli.py copy-local \
|
||||
--quality-tier default \
|
||||
--device-folder L40S_reference_videos \
|
||||
--generated-dir ./generated_videos_modal/default/generated_videos/L40S_reference_videos
|
||||
```
|
||||
|
||||
Result: `fastvideo/tests/ssim/reference_videos/default/L40S_reference_videos/<model_id>/<backend>/<prompt>.mp4`.
|
||||
|
||||
### 7. Upload with `--force`, scoped to `--model-id`
|
||||
|
||||
The `--force` flag is what makes this skill different from `seed-ssim-references`.
|
||||
Always pair it with `--model-id` so a typo cannot accidentally overwrite a
|
||||
neighboring model's refs.
|
||||
|
||||
```bash
|
||||
python fastvideo/tests/ssim/reference_videos_cli.py upload \
|
||||
--quality-tier default \
|
||||
--device-folder L40S_reference_videos \
|
||||
--model-id "<model_id>" \
|
||||
--force
|
||||
```
|
||||
|
||||
The CLI's overwrite guard refuses without `--force`; with `--force` it
|
||||
overwrites only files under
|
||||
`reference_videos/default/L40S_reference_videos/<model_id>/`.
|
||||
|
||||
### 8. Report success and retention guidance
|
||||
|
||||
Print:
|
||||
|
||||
- The HF path that was overwritten (`<repo>/reference_videos/default/L40S_reference_videos/<model_id>/`).
|
||||
- The local backup directory path.
|
||||
- The new SSIM numbers from step 5.
|
||||
- This restore command, in case the PR review surfaces a problem after
|
||||
upload:
|
||||
|
||||
```bash
|
||||
python fastvideo/tests/ssim/reference_videos_cli.py upload \
|
||||
--quality-tier default \
|
||||
--device-folder L40S_reference_videos \
|
||||
--model-id "<model_id>" \
|
||||
--reference-dir "$BACKUP_DIR/reference_videos/default/L40S_reference_videos" \
|
||||
--force
|
||||
```
|
||||
|
||||
- This PR-description checklist (see `fastvideo/tests/ssim/AGENTS.md` →
|
||||
*Updating Reference Videos*):
|
||||
1. Source commit that produced the new refs (HEAD at re-seed time).
|
||||
2. Test command and GPU SKU (`L40S`).
|
||||
3. Before/after SSIM numbers.
|
||||
4. The `<intent_rationale>` from step 1.
|
||||
5. A note that the backup lives at `$BACKUP_DIR` and should be retained
|
||||
until CI on the PR is green.
|
||||
|
||||
Do **not** auto-rerun the SSIM test — the user does that as part of the PR.
|
||||
|
||||
## Failure modes and how to handle them
|
||||
|
||||
- **`HF_API_KEY` unset.** Stop before step 2.
|
||||
- **Backup is empty (zero mp4s).** Stop before step 3 — the model id is
|
||||
wrong or the refs don't exist yet (use `seed-ssim-references`).
|
||||
- **Modal run fails before generation.** No mp4s on the volume. Don't
|
||||
upload. Investigate the failure (test crash, OOM, partition exhaustion),
|
||||
fix, then retry from step 3. Backup is still intact.
|
||||
- **Quality regressed (visual or metric).** User aborts at step 5. Backup
|
||||
retained. New videos retained on disk for inspection. Nothing on HF
|
||||
changed. Either fix the underlying code change or abandon the re-seed.
|
||||
- **User confirmed `upload` but later realized the new refs are wrong.**
|
||||
Run the restore command from step 8 with the backup `--reference-dir`.
|
||||
This is exactly why the backup exists.
|
||||
- **Multi-model test, only one model is being re-seeded.** Run the skill
|
||||
once per model id. The `--model-id` scope on upload guarantees the others
|
||||
are untouched.
|
||||
|
||||
## Design notes (for future skill maintainers)
|
||||
|
||||
- Per-`model_id` scope is mandatory. The dataset houses many model subtrees;
|
||||
re-seeding the wrong one is hard to undo without backup.
|
||||
- `default` tier only; `full_quality` is a separate, deliberate operation
|
||||
with different params and ~doubled runtime, and isn't what CI gates on.
|
||||
- The skill deliberately does **not** pass `--skip-reference-download` to
|
||||
Modal so we get pre-reseed SSIM numbers for the PR. The `seed`-skill
|
||||
passes it because no refs exist yet; for re-seed, refs do exist and
|
||||
exposing the comparison is informative.
|
||||
- The two-token confirm (`confirm reseed`, then `upload`) is intentional.
|
||||
Re-seeding is high-blast-radius and should not be one-keystroke.
|
||||
- The backup directory is plain mp4s + `PROVENANCE.txt`. No HF metadata is
|
||||
preserved; the restore path uses `reference_videos_cli.py upload
|
||||
--reference-dir` which doesn't need it.
|
||||
|
||||
## References
|
||||
|
||||
- `.agents/skills/seed-ssim-references/SKILL.md` — the first-time seed
|
||||
skill this one parallels. Read it for the Modal flag rationale shared
|
||||
between the two flows.
|
||||
- `fastvideo/tests/ssim/AGENTS.md` — directory rules, including the PR
|
||||
expectations for any reference-video change (rationale, before/after
|
||||
SSIM, source commit/model/backend).
|
||||
- `fastvideo/tests/ssim/reference_videos_cli.py` — `copy-local`, `upload`
|
||||
(with `--model-id`, `--force`), `download`. The overwrite guard at
|
||||
`upload_reference_videos` is the safety net this skill leans on.
|
||||
- `fastvideo/tests/modal/ssim_test.py` — Modal orchestrator;
|
||||
`--sync-generated-to-volume`, `--generated-volume-subdir`,
|
||||
`--skip-reference-download`, `--no-fail-fast`.
|
||||
|
||||
## Changelog
|
||||
|
||||
| Date | Change |
|
||||
|------|--------|
|
||||
| 2026-05-02 | Initial version. Sister skill to `seed-ssim-references`, scoped to single `(test_file, model_id)` re-seeds, with mandatory backup and two-token confirm. |
|
||||
@@ -0,0 +1,380 @@
|
||||
---
|
||||
name: seed-ssim-references
|
||||
description: Seed HF reference artefacts for a single newly-added SSIM test (pixel `.mp4` for `run_text_to_video_similarity_test`-style tests, or latent `.pt` for `run_text_to_latent_similarity_test`-style tests). Runs the test on Modal L40S, downloads the generated artefacts via `modal volume get`, pauses for the user to verify (visual eyeball for mp4, numerics dump for pt), then uploads only that test's files to `FastVideo/ssim-reference-videos`. Use when a new `fastvideo/tests/ssim/test_*_similarity.py` has just been added and has no references on HF yet.
|
||||
---
|
||||
|
||||
# Seed SSIM Reference Artefacts (mp4 or pt)
|
||||
|
||||
## Purpose
|
||||
|
||||
A brand-new SSIM test in `fastvideo/tests/ssim/` fails forever until its
|
||||
reference artefacts exist on the HF dataset
|
||||
(`FastVideo/ssim-reference-videos`). The dataset hosts two kinds of artefacts
|
||||
side-by-side per `(model_id, backend, prompt)`:
|
||||
|
||||
- **`.mp4`** — pixel ground-truth for tests that call
|
||||
`run_text_to_video_similarity_test` / `run_image_to_video_similarity_test`
|
||||
in `inference_similarity_utils.py`. Compared via SSIM.
|
||||
- **`.pt`** — pre-VAE latent bundle (fp16 full latent + fp32 slice +
|
||||
metadata + `slice_spec` + `format_version`) for tests that call
|
||||
`run_text_to_latent_similarity_test` in `latent_similarity_utils.py`.
|
||||
Compared via cosine distance on the slice and the full tensor.
|
||||
|
||||
This skill:
|
||||
|
||||
1. Detects which artefact type the test produces (pixel vs latent).
|
||||
2. Runs the test on Modal's L40S pool to generate the artefacts.
|
||||
3. Downloads them to the local repo via `modal volume get`.
|
||||
4. Pauses so the user can verify quality:
|
||||
- **mp4**: visual eyeball in a video player.
|
||||
- **pt**: numerics dump (shape, slice stats, NaN/Inf check, metadata).
|
||||
5. Uploads only the new test's files to HF, with a guard that refuses to
|
||||
overwrite anything already present.
|
||||
|
||||
The skill is run **manually**, once per new test. Before invoking it, the user
|
||||
has already sanity-tested the new test locally — it launches `VideoGenerator`
|
||||
and writes an artefact without crashing (the missing-reference assertion at
|
||||
the end is expected). The skill does not re-test locally; it goes straight
|
||||
to the manual legacy Modal L40S reference-maintenance target. Active CI runs
|
||||
on the Slinky Slurm cluster and only consumes the resulting references.
|
||||
|
||||
## When to use
|
||||
|
||||
- A new `test_*_similarity.py` file has been added in `fastvideo/tests/ssim/`
|
||||
and the HF dataset has no `reference_videos/default/L40S_reference_videos/<model_id>/`
|
||||
subtree for it yet.
|
||||
|
||||
## When not to use
|
||||
|
||||
- Regular CI runs — once refs exist, `pytest fastvideo/tests/ssim/` downloads
|
||||
them automatically.
|
||||
- Re-seeding an existing test. That requires `--force` on the upload step, and
|
||||
is out of scope here; treat as a separate, deliberate operation.
|
||||
|
||||
## Inputs
|
||||
|
||||
The skill has **one required input**: the path to the new SSIM test file.
|
||||
Prompt the user for it if they didn't supply it.
|
||||
|
||||
| Parameter | Required | Description |
|
||||
|-----------|----------|-------------|
|
||||
| `test_file` | Yes | e.g. `fastvideo/tests/ssim/test_ltx2_similarity.py`. The skill's first action is to ask for this if missing. |
|
||||
|
||||
Everything else is fixed:
|
||||
|
||||
- Modal maintenance GPU: **L40S** (hardcoded in
|
||||
`fastvideo/tests/modal/ssim_test.py`; this is not the active CI compute path).
|
||||
- Device folder: `L40S_reference_videos`.
|
||||
- Quality tier: `default` (the tier CI runs). The `full_quality` tier is not
|
||||
seeded by this skill.
|
||||
- HF repo: `FastVideo/ssim-reference-videos` (dataset).
|
||||
- Multi-model test files: all model ids in `*_MODEL_TO_PARAMS` are seeded
|
||||
together; the Modal run produces one mp4 per (model, prompt, backend) and
|
||||
the upload scopes by `--model-id`, looping if there is more than one.
|
||||
|
||||
## Prerequisites
|
||||
|
||||
The user has confirmed:
|
||||
|
||||
- `modal` CLI authenticated.
|
||||
- `HF_API_KEY` (or `HUGGINGFACE_HUB_TOKEN` / `HF_TOKEN`) exported with write
|
||||
access to `FastVideo/ssim-reference-videos`.
|
||||
- The test file runs locally end-to-end (generates an mp4; SSIM assertion
|
||||
failure due to missing reference is expected and fine).
|
||||
|
||||
Fail fast if the token env var is missing.
|
||||
|
||||
## Steps
|
||||
|
||||
### 1. Ask for the test file, then detect artefact type
|
||||
|
||||
If the user didn't name one, ask: *"Which SSIM test file do you want to seed
|
||||
references for? (e.g. `fastvideo/tests/ssim/test_ltx2_similarity.py`)"*.
|
||||
|
||||
Validate:
|
||||
|
||||
- Path exists and matches `fastvideo/tests/ssim/test_*_similarity.py`.
|
||||
- File defines a `*_MODEL_TO_PARAMS` dict — grep it to extract the set of
|
||||
model ids. Those ids drive step 5.
|
||||
|
||||
Detect artefact type by inspecting the file's imports / helper call:
|
||||
|
||||
- **latent** (`.pt`) — file imports `run_text_to_latent_similarity_test`
|
||||
from `fastvideo.tests.ssim.latent_similarity_utils` (or any other helper
|
||||
that ends with `_latent_similarity_test`).
|
||||
- **pixel** (`.mp4`) — file imports
|
||||
`run_text_to_video_similarity_test` / `run_image_to_video_similarity_test`
|
||||
from `fastvideo.tests.ssim.inference_similarity_utils`, OR uses the
|
||||
legacy custom-inline helper pattern (see `test_gamecraft`,
|
||||
`test_longcat`, etc.). Default to pixel when both heuristics fail.
|
||||
|
||||
Record `ARTEFACT_TYPE ∈ {pixel, latent}` for use in step 4. Steps 2, 3, 5,
|
||||
and 6 are artefact-type-agnostic — `_iter_reference_files`,
|
||||
`copy_generated_to_reference`, and `upload_reference_videos` already walk
|
||||
both `.mp4` and `.pt` (see `reference_videos_cli.py`).
|
||||
|
||||
If either check fails, stop and tell the user what's wrong.
|
||||
|
||||
### 2. Run the test on Modal L40S
|
||||
|
||||
Pick a subdir name so repeated runs don't collide:
|
||||
|
||||
```bash
|
||||
SHORT_COMMIT=$(git rev-parse --short=12 HEAD)
|
||||
TIMESTAMP=$(date -u +%Y%m%d_%H%M%S)
|
||||
SUBDIR="${TIMESTAMP}_${SHORT_COMMIT}"
|
||||
```
|
||||
|
||||
Then launch the Modal run. The `IMAGE_VERSION` and `BUILDKITE_*` env-prefix
|
||||
**must** match what CI exports in `.buildkite/scripts/pr_test.sh`, otherwise
|
||||
`fastvideo/tests/modal/ssim_test.py` resolves a different GHCR image tag
|
||||
(default is `latest`, CI is `py3.12-latest`) and bakes different values into
|
||||
the image's frozen env block (`ssim_test.py:17-18, 38-46`). Mismatched image
|
||||
or env produces SSIM drift that doesn't show up until the same commit runs
|
||||
in CI.
|
||||
|
||||
```bash
|
||||
IMAGE_VERSION="py3.12-latest" \
|
||||
BUILDKITE_REPO="$(git config --get remote.origin.url)" \
|
||||
BUILDKITE_COMMIT="$(git rev-parse HEAD)" \
|
||||
BUILDKITE_PULL_REQUEST="${BUILDKITE_PULL_REQUEST:-false}" \
|
||||
modal run fastvideo/tests/modal/ssim_test.py \
|
||||
--git-repo="$(git config --get remote.origin.url)" \
|
||||
--git-commit="$(git rev-parse HEAD)" \
|
||||
--hf-api-key="$HF_API_KEY" \
|
||||
--test-files="<test_file>" \
|
||||
--sync-generated-to-volume \
|
||||
--generated-volume-subdir="$SUBDIR" \
|
||||
--skip-reference-download \
|
||||
--no-fail-fast
|
||||
```
|
||||
|
||||
Env prefix rationale (parity with CI; see `.buildkite/pipeline.yml:1-3` and
|
||||
`.buildkite/scripts/pr_test.sh:62-83`):
|
||||
- `IMAGE_VERSION=py3.12-latest`: pins the Modal image tag to the same one CI
|
||||
uses. The published `py3.12-latest` and `latest` tags point at Python 3.12 /
|
||||
CUDA 12.6.3 / cu126; `py3.12-cuda12.6.3-latest` is the explicit alias for the
|
||||
same image. CUDA 13 / cu130 is available under the explicit
|
||||
`py3.12-cuda13.0.0-latest` tag. This tag policy comes from
|
||||
`infra-build-image.yml`; the unparameterized `docker/Dockerfile` build itself
|
||||
still defaults to CUDA 13 / cu130.
|
||||
- `BUILDKITE_REPO`/`BUILDKITE_COMMIT`/`BUILDKITE_PULL_REQUEST`: mirror what
|
||||
Buildkite exports. `ssim_test.py:38-46` bakes these into the image's
|
||||
`.env(...)` block; mismatched values can perturb in-container code paths
|
||||
that branch on PR-vs-non-PR. `false` for `BUILDKITE_PULL_REQUEST` matches
|
||||
Buildkite's "non-PR build" sentinel.
|
||||
|
||||
Flag rationale:
|
||||
- `--skip-reference-download`: no refs exist yet, so conftest must not try to
|
||||
pull them.
|
||||
- `--no-fail-fast`: lets the test finish generation before `_assert_similarity`
|
||||
raises `FileNotFoundError: Reference video folder does not exist`. The
|
||||
expected failure is what we want — the mp4 has already been written.
|
||||
- `--sync-generated-to-volume` + `--generated-volume-subdir`: copies the
|
||||
generated mp4s to the `hf-model-weights` Modal volume under
|
||||
`ssim_generated_videos/default/<SUBDIR>/generated_videos/` so we can pull
|
||||
them locally.
|
||||
|
||||
The Modal run will end with a nonzero exit (expected) and print a
|
||||
`modal volume get hf-model-weights ssim_generated_videos/default/<SUBDIR>/generated_videos ./generated_videos_modal/default`
|
||||
command. Capture that `<SUBDIR>` — you need it for step 3.
|
||||
|
||||
### 3. Download generated videos locally
|
||||
|
||||
```bash
|
||||
modal volume get --force hf-model-weights \
|
||||
ssim_generated_videos/default/"$SUBDIR"/generated_videos \
|
||||
./generated_videos_modal/default
|
||||
```
|
||||
|
||||
`--force` is required when the parent `./generated_videos_modal/default`
|
||||
already exists; without it, `modal volume get` errors with `[Errno 21] Is a
|
||||
directory`. Safe to pass on the first run too.
|
||||
|
||||
After this, the mp4s live at
|
||||
`./generated_videos_modal/default/generated_videos/L40S_reference_videos/<model_id>/<backend>/<prompt>.mp4`.
|
||||
The extra `generated_videos/` level comes from the volume layout in
|
||||
`_sync_generated_videos_to_volume` (`ssim_test.py`) — the command copies
|
||||
`<repo>/fastvideo/tests/ssim/generated_videos/<tier>` to
|
||||
`ssim_generated_videos/<tier>/<SUBDIR>/generated_videos/`, and `modal volume
|
||||
get` preserves that trailing `generated_videos/` segment.
|
||||
|
||||
### 4. PAUSE — user reviews quality
|
||||
|
||||
Type-aware verification.
|
||||
|
||||
**For `ARTEFACT_TYPE = pixel`** — list the downloaded mp4s and ask the user to
|
||||
open them in a video player:
|
||||
|
||||
> "Generated videos downloaded to `./generated_videos_modal/default/generated_videos/L40S_reference_videos/`. Please open them and confirm the quality looks correct. Reply **`upload`** to continue, or anything else to abort."
|
||||
|
||||
**For `ARTEFACT_TYPE = latent`** — `.pt` files are not human-watchable. Print
|
||||
a numerics dump for each `.pt` so the user can sanity-check shape, distribution,
|
||||
and metadata:
|
||||
|
||||
```python
|
||||
import torch
|
||||
from pathlib import Path
|
||||
ROOT = Path("./generated_videos_modal/default/generated_videos/L40S_reference_videos")
|
||||
for p in sorted(ROOT.rglob("*.pt")):
|
||||
d = torch.load(p, map_location="cpu", weights_only=False)
|
||||
s = d["expected_slice"]
|
||||
L = d["latent"].float()
|
||||
print(f"=== {p.relative_to(ROOT)} ===")
|
||||
print(f" format_version: {d['format_version']}")
|
||||
print(f" shape: {d['shape']}")
|
||||
print(f" dtype_original: {d['dtype_original']}")
|
||||
print(f" slice_spec: {d['slice_spec']}")
|
||||
print(f" slice shape={tuple(s.shape)} mean={s.mean():+.4f} std={s.std():.4f} min={s.min():+.4f} max={s.max():+.4f}")
|
||||
print(f" latent shape={tuple(L.shape)} mean={L.mean():+.4f} std={L.std():.4f} min={L.min():+.4f} max={L.max():+.4f}")
|
||||
print(f" finite: latent NaN={torch.isnan(L).any().item()} Inf={torch.isinf(L).any().item()}; "
|
||||
f"slice NaN={torch.isnan(s).any().item()} Inf={torch.isinf(s).any().item()}")
|
||||
print(f" metadata: {d['metadata']}\n")
|
||||
```
|
||||
|
||||
Sanity criteria:
|
||||
- `format_version == 1` (matches `LATENT_REFERENCE_FORMAT_VERSION`).
|
||||
- `shape` matches what the model produces (e.g. LTX-2 distilled =
|
||||
`[1, 128, T_lat, H_lat, W_lat]`; Stable Audio Open 1.0 = `[1, 64, 1024]`).
|
||||
- `slice_spec.kind` matches a registered kind (`corner_3x3_first_frame`
|
||||
for video, `audio_first_8_timesteps` for audio).
|
||||
- No `NaN`/`Inf`. `mean ≈ 0`, `std ≈ 1` (denoised latents stay close to
|
||||
the initial Gaussian distribution; very wide deviations suggest
|
||||
numerical drift).
|
||||
- `metadata.prompt` matches the test's prompt.
|
||||
|
||||
Then ask:
|
||||
|
||||
> "Numerics look right? Reply **`upload`** to continue, or anything else to abort."
|
||||
|
||||
Do not proceed until the user explicitly says `upload`. If they abort, leave
|
||||
everything on disk so they can inspect further — no cleanup.
|
||||
|
||||
### 5. Copy into the local reference layout
|
||||
|
||||
Scoped copy — only the new test's artefacts. Single command works for both
|
||||
artefact types because `_iter_reference_files` walks `.mp4` and `.pt`:
|
||||
|
||||
```bash
|
||||
python fastvideo/tests/ssim/reference_videos_cli.py copy-local \
|
||||
--quality-tier default \
|
||||
--device-folder L40S_reference_videos \
|
||||
--generated-dir ./generated_videos_modal/default/generated_videos/L40S_reference_videos
|
||||
```
|
||||
|
||||
(The `--generated-dir` points at the device-folder root inside the
|
||||
downloaded tree; `copy-local` walks all `<model>/<backend>/*.{mp4,pt}`
|
||||
underneath it. Since the Modal run was scoped to a single test file via
|
||||
`--test-files`, only that test's model(s) are present — so the copy is
|
||||
implicitly per-test.)
|
||||
|
||||
Result for pixel: `fastvideo/tests/ssim/reference_videos/default/L40S_reference_videos/<model_id>/<backend>/<prompt>.mp4`.
|
||||
Result for latent: same path with `.pt` extension.
|
||||
|
||||
### 6. Upload to HF — scoped per model_id, with overwrite guard
|
||||
|
||||
For each `<model_id>`:
|
||||
|
||||
```bash
|
||||
python fastvideo/tests/ssim/reference_videos_cli.py upload \
|
||||
--quality-tier default \
|
||||
--device-folder L40S_reference_videos \
|
||||
--model-id "<model_id>"
|
||||
```
|
||||
|
||||
The upload command:
|
||||
|
||||
- Uploads **only** `reference_videos/default/L40S_reference_videos/<model_id>/`.
|
||||
- **Refuses** if any file already exists at that path on HF (this is the
|
||||
guard — seeding a new test should never clobber existing refs). To override,
|
||||
the user must re-run with `--force`. If the guard fires, stop and report
|
||||
exactly which files exist; do not silently `--force`.
|
||||
|
||||
Reads the HF token from `HF_API_KEY` / `HUGGINGFACE_HUB_TOKEN` / `HF_TOKEN`.
|
||||
|
||||
### 7. Report success
|
||||
|
||||
List what was uploaded (paths in repo) and remind the user to push any
|
||||
related code changes. Do **not** auto-verify by re-running Modal — the user
|
||||
can run `pytest fastvideo/tests/ssim/<test_file>` later to confirm end-to-end;
|
||||
it will auto-download the refs they just uploaded.
|
||||
|
||||
## Failure modes and how to handle them
|
||||
|
||||
- **`HF_API_KEY` unset.** Stop before step 2. The Modal run needs it (passed
|
||||
via `--hf-api-key`), and step 6 needs it for upload. If the user
|
||||
ran `hf auth login` instead of exporting an env var, read the cached
|
||||
token via `huggingface_hub.get_token()` and forward it to Modal as
|
||||
`--hf-api-key="$CACHED_TOKEN"`.
|
||||
- **Modal run fails before generation.** No artefacts on the volume — nothing
|
||||
to download. Fix the test locally (`pytest fastvideo/tests/ssim/<test_file>`)
|
||||
and retry from step 2.
|
||||
- **`./generated_videos_modal/default/L40S_reference_videos/` missing after
|
||||
`modal volume get`.** The run didn't produce artefacts (most likely the
|
||||
test crashed before writing, or `REQUIRED_GPUS` exceeded the partition
|
||||
capacity — see Modal logs).
|
||||
- **Latent test crashed with FSDP / inference_mode error
|
||||
(`RuntimeError: Inference tensors do not track version counter`).** The
|
||||
test must pass `init_kwargs_override={"use_fsdp_inference": False}` when
|
||||
`sp_size == 1` — see `test_stable_audio_similarity.py` for the pattern.
|
||||
Fix in the test, push, retry.
|
||||
- **Upload guard fires (files already exist).** The test name / model id
|
||||
collides with something already on HF. Verify the user actually wants to
|
||||
replace existing refs; if so, re-run the upload with `--force`. If not,
|
||||
rename the model id in `*_MODEL_TO_PARAMS` and re-seed.
|
||||
- **Quality looks wrong in step 4.** Abort. The artefacts stay on disk for
|
||||
inspection. The fix is usually in the test's params (resolution, steps,
|
||||
seed) — edit the test, then re-run the skill.
|
||||
- For latent: also check `slice_spec.kind` matches the latent rank
|
||||
(`corner_3x3_first_frame` requires 5-D, `audio_first_8_timesteps`
|
||||
requires 3-D); a rank/kind mismatch raises in `_extract_expected_slice`.
|
||||
|
||||
## Design notes (for future skill maintainers)
|
||||
|
||||
- The skill deliberately runs on Modal, **not** locally, because the CI
|
||||
runner is L40S. Seeding from a different GPU SKU produces refs that CI's
|
||||
L40S runs can't match (pixel SSIM drifts across SKUs; latent cosine has
|
||||
tighter cross-SKU bf16 drift but the configured tolerances assume
|
||||
same-SKU seed → same-SKU verify).
|
||||
- The skill is default-tier only. `full_quality` refs are seeded by a
|
||||
separate, deliberate operation — they double runtime and aren't what CI
|
||||
gates on.
|
||||
- The overwrite guard in `reference_videos_cli.py upload` is default-on
|
||||
specifically because this skill exists. Re-seeding is a distinct operation
|
||||
that requires explicit `--force`.
|
||||
- Both artefact types share the same Modal flow: the orchestrator sets
|
||||
`--skip-reference-download` + `--no-fail-fast`, runs pytest, the test's
|
||||
helper writes the artefact (`.mp4` via `imageio` for pixel,
|
||||
`save_latent_reference` → `torch.save` for latent) BEFORE the
|
||||
missing-reference assertion raises. `_sync_generated_videos_to_volume` in
|
||||
`ssim_test.py` does a `shutil.copytree` of the whole `generated_videos/`
|
||||
tree, picking up `.mp4`, `.pt`, and the `*_ssim.json` / `*_latent.json`
|
||||
metric files alongside.
|
||||
|
||||
## References
|
||||
|
||||
- `fastvideo/tests/modal/ssim_test.py` — Modal orchestrator; see
|
||||
`--sync-generated-to-volume`, `--generated-volume-subdir`,
|
||||
`--skip-reference-download`, `--no-fail-fast`.
|
||||
- `fastvideo/tests/ssim/reference_videos_cli.py` — `copy-local`, `upload`
|
||||
(with `--model-id`, `--force`), `download`, `ensure` subcommands.
|
||||
Extension allowlist is `REFERENCE_EXTENSIONS = VIDEO_EXTENSIONS +
|
||||
LATENT_EXTENSIONS` (`.pt`).
|
||||
- `fastvideo/tests/ssim/README.md` — reference layout, HF repo conventions.
|
||||
- `fastvideo/tests/ssim/inference_similarity_utils.py` — pixel helpers
|
||||
(`run_text_to_video_similarity_test`,
|
||||
`run_image_to_video_similarity_test`, `build_init_kwargs`).
|
||||
- `fastvideo/tests/ssim/latent_similarity_utils.py` — latent helper
|
||||
(`run_text_to_latent_similarity_test`), slice spec dispatch
|
||||
(`_extract_expected_slice`), reference schema
|
||||
(`save_latent_reference` / `load_latent_reference`),
|
||||
`LATENT_REFERENCE_FORMAT_VERSION`.
|
||||
|
||||
## Changelog
|
||||
|
||||
| Date | Change |
|
||||
|------|--------|
|
||||
| 2026-04-17 | Initial version (Modal sync-to-volume flow). |
|
||||
| 2026-04-21 | Rewrite: single-test scope, explicit user-review pause, per-`model_id` upload, HF overwrite guard. Dropped `scripts/seed_ssim.sh`. |
|
||||
| 2026-04-21 | Post-first-run fixes: `modal volume get` needs `--force` when parent exists; download tree has an extra `generated_videos/` level so `--generated-dir` must reflect it. |
|
||||
| 2026-05-01 | Latent (`*.pt`) artefact support: artefact-type detection in step 1, type-aware verification (visual eyeball for mp4, numerics dump for pt) in step 4, FSDP+inference_mode failure-mode added, design notes for the unified Modal flow. Triggered by PR #1253 (LTX-2 latent migration + Stable Audio latent test). |
|
||||
@@ -0,0 +1,51 @@
|
||||
{
|
||||
"benchmark_id": "wan-t2v-1.3b-1gpu-gb10",
|
||||
"config_schema_version": 2,
|
||||
"workload_id": "wan-t2v",
|
||||
"variant_id": "1.3b-sp1",
|
||||
"benchmark_version": 3,
|
||||
"description": "Wan2.1 T2V 1.3B single-GPU inference performance on NVIDIA DGX Spark (GB10). Single-GPU variant of wan-t2v-1.3b (same workload_id for dashboard comparability). Gated to the GB10 via run_config.gpu_types so it does not run on the shared H100/L40S lanes.",
|
||||
"model": {
|
||||
"model_path": "Wan-AI/Wan2.1-T2V-1.3B-Diffusers",
|
||||
"model_short_name": "Wan2.1-T2V-1.3B"
|
||||
},
|
||||
"init_kwargs": {
|
||||
"num_gpus": 1,
|
||||
"flow_shift": 7.0,
|
||||
"sp_size": 1,
|
||||
"tp_size": 1,
|
||||
"vae_sp": false,
|
||||
"vae_tiling": true,
|
||||
"text_encoder_precisions": ["fp32"]
|
||||
},
|
||||
"generation_kwargs": {
|
||||
"height": 480,
|
||||
"width": 832,
|
||||
"num_frames": 45,
|
||||
"num_inference_steps": 4,
|
||||
"guidance_scale": 3,
|
||||
"embedded_cfg_scale": 6,
|
||||
"seed": 1024,
|
||||
"fps": 24,
|
||||
"neg_prompt": "Bright tones, overexposed, static, blurred details, subtitles, style, works, paintings, images, static, overall gray, worst quality, low quality, JPEG compression residue, ugly, incomplete, extra fingers, poorly drawn hands, poorly drawn faces, deformed, disfigured, misshapen limbs, fused fingers, still picture, messy background, three legs, many people in the background, walking backwards"
|
||||
},
|
||||
"test_prompts": [
|
||||
"Will Smith casually eats noodles, his relaxed demeanor contrasting with the energetic background of a bustling street food market. The scene captures a mix of humor and authenticity. Mid-shot framing, vibrant lighting."
|
||||
],
|
||||
"run_config": {
|
||||
"num_warmup_runs": 2,
|
||||
"num_measurement_runs": 5,
|
||||
"required_gpus": 1,
|
||||
"gpu_types": ["GB10"]
|
||||
},
|
||||
"thresholds": {
|
||||
"GB10": {
|
||||
"max_generation_time_s": 55.0,
|
||||
"max_peak_memory_mb": 12000.0
|
||||
},
|
||||
"default": {
|
||||
"max_generation_time_s": 120.0,
|
||||
"max_peak_memory_mb": 40000.0
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,53 @@
|
||||
{
|
||||
"benchmark_id": "wan-t2v-1.3b-2gpu",
|
||||
"config_schema_version": 2,
|
||||
"workload_id": "wan-t2v",
|
||||
"variant_id": "1.3b-sp2",
|
||||
"benchmark_version": 3,
|
||||
"description": "Wan2.1 T2V 1.3B inference performance",
|
||||
"model": {
|
||||
"model_path": "Wan-AI/Wan2.1-T2V-1.3B-Diffusers",
|
||||
"model_short_name": "Wan2.1-T2V-1.3B"
|
||||
},
|
||||
"init_kwargs": {
|
||||
"num_gpus": 2,
|
||||
"flow_shift": 7.0,
|
||||
"sp_size": 2,
|
||||
"tp_size": 1,
|
||||
"vae_sp": true,
|
||||
"vae_tiling": true,
|
||||
"text_encoder_precisions": ["fp32"]
|
||||
},
|
||||
"generation_kwargs": {
|
||||
"height": 480,
|
||||
"width": 832,
|
||||
"num_frames": 45,
|
||||
"num_inference_steps": 4,
|
||||
"guidance_scale": 3,
|
||||
"embedded_cfg_scale": 6,
|
||||
"seed": 1024,
|
||||
"fps": 24,
|
||||
"neg_prompt": "Bright tones, overexposed, static, blurred details, subtitles, style, works, paintings, images, static, overall gray, worst quality, low quality, JPEG compression residue, ugly, incomplete, extra fingers, poorly drawn hands, poorly drawn faces, deformed, disfigured, misshapen limbs, fused fingers, still picture, messy background, three legs, many people in the background, walking backwards"
|
||||
},
|
||||
"test_prompts": [
|
||||
"Will Smith casually eats noodles, his relaxed demeanor contrasting with the energetic background of a bustling street food market. The scene captures a mix of humor and authenticity. Mid-shot framing, vibrant lighting."
|
||||
],
|
||||
"run_config": {
|
||||
"num_warmup_runs": 2,
|
||||
"num_measurement_runs": 5,
|
||||
"required_gpus": 2
|
||||
},
|
||||
"thresholds": {
|
||||
"L40S": {
|
||||
"max_generation_time_s": 34.0,
|
||||
"max_peak_memory_mb": 11000.0,
|
||||
"max_text_encoder_time_s": 5.0,
|
||||
"max_dit_time_s": 10.0,
|
||||
"max_vae_decode_time_s": 10.0
|
||||
},
|
||||
"default": {
|
||||
"max_generation_time_s": 120.0,
|
||||
"max_peak_memory_mb": 30000.0
|
||||
}
|
||||
}
|
||||
}
|
||||
+476
-145
@@ -1,153 +1,484 @@
|
||||
env:
|
||||
IMAGE_VERSION: "py3.12-latest"
|
||||
BUILDKITE_CLEAN_CHECKOUT: true
|
||||
# Slurm workers clone the immutable commit and initialize submodules inside
|
||||
# their isolated container. The Buildkite login-plane checkout is a no-op.
|
||||
BUILDKITE_GIT_SUBMODULES: false
|
||||
|
||||
notify:
|
||||
- github_commit_status:
|
||||
context: "fastcheck-passed"
|
||||
if: build.env("TEST_SCOPE") == "fastcheck" || build.env("TEST_SCOPE") == null
|
||||
- github_commit_status:
|
||||
context: "full-suite-passed"
|
||||
if: build.env("TEST_SCOPE") == "full" || build.env("TEST_SCOPE") == "merge"
|
||||
- github_commit_status:
|
||||
context: "direct-test-completed"
|
||||
if: build.env("TEST_SCOPE") == "direct"
|
||||
- github_commit_status:
|
||||
context: "scheduled-ssim-passed"
|
||||
if: build.env("TEST_SCOPE") == "scheduled"
|
||||
|
||||
# This is the complete active GPU CI surface. Every command is a trusted host
|
||||
# dispatcher, and every test payload executes inside the Slinky Slurm tray.
|
||||
# fastvideo/tests/modal remains available only for an explicit manual rollback;
|
||||
# no active pipeline or slash-command route invokes it.
|
||||
# Buildkite hands jobs to free agents in the order they appear here. Golden-gate comes first
|
||||
# because every later merge lane waits for it; the fastcheck lanes follow from longest to
|
||||
# shortest measured runtime, so the longest lane never starts last and stretches the build.
|
||||
steps:
|
||||
- label: "pre-commit"
|
||||
command: ".buildkite/scripts/pre_commit.sh"
|
||||
- label: ":test_tube: Golden-Gate Tests"
|
||||
key: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
build.env("MERGE_TEST_PLAN") =~ /,golden-gate,/) ||
|
||||
(build.env("TEST_SCOPE") == "direct" &&
|
||||
(build.env("TEST_TYPE") == "golden_gate" || build.env("TEST_TYPE") == "golden_gate_ci"))
|
||||
command: "/opt/fastvideo-ci-runner/run-ci"
|
||||
timeout_in_minutes: 90
|
||||
env:
|
||||
TEST_TYPE: "golden_gate_ci"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
queue: "ci-runner"
|
||||
|
||||
- wait
|
||||
- label: ":microscope: Unit Tests"
|
||||
key: "unit"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
build.env("MERGE_TEST_PLAN") =~ /,unit,/) ||
|
||||
build.env("TEST_SCOPE") == "fastcheck" ||
|
||||
build.env("TEST_SCOPE") == null ||
|
||||
(build.env("TEST_SCOPE") == "direct" &&
|
||||
(build.env("TEST_TYPE") == "unit_test" || build.env("TEST_TYPE") == "unit_test_ci"))
|
||||
command: "/opt/fastvideo-ci-runner/run-unit"
|
||||
timeout_in_minutes: 90
|
||||
env:
|
||||
TEST_TYPE: "unit_test_ci"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "ci-runner"
|
||||
|
||||
- label: "Trigger Tests"
|
||||
plugins:
|
||||
- monorepo-diff#v1.4.0:
|
||||
diff: "git diff --name-only $BUILDKITE_PULL_REQUEST_BASE_BRANCH...HEAD"
|
||||
watch:
|
||||
- path:
|
||||
- "fastvideo/v1/models/encoders/**"
|
||||
- "fastvideo/v1/models/loader/**"
|
||||
- "fastvideo/v1/tests/encoders/**"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile.python3.12"
|
||||
config:
|
||||
command: "timeout 15m .buildkite/scripts/pr_test.sh"
|
||||
label: "Encoder Tests"
|
||||
env:
|
||||
- TEST_TYPE=encoder
|
||||
agents:
|
||||
queue: "default"
|
||||
- path:
|
||||
- "fastvideo/v1/models/vaes/**"
|
||||
- "fastvideo/v1/models/loader/**"
|
||||
- "fastvideo/v1/tests/vaes/**"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile.python3.12"
|
||||
config:
|
||||
command: "timeout 15m .buildkite/scripts/pr_test.sh"
|
||||
label: "VAE Tests"
|
||||
env:
|
||||
- TEST_TYPE=vae
|
||||
agents:
|
||||
queue: "default"
|
||||
- path:
|
||||
- "fastvideo/v1/models/dits/**"
|
||||
- "fastvideo/v1/models/loader/**"
|
||||
- "fastvideo/v1/tests/transformers/**"
|
||||
- "fastvideo/v1/layers/**"
|
||||
- "fastvideo/v1/attention/**"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile.python3.12"
|
||||
config:
|
||||
command: "timeout 15m .buildkite/scripts/pr_test.sh"
|
||||
label: "Transformer Tests"
|
||||
env:
|
||||
- TEST_TYPE=transformer
|
||||
agents:
|
||||
queue: "default"
|
||||
- path:
|
||||
- "fastvideo/v1/**/*.py"
|
||||
config:
|
||||
command: "timeout 30m .buildkite/scripts/pr_test.sh"
|
||||
label: "SSIM Tests"
|
||||
env:
|
||||
- TEST_TYPE=ssim
|
||||
agents:
|
||||
queue: "default"
|
||||
- path:
|
||||
- "fastvideo/v1/tests/lora/**"
|
||||
- "fastvideo/v1/models/loader/**"
|
||||
- "fastvideo/v1/tests/transformers/**"
|
||||
- "fastvideo/v1/pipelines/**"
|
||||
- "fastvideo/v1/layers/lora/**"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile.python3.12"
|
||||
config:
|
||||
command: "timeout 15m .buildkite/scripts/pr_test.sh"
|
||||
label: "LoRA Inference Tests"
|
||||
env:
|
||||
- TEST_TYPE=inference_lora
|
||||
agents:
|
||||
queue: "default"
|
||||
- path:
|
||||
- "fastvideo/v1/**"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile.python3.12"
|
||||
config:
|
||||
command: "timeout 15m .buildkite/scripts/pr_test.sh"
|
||||
label: "Training Tests"
|
||||
env:
|
||||
- TEST_TYPE=training
|
||||
agents:
|
||||
queue: "default"
|
||||
- path:
|
||||
- "fastvideo/v1/**"
|
||||
- "csrc/attn/vsa/**"
|
||||
- "csrc/attn/tk/**"
|
||||
- "csrc/attn/setup_vsa.py"
|
||||
- "csrc/attn/config_vsa.py"
|
||||
- "csrc/attn/vsa.cpp"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile.python3.12"
|
||||
config:
|
||||
command: "timeout 15m .buildkite/scripts/pr_test.sh"
|
||||
label: "Training Tests VSA"
|
||||
env:
|
||||
- TEST_TYPE=training_vsa
|
||||
agents:
|
||||
queue: "default"
|
||||
- path:
|
||||
- "fastvideo/v1/**"
|
||||
- "csrc/attn/st_attn/**"
|
||||
- "csrc/attn/setup_sta.py"
|
||||
- "csrc/attn/config_sta.py"
|
||||
- "csrc/attn/st_attn.cpp"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile.python3.12"
|
||||
config:
|
||||
command: "timeout 15m .buildkite/scripts/pr_test.sh"
|
||||
label: "Inference Tests STA"
|
||||
env:
|
||||
- TEST_TYPE=inference_sta
|
||||
agents:
|
||||
queue: "default"
|
||||
- path:
|
||||
- "csrc/attn/st_attn/**"
|
||||
- "csrc/attn/setup_sta.py"
|
||||
- "csrc/attn/config_sta.py"
|
||||
- "csrc/attn/st_attn.cpp"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile.python3.12"
|
||||
config:
|
||||
command: "timeout 15m .buildkite/scripts/pr_test.sh"
|
||||
label: "Precision Tests STA"
|
||||
env:
|
||||
- TEST_TYPE=precision_sta
|
||||
agents:
|
||||
queue: "default"
|
||||
- path:
|
||||
- "csrc/attn/vsa/**"
|
||||
- "csrc/attn/tk/**"
|
||||
- "csrc/attn/setup_vsa.py"
|
||||
- "csrc/attn/config_vsa.py"
|
||||
- "csrc/attn/vsa.cpp"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile.python3.12"
|
||||
config:
|
||||
command: "timeout 15m .buildkite/scripts/pr_test.sh"
|
||||
label: "Precision Tests VSA"
|
||||
env:
|
||||
- TEST_TYPE=precision_vsa
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":microscope: Kernel Tests"
|
||||
key: "kernel-tests"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
build.env("MERGE_TEST_PLAN") =~ /,kernel-tests,/) ||
|
||||
build.env("TEST_SCOPE") == "fastcheck" ||
|
||||
build.env("TEST_SCOPE") == null ||
|
||||
(build.env("TEST_SCOPE") == "direct" &&
|
||||
(build.env("TEST_TYPE") == "kernel_tests" || build.env("TEST_TYPE") == "kernel_tests_ci"))
|
||||
command: "/opt/fastvideo-ci-runner/run-ci"
|
||||
timeout_in_minutes: 90
|
||||
env:
|
||||
TEST_TYPE: "kernel_tests_ci"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "ci-runner"
|
||||
|
||||
- label: ":microscope: DreamVerse App Tests"
|
||||
key: "dreamverse"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
build.env("MERGE_TEST_PLAN") =~ /,dreamverse,/) ||
|
||||
build.env("TEST_SCOPE") == "fastcheck" ||
|
||||
build.env("TEST_SCOPE") == null ||
|
||||
(build.env("TEST_SCOPE") == "direct" &&
|
||||
(build.env("TEST_TYPE") == "dreamverse_app" || build.env("TEST_TYPE") == "dreamverse_app_ci"))
|
||||
command: "/opt/fastvideo-ci-runner/run-ci"
|
||||
timeout_in_minutes: 90
|
||||
env:
|
||||
TEST_TYPE: "dreamverse_app_ci"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "ci-runner"
|
||||
|
||||
- label: ":microscope: Encoder Tests"
|
||||
key: "encoder"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
build.env("MERGE_TEST_PLAN") =~ /,encoder,/) ||
|
||||
build.env("TEST_SCOPE") == "fastcheck" ||
|
||||
build.env("TEST_SCOPE") == null ||
|
||||
(build.env("TEST_SCOPE") == "direct" &&
|
||||
(build.env("TEST_TYPE") == "encoder" || build.env("TEST_TYPE") == "encoder_ci"))
|
||||
command: "/opt/fastvideo-ci-runner/run-ci"
|
||||
timeout_in_minutes: 90
|
||||
env:
|
||||
TEST_TYPE: "encoder_ci"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "ci-runner"
|
||||
|
||||
- label: ":microscope: VAE Tests"
|
||||
key: "vae"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
build.env("MERGE_TEST_PLAN") =~ /,vae,/) ||
|
||||
build.env("TEST_SCOPE") == "fastcheck" ||
|
||||
build.env("TEST_SCOPE") == null ||
|
||||
(build.env("TEST_SCOPE") == "direct" &&
|
||||
(build.env("TEST_TYPE") == "vae" || build.env("TEST_TYPE") == "vae_ci"))
|
||||
command: "/opt/fastvideo-ci-runner/run-ci"
|
||||
timeout_in_minutes: 90
|
||||
env:
|
||||
TEST_TYPE: "vae_ci"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "ci-runner"
|
||||
|
||||
- label: ":microscope: Transformer Tests"
|
||||
key: "transformer"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
build.env("MERGE_TEST_PLAN") =~ /,transformer,/) ||
|
||||
build.env("TEST_SCOPE") == "fastcheck" ||
|
||||
build.env("TEST_SCOPE") == null ||
|
||||
(build.env("TEST_SCOPE") == "direct" &&
|
||||
(build.env("TEST_TYPE") == "transformer" || build.env("TEST_TYPE") == "transformer_ci"))
|
||||
command: "/opt/fastvideo-ci-runner/run-ci"
|
||||
timeout_in_minutes: 90
|
||||
env:
|
||||
TEST_TYPE: "transformer_ci"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "ci-runner"
|
||||
|
||||
- label: ":bar_chart: SSIM Tests"
|
||||
key: "ssim"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
build.env("TEST_SCOPE") == "scheduled" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
build.env("MERGE_TEST_PLAN") =~ /,ssim,/) ||
|
||||
(build.env("TEST_SCOPE") == "direct" &&
|
||||
(build.env("TEST_TYPE") == "ssim" || build.env("TEST_TYPE") == "ssim_ci"))
|
||||
command: "/opt/fastvideo-ci-runner/run-ci"
|
||||
timeout_in_minutes: 90
|
||||
concurrency: 1
|
||||
concurrency_group: "fastvideo/slinky/whole-tray"
|
||||
env:
|
||||
TEST_TYPE: "ssim_ci"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
- exit_status: 1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "ci-runner"
|
||||
|
||||
- label: ":test_tube: LoRA Inference Tests"
|
||||
key: "lora-inference"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
build.env("MERGE_TEST_PLAN") =~ /,lora-inference,/) ||
|
||||
(build.env("TEST_SCOPE") == "direct" &&
|
||||
(build.env("TEST_TYPE") == "inference_lora" || build.env("TEST_TYPE") == "inference_lora_ci"))
|
||||
command: "/opt/fastvideo-ci-runner/run-ci"
|
||||
timeout_in_minutes: 90
|
||||
env:
|
||||
TEST_TYPE: "inference_lora_ci"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "ci-runner"
|
||||
|
||||
- label: ":test_tube: LoRA Extraction Tests"
|
||||
key: "lora-extraction"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
build.env("MERGE_TEST_PLAN") =~ /,lora-extraction,/) ||
|
||||
(build.env("TEST_SCOPE") == "direct" &&
|
||||
(build.env("TEST_TYPE") == "lora_extraction" || build.env("TEST_TYPE") == "lora_extraction_ci"))
|
||||
command: "/opt/fastvideo-ci-runner/run-ci"
|
||||
timeout_in_minutes: 90
|
||||
env:
|
||||
TEST_TYPE: "lora_extraction_ci"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "ci-runner"
|
||||
|
||||
- label: ":test_tube: Training Tests"
|
||||
key: "training"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
build.env("MERGE_TEST_PLAN") =~ /,training,/) ||
|
||||
(build.env("TEST_SCOPE") == "direct" &&
|
||||
(build.env("TEST_TYPE") == "training" || build.env("TEST_TYPE") == "training_ci"))
|
||||
command: "/opt/fastvideo-ci-runner/run-ci"
|
||||
timeout_in_minutes: 90
|
||||
concurrency: 1
|
||||
concurrency_group: "fastvideo/slinky/whole-tray"
|
||||
env:
|
||||
TEST_TYPE: "training_ci"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "ci-runner"
|
||||
|
||||
- label: ":test_tube: Distillation DMD Tests"
|
||||
key: "distillation"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
build.env("MERGE_TEST_PLAN") =~ /,distillation,/) ||
|
||||
(build.env("TEST_SCOPE") == "direct" &&
|
||||
(build.env("TEST_TYPE") == "distillation_dmd" || build.env("TEST_TYPE") == "distillation_dmd_ci"))
|
||||
command: "/opt/fastvideo-ci-runner/run-ci"
|
||||
timeout_in_minutes: 90
|
||||
env:
|
||||
TEST_TYPE: "distillation_dmd_ci"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "ci-runner"
|
||||
|
||||
- label: ":test_tube: Self-Forcing Tests"
|
||||
key: "self-forcing"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
build.env("MERGE_TEST_PLAN") =~ /,self-forcing,/) ||
|
||||
(build.env("TEST_SCOPE") == "direct" &&
|
||||
(build.env("TEST_TYPE") == "self_forcing" || build.env("TEST_TYPE") == "self_forcing_ci"))
|
||||
command: "/opt/fastvideo-ci-runner/run-ci"
|
||||
timeout_in_minutes: 90
|
||||
env:
|
||||
TEST_TYPE: "self_forcing_ci"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "ci-runner"
|
||||
|
||||
- label: ":test_tube: LoRA Training Tests"
|
||||
key: "lora-training"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
build.env("MERGE_TEST_PLAN") =~ /,lora-training,/) ||
|
||||
(build.env("TEST_SCOPE") == "direct" &&
|
||||
(build.env("TEST_TYPE") == "training_lora" || build.env("TEST_TYPE") == "training_lora_ci"))
|
||||
command: "/opt/fastvideo-ci-runner/run-ci"
|
||||
timeout_in_minutes: 90
|
||||
env:
|
||||
TEST_TYPE: "training_lora_ci"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
- exit_status: 1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "ci-runner"
|
||||
|
||||
- label: ":test_tube: Training Tests VSA"
|
||||
key: "training-vsa"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
build.env("MERGE_TEST_PLAN") =~ /,training-vsa,/) ||
|
||||
(build.env("TEST_SCOPE") == "direct" &&
|
||||
(build.env("TEST_TYPE") == "training_vsa" || build.env("TEST_TYPE") == "training_vsa_ci"))
|
||||
command: "/opt/fastvideo-ci-runner/run-ci"
|
||||
timeout_in_minutes: 90
|
||||
env:
|
||||
TEST_TYPE: "training_vsa_ci"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
- exit_status: 1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "ci-runner"
|
||||
|
||||
- label: ":test_tube: Inference Tests VMoBA"
|
||||
key: "inference-vmoba"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
build.env("MERGE_TEST_PLAN") =~ /,inference-vmoba,/) ||
|
||||
(build.env("TEST_SCOPE") == "direct" &&
|
||||
(build.env("TEST_TYPE") == "inference_vmoba" || build.env("TEST_TYPE") == "inference_vmoba_ci"))
|
||||
command: "/opt/fastvideo-ci-runner/run-ci"
|
||||
timeout_in_minutes: 90
|
||||
env:
|
||||
TEST_TYPE: "inference_vmoba_ci"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "ci-runner"
|
||||
|
||||
- label: ":test_tube: Performance Tests"
|
||||
key: "performance"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
build.env("MERGE_TEST_PLAN") =~ /,performance,/) ||
|
||||
(build.env("TEST_SCOPE") == "direct" &&
|
||||
(build.env("TEST_TYPE") == "performance" || build.env("TEST_TYPE") == "performance_ci"))
|
||||
command: "/opt/fastvideo-ci-runner/run-ci"
|
||||
timeout_in_minutes: 90
|
||||
env:
|
||||
TEST_TYPE: "performance_ci"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "ci-runner"
|
||||
|
||||
- label: ":test_tube: API Server Tests"
|
||||
key: "api-server"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
build.env("MERGE_TEST_PLAN") =~ /,api-server,/) ||
|
||||
(build.env("TEST_SCOPE") == "direct" &&
|
||||
(build.env("TEST_TYPE") == "api_server" || build.env("TEST_TYPE") == "api_server_ci"))
|
||||
command: "/opt/fastvideo-ci-runner/run-ci"
|
||||
timeout_in_minutes: 90
|
||||
env:
|
||||
TEST_TYPE: "api_server_ci"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "ci-runner"
|
||||
|
||||
- label: ":test_tube: Train Framework Tests"
|
||||
key: "train-framework"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
build.env("MERGE_TEST_PLAN") =~ /,train-framework,/) ||
|
||||
(build.env("TEST_SCOPE") == "direct" &&
|
||||
(build.env("TEST_TYPE") == "train_framework" || build.env("TEST_TYPE") == "train_framework_ci"))
|
||||
command: "/opt/fastvideo-ci-runner/run-ci"
|
||||
timeout_in_minutes: 90
|
||||
env:
|
||||
TEST_TYPE: "train_framework_ci"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "ci-runner"
|
||||
|
||||
- label: ":test_tube: Eval Metrics Tests"
|
||||
key: "eval"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
build.env("MERGE_TEST_PLAN") =~ /,eval,/) ||
|
||||
(build.env("TEST_SCOPE") == "direct" &&
|
||||
(build.env("TEST_TYPE") == "eval" || build.env("TEST_TYPE") == "eval_ci"))
|
||||
command: "/opt/fastvideo-ci-runner/run-ci"
|
||||
timeout_in_minutes: 90
|
||||
env:
|
||||
TEST_TYPE: "eval_ci"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "ci-runner"
|
||||
|
||||
Executable
+5
@@ -0,0 +1,5 @@
|
||||
#!/usr/bin/env bash
|
||||
# Canonical Slurm CI selection for the OpenAI-compatible API lane.
|
||||
set -euo pipefail
|
||||
|
||||
exec pytest ./fastvideo/tests/entrypoints/test_openai_api_integration.py -vs
|
||||
Executable
+5
@@ -0,0 +1,5 @@
|
||||
#!/usr/bin/env bash
|
||||
# Canonical Slurm CI selection for the distillation-DMD lane.
|
||||
set -euo pipefail
|
||||
|
||||
exec pytest ./fastvideo/tests/training/distill/test_distill_dmd.py -vs
|
||||
Executable
+87
@@ -0,0 +1,87 @@
|
||||
#!/usr/bin/env bash
|
||||
# DreamVerse needs a GPU for import-time device resolution, but it does not
|
||||
# build or exercise fastvideo-kernel. A checksummed Node archive is installed
|
||||
# in the disposable Slurm container because the shared CI image is
|
||||
# Python/CUDA focused.
|
||||
set -euo pipefail
|
||||
|
||||
node_version=v22.23.2
|
||||
case $(uname -m) in
|
||||
aarch64 | arm64)
|
||||
node_arch=arm64
|
||||
node_archive_sha256=013b59cfd2819703a6f4a14ab891fc46fc2a4e3f5bcd92de3fb4929b43e35b30
|
||||
;;
|
||||
x86_64 | amd64)
|
||||
node_arch=x64
|
||||
node_archive_sha256=b294a556e639d64338823920e5866c21c02741742d2e1529ee1a225c1ec9252a
|
||||
;;
|
||||
*)
|
||||
echo "Unsupported architecture for DreamVerse Node runtime: $(uname -m)" >&2
|
||||
exit 2
|
||||
;;
|
||||
esac
|
||||
node_archive="node-${node_version}-linux-${node_arch}.tar.gz"
|
||||
node_runtime_root=$(mktemp -d -t fastvideo-node.XXXXXX)
|
||||
node_archive_path="${node_runtime_root}/${node_archive}"
|
||||
node_install_dir="${node_runtime_root}/${node_archive%.tar.gz}"
|
||||
curl --proto '=https' --tlsv1.2 --retry 5 --retry-all-errors \
|
||||
--location --fail --silent --show-error \
|
||||
"https://nodejs.org/dist/${node_version}/${node_archive}" \
|
||||
--output "$node_archive_path"
|
||||
printf '%s %s\n' "$node_archive_sha256" "$node_archive_path" | sha256sum --check --status
|
||||
tar -xzf "$node_archive_path" -C "$node_runtime_root"
|
||||
export PATH="${node_install_dir}/bin:${PATH}"
|
||||
node --version
|
||||
npm --version
|
||||
|
||||
export PYTHONPATH="$(pwd)/apps/dreamverse${PYTHONPATH:+:$PYTHONPATH}"
|
||||
pytest apps/dreamverse/dreamverse/tests -q
|
||||
|
||||
cd apps/dreamverse/web
|
||||
npm ci
|
||||
npm run typecheck
|
||||
npm test
|
||||
machine_arch=$(uname -m)
|
||||
if [[ $machine_arch =~ ^(aarch64|arm64)$ ]]; then
|
||||
npx playwright install --with-deps chromium firefox
|
||||
else
|
||||
npx playwright install --with-deps chromium webkit firefox
|
||||
fi
|
||||
|
||||
master_port=${MASTER_PORT:-7959}
|
||||
BACKEND_PORT=${BACKEND_PORT:-$((master_port + 50))}
|
||||
python -m uvicorn dreamverse.mock_server:app --host 127.0.0.1 --port "$BACKEND_PORT" &
|
||||
mock_server_pid=$!
|
||||
cleanup() {
|
||||
kill "$mock_server_pid" 2>/dev/null || true
|
||||
wait "$mock_server_pid" 2>/dev/null || true
|
||||
}
|
||||
trap cleanup EXIT INT TERM
|
||||
|
||||
for _ in {1..30}; do
|
||||
curl -fsS "http://127.0.0.1:$BACKEND_PORT/healthz" && break
|
||||
sleep 1
|
||||
done
|
||||
curl -fsS "http://127.0.0.1:$BACKEND_PORT/healthz"
|
||||
|
||||
if [[ $machine_arch =~ ^(aarch64|arm64)$ ]]; then
|
||||
# Playwright WebKit traps before opening a page on Linux ARM64, and its
|
||||
# bundled Chromium lacks the H.264/AAC codecs used by the fMP4 assertions.
|
||||
# Firefox covers every flow, including streaming. Chromium and its mobile
|
||||
# profile still cover all codec-independent UI behavior on GB200.
|
||||
BACKEND_HOST=127.0.0.1 BACKEND_PORT="$BACKEND_PORT" CI=1 \
|
||||
npm run e2e -- --project=firefox
|
||||
BACKEND_HOST=127.0.0.1 BACKEND_PORT="$BACKEND_PORT" CI=1 \
|
||||
npm run e2e -- \
|
||||
--project=chromium \
|
||||
--project=mobile-chromium \
|
||||
--grep-invert='streams, plays, and surfaces a downloadable clip|starts a new project and switches back to the prior session|saved projects persist across a page reload'
|
||||
else
|
||||
BACKEND_HOST=127.0.0.1 BACKEND_PORT="$BACKEND_PORT" CI=1 \
|
||||
npm run e2e -- \
|
||||
--project=chromium \
|
||||
--project=webkit \
|
||||
--project=firefox \
|
||||
--project=mobile-safari \
|
||||
--project=mobile-chromium
|
||||
fi
|
||||
Executable
+5
@@ -0,0 +1,5 @@
|
||||
#!/usr/bin/env bash
|
||||
# Canonical Slurm CI selection for the encoder lane.
|
||||
set -euo pipefail
|
||||
|
||||
exec pytest ./fastvideo/tests/encoders -vs
|
||||
Executable
+5
@@ -0,0 +1,5 @@
|
||||
#!/usr/bin/env bash
|
||||
# Canonical Slurm CI selection for the evaluation lane.
|
||||
set -euo pipefail
|
||||
|
||||
exec pytest ./fastvideo/tests/eval -vs
|
||||
Executable
+35
@@ -0,0 +1,35 @@
|
||||
#!/usr/bin/env bash
|
||||
# Canonical Slurm CI selection for the golden-gate lane. Environment (HF_HOME
|
||||
# and authentication) is the runner's responsibility.
|
||||
set -euo pipefail
|
||||
|
||||
golden_root=./fastvideo/tests/golden_gate
|
||||
selected=${FASTVIDEO_GOLDEN_TEST_FILES-}
|
||||
if [ -z "$selected" ]; then
|
||||
if [ "${TEST_SCOPE:-}" = merge ]; then
|
||||
echo "Missing FASTVIDEO_GOLDEN_TEST_FILES for merge scope" >&2
|
||||
exit 2
|
||||
fi
|
||||
selected=all
|
||||
fi
|
||||
if [ "$selected" = all ]; then
|
||||
exec pytest "$golden_root" -xvs
|
||||
fi
|
||||
|
||||
[[ $selected =~ ^test_[a-z0-9_]+\.py(,test_[a-z0-9_]+\.py)*$ ]] || {
|
||||
echo "Invalid FASTVIDEO_GOLDEN_TEST_FILES selection" >&2
|
||||
exit 2
|
||||
}
|
||||
|
||||
IFS=, read -r -a golden_files <<< "$selected"
|
||||
golden_paths=()
|
||||
for golden_file in "${golden_files[@]}"; do
|
||||
golden_path="$golden_root/$golden_file"
|
||||
[ -f "$golden_path" ] || {
|
||||
echo "Selected golden test does not exist: $golden_file" >&2
|
||||
exit 2
|
||||
}
|
||||
golden_paths+=("$golden_path")
|
||||
done
|
||||
|
||||
exec pytest "${golden_paths[@]}" -xvs
|
||||
Executable
+5
@@ -0,0 +1,5 @@
|
||||
#!/usr/bin/env bash
|
||||
# Canonical Slurm CI selection for the LoRA-inference lane.
|
||||
set -euo pipefail
|
||||
|
||||
exec pytest ./fastvideo/tests/inference/lora/test_lora_inference_similarity.py -vs
|
||||
Executable
+5
@@ -0,0 +1,5 @@
|
||||
#!/usr/bin/env bash
|
||||
# Canonical Slurm CI selection for the VMoBA-inference lane.
|
||||
set -euo pipefail
|
||||
|
||||
exec python fastvideo/tests/inference/vmoba/test_vmoba_inference.py
|
||||
Executable
+5
@@ -0,0 +1,5 @@
|
||||
#!/usr/bin/env bash
|
||||
# Canonical Slurm CI selection for the custom-kernel lane.
|
||||
set -euo pipefail
|
||||
|
||||
exec pytest fastvideo-kernel/tests/ -vs
|
||||
Executable
+5
@@ -0,0 +1,5 @@
|
||||
#!/usr/bin/env bash
|
||||
# Canonical Slurm CI selection for the LoRA-extraction lane.
|
||||
set -euo pipefail
|
||||
|
||||
exec pytest ./fastvideo/tests/lora_extraction/ -vs
|
||||
Executable
+64
@@ -0,0 +1,64 @@
|
||||
#!/usr/bin/env bash
|
||||
# Canonical Slurm performance lane. Reports are written outside the checkout
|
||||
# so the trusted host driver can upload them after untrusted code exits.
|
||||
set -uo pipefail
|
||||
|
||||
export PERFORMANCE_TRACKING_ROOT=/tmp/perf-tracking
|
||||
export PERF_REPORTS_DIR=/workspace/artifacts/performance
|
||||
mkdir -p "$PERF_REPORTS_DIR"
|
||||
|
||||
if [[ ${BUILDKITE_PULL_REQUEST:-false} =~ ^[1-9][0-9]*$ ]]; then
|
||||
export PERF_RUN_SOURCE=pr
|
||||
export PERF_UPLOAD_POLICY=pass
|
||||
elif [ "${BUILDKITE_BRANCH:-}" = main ] \
|
||||
&& { [ "${BUILDKITE_SOURCE:-}" = schedule ] || [ "${TEST_SCOPE:-}" = full ]; }; then
|
||||
export PERF_RUN_SOURCE=scheduled_main
|
||||
export PERF_UPLOAD_POLICY=always
|
||||
elif [ "${TEST_SCOPE:-}" = direct ]; then
|
||||
export PERF_RUN_SOURCE=unknown
|
||||
export PERF_UPLOAD_POLICY=pass
|
||||
else
|
||||
export PERF_RUN_SOURCE=unknown
|
||||
export PERF_UPLOAD_POLICY=never
|
||||
fi
|
||||
|
||||
nvidia-smi \
|
||||
--query-gpu=index,timestamp,clocks.sm,clocks.max.sm,power.draw,power.limit,temperature.gpu \
|
||||
--format=csv -l 10 > "$PERF_REPORTS_DIR/gpu_telemetry.csv" 2>/dev/null &
|
||||
telemetry_pid=$!
|
||||
cleanup() {
|
||||
kill "$telemetry_pid" 2>/dev/null || true
|
||||
wait "$telemetry_pid" 2>/dev/null || true
|
||||
}
|
||||
trap cleanup EXIT INT TERM
|
||||
|
||||
pytest ./fastvideo/tests/performance/test_inference_performance.py -vs
|
||||
pytest_rc=$?
|
||||
compare_rc=0
|
||||
if [ "$pytest_rc" -eq 0 ] || [ "$PERF_UPLOAD_POLICY" = always ]; then
|
||||
PERF_PYTEST_RC=$pytest_rc python ./fastvideo/tests/performance/compare_baseline.py
|
||||
compare_rc=$?
|
||||
fi
|
||||
python ./fastvideo/tests/performance/dashboard.py || true
|
||||
cp -f fastvideo/tests/performance/results/*.json "$PERF_REPORTS_DIR/" 2>/dev/null || true
|
||||
# The trusted host relays only .md/.html/.json/.csv from PERF_REPORTS_DIR, so
|
||||
# mirror each captured worker log with an allowlisted extension.
|
||||
for worker_log in fastvideo/tests/performance/results/worker_logs/*.log; do
|
||||
[ -f "$worker_log" ] || continue
|
||||
base=$(basename "${worker_log%.log}")
|
||||
# WorkerLogCapture keeps a .log.1 backup after rollover, and read_log_tail
|
||||
# includes it; mirror that retained history too so the artifact is complete.
|
||||
if [ -f "$worker_log.1" ]; then
|
||||
cp -f "$worker_log.1" "$PERF_REPORTS_DIR/${base}.1.md" 2>/dev/null || true
|
||||
fi
|
||||
cp -f "$worker_log" "$PERF_REPORTS_DIR/${base}.md" 2>/dev/null || true
|
||||
done
|
||||
|
||||
echo "--- GPU telemetry (clocks.sm vs clocks.max.sm reveals capped hosts) ---"
|
||||
cat "$PERF_REPORTS_DIR/gpu_telemetry.csv" || true
|
||||
|
||||
final_rc=$pytest_rc
|
||||
if [ "$final_rc" -eq 0 ]; then
|
||||
final_rc=$compare_rc
|
||||
fi
|
||||
exit "$final_rc"
|
||||
Executable
+6
@@ -0,0 +1,6 @@
|
||||
#!/usr/bin/env bash
|
||||
# Canonical Slurm CI selection for the self-forcing lane.
|
||||
set -euo pipefail
|
||||
|
||||
export WANDB_MODE=offline
|
||||
exec pytest ./fastvideo/tests/training/self-forcing/test_self_forcing.py -vs
|
||||
Executable
+40
@@ -0,0 +1,40 @@
|
||||
#!/usr/bin/env bash
|
||||
# Canonical four-GPU SSIM lane for the Slinky Slurm worker.
|
||||
set -euo pipefail
|
||||
|
||||
args=()
|
||||
if [ "${FASTVIDEO_SSIM_BOOTSTRAP_MODE:-0}" = 1 ]; then
|
||||
args+=(--bootstrap-mode)
|
||||
fi
|
||||
selected=${FASTVIDEO_SSIM_TEST_FILES-}
|
||||
if [ -z "$selected" ]; then
|
||||
if [ "${TEST_SCOPE:-}" = merge ]; then
|
||||
echo "Missing FASTVIDEO_SSIM_TEST_FILES for merge scope" >&2
|
||||
exit 2
|
||||
fi
|
||||
selected=all
|
||||
fi
|
||||
if [ "$selected" != all ]; then
|
||||
[[ $selected =~ ^test_[a-z0-9_]+\.py(,test_[a-z0-9_]+\.py)*$ ]] || {
|
||||
echo "Invalid FASTVIDEO_SSIM_TEST_FILES selection" >&2
|
||||
exit 2
|
||||
}
|
||||
IFS=, read -r -a ssim_files <<< "$selected"
|
||||
for ssim_file in "${ssim_files[@]}"; do
|
||||
args+=(--test-file "$ssim_file")
|
||||
done
|
||||
fi
|
||||
|
||||
# MoGe's utils3d dependency builds glcontext from source on ARM64. The current
|
||||
# runner image predates the baked-in X11 headers below, so keep this guarded
|
||||
# bootstrap until every deployed image digest contains libx11-dev.
|
||||
if [ ! -f /usr/include/X11/Xlib.h ]; then
|
||||
apt-get -o Acquire::Retries=5 update
|
||||
apt-get -o Acquire::Retries=5 install -y --no-install-recommends libx11-dev
|
||||
rm -rf /var/lib/apt/lists/*
|
||||
fi
|
||||
|
||||
uv pip install git+https://github.com/microsoft/MoGe.git
|
||||
uv pip install k_diffusion einops_exts alias_free_torch torchsde
|
||||
|
||||
exec python fastvideo/tests/ssim/ci_runner.py "${args[@]}"
|
||||
Executable
+5
@@ -0,0 +1,5 @@
|
||||
#!/usr/bin/env bash
|
||||
# Canonical Slurm CI selection for the modular training-framework lane.
|
||||
set -euo pipefail
|
||||
|
||||
exec pytest ./fastvideo/tests/train/models ./fastvideo/tests/train/methods -vs
|
||||
Executable
+6
@@ -0,0 +1,6 @@
|
||||
#!/usr/bin/env bash
|
||||
# Canonical Slurm CI selection for the legacy vanilla-training lane.
|
||||
set -euo pipefail
|
||||
|
||||
export WANDB_MODE=offline
|
||||
exec pytest ./fastvideo/tests/training/Vanilla -srP
|
||||
Executable
+6
@@ -0,0 +1,6 @@
|
||||
#!/usr/bin/env bash
|
||||
# Canonical Slurm CI selection for the legacy LoRA-training lane.
|
||||
set -euo pipefail
|
||||
|
||||
export WANDB_MODE=offline
|
||||
exec pytest ./fastvideo/tests/training/lora/test_lora_training.py -srP
|
||||
Executable
+6
@@ -0,0 +1,6 @@
|
||||
#!/usr/bin/env bash
|
||||
# Canonical Slurm CI selection for the legacy VSA-training lane.
|
||||
set -euo pipefail
|
||||
|
||||
export WANDB_MODE=offline
|
||||
exec pytest ./fastvideo/tests/training/VSA -srP
|
||||
Executable
+9
@@ -0,0 +1,9 @@
|
||||
#!/usr/bin/env bash
|
||||
# Canonical Slurm CI selection for the transformer lane.
|
||||
set -euo pipefail
|
||||
|
||||
# The existing block reference records an absent FASTVIDEO_FA4 (FA2). Keep
|
||||
# that reference identity; the component lane also selects FA2 explicitly.
|
||||
env -u FASTVIDEO_FA4 pytest ./fastvideo/tests/golden_gate/test_wan_t2v.py -xvs
|
||||
pytest ./fastvideo/tests/golden_gate/test_wan_causal.py -xvs
|
||||
exec pytest ./fastvideo/tests/transformers -vs
|
||||
Executable
+6
@@ -0,0 +1,6 @@
|
||||
#!/usr/bin/env bash
|
||||
# Canonical Slurm CI selection for the VAE lane.
|
||||
set -euo pipefail
|
||||
|
||||
pytest ./fastvideo/tests/golden_gate/test_wan_vae.py -xvs
|
||||
exec pytest ./fastvideo/tests/vaes -vs
|
||||
+200
-21
@@ -1,6 +1,19 @@
|
||||
#!/bin/bash
|
||||
set -uo pipefail
|
||||
|
||||
# DORMANT ROLLBACK ONLY. Active CI is Slurm-only and pipeline.yml never calls
|
||||
# this launcher. Refuse every Buildkite invocation even if a stale step or
|
||||
# operator typo reaches this file; local rollback experiments require an
|
||||
# explicit opt-in.
|
||||
if [ -n "${BUILDKITE:-}" ]; then
|
||||
echo "Legacy Modal CI is disabled; use the Slinky Slurm runner." >&2
|
||||
exit 2
|
||||
fi
|
||||
if [ "${FASTVIDEO_ENABLE_LEGACY_MODAL_CI:-0}" != 1 ]; then
|
||||
echo "Legacy Modal CI is dormant. Set FASTVIDEO_ENABLE_LEGACY_MODAL_CI=1 only for a manual rollback test." >&2
|
||||
exit 2
|
||||
fi
|
||||
|
||||
log() {
|
||||
echo "[$(date '+%Y-%m-%d %H:%M:%S')] $1"
|
||||
}
|
||||
@@ -15,8 +28,21 @@ log "Project root: $PROJECT_ROOT"
|
||||
# Install Modal if not available
|
||||
if ! python3 -m modal --version &> /dev/null; then
|
||||
log "Modal not found, installing..."
|
||||
python3 -m pip install modal
|
||||
|
||||
if ! command -v uv &> /dev/null; then
|
||||
log "uv not found, bootstrapping..."
|
||||
if ! curl -LsSf https://astral.sh/uv/install.sh | sh; then
|
||||
log "Error: Failed to bootstrap uv via astral.sh installer."
|
||||
exit 1
|
||||
fi
|
||||
export PATH="$HOME/.local/bin:$PATH"
|
||||
if ! command -v uv &> /dev/null; then
|
||||
log "Error: uv still not on PATH after bootstrap."
|
||||
exit 1
|
||||
fi
|
||||
fi
|
||||
# --break-system-packages preserves prior `pip install --user` semantics on PEP 668 agents.
|
||||
uv pip install --system --break-system-packages modal
|
||||
|
||||
# Verify installation
|
||||
if ! python3 -m modal --version &> /dev/null; then
|
||||
log "Error: Failed to install modal. Please install it manually."
|
||||
@@ -31,9 +57,9 @@ log "Setting up Modal authentication from Buildkite secrets..."
|
||||
MODAL_TOKEN_ID=$(buildkite-agent secret get modal_token_id)
|
||||
MODAL_TOKEN_SECRET=$(buildkite-agent secret get modal_token_secret)
|
||||
|
||||
# Retrieve other secrets
|
||||
WANDB_API_KEY=$(buildkite-agent secret get wandb_api_key)
|
||||
|
||||
WANDB_API_KEY=$(buildkite-agent secret get wandb_api_key)
|
||||
HF_API_KEY=$(buildkite-agent secret get hf_api_key)
|
||||
|
||||
if [ -n "$MODAL_TOKEN_ID" ] && [ -n "$MODAL_TOKEN_SECRET" ]; then
|
||||
log "Retrieved Modal credentials from Buildkite secrets"
|
||||
@@ -50,7 +76,8 @@ else
|
||||
exit 1
|
||||
fi
|
||||
|
||||
MODAL_TEST_FILE="fastvideo/v1/tests/modal/pr_test.py"
|
||||
MODAL_TEST_FILE="fastvideo/tests/modal/pr_test.py"
|
||||
MODAL_SSIM_TEST_FILE="fastvideo/tests/modal/ssim_test.py"
|
||||
|
||||
if [ -z "${TEST_TYPE:-}" ]; then
|
||||
log "Error: TEST_TYPE environment variable is not set"
|
||||
@@ -58,49 +85,196 @@ if [ -z "${TEST_TYPE:-}" ]; then
|
||||
fi
|
||||
log "Test type: $TEST_TYPE"
|
||||
|
||||
MODAL_ENV="BUILDKITE_REPO=$BUILDKITE_REPO BUILDKITE_COMMIT=$BUILDKITE_COMMIT BUILDKITE_PULL_REQUEST=$BUILDKITE_PULL_REQUEST IMAGE_VERSION=$IMAGE_VERSION"
|
||||
EFFECTIVE_PR=${BUILDKITE_PULL_REQUEST:-false}
|
||||
if [ "$EFFECTIVE_PR" = "false" ] && [ -n "${PR_NUMBER:-}" ]; then
|
||||
EFFECTIVE_PR=$PR_NUMBER
|
||||
fi
|
||||
MODAL_ENV="BUILDKITE_REPO=$BUILDKITE_REPO BUILDKITE_COMMIT=$BUILDKITE_COMMIT BUILDKITE_PULL_REQUEST=$EFFECTIVE_PR BUILDKITE_BRANCH=${BUILDKITE_BRANCH:-} BUILDKITE_SOURCE=${BUILDKITE_SOURCE:-} TEST_SCOPE=${TEST_SCOPE:-} BUILDKITE_BUILD_URL=${BUILDKITE_BUILD_URL:-} BUILDKITE_BUILD_ID=${BUILDKITE_BUILD_ID:-} BUILDKITE_JOB_ID=${BUILDKITE_JOB_ID:-} IMAGE_VERSION=$IMAGE_VERSION"
|
||||
|
||||
POST_RUN_HOOK=""
|
||||
|
||||
is_truthy() {
|
||||
case "${1:-}" in
|
||||
1|true|TRUE|yes|YES|on|ON) return 0 ;;
|
||||
*) return 1 ;;
|
||||
esac
|
||||
}
|
||||
|
||||
ssim_bootstrap_args() {
|
||||
local title="${PR_TITLE:-}"
|
||||
local message="${BUILDKITE_MESSAGE:-}"
|
||||
if is_truthy "${FASTVIDEO_SSIM_BOOTSTRAP_MODE:-}" \
|
||||
|| [[ "$title" == *"[new-model]"* ]] \
|
||||
|| [[ "$message" == *"[new-model]"* ]]; then
|
||||
printf ' --bootstrap-mode'
|
||||
fi
|
||||
}
|
||||
|
||||
upload_performance_artifacts() {
|
||||
SHORT_SHA=${BUILDKITE_COMMIT:0:7}
|
||||
LOCAL_DIR="downloaded_reports"
|
||||
|
||||
_download_reports() {
|
||||
log "Downloading perf_reports/ from Modal Volume..."
|
||||
mkdir -p "$LOCAL_DIR"
|
||||
if ! modal volume get hf-model-weights "perf_reports/" "$LOCAL_DIR"; then
|
||||
log "Error: Failed to download perf_reports/ from Modal Volume."
|
||||
return 1
|
||||
fi
|
||||
}
|
||||
|
||||
_upload_dashboard() {
|
||||
local target
|
||||
target=$(find "$LOCAL_DIR" -name "dashboard_${SHORT_SHA}_*" | head -n 1)
|
||||
log "TARGET dashboard: '$target'"
|
||||
|
||||
if [ -n "$target" ]; then
|
||||
log "Found dashboard: $target. Uploading to Buildkite..."
|
||||
buildkite-agent artifact upload "$target"
|
||||
buildkite-agent annotate --style info --context "perf-dashboard" < "$target"
|
||||
else
|
||||
log "Warning: Could not find a dashboard file matching $SHORT_SHA"
|
||||
fi
|
||||
}
|
||||
|
||||
_upload_perf_summary() {
|
||||
local target
|
||||
target=$(find "$LOCAL_DIR" -name "perf_${SHORT_SHA}_*" | head -n 1)
|
||||
log "TARGET perf summary: '$target'"
|
||||
|
||||
if [ -n "$target" ]; then
|
||||
log "Found perf summary: $target. Uploading to Buildkite..."
|
||||
buildkite-agent artifact upload "$target"
|
||||
buildkite-agent annotate --style info --context "perf-summary" < "$target"
|
||||
else
|
||||
log "Warning: Could not find a perf summary file matching $SHORT_SHA"
|
||||
fi
|
||||
}
|
||||
|
||||
_upload_normalized_perf_results() {
|
||||
local found=0
|
||||
while IFS= read -r -d '' target; do
|
||||
found=1
|
||||
log "Found normalized performance result: $target. Uploading to Buildkite..."
|
||||
buildkite-agent artifact upload "$target"
|
||||
done < <(find "$LOCAL_DIR" -path "*/results/normalized_perf_*.json" -print0)
|
||||
|
||||
if [ "$found" -eq 0 ]; then
|
||||
log "No normalized performance result artifacts found. This is expected when the rolling performance comparison did not run."
|
||||
fi
|
||||
}
|
||||
|
||||
_cleanup_modal_volume() {
|
||||
log "Cleaning up perf_reports/ from Modal Volume..."
|
||||
if modal volume rm hf-model-weights "perf_reports/" --recursive; then
|
||||
log "Successfully deleted perf_reports/ from Modal Volume."
|
||||
else
|
||||
log "Warning: Failed to delete perf_reports/ from Modal Volume. Manual cleanup may be required."
|
||||
fi
|
||||
}
|
||||
|
||||
_cleanup_local() {
|
||||
log "Cleaning up local download directory..."
|
||||
rm -rf "$LOCAL_DIR"
|
||||
}
|
||||
|
||||
# --- Main flow ---
|
||||
_download_reports || { _cleanup_local; return 1; }
|
||||
_upload_dashboard
|
||||
_upload_perf_summary
|
||||
_upload_normalized_perf_results
|
||||
_cleanup_modal_volume
|
||||
_cleanup_local
|
||||
}
|
||||
|
||||
case "$TEST_TYPE" in
|
||||
"encoder")
|
||||
log "Running encoder tests..."
|
||||
MODAL_COMMAND="$MODAL_ENV python3 -m modal run $MODAL_TEST_FILE::run_encoder_tests"
|
||||
MODAL_COMMAND="$MODAL_ENV HF_API_KEY=$HF_API_KEY python3 -m modal run $MODAL_TEST_FILE::run_encoder_tests"
|
||||
;;
|
||||
"vae")
|
||||
log "Running VAE tests..."
|
||||
MODAL_COMMAND="$MODAL_ENV python3 -m modal run $MODAL_TEST_FILE::run_vae_tests"
|
||||
MODAL_COMMAND="$MODAL_ENV HF_API_KEY=$HF_API_KEY python3 -m modal run $MODAL_TEST_FILE::run_vae_tests"
|
||||
;;
|
||||
"transformer")
|
||||
log "Running transformer tests..."
|
||||
MODAL_COMMAND="$MODAL_ENV python3 -m modal run $MODAL_TEST_FILE::run_transformer_tests"
|
||||
MODAL_COMMAND="$MODAL_ENV HF_API_KEY=$HF_API_KEY python3 -m modal run $MODAL_TEST_FILE::run_transformer_tests"
|
||||
;;
|
||||
"golden_gate")
|
||||
log "Running golden-gate tests..."
|
||||
MODAL_COMMAND="$MODAL_ENV HF_API_KEY=$HF_API_KEY python3 -m modal run $MODAL_TEST_FILE::run_golden_gate_tests"
|
||||
;;
|
||||
"ssim")
|
||||
log "Running SSIM tests..."
|
||||
MODAL_COMMAND="$MODAL_ENV python3 -m modal run $MODAL_TEST_FILE::run_ssim_tests"
|
||||
SSIM_BOOTSTRAP_ARGS=$(ssim_bootstrap_args)
|
||||
if [ -n "$SSIM_BOOTSTRAP_ARGS" ]; then
|
||||
log "SSIM bootstrap mode enabled for new-model reference draft generation"
|
||||
fi
|
||||
MODAL_COMMAND="$MODAL_ENV HF_API_KEY=$HF_API_KEY python3 -m modal run "
|
||||
MODAL_COMMAND+="$MODAL_SSIM_TEST_FILE::run_ssim_tests$SSIM_BOOTSTRAP_ARGS"
|
||||
;;
|
||||
"training")
|
||||
log "Running training tests..."
|
||||
MODAL_COMMAND="$MODAL_ENV WANDB_API_KEY=$WANDB_API_KEY python3 -m modal run $MODAL_TEST_FILE::run_training_tests"
|
||||
;;
|
||||
"training_lora")
|
||||
log "Running LoRA training tests..."
|
||||
MODAL_COMMAND="$MODAL_ENV WANDB_API_KEY=$WANDB_API_KEY python3 -m modal run $MODAL_TEST_FILE::run_training_lora_tests"
|
||||
;;
|
||||
"training_vsa")
|
||||
log "Running training VSA tests..."
|
||||
MODAL_COMMAND="$MODAL_ENV WANDB_API_KEY=$WANDB_API_KEY python3 -m modal run $MODAL_TEST_FILE::run_training_tests_VSA"
|
||||
;;
|
||||
"inference_sta")
|
||||
log "Running inference STA tests..."
|
||||
MODAL_COMMAND="$MODAL_ENV python3 -m modal run $MODAL_TEST_FILE::run_inference_tests_STA"
|
||||
;;
|
||||
"precision_sta")
|
||||
log "Running precision STA tests..."
|
||||
MODAL_COMMAND="$MODAL_ENV python3 -m modal run $MODAL_TEST_FILE::run_precision_tests_STA"
|
||||
;;
|
||||
"precision_vsa")
|
||||
log "Running precision VSA tests..."
|
||||
MODAL_COMMAND="$MODAL_ENV python3 -m modal run $MODAL_TEST_FILE::run_precision_tests_VSA"
|
||||
"kernel_tests")
|
||||
log "Running kernel tests..."
|
||||
MODAL_COMMAND="$MODAL_ENV python3 -m modal run $MODAL_TEST_FILE::run_kernel_tests"
|
||||
;;
|
||||
"inference_lora")
|
||||
log "Running LoRA tests..."
|
||||
MODAL_COMMAND="$MODAL_ENV python3 -m modal run $MODAL_TEST_FILE::run_inference_lora_tests"
|
||||
;;
|
||||
"distillation_dmd")
|
||||
log "Running distillation DMD tests..."
|
||||
MODAL_COMMAND="$MODAL_ENV WANDB_API_KEY=$WANDB_API_KEY python3 -m modal run $MODAL_TEST_FILE::run_distill_dmd_tests"
|
||||
;;
|
||||
# run_inference_tests_vmoba
|
||||
"self_forcing")
|
||||
log "Running self-forcing tests..."
|
||||
MODAL_COMMAND="$MODAL_ENV WANDB_API_KEY=$WANDB_API_KEY python3 -m modal run $MODAL_TEST_FILE::run_self_forcing_tests"
|
||||
;;
|
||||
"inference_vmoba")
|
||||
log "Running V-MoBA inference tests..."
|
||||
MODAL_COMMAND="$MODAL_ENV python3 -m modal run $MODAL_TEST_FILE::run_inference_tests_vmoba"
|
||||
;;
|
||||
"unit_test")
|
||||
log "Running unit tests..."
|
||||
MODAL_COMMAND="$MODAL_ENV python3 -m modal run $MODAL_TEST_FILE::run_unit_test"
|
||||
;;
|
||||
"dreamverse_app")
|
||||
log "Running DreamVerse app tests..."
|
||||
MODAL_COMMAND="$MODAL_ENV python3 -m modal run $MODAL_TEST_FILE::run_dreamverse_app_tests"
|
||||
;;
|
||||
"train_framework")
|
||||
log "Running fastvideo.train framework tests..."
|
||||
MODAL_COMMAND="$MODAL_ENV HF_API_KEY=$HF_API_KEY python3 -m modal run $MODAL_TEST_FILE::run_train_framework_tests"
|
||||
;;
|
||||
"eval")
|
||||
log "Running eval metric tests..."
|
||||
MODAL_COMMAND="$MODAL_ENV HF_API_KEY=$HF_API_KEY python3 -m modal run $MODAL_TEST_FILE::run_eval_tests"
|
||||
;;
|
||||
"lora_extraction")
|
||||
log "Running LoRA extraction tests..."
|
||||
MODAL_COMMAND="$MODAL_ENV HF_API_KEY=$HF_API_KEY python3 -m modal run $MODAL_TEST_FILE::run_lora_extraction_tests"
|
||||
;;
|
||||
"performance")
|
||||
log "Running performance tests on Modal..."
|
||||
MODAL_COMMAND="$MODAL_ENV HF_API_KEY=$HF_API_KEY python3 -m modal run $MODAL_TEST_FILE::run_performance_tests"
|
||||
POST_RUN_HOOK="upload_performance_artifacts"
|
||||
;;
|
||||
"api_server")
|
||||
log "Running API server integration tests..."
|
||||
MODAL_COMMAND="$MODAL_ENV HF_API_KEY=$HF_API_KEY python3 -m modal run $MODAL_TEST_FILE::run_api_server_tests"
|
||||
;;
|
||||
*)
|
||||
log "Error: Unknown test type: $TEST_TYPE"
|
||||
exit 1
|
||||
@@ -117,5 +291,10 @@ else
|
||||
log "Error: Modal test failed with exit code: $TEST_EXIT_CODE"
|
||||
fi
|
||||
|
||||
if [ -n "$POST_RUN_HOOK" ]; then
|
||||
log "Executing post-run hook: $POST_RUN_HOOK"
|
||||
"$POST_RUN_HOOK"
|
||||
fi
|
||||
|
||||
log "=== Test execution completed with exit code: $TEST_EXIT_CODE ==="
|
||||
exit $TEST_EXIT_CODE
|
||||
|
||||
@@ -13,8 +13,21 @@ log "Project root: $PROJECT_ROOT"
|
||||
|
||||
if ! python3 -m pre_commit --version &> /dev/null; then
|
||||
log "pre-commit not found, installing..."
|
||||
python3 -m pip install --user pre-commit==4.0.1
|
||||
|
||||
if ! command -v uv &> /dev/null; then
|
||||
log "uv not found, bootstrapping..."
|
||||
if ! curl -LsSf https://astral.sh/uv/install.sh | sh; then
|
||||
log "Error: Failed to bootstrap uv via astral.sh installer."
|
||||
exit 1
|
||||
fi
|
||||
export PATH="$HOME/.local/bin:$PATH"
|
||||
if ! command -v uv &> /dev/null; then
|
||||
log "Error: uv still not on PATH after bootstrap."
|
||||
exit 1
|
||||
fi
|
||||
fi
|
||||
# --break-system-packages preserves prior `pip install --user` semantics on PEP 668 agents.
|
||||
uv pip install --system --break-system-packages pre-commit==4.0.1
|
||||
|
||||
if ! python3 -m pre_commit --version &> /dev/null; then
|
||||
log "Error: Failed to install pre-commit."
|
||||
exit 1
|
||||
|
||||
@@ -0,0 +1,41 @@
|
||||
#!/usr/bin/env bash
|
||||
set -euo pipefail
|
||||
|
||||
# Collect the whole attention directory so new files cannot land uncovered.
|
||||
# Its FA2/FA3 regression files skip when FA4 is selected (the Modal image
|
||||
# enables FA4 by default), so pin FA4 off for the directory to be real
|
||||
# coverage on every runner rather than a nominal collection.
|
||||
export FASTVIDEO_FA4=0
|
||||
|
||||
# The livestream app's tests are CPU-only; its single gpu-marked module is
|
||||
# deselected, and DreamVerse's GPU tests have their own lane.
|
||||
exec pytest \
|
||||
./apps/infinite_livestream/infinite_livestream/tests \
|
||||
./fastvideo/tests/api/ \
|
||||
./fastvideo/tests/contract/ \
|
||||
./fastvideo/tests/dataset/ \
|
||||
./fastvideo/tests/workflow/ \
|
||||
./fastvideo/tests/entrypoints/ \
|
||||
./fastvideo/tests/loader/ \
|
||||
./fastvideo/tests/pipelines/ \
|
||||
./fastvideo/tests/platforms/ \
|
||||
./fastvideo/tests/schedulers/ \
|
||||
./fastvideo/tests/train/ \
|
||||
./fastvideo/tests/stages/ \
|
||||
./fastvideo/tests/ops/ \
|
||||
./fastvideo/tests/worker/ \
|
||||
./fastvideo/tests/training/test_runner.py \
|
||||
./fastvideo/tests/training/test_trackers.py \
|
||||
./fastvideo/tests/inference/test_basic_fasth3_omniref_pdd.py \
|
||||
./fastvideo/tests/inference/test_inference_regional_compile.py \
|
||||
./fastvideo/tests/attention/ \
|
||||
./fastvideo/tests/layers/test_pdd_linear.py \
|
||||
./fastvideo/tests/layers/test_triton_fused_norm.py \
|
||||
./fastvideo/tests/modal/test_kernel_build_cache.py \
|
||||
./fastvideo/tests/modal/test_pr_test.py \
|
||||
./fastvideo/tests/modal/test_ssim_test.py \
|
||||
--ignore=./fastvideo/tests/entrypoints/test_openai_api_integration.py \
|
||||
--ignore=./fastvideo/tests/train/models \
|
||||
--ignore=./fastvideo/tests/train/methods \
|
||||
-m "not gpu" \
|
||||
-vs
|
||||
@@ -0,0 +1,31 @@
|
||||
# Build-context excludes: keep the context small and the `COPY . .` layer cache
|
||||
# stable. Docker uploads everything here to the daemon and bakes it into a layer;
|
||||
# without this, the 6.5 GB host .venv alone is shipped + cached on every build.
|
||||
#
|
||||
# IMPORTANT: do NOT ignore .git — fastvideo-kernel's build runs
|
||||
# `git submodule update --init --recursive`, which needs the repo metadata.
|
||||
|
||||
# Virtualenvs — the image builds its own /opt/venv
|
||||
.venv/
|
||||
venv/
|
||||
env/
|
||||
|
||||
# Python caches & build/test/lint artifacts
|
||||
**/__pycache__/
|
||||
*.py[cod]
|
||||
*.egg-info/
|
||||
.eggs/
|
||||
.pytest_cache/
|
||||
.mypy_cache/
|
||||
.ruff_cache/
|
||||
.cache/
|
||||
|
||||
# Local run outputs / logs (not needed in the image)
|
||||
outputs/
|
||||
wandb/
|
||||
*.log
|
||||
|
||||
# Editor / OS cruft
|
||||
.DS_Store
|
||||
.idea/
|
||||
.vscode/
|
||||
@@ -23,7 +23,7 @@ body:
|
||||
attributes:
|
||||
label: Environment
|
||||
description: |
|
||||
Please share your environment with us. You can run the command **python fastvideo/utils/collect_env.py** and copy-paste its output below.
|
||||
Please share your environment with us. You can run the command **python collect_env.py** and copy-paste its output below.
|
||||
placeholder: FastVideo version, platform, python version, cuda version...
|
||||
validations:
|
||||
required: true
|
||||
@@ -0,0 +1,56 @@
|
||||
name: 💬 Request for comments (RFC).
|
||||
description: Ask for feedback on major architectural changes or design choices.
|
||||
title: "[RFC]: "
|
||||
labels: ["RFC"]
|
||||
|
||||
body:
|
||||
- type: markdown
|
||||
attributes:
|
||||
value: >
|
||||
#### Please take a look at previous [RFCs](https://github.com/hao-ai-lab/FastVideo/issues?q=label%3ARFC+sort%3Aupdated-desc) for reference.
|
||||
- type: textarea
|
||||
attributes:
|
||||
label: Motivation.
|
||||
description: >
|
||||
The motivation of the RFC.
|
||||
validations:
|
||||
required: true
|
||||
- type: textarea
|
||||
attributes:
|
||||
label: Proposed Change.
|
||||
description: >
|
||||
The proposed change of the RFC.
|
||||
validations:
|
||||
required: true
|
||||
- type: textarea
|
||||
attributes:
|
||||
label: Feedback Period.
|
||||
description: >
|
||||
The feedback period of the RFC. Usually at least one week.
|
||||
validations:
|
||||
required: false
|
||||
- type: textarea
|
||||
attributes:
|
||||
label: CC List.
|
||||
description: >
|
||||
The list of people you want to CC.
|
||||
validations:
|
||||
required: false
|
||||
- type: textarea
|
||||
attributes:
|
||||
label: Any Other Things.
|
||||
description: >
|
||||
Any other things you would like to mention.
|
||||
validations:
|
||||
required: false
|
||||
- type: markdown
|
||||
attributes:
|
||||
value: >
|
||||
Thanks for contributing 🎉!
|
||||
- type: checkboxes
|
||||
id: askllm
|
||||
attributes:
|
||||
label: Before submitting a new issue...
|
||||
options:
|
||||
- label: Make sure you already searched for relevant issues.
|
||||
required: true
|
||||
@@ -0,0 +1,63 @@
|
||||
<!--
|
||||
PR TITLE: Must start with a type tag, e.g.:
|
||||
[feat] Add new model [bugfix] Fix VAE tiling [refactor] Restructure pipeline
|
||||
[perf] Optimize kernel [ci] Update tests [docs] Add guide
|
||||
[misc] Cleanup configs [new-model] Port Flux2 [infra] Add trace hooks
|
||||
[skill] Add agent skill
|
||||
|
||||
MERGE WORKFLOW:
|
||||
1. Ensure pre-commit passes and you have at least 1 approval
|
||||
2. Comment /merge (or add the "ready" label) to enter the Merge Queue
|
||||
3. A path-aware merge gate runs only relevant integration tests → auto-merge on success
|
||||
|
||||
ON-DEMAND TESTING (write access required):
|
||||
/test full — Explicit all-lane run /test ssim — Full SSIM regression
|
||||
/test training — Training pipeline /test encoder — Encoder tests
|
||||
/test transformer — Transformer tests /test vae — VAE tests
|
||||
/test kernel — CUDA kernel tests /test unit — Unit tests
|
||||
See docs/contributing/pull_requests.md for all 17 test commands
|
||||
-->
|
||||
|
||||
## Purpose
|
||||
|
||||
<!-- What does this PR do? Link the related issue if applicable. -->
|
||||
|
||||
Fixes #
|
||||
|
||||
## Changes
|
||||
|
||||
<!-- Describe your changes concisely. What approach did you take? -->
|
||||
|
||||
-
|
||||
|
||||
## Test Plan
|
||||
|
||||
<!-- How did you verify your changes? Paste exact commands and output. -->
|
||||
|
||||
```bash
|
||||
# Commands you ran
|
||||
```
|
||||
|
||||
## Test Results
|
||||
|
||||
<!-- Paste test output, before/after comparisons, or SSIM scores for model changes. -->
|
||||
|
||||
<details>
|
||||
<summary>Test output</summary>
|
||||
|
||||
```
|
||||
# Paste output here
|
||||
```
|
||||
|
||||
</details>
|
||||
|
||||
## Checklist
|
||||
|
||||
- [ ] I ran `pre-commit run --all-files` and fixed all issues
|
||||
- [ ] I added or updated tests for my changes
|
||||
- [ ] I updated documentation if needed
|
||||
- [ ] I considered GPU memory impact of my changes
|
||||
|
||||
**For model/pipeline changes, also check:**
|
||||
- [ ] I verified SSIM regression tests pass
|
||||
- [ ] I updated the support matrix if adding a new model
|
||||
@@ -0,0 +1,324 @@
|
||||
merge_protections:
|
||||
- name: PR merge requirements
|
||||
if:
|
||||
- base = main
|
||||
success_conditions:
|
||||
- "title~=(?i)^\\[(feat|feature|bugfix|fix|refactor|perf|ci|doc|docs|misc|chore|kernel|new.?model|skill|skills|infra)\\]"
|
||||
- "#approved-reviews-by>=1"
|
||||
- check-success~=pre-commit
|
||||
- check-success=fastcheck-passed
|
||||
- check-success=full-suite-passed
|
||||
|
||||
pull_request_rules:
|
||||
|
||||
# ============================================================
|
||||
# Type labels (from PR title prefix)
|
||||
# ============================================================
|
||||
|
||||
- name: "label type: feat"
|
||||
conditions:
|
||||
- "title~=(?i)^\\[(feat|feature)\\]"
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["type: feat"]
|
||||
|
||||
- name: "label type: bugfix"
|
||||
conditions:
|
||||
- "title~=(?i)^\\[(bug)?fix\\]"
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["type: bugfix"]
|
||||
|
||||
- name: "label type: refactor"
|
||||
conditions:
|
||||
- "title~=(?i)^\\[refactor\\]"
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["type: refactor"]
|
||||
|
||||
- name: "label type: perf"
|
||||
conditions:
|
||||
- "title~=(?i)^\\[perf\\]"
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["type: perf"]
|
||||
|
||||
- name: "label type: ci"
|
||||
conditions:
|
||||
- "title~=(?i)^\\[ci\\]"
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["type: ci"]
|
||||
|
||||
- name: "label type: docs"
|
||||
conditions:
|
||||
- "title~=(?i)^\\[(doc|docs)\\]"
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["type: docs"]
|
||||
|
||||
- name: "label type: misc"
|
||||
conditions:
|
||||
- "title~=(?i)^\\[(misc|chore)\\]"
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["type: misc"]
|
||||
|
||||
- name: "label type: new-model"
|
||||
conditions:
|
||||
- "title~=(?i)^\\[new.?model\\]"
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["type: new-model"]
|
||||
|
||||
- name: "label type: infra"
|
||||
conditions:
|
||||
- "title~=(?i)^\\[infra\\]"
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["type: infra"]
|
||||
|
||||
- name: "label type: skill"
|
||||
conditions:
|
||||
- "title~=(?i)^\\[skills?\\]"
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["type: skill"]
|
||||
|
||||
# ============================================================
|
||||
# Scope labels (from changed files)
|
||||
# ============================================================
|
||||
|
||||
- name: "label scope: training"
|
||||
conditions:
|
||||
- or:
|
||||
- files~=^fastvideo/train/
|
||||
- files~=^fastvideo/training/
|
||||
- files~=^fastvideo/distillation/
|
||||
- files~=^examples/train/
|
||||
- files~=^examples/training/
|
||||
- files~=^examples/distill/
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["scope: training"]
|
||||
|
||||
- name: "label scope: inference"
|
||||
conditions:
|
||||
- or:
|
||||
- files~=^fastvideo/pipelines/basic/
|
||||
- files~=^fastvideo/pipelines/stages/
|
||||
- files~=^fastvideo/pipelines/samplers/
|
||||
- files~=^fastvideo/entrypoints/
|
||||
- files~=^fastvideo/worker/
|
||||
- files~=^fastvideo/api/sampling_param
|
||||
- files~=^fastvideo/configs/pipelines/
|
||||
- files~=^examples/inference/
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["scope: inference"]
|
||||
|
||||
- name: "label scope: attention"
|
||||
conditions:
|
||||
- files~=^fastvideo/attention/
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["scope: attention"]
|
||||
|
||||
- name: "label scope: kernel"
|
||||
conditions:
|
||||
- or:
|
||||
- files~=^fastvideo-kernel/
|
||||
- files~=^csrc/
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["scope: kernel"]
|
||||
|
||||
- name: "label scope: data"
|
||||
conditions:
|
||||
- or:
|
||||
- files~=^fastvideo/dataset/
|
||||
- files~=^fastvideo/pipelines/preprocess/
|
||||
- files~=^examples/preprocessing/
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["scope: data"]
|
||||
|
||||
- name: "label scope: infra"
|
||||
conditions:
|
||||
- or:
|
||||
- files~=^\.github/
|
||||
- files~=^\.buildkite/
|
||||
- files~=^fastvideo/tests/
|
||||
- files~=^docker/
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["scope: infra"]
|
||||
|
||||
- name: "label scope: distributed"
|
||||
conditions:
|
||||
- files~=^fastvideo/distributed/
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["scope: distributed"]
|
||||
|
||||
- name: "label scope: docs"
|
||||
conditions:
|
||||
- files~=^docs/
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["scope: docs"]
|
||||
|
||||
- name: "label scope: studio"
|
||||
conditions:
|
||||
- files~=^apps/fastvideo_studio/
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["scope: studio"]
|
||||
|
||||
- name: "label scope: model"
|
||||
conditions:
|
||||
- or:
|
||||
- files~=^fastvideo/models/
|
||||
- files~=^fastvideo/layers/
|
||||
- files~=^fastvideo/configs/models/
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["scope: model"]
|
||||
|
||||
# ============================================================
|
||||
# Pre-commit failure help comment
|
||||
# ============================================================
|
||||
|
||||
- name: comment on pre-commit failure
|
||||
conditions:
|
||||
- check-failure~=pre-commit
|
||||
- -closed
|
||||
actions:
|
||||
comment:
|
||||
message: |
|
||||
## Pre-commit checks failed
|
||||
|
||||
Hi @{{author}}, the pre-commit checks have failed. To fix them locally:
|
||||
|
||||
```bash
|
||||
# Install pre-commit if you haven't already
|
||||
uv pip install pre-commit
|
||||
pre-commit install
|
||||
|
||||
# Run all checks and auto-fix what's possible
|
||||
pre-commit run --all-files
|
||||
```
|
||||
|
||||
Common fixes:
|
||||
- **yapf**: `yapf -i <file>` (formatting)
|
||||
- **ruff**: `ruff check --fix <file>` (linting)
|
||||
- **codespell**: `codespell --write-changes <file>` (spelling)
|
||||
|
||||
After fixing, commit and push the changes. The checks will re-run automatically.
|
||||
|
||||
For future commits, `pre-commit` will run automatically on changed files before each commit.
|
||||
|
||||
|
||||
# ============================================================
|
||||
# Merge conflict detection
|
||||
# ============================================================
|
||||
|
||||
- name: label conflicting PRs
|
||||
conditions:
|
||||
- conflict
|
||||
- -closed
|
||||
- label!=stale
|
||||
actions:
|
||||
label:
|
||||
add: [needs-rebase]
|
||||
comment:
|
||||
message: |
|
||||
This PR has merge conflicts with the base branch. Please rebase:
|
||||
|
||||
```bash
|
||||
git fetch origin main
|
||||
git rebase origin/main
|
||||
# Resolve any conflicts, then:
|
||||
git push --force-with-lease
|
||||
```
|
||||
|
||||
- name: remove conflict label when resolved
|
||||
conditions:
|
||||
- -conflict
|
||||
- -closed
|
||||
- label=needs-rebase
|
||||
actions:
|
||||
label:
|
||||
remove: [needs-rebase]
|
||||
|
||||
# ============================================================
|
||||
# Auto-merge
|
||||
# ============================================================
|
||||
|
||||
- name: auto-merge when ready and all checks pass
|
||||
conditions:
|
||||
- label=ready
|
||||
- "title~=(?i)^\\[(feat|feature|bugfix|fix|refactor|perf|ci|doc|docs|misc|chore|kernel|new.?model|skill|skills|infra)\\]"
|
||||
- "#approved-reviews-by>=1"
|
||||
- check-success~=pre-commit
|
||||
- check-success=fastcheck-passed
|
||||
- check-success=full-suite-passed
|
||||
- -conflict
|
||||
- -closed
|
||||
- -draft
|
||||
actions:
|
||||
merge:
|
||||
method: squash
|
||||
|
||||
# ============================================================
|
||||
# PR title format help
|
||||
# ============================================================
|
||||
|
||||
- name: comment on invalid PR title format
|
||||
conditions:
|
||||
- -closed
|
||||
- -draft
|
||||
- "-title~=(?i)^\\[(feat|feature|bugfix|fix|refactor|perf|ci|doc|docs|misc|chore|kernel|new.?model|skill|skills|infra)\\]"
|
||||
actions:
|
||||
comment:
|
||||
message: |
|
||||
## ⚠️ PR title format required
|
||||
|
||||
Your PR title must start with a type tag in brackets. Examples:
|
||||
- `[feat] Add new model support`
|
||||
- `[bugfix] Fix VAE tiling corruption`
|
||||
- `[refactor] Restructure training pipeline`
|
||||
- `[perf] Optimize attention kernel`
|
||||
- `[ci] Update test infrastructure`
|
||||
- `[infra] Add activation trace hooks`
|
||||
- `[docs] Add inference guide`
|
||||
- `[misc] Clean up configs`
|
||||
- `[new-model] Port Flux2 to FastVideo`
|
||||
- `[skill] Add add-model agent skill`
|
||||
|
||||
Valid tags: `feat`, `feature`, `bugfix`, `fix`, `refactor`, `perf`, `ci`, `infra`, `doc`, `docs`, `misc`, `chore`, `kernel`, `new-model`, `skill`, `skills`
|
||||
|
||||
Please update your PR title and the merge protection check will pass automatically.
|
||||
|
||||
merge_protections_settings:
|
||||
reporting_method: check-runs
|
||||
Executable
+133
@@ -0,0 +1,133 @@
|
||||
#!/usr/bin/env bash
|
||||
# Gate the path-aware Buildkite merge plan on the cheap GitHub checks.
|
||||
#
|
||||
# Polls the workflow runs for the PR head commit and only exits 0 once the
|
||||
# watched cheap workflows (pre-commit, docs build) have succeeded, so the
|
||||
# 'ready' label cannot burn path-selected GPU lanes on a head that a cheap
|
||||
# check has already doomed.
|
||||
#
|
||||
# Semantics:
|
||||
# - watched run completed with a bad conclusion -> exit 1 (fail CLOSED:
|
||||
# no merge gate; the next push re-arms via the 'synchronize' trigger)
|
||||
# - watched run cancelled -> still pending: the docs
|
||||
# workflow's repo-global 'pages' concurrency group cancels runs superseded
|
||||
# by unrelated pushes, so 'cancelled' is not a verdict on this PR
|
||||
# - watched runs pending -> poll until done
|
||||
# - docs run absent -> not applicable after a
|
||||
# short grace period ('Deploy Documentation' is path-filtered on PRs)
|
||||
# - pre-commit run absent -> keep polling: pre-commit
|
||||
# is never path-filtered, so its absence is always anomalous
|
||||
# - 'ready' label removed while waiting -> exit 1 (fail CLOSED:
|
||||
# un-labeling is a deliberate maintainer action)
|
||||
# - GitHub API unreachable or timeout -> exit 0 (fail OPEN,
|
||||
# loud warning: never brick CI on a GitHub outage)
|
||||
#
|
||||
# Required env: PR_SHA (PR head commit), PR_NUMBER, GITHUB_REPOSITORY, GH_TOKEN.
|
||||
set -euo pipefail
|
||||
|
||||
: "${PR_SHA:?PR_SHA (PR head commit) is required}"
|
||||
: "${PR_NUMBER:?PR_NUMBER (pull request number) is required}"
|
||||
: "${GITHUB_REPOSITORY:?GITHUB_REPOSITORY is required}"
|
||||
|
||||
# Workflow-level `name:` values that must be green before the merge gate
|
||||
# may start. "Deploy Documentation" is path-filtered on PRs, so its run may
|
||||
# legitimately never exist; pre-commit always runs, so it must appear.
|
||||
WATCHED_NAMES='["pre-commit", "Deploy Documentation"]'
|
||||
WATCHED_REGEX='^(pre-commit|Deploy Documentation)$'
|
||||
POLL_SECS="${POLL_SECS:-20}"
|
||||
GRACE_SECS="${GRACE_SECS:-60}"
|
||||
MAX_WAIT_SECS="${MAX_WAIT_SECS:-1500}"
|
||||
|
||||
# Bound each API call so a hung connection hits the 3-strike fail-open path
|
||||
# instead of pinning the loop until the job timeout (which would fail closed
|
||||
# on exactly the GitHub-outage case this script is meant to survive).
|
||||
if command -v timeout >/dev/null 2>&1; then
|
||||
gh_api() { timeout 30 gh api "$@"; }
|
||||
else
|
||||
gh_api() { gh api "$@"; } # macOS dev boxes; CI always has coreutils timeout
|
||||
fi
|
||||
|
||||
# The workflow checked the label before starting the gate, but the wait can
|
||||
# last ~25 min: re-check once before any exit 0 and fail closed if 'ready'
|
||||
# was removed in the meantime. An API error here proceeds (the label was
|
||||
# present when the gate started; never brick CI on an outage).
|
||||
recheck_ready_label() {
|
||||
local pr_json
|
||||
if pr_json=$(gh_api "repos/${GITHUB_REPOSITORY}/pulls/${PR_NUMBER}" 2>/dev/null); then
|
||||
if ! jq -e '[.labels[]?.name] | index("ready")' <<<"$pr_json" >/dev/null 2>&1; then
|
||||
echo "::error::PR #${PR_NUMBER} no longer has the 'ready' label —" \
|
||||
"NOT triggering the Buildkite merge gate. Re-add the label to re-arm."
|
||||
exit 1
|
||||
fi
|
||||
else
|
||||
echo "::warning::Could not re-check the 'ready' label on PR #${PR_NUMBER}; proceeding (it was present when the gate started)."
|
||||
fi
|
||||
}
|
||||
|
||||
start=$(date +%s)
|
||||
api_fails=0
|
||||
missing=""
|
||||
|
||||
while true; do
|
||||
elapsed=$(( $(date +%s) - start ))
|
||||
|
||||
if runs_json=$(gh_api "repos/${GITHUB_REPOSITORY}/actions/runs?head_sha=${PR_SHA}&per_page=100" 2>/dev/null) \
|
||||
&& state=$(jq --arg re "$WATCHED_REGEX" '
|
||||
[.workflow_runs[]? | select(.name // "" | test($re))]
|
||||
| group_by(.name) | map(max_by(.id))
|
||||
| map({name, status, conclusion})' <<<"$runs_json" 2>/dev/null); then
|
||||
api_fails=0
|
||||
echo "t+${elapsed}s watched checks: $(jq -c . <<<"$state")"
|
||||
|
||||
failed=$(jq -r '[.[] | select(.status == "completed"
|
||||
and (.conclusion | IN("success", "skipped", "neutral", "cancelled") | not))]
|
||||
| map(.name) | join(", ")' <<<"$state")
|
||||
if [ -n "$failed" ]; then
|
||||
echo "::error::Cheap check(s) failed on ${PR_SHA}: ${failed}." \
|
||||
"NOT triggering the Buildkite merge gate. Push a fix (the 'ready'" \
|
||||
"label re-arms on every push), or re-run the failed check and then" \
|
||||
"re-run this workflow."
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# 'cancelled' counts as pending: wait for a re-run to reach a real verdict
|
||||
# (bounded by MAX_WAIT, then the fail-open below).
|
||||
pending=$(jq '[.[] | select(.status != "completed" or .conclusion == "cancelled")] | length' <<<"$state")
|
||||
missing=$(jq -r --argjson watched "$WATCHED_NAMES" '($watched - map(.name)) | join(", ")' <<<"$state")
|
||||
if [ "$pending" -eq 0 ]; then
|
||||
if [ -z "$missing" ]; then
|
||||
recheck_ready_label
|
||||
echo "All watched cheap checks are green — merge gate may proceed."
|
||||
exit 0
|
||||
fi
|
||||
case "$missing" in
|
||||
*pre-commit*)
|
||||
echo "pre-commit run not found for ${PR_SHA} yet; waiting (pre-commit is never path-filtered, so its absence is anomalous)."
|
||||
;;
|
||||
*)
|
||||
if [ "$elapsed" -ge "$GRACE_SECS" ]; then
|
||||
recheck_ready_label
|
||||
echo "::warning::Watched run(s) never appeared for ${PR_SHA}: ${missing} (path-filtered, likely not applicable). Proceeding on the checks that did run."
|
||||
exit 0
|
||||
fi
|
||||
echo "Waiting up to ${GRACE_SECS}s grace for path-filtered run(s) to appear: ${missing}."
|
||||
;;
|
||||
esac
|
||||
fi
|
||||
else
|
||||
api_fails=$(( api_fails + 1 ))
|
||||
echo "::warning::GitHub API error querying workflow runs for ${PR_SHA} (attempt ${api_fails}/3)."
|
||||
if [ "$api_fails" -ge 3 ]; then
|
||||
recheck_ready_label
|
||||
echo "::warning::FAILING OPEN: cannot query GitHub check status — triggering the merge gate WITHOUT the cheap-check gate."
|
||||
exit 0
|
||||
fi
|
||||
fi
|
||||
|
||||
if [ "$elapsed" -ge "$MAX_WAIT_SECS" ]; then
|
||||
recheck_ready_label
|
||||
echo "::warning::FAILING OPEN: watched checks still pending after $(( MAX_WAIT_SECS / 60 )) min${missing:+ (never appeared: ${missing})} — triggering the merge gate anyway."
|
||||
exit 0
|
||||
fi
|
||||
sleep "$POLL_SECS"
|
||||
done
|
||||
@@ -0,0 +1,592 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Select the additive GPU integration lanes needed by a PR diff.
|
||||
|
||||
Fastcheck is the universal six-lane baseline and is intentionally not repeated
|
||||
here. This planner selects only the more expensive merge-gate lanes. Unknown
|
||||
source/build paths fail closed to the complete integration set, while explicit
|
||||
documentation and repository-metadata paths require no additional GPU work.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import fnmatch
|
||||
import re
|
||||
from dataclasses import dataclass, field
|
||||
from pathlib import Path
|
||||
from typing import TextIO
|
||||
|
||||
MERGE_LANES = (
|
||||
"golden-gate",
|
||||
"ssim",
|
||||
"lora-inference",
|
||||
"lora-extraction",
|
||||
"training",
|
||||
"distillation",
|
||||
"self-forcing",
|
||||
"lora-training",
|
||||
"training-vsa",
|
||||
"inference-vmoba",
|
||||
"performance",
|
||||
"api-server",
|
||||
"train-framework",
|
||||
"eval",
|
||||
)
|
||||
|
||||
LANE_SCRIPT_TO_KEY = {
|
||||
"api_server.sh": "api-server",
|
||||
"distillation_dmd.sh": "distillation",
|
||||
"eval.sh": "eval",
|
||||
"golden_gate.sh": "golden-gate",
|
||||
"inference_lora.sh": "lora-inference",
|
||||
"inference_vmoba.sh": "inference-vmoba",
|
||||
"lora_extraction.sh": "lora-extraction",
|
||||
"performance.sh": "performance",
|
||||
"self_forcing.sh": "self-forcing",
|
||||
"ssim.sh": "ssim",
|
||||
"train_framework.sh": "train-framework",
|
||||
"training.sh": "training",
|
||||
"training_lora.sh": "lora-training",
|
||||
"training_vsa.sh": "training-vsa",
|
||||
}
|
||||
|
||||
FASTCHECK_LANE_SCRIPTS = {
|
||||
"dreamverse.sh",
|
||||
"encoder.sh",
|
||||
"kernel_tests.sh",
|
||||
"transformer.sh",
|
||||
"vae.sh",
|
||||
}
|
||||
|
||||
LEGACY_TRAINING_LANES = (
|
||||
"training",
|
||||
"distillation",
|
||||
"self-forcing",
|
||||
"lora-training",
|
||||
"training-vsa",
|
||||
)
|
||||
|
||||
ALL_TRAINING_LANES = (*LEGACY_TRAINING_LANES, "train-framework")
|
||||
|
||||
SSIM_SMOKE_TESTS = (
|
||||
"test_flux_t2i_similarity.py",
|
||||
"test_wan_t2v_similarity.py",
|
||||
)
|
||||
|
||||
SAFE_PATTERNS = (
|
||||
"*.md",
|
||||
"*.rst",
|
||||
".agents/**",
|
||||
".claude/**",
|
||||
".codex/**",
|
||||
".github/ISSUE_TEMPLATE/**",
|
||||
".github/PULL_REQUEST_TEMPLATE.md",
|
||||
".github/dependabot.yml",
|
||||
".github/mergify.yml",
|
||||
".github/scripts/**",
|
||||
".github/workflows/**",
|
||||
".buildkite/scripts/pre_commit.sh",
|
||||
".git-blame-ignore-revs",
|
||||
".gitattributes",
|
||||
".gitignore",
|
||||
".pre-commit-config.yaml",
|
||||
"AGENTS.md",
|
||||
"CITATION.cff",
|
||||
"CODE_OF_CONDUCT.md",
|
||||
"CONTRIBUTING.md",
|
||||
"LICENSE",
|
||||
"NOTICE",
|
||||
"__init__.py",
|
||||
"collect_env.py",
|
||||
"SECURITY.md",
|
||||
"assets/**",
|
||||
"comfyui/**",
|
||||
"docs/**",
|
||||
"examples/**",
|
||||
"mkdocs.yml",
|
||||
"requirements-mkdocs.in",
|
||||
"requirements-mkdocs.txt",
|
||||
"scripts/**",
|
||||
"tests/__init__.py",
|
||||
"tests/local_tests/**",
|
||||
)
|
||||
|
||||
ALL_IMPACT_PATTERNS = (
|
||||
".buildkite/pipeline.yml",
|
||||
"docker/**",
|
||||
"pyproject.toml",
|
||||
"requirements*.txt",
|
||||
"setup.cfg",
|
||||
"setup.py",
|
||||
"uv.lock",
|
||||
)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class FamilyCoverage:
|
||||
pattern: re.Pattern[str]
|
||||
golden_tests: tuple[str, ...]
|
||||
ssim_tests: tuple[str, ...]
|
||||
|
||||
|
||||
FAMILY_COVERAGE = (
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])dreamx(_world)?([/_.-]|$)"),
|
||||
("test_dreamx.py", ),
|
||||
("test_dreamx_world_similarity.py", ),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])flux[_-]?2([/_.-]|$)"),
|
||||
("test_flux2_klein.py", ),
|
||||
("test_flux2_similarity.py", ),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])flux(?![_-]?2)([/_.-]|$)"),
|
||||
("test_flux.py", ),
|
||||
("test_flux_t2i_similarity.py", ),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])(hunyuan)?gamecraft([/_.-]|$)"),
|
||||
("test_gamecraft.py", ),
|
||||
("test_gamecraft_similarity.py", ),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])gen3c([/_.-]|$)"),
|
||||
("test_gen3c.py", ),
|
||||
("test_gen3c_similarity.py", ),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])glm[_-]?image([/_.-]|$)"),
|
||||
("test_glm_image.py", ),
|
||||
("test_glm_image_similarity.py", ),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])hunyuan(video)?15([a-z0-9_-]*)([/_.-]|$)"),
|
||||
(),
|
||||
("test_hunyuan15_i2v_similarity.py", ),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])kandinsky[_-]?5([/_.-]|$)"),
|
||||
("test_kandinsky5.py", ),
|
||||
("test_kandinsky5_similarity.py", ),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])lingbot([a-z0-9_-]*)([/_.-]|$)"),
|
||||
("test_lingbot.py", ),
|
||||
("test_lingbot_similarity.py", ),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])longcat([/_.-]|$)"),
|
||||
("test_longcat.py", ),
|
||||
("test_longcat_similarity.py", ),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])ltx[_-]?2([/_.-]|$)"),
|
||||
("test_ltx2.py", ),
|
||||
("test_ltx2_similarity.py", ),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])matrixgame[_-]?2([/_.-]|$)"),
|
||||
("test_matrixgame.py", ),
|
||||
("test_matrixgame2_similarity.py", ),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])matrixgame[_-]?3([/_.-]|$)"),
|
||||
("test_matrixgame.py", ),
|
||||
("test_matrixgame3_similarity.py", ),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])minimax[_-]?h3([/_.-]|$)"),
|
||||
("test_minimax_h3_t2v.py", ),
|
||||
("test_minimax_h3_similarity.py", ),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])sd[_-]?3([._-]?5)?([/_.-]|$)"),
|
||||
("test_sd35.py", ),
|
||||
("test_sd35_similarity.py", ),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])stable[_-]?audio([/_.-]|$)"),
|
||||
("test_stable_audio.py", ),
|
||||
("test_stable_audio_similarity.py", ),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])turbo(diffusion)?([/_.-]|$)"),
|
||||
(),
|
||||
("test_turbodiffusion_similarity.py", ),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])wan(video|vae)?([/_.-]|$)"),
|
||||
("test_wan_t2v.py", "test_wan_vae.py", "test_wan_causal.py", "test_wan_denoising.py"),
|
||||
(
|
||||
"test_causal_similarity.py",
|
||||
"test_wan_i2v_similarity.py",
|
||||
"test_wan_t2v_similarity.py",
|
||||
"test_wan_ti2v_similarity.py",
|
||||
),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])z[_-]?image([/_.-]|$)"),
|
||||
("test_zimage.py", ),
|
||||
("test_zimage_similarity.py", ),
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
@dataclass
|
||||
class MergePlan:
|
||||
lanes: set[str] = field(default_factory=set)
|
||||
golden_tests: set[str] = field(default_factory=set)
|
||||
ssim_tests: set[str] = field(default_factory=set)
|
||||
golden_all: bool = False
|
||||
ssim_all: bool = False
|
||||
reasons: list[str] = field(default_factory=list)
|
||||
|
||||
def add_lanes(self, *lanes: str, reason: str) -> None:
|
||||
unknown = set(lanes) - set(MERGE_LANES)
|
||||
if unknown:
|
||||
raise ValueError(f"Unknown merge lanes: {sorted(unknown)}")
|
||||
self.lanes.update(lanes)
|
||||
self.reasons.append(reason)
|
||||
|
||||
def add_golden(self, tests: tuple[str, ...], reason: str) -> None:
|
||||
self.add_lanes("golden-gate", reason=reason)
|
||||
self.golden_tests.update(tests)
|
||||
|
||||
def add_ssim(self, tests: tuple[str, ...], reason: str) -> None:
|
||||
self.add_lanes("ssim", reason=reason)
|
||||
self.ssim_tests.update(tests)
|
||||
|
||||
def require_all(self, reason: str) -> None:
|
||||
self.lanes.update(MERGE_LANES)
|
||||
self.golden_all = True
|
||||
self.ssim_all = True
|
||||
self.reasons.append(reason)
|
||||
|
||||
def ordered_lanes(self) -> tuple[str, ...]:
|
||||
return tuple(lane for lane in MERGE_LANES if lane in self.lanes)
|
||||
|
||||
def encoded_lanes(self) -> str:
|
||||
lanes = self.ordered_lanes()
|
||||
return "," + ",".join(lanes or ("none", )) + ","
|
||||
|
||||
def encoded_golden_tests(self) -> str:
|
||||
if "golden-gate" not in self.lanes:
|
||||
return "none"
|
||||
if self.golden_all or not self.golden_tests:
|
||||
return "all"
|
||||
return ",".join(sorted(self.golden_tests))
|
||||
|
||||
def encoded_ssim_tests(self) -> str:
|
||||
if "ssim" not in self.lanes:
|
||||
return "none"
|
||||
if self.ssim_all or not self.ssim_tests:
|
||||
return "all"
|
||||
return ",".join(sorted(self.ssim_tests))
|
||||
|
||||
|
||||
def _matches_any(path: str, patterns: tuple[str, ...]) -> bool:
|
||||
return any(fnmatch.fnmatchcase(path, pattern) for pattern in patterns)
|
||||
|
||||
|
||||
def _family_coverage(path: str) -> tuple[set[str], set[str]]:
|
||||
normalized = path.lower()
|
||||
golden: set[str] = set()
|
||||
ssim: set[str] = set()
|
||||
for family in FAMILY_COVERAGE:
|
||||
if family.pattern.search(normalized):
|
||||
golden.update(family.golden_tests)
|
||||
ssim.update(family.ssim_tests)
|
||||
# Select the component actually touched, including compatibility paths.
|
||||
# Family configs/pipeline wiring can affect all four Wan gates.
|
||||
if re.search(r"(^|[/_.-])wan(video|vae)?([/_.-]|$)", normalized):
|
||||
if (normalized.endswith(("/wan/vae.py", "/wan/vae_config.py", "/vaes/wanvae.py"))
|
||||
or normalized.endswith("/wan/stages/conditioning.py")):
|
||||
golden = {"test_wan_vae.py"}
|
||||
elif normalized.endswith(("/wan/causal_transformer.py", "/dits/causal_wanvideo.py",
|
||||
"/wan/stages/causal_denoising.py")):
|
||||
golden = {"test_wan_causal.py"}
|
||||
elif (normalized == "fastvideo/models/dits/wanvideo.py"
|
||||
or normalized.endswith(("/wan/transformer.py", "/wan/stages/denoising.py", "/wan/stages/dmd.py"))):
|
||||
golden = {"test_wan_t2v.py", "test_wan_denoising.py"}
|
||||
return golden, ssim
|
||||
|
||||
|
||||
def _select_output_coverage(plan: MergePlan, path: str) -> None:
|
||||
golden, ssim = _family_coverage(path)
|
||||
if golden:
|
||||
plan.add_golden(tuple(sorted(golden)), reason=f"model-family golden coverage: {path}")
|
||||
else:
|
||||
plan.golden_all = True
|
||||
plan.add_lanes("golden-gate", reason=f"shared output golden coverage: {path}")
|
||||
if ssim:
|
||||
plan.add_ssim(tuple(sorted(ssim)), reason=f"model-family SSIM coverage: {path}")
|
||||
else:
|
||||
plan.add_ssim(SSIM_SMOKE_TESTS, reason=f"shared output SSIM smoke coverage: {path}")
|
||||
|
||||
|
||||
def classify_paths(paths: list[str]) -> MergePlan:
|
||||
plan = MergePlan()
|
||||
normalized_paths: list[str] = []
|
||||
for raw_path in paths:
|
||||
path = raw_path.strip()
|
||||
while path.startswith("./"):
|
||||
path = path[2:]
|
||||
if path:
|
||||
normalized_paths.append(path)
|
||||
normalized_paths = sorted(set(normalized_paths))
|
||||
if not normalized_paths:
|
||||
plan.require_all("changed-file list was empty; failing closed")
|
||||
return plan
|
||||
|
||||
for path in normalized_paths:
|
||||
if path == "__FASTVIDEO_CI_PLAN_ALL__":
|
||||
plan.require_all("changed-file API failed; failing closed")
|
||||
continue
|
||||
|
||||
if path in {"requirements-mkdocs.in", "requirements-mkdocs.txt"}:
|
||||
plan.reasons.append(f"documentation dependencies need no GPU integration: {path}")
|
||||
continue
|
||||
|
||||
if _matches_any(path, ALL_IMPACT_PATTERNS):
|
||||
plan.require_all(f"cross-cutting build/runtime surface: {path}")
|
||||
continue
|
||||
|
||||
lane_script_prefix = ".buildkite/scripts/lanes/"
|
||||
if path.startswith(lane_script_prefix):
|
||||
script_name = Path(path).name
|
||||
lane = LANE_SCRIPT_TO_KEY.get(script_name)
|
||||
if lane is None:
|
||||
if script_name in FASTCHECK_LANE_SCRIPTS:
|
||||
plan.reasons.append(f"covered by automatic Fastcheck lane: {path}")
|
||||
else:
|
||||
plan.require_all(f"unknown lane script: {path}")
|
||||
elif lane == "golden-gate":
|
||||
plan.golden_all = True
|
||||
plan.add_lanes(lane, reason=f"golden lane implementation: {path}")
|
||||
elif lane == "ssim":
|
||||
plan.ssim_all = True
|
||||
plan.add_lanes(lane, reason=f"SSIM lane implementation: {path}")
|
||||
else:
|
||||
plan.add_lanes(lane, reason=f"lane implementation: {path}")
|
||||
continue
|
||||
|
||||
if path.startswith("fastvideo/tests/golden_gate/"):
|
||||
name = Path(path).name
|
||||
if name.startswith("test_") and name.endswith(".py"):
|
||||
plan.add_golden((name, ), reason=f"changed golden test: {path}")
|
||||
elif name in {"AGENTS.md", "README.md"}:
|
||||
plan.reasons.append(f"golden documentation only: {path}")
|
||||
else:
|
||||
plan.golden_all = True
|
||||
plan.add_lanes("golden-gate", reason=f"shared golden harness/reference: {path}")
|
||||
continue
|
||||
|
||||
if path.startswith("fastvideo/tests/ssim/"):
|
||||
name = Path(path).name
|
||||
if name.startswith("test_") and name.endswith(".py"):
|
||||
plan.add_ssim((name, ), reason=f"changed SSIM test: {path}")
|
||||
elif path.endswith((".py", ".json", ".pt", ".png", ".mp4")):
|
||||
plan.ssim_all = True
|
||||
plan.add_lanes("ssim", reason=f"shared SSIM harness/reference: {path}")
|
||||
continue
|
||||
|
||||
if path.startswith("fastvideo/tests/performance/") or path.startswith(".buildkite/performance-benchmarks/"):
|
||||
plan.add_lanes("performance", reason=f"performance coverage: {path}")
|
||||
continue
|
||||
if path.startswith(("fastvideo/performance/", "fastvideo/performance_dashboard/",
|
||||
"apps/performance_dashboard/")):
|
||||
plan.add_lanes("performance", reason=f"performance implementation: {path}")
|
||||
continue
|
||||
if path.startswith("fastvideo/benchmarks/"):
|
||||
if "/mlx_" in path or Path(path).name.startswith("mlx_"):
|
||||
plan.reasons.append(f"covered by the path-filtered macOS MLX workflow: {path}")
|
||||
else:
|
||||
plan.add_lanes("performance", reason=f"benchmark implementation: {path}")
|
||||
continue
|
||||
if path.startswith("fastvideo/tests/eval/") or path.startswith("fastvideo/eval/"):
|
||||
plan.add_lanes("eval", reason=f"evaluation coverage: {path}")
|
||||
continue
|
||||
if path.startswith("fastvideo/third_party/eval/"):
|
||||
plan.add_lanes("eval", reason=f"vendored evaluation implementation: {path}")
|
||||
continue
|
||||
if path.startswith("fastvideo/tests/lora_extraction/") or path.startswith("scripts/lora_extraction/"):
|
||||
plan.add_lanes("lora-extraction", reason=f"LoRA extraction coverage: {path}")
|
||||
continue
|
||||
if path.startswith("fastvideo/tests/inference/lora/"):
|
||||
plan.add_lanes("lora-inference", reason=f"LoRA inference coverage: {path}")
|
||||
continue
|
||||
if path.startswith("fastvideo/tests/inference/vmoba/"):
|
||||
plan.add_lanes("inference-vmoba", reason=f"VMoBA inference coverage: {path}")
|
||||
continue
|
||||
if path.startswith(("fastvideo/dataset/", "fastvideo/workflow/", "fastvideo/pipelines/preprocess/",
|
||||
"fastvideo/pipelines/training/")):
|
||||
plan.add_lanes(*ALL_TRAINING_LANES, reason=f"shared data/training input surface: {path}")
|
||||
continue
|
||||
if path.startswith("fastvideo/tests/train/") or path.startswith("fastvideo/train/"):
|
||||
plan.add_lanes("train-framework", reason=f"modular training coverage: {path}")
|
||||
continue
|
||||
|
||||
if path.startswith("fastvideo/tests/training/"):
|
||||
lowered = path.lower()
|
||||
if "/vanilla/" in lowered:
|
||||
plan.add_lanes("training", reason=f"vanilla training coverage: {path}")
|
||||
elif "/distill/" in lowered:
|
||||
plan.add_lanes("distillation", reason=f"distillation coverage: {path}")
|
||||
elif "/self-forcing/" in lowered:
|
||||
plan.add_lanes("self-forcing", reason=f"self-forcing coverage: {path}")
|
||||
elif "/lora/" in lowered:
|
||||
plan.add_lanes("lora-training", reason=f"LoRA training coverage: {path}")
|
||||
elif "/vsa/" in lowered:
|
||||
plan.add_lanes("training-vsa", reason=f"VSA training coverage: {path}")
|
||||
else:
|
||||
plan.add_lanes(*LEGACY_TRAINING_LANES, reason=f"shared legacy training coverage: {path}")
|
||||
continue
|
||||
|
||||
if path.startswith("fastvideo/training/"):
|
||||
lowered = path.lower()
|
||||
if "self_forcing" in lowered:
|
||||
plan.add_lanes("self-forcing", reason=f"self-forcing implementation: {path}")
|
||||
elif "distill" in lowered:
|
||||
plan.add_lanes("distillation", reason=f"distillation implementation: {path}")
|
||||
elif "lora" in lowered:
|
||||
plan.add_lanes("lora-training", reason=f"LoRA training implementation: {path}")
|
||||
else:
|
||||
plan.add_lanes(*LEGACY_TRAINING_LANES, reason=f"shared legacy training implementation: {path}")
|
||||
continue
|
||||
|
||||
lowered = path.lower()
|
||||
if "vmoba" in lowered and path.startswith(("fastvideo/", ".buildkite/")):
|
||||
plan.add_lanes("inference-vmoba", reason=f"VMoBA implementation: {path}")
|
||||
plan.add_golden(("test_wan_t2v.py", ), reason=f"VMoBA end-to-end coverage: {path}")
|
||||
continue
|
||||
if "lora" in lowered and path.startswith("fastvideo/"):
|
||||
plan.add_lanes(
|
||||
"lora-inference",
|
||||
"lora-extraction",
|
||||
"lora-training",
|
||||
reason=f"shared LoRA implementation: {path}",
|
||||
)
|
||||
_select_output_coverage(plan, path)
|
||||
continue
|
||||
|
||||
if path.startswith("fastvideo/entrypoints/") or path.startswith("fastvideo/api/"):
|
||||
plan.add_lanes("api-server", reason=f"API/entrypoint integration: {path}")
|
||||
if "openai" not in lowered and "/cli/" not in lowered:
|
||||
_select_output_coverage(plan, path)
|
||||
continue
|
||||
if path.startswith("fastvideo/worker/"):
|
||||
plan.add_lanes("api-server", reason=f"worker/API integration: {path}")
|
||||
_select_output_coverage(plan, path)
|
||||
continue
|
||||
if path.startswith("fastvideo/distributed/"):
|
||||
plan.add_lanes(
|
||||
"training",
|
||||
"train-framework",
|
||||
reason=f"distributed runtime integration: {path}",
|
||||
)
|
||||
_select_output_coverage(plan, path)
|
||||
continue
|
||||
if path.startswith(("fastvideo/hooks/", "fastvideo/platforms/", "fastvideo/third_party/")):
|
||||
_select_output_coverage(plan, path)
|
||||
continue
|
||||
if path.startswith(("fastvideo/models/", "fastvideo/pipelines/", "fastvideo/configs/",
|
||||
"fastvideo/layers/", "fastvideo/attention/")):
|
||||
_select_output_coverage(plan, path)
|
||||
continue
|
||||
if path in {
|
||||
"fastvideo/fastvideo_args.py",
|
||||
"fastvideo/forward_context.py",
|
||||
"fastvideo/image_processor.py",
|
||||
"fastvideo/registry.py",
|
||||
"fastvideo/utils.py",
|
||||
}:
|
||||
_select_output_coverage(plan, path)
|
||||
continue
|
||||
if path.startswith("fastvideo/mlx_runtime/"):
|
||||
plan.reasons.append(f"covered by the path-filtered macOS MLX workflow: {path}")
|
||||
continue
|
||||
if path.startswith("fastvideo/logging_utils/") or path in {
|
||||
"fastvideo/__init__.py",
|
||||
"fastvideo/envs.py",
|
||||
"fastvideo/logger.py",
|
||||
"fastvideo/profiler.py",
|
||||
"fastvideo/version.py",
|
||||
}:
|
||||
plan.reasons.append(f"covered by automatic Fastcheck: {path}")
|
||||
continue
|
||||
if path.startswith(("fastvideo-kernel/", "csrc/")):
|
||||
plan.add_golden(("test_wan_t2v.py", ), reason=f"kernel integration smoke: {path}")
|
||||
plan.add_ssim(("test_wan_t2v_similarity.py", ), reason=f"kernel numerical smoke: {path}")
|
||||
continue
|
||||
|
||||
if path.startswith("apps/dreamverse/"):
|
||||
# DreamVerse is already one of the six automatic Fastcheck lanes.
|
||||
plan.reasons.append(f"covered by automatic DreamVerse Fastcheck: {path}")
|
||||
continue
|
||||
if path.startswith("apps/infinite_livestream/"):
|
||||
# The app's CPU-only tests run in the automatic unit Fastcheck lane.
|
||||
plan.reasons.append(f"covered by automatic unit Fastcheck: {path}")
|
||||
continue
|
||||
if path.startswith("fastvideo/tests/"):
|
||||
# The automatic unit/component Fastcheck lanes own the remaining
|
||||
# package tests. Domain-specific expensive test roots were handled
|
||||
# above.
|
||||
plan.reasons.append(f"covered by automatic Fastcheck: {path}")
|
||||
continue
|
||||
if path in {".buildkite/scripts/unit_test.sh", ".buildkite/scripts/pr_test.sh"}:
|
||||
plan.reasons.append(f"covered by automatic unit Fastcheck: {path}")
|
||||
continue
|
||||
if _matches_any(path, SAFE_PATTERNS):
|
||||
plan.reasons.append(f"no additional GPU integration needed: {path}")
|
||||
continue
|
||||
|
||||
plan.require_all(f"unclassified path; failing closed: {path}")
|
||||
|
||||
return plan
|
||||
|
||||
|
||||
def _write_github_output(output: TextIO, plan: MergePlan) -> None:
|
||||
output.write(f"merge_test_plan={plan.encoded_lanes()}\n")
|
||||
output.write(f"merge_golden_tests={plan.encoded_golden_tests()}\n")
|
||||
output.write(f"merge_ssim_tests={plan.encoded_ssim_tests()}\n")
|
||||
output.write(f"merge_plan_label={','.join(plan.ordered_lanes()) or 'none'}\n")
|
||||
|
||||
|
||||
def _write_summary(output: TextIO, plan: MergePlan) -> None:
|
||||
output.write("## Change-aware merge test plan\n\n")
|
||||
output.write("| Selection | Value |\n|---|---|\n")
|
||||
output.write(f"| Additional Slurm lanes | `{','.join(plan.ordered_lanes()) or 'none'}` |\n")
|
||||
output.write(f"| Golden tests | `{plan.encoded_golden_tests()}` |\n")
|
||||
output.write(f"| SSIM tests | `{plan.encoded_ssim_tests()}` |\n\n")
|
||||
output.write("Fastcheck remains the universal six-lane baseline.\n")
|
||||
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("--paths-file", type=Path, required=True)
|
||||
parser.add_argument("--github-output", type=Path)
|
||||
parser.add_argument("--summary-file", type=Path)
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main() -> int:
|
||||
args = parse_args()
|
||||
paths = args.paths_file.read_text(encoding="utf-8").splitlines()
|
||||
plan = classify_paths(paths)
|
||||
print(f"MERGE_TEST_PLAN={plan.encoded_lanes()}")
|
||||
print(f"MERGE_GOLDEN_TESTS={plan.encoded_golden_tests()}")
|
||||
print(f"MERGE_SSIM_TESTS={plan.encoded_ssim_tests()}")
|
||||
for reason in plan.reasons:
|
||||
print(f"- {reason}")
|
||||
if args.github_output:
|
||||
with args.github_output.open("a", encoding="utf-8") as output:
|
||||
_write_github_output(output, plan)
|
||||
if args.summary_file:
|
||||
with args.summary_file.open("a", encoding="utf-8") as output:
|
||||
_write_summary(output, plan)
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -1,249 +0,0 @@
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
|
||||
import requests
|
||||
|
||||
|
||||
def parse_arguments():
|
||||
"""Parse command line arguments"""
|
||||
parser = argparse.ArgumentParser(description='Run tests on RunPod GPU')
|
||||
parser.add_argument('--gpu-type', type=str, help='GPU type to use')
|
||||
parser.add_argument('--gpu-count',
|
||||
type=int,
|
||||
help='Number of GPUs to use',
|
||||
default=1)
|
||||
parser.add_argument('--test-command', type=str, help='Test command to run')
|
||||
parser.add_argument('--disk-size',
|
||||
type=int,
|
||||
default=20,
|
||||
help='Container disk size in GB (default: 20)')
|
||||
parser.add_argument('--volume-size',
|
||||
type=int,
|
||||
default=20,
|
||||
help='Persistent volume size in GB (default: 20)')
|
||||
parser.add_argument(
|
||||
'--image',
|
||||
type=str,
|
||||
required=True,
|
||||
help='Docker image to use')
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
args = parse_arguments()
|
||||
API_KEY = os.environ['RUNPOD_API_KEY']
|
||||
RUN_ID = os.environ['GITHUB_RUN_ID']
|
||||
JOB_ID = os.environ['JOB_ID']
|
||||
PODS_API = "https://rest.runpod.io/v1/pods"
|
||||
HEADERS = {
|
||||
"Content-Type": "application/json",
|
||||
"Authorization": f"Bearer {API_KEY}"
|
||||
}
|
||||
|
||||
|
||||
def create_pod():
|
||||
"""Create a RunPod instance"""
|
||||
# Ensure image name is lowercase (Docker requirement)
|
||||
image_name = args.image.lower()
|
||||
print(f"Using specified image: {image_name}")
|
||||
|
||||
docker_start_cmd = [
|
||||
"bash",
|
||||
"-c",
|
||||
"apt update;DEBIAN_FRONTEND=noninteractive apt-get install openssh-server -y;mkdir -p ~/.ssh;cd $_;chmod 700 ~/.ssh;echo \"$PUBLIC_KEY\" >> authorized_keys;chmod 700 authorized_keys;service ssh start;sleep infinity"
|
||||
]
|
||||
|
||||
print(f"Creating RunPod instance with GPU: {args.gpu_type}...")
|
||||
payload = {
|
||||
"name": f"fastvideo-{JOB_ID}-{RUN_ID}",
|
||||
"containerDiskInGb": args.disk_size,
|
||||
"volumeInGb": args.volume_size,
|
||||
"gpuTypeIds": [args.gpu_type],
|
||||
"gpuCount": args.gpu_count,
|
||||
"imageName": image_name,
|
||||
"allowedCudaVersions": ["12.4"],
|
||||
"dockerStartCmd": docker_start_cmd
|
||||
}
|
||||
|
||||
response = requests.post(PODS_API, headers=HEADERS, json=payload)
|
||||
response_data = response.json()
|
||||
print(f"Response: {json.dumps(response_data, indent=2)}")
|
||||
|
||||
return response_data["id"]
|
||||
|
||||
|
||||
def wait_for_pod(pod_id):
|
||||
"""Wait for pod to be in RUNNING state and fully ready with SSH access"""
|
||||
print("Waiting for RunPod to be ready...")
|
||||
|
||||
# First wait for RUNNING status
|
||||
max_attempts = 10
|
||||
attempts = 0
|
||||
while attempts < max_attempts:
|
||||
response = requests.get(f"{PODS_API}/{pod_id}", headers=HEADERS)
|
||||
pod_data = response.json()
|
||||
status = pod_data["desiredStatus"]
|
||||
|
||||
if status == "RUNNING":
|
||||
print("RunPod is running! Now waiting for ports to be assigned...")
|
||||
break
|
||||
|
||||
print(
|
||||
f"Current status: {status}, waiting... (attempt {attempts+1}/{max_attempts})"
|
||||
)
|
||||
time.sleep(2)
|
||||
attempts += 1
|
||||
|
||||
if attempts >= max_attempts:
|
||||
raise TimeoutError(
|
||||
"Timed out waiting for RunPod to reach RUNNING state")
|
||||
|
||||
# Wait for ports to be assigned
|
||||
max_attempts = 50
|
||||
attempts = 0
|
||||
while attempts < max_attempts:
|
||||
response = requests.get(f"{PODS_API}/{pod_id}", headers=HEADERS)
|
||||
pod_data = response.json()
|
||||
port_mappings = pod_data.get("portMappings")
|
||||
|
||||
if (port_mappings is not None and "22" in port_mappings
|
||||
and pod_data.get("publicIp", "") != ""):
|
||||
print("RunPod is ready with SSH access!")
|
||||
print(f"SSH IP: {pod_data['publicIp']}")
|
||||
print(f"SSH Port: {port_mappings['22']}")
|
||||
break
|
||||
|
||||
print(
|
||||
f"Waiting for SSH port and public IP to be available... (attempt {attempts+1}/{max_attempts})"
|
||||
)
|
||||
time.sleep(20)
|
||||
attempts += 1
|
||||
|
||||
if attempts >= max_attempts:
|
||||
raise TimeoutError("Timed out waiting for RunPod SSH access")
|
||||
|
||||
|
||||
def execute_command(pod_id):
|
||||
"""Execute command on the pod via SSH using system SSH client"""
|
||||
print(f"Running command: {args.test_command}")
|
||||
|
||||
response = requests.get(f"{PODS_API}/{pod_id}", headers=HEADERS)
|
||||
pod_data = response.json()
|
||||
ssh_ip = pod_data["publicIp"]
|
||||
ssh_port = pod_data["portMappings"]["22"]
|
||||
|
||||
# Copy the repository to the pod using scp
|
||||
repo_dir = os.path.abspath(os.getcwd())
|
||||
repo_name = os.path.basename(repo_dir)
|
||||
|
||||
print(f"Copying repository from {repo_dir} to RunPod...")
|
||||
|
||||
tar_command = [
|
||||
"tar", "-czf", "/tmp/repo.tar.gz", "-C",
|
||||
os.path.dirname(repo_dir), repo_name
|
||||
]
|
||||
subprocess.run(tar_command, check=True)
|
||||
|
||||
# Copy the tarball to the pod
|
||||
scp_command = [
|
||||
"scp", "-o", "StrictHostKeyChecking=no", "-o",
|
||||
"UserKnownHostsFile=/dev/null", "-o", "ServerAliveInterval=60", "-o",
|
||||
"ServerAliveCountMax=10", "-P",
|
||||
str(ssh_port), "/tmp/repo.tar.gz", f"root@{ssh_ip}:/tmp/"
|
||||
]
|
||||
subprocess.run(scp_command, check=True)
|
||||
|
||||
# For custom image, we can use the pre-configured environment
|
||||
setup_steps = [
|
||||
"tar -xzf /tmp/repo.tar.gz --no-same-owner -C /workspace/",
|
||||
f"cd /workspace/{repo_name}",
|
||||
"source $HOME/.local/bin/env && source /opt/venv/bin/activate",
|
||||
args.test_command
|
||||
]
|
||||
|
||||
remote_command = " && ".join(setup_steps)
|
||||
|
||||
ssh_command = [
|
||||
"ssh", "-o", "StrictHostKeyChecking=no", "-o",
|
||||
"UserKnownHostsFile=/dev/null", "-o", "ServerAliveInterval=60", "-o",
|
||||
"ServerAliveCountMax=10", "-p",
|
||||
str(ssh_port), f"root@{ssh_ip}", remote_command
|
||||
]
|
||||
|
||||
print(f"Connecting to {ssh_ip}:{ssh_port}...")
|
||||
|
||||
try:
|
||||
process = subprocess.Popen(ssh_command,
|
||||
stdout=subprocess.PIPE,
|
||||
stderr=subprocess.STDOUT,
|
||||
universal_newlines=True,
|
||||
bufsize=0)
|
||||
|
||||
stdout_lines = []
|
||||
|
||||
print("Command output:")
|
||||
|
||||
for line in iter(process.stdout.readline, ''):
|
||||
print(line.strip())
|
||||
stdout_lines.append(line)
|
||||
|
||||
process.wait()
|
||||
|
||||
return_code = process.returncode
|
||||
success = return_code == 0
|
||||
|
||||
stdout_str = "".join(stdout_lines)
|
||||
|
||||
if success:
|
||||
print("Command executed successfully")
|
||||
else:
|
||||
print(f"Command failed with exit code {return_code}")
|
||||
|
||||
result = {
|
||||
"success": success,
|
||||
"return_code": return_code,
|
||||
"stdout": stdout_str,
|
||||
"stderr": ""
|
||||
}
|
||||
return result
|
||||
|
||||
except Exception as e:
|
||||
print(f"Error executing SSH command: {str(e)}")
|
||||
result = {"success": False, "error": str(e), "stdout": "", "stderr": ""}
|
||||
return result
|
||||
|
||||
|
||||
def terminate_pod(pod_id):
|
||||
"""Terminate the pod"""
|
||||
print("Terminating RunPod...")
|
||||
requests.delete(f"{PODS_API}/{pod_id}", headers=HEADERS)
|
||||
print(f"Terminated pod {pod_id}")
|
||||
|
||||
|
||||
def main():
|
||||
pod_id = None
|
||||
try:
|
||||
pod_id = create_pod()
|
||||
wait_for_pod(pod_id)
|
||||
result = execute_command(pod_id)
|
||||
|
||||
if result.get("error") is not None:
|
||||
print(f"Error executing command: {result['error']}")
|
||||
sys.exit(1)
|
||||
|
||||
if not result.get("success", False):
|
||||
print(
|
||||
"Tests failed - check the output above for details on which tests failed"
|
||||
)
|
||||
sys.exit(1)
|
||||
|
||||
finally:
|
||||
if pod_id:
|
||||
terminate_pod(pod_id)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -1,90 +0,0 @@
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
import uuid
|
||||
|
||||
import requests
|
||||
|
||||
API_KEY = os.environ['RUNPOD_API_KEY']
|
||||
RUN_ID = os.environ.get('GITHUB_RUN_ID', str(uuid.uuid4()))
|
||||
PODS_API = "https://rest.runpod.io/v1/pods"
|
||||
HEADERS = {
|
||||
"Content-Type": "application/json",
|
||||
"Authorization": f"Bearer {API_KEY}"
|
||||
}
|
||||
|
||||
|
||||
def get_job_ids():
|
||||
"""Parse job IDs from environment variable"""
|
||||
job_ids_str = os.environ.get('JOB_IDS')
|
||||
try:
|
||||
job_ids = json.loads(job_ids_str)
|
||||
if not isinstance(job_ids, list):
|
||||
print("Error: JOB_IDS is not a list.")
|
||||
sys.exit(1)
|
||||
return job_ids
|
||||
except json.JSONDecodeError as e:
|
||||
print(f"Error parsing JOB_IDS: {e}")
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
def cleanup_pods():
|
||||
"""Find and terminate RunPod instances"""
|
||||
print(f"Run ID: {RUN_ID}")
|
||||
|
||||
single_job_id = os.environ.get('JOB_ID')
|
||||
|
||||
if single_job_id:
|
||||
job_ids = [single_job_id]
|
||||
print(f"Job ID: {single_job_id}")
|
||||
else:
|
||||
job_ids = get_job_ids()
|
||||
print(f"Job IDs: {job_ids}")
|
||||
|
||||
# Get all pods associated with RunPod API_KEY
|
||||
try:
|
||||
response = requests.get(PODS_API, headers=HEADERS)
|
||||
response.raise_for_status()
|
||||
pods = response.json()
|
||||
except requests.exceptions.RequestException as e:
|
||||
print(f"Error getting pods: {e}")
|
||||
sys.exit(1)
|
||||
|
||||
# Find and terminate pods created by this workflow run
|
||||
terminated_pods = []
|
||||
for pod in pods:
|
||||
pod_name = pod.get("name", "")
|
||||
pod_id = pod.get("id")
|
||||
|
||||
# Check if this pod was created by one of our jobs
|
||||
if any(f"{job_id}-{RUN_ID}" in pod_name for job_id in job_ids):
|
||||
print(f"Found pod: {pod_id} ({pod_name})")
|
||||
try:
|
||||
print(f"Terminating pod {pod_id}...")
|
||||
term_response = requests.delete(f"{PODS_API}/{pod_id}",
|
||||
headers=HEADERS)
|
||||
term_response.raise_for_status()
|
||||
terminated_pods.append(pod_id)
|
||||
print(f"Successfully terminated pod {pod_id}")
|
||||
except requests.exceptions.RequestException as e:
|
||||
print(f"Error terminating pod {pod_id}: {e}")
|
||||
sys.exit(1)
|
||||
|
||||
if terminated_pods:
|
||||
if single_job_id:
|
||||
print(f"Terminated pod: {terminated_pods[0]}")
|
||||
else:
|
||||
print(f"Terminated {len(terminated_pods)} pods: {terminated_pods}")
|
||||
else:
|
||||
if single_job_id:
|
||||
print(f"No pod found matching pattern: {single_job_id}-{RUN_ID}")
|
||||
else:
|
||||
print("No pods found to terminate.")
|
||||
|
||||
|
||||
def main():
|
||||
cleanup_pods()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Executable
+122
@@ -0,0 +1,122 @@
|
||||
#!/usr/bin/env bash
|
||||
# Self-test for gate_full_suite.sh using a mocked `gh`. No network, runs on
|
||||
# any dev box: bash .github/scripts/test_gate_full_suite.sh
|
||||
set -u
|
||||
here=$(cd "$(dirname "$0")" && pwd)
|
||||
tmp=$(mktemp -d)
|
||||
trap 'rm -rf "$tmp"' EXIT
|
||||
|
||||
# Mock gh. Asserts the exact endpoint (including head_sha) it is called
|
||||
# with — an endpoint typo in the gate script fails the test rather than
|
||||
# silently serving canned data. On the runs endpoint it serves
|
||||
# $MOCK_DIR/response_<call#>.json, sticking on the highest existing file,
|
||||
# and exits 1 if none exist (simulates a GitHub API outage). On the pulls
|
||||
# endpoint it serves $MOCK_DIR/pr.json, defaulting to a 'ready'-labeled PR.
|
||||
cat > "$tmp/gh" <<'EOF'
|
||||
#!/usr/bin/env bash
|
||||
if [ "${1:-}" != "api" ]; then
|
||||
echo "unexpected gh invocation: $*" >> "$MOCK_DIR/endpoint_error"
|
||||
exit 2
|
||||
fi
|
||||
case "${2:-}" in
|
||||
"repos/o/r/actions/runs?head_sha=deadbeef&per_page=100")
|
||||
n=$(( $(cat "$MOCK_DIR/count" 2>/dev/null || echo 0) + 1 ))
|
||||
echo "$n" > "$MOCK_DIR/count"
|
||||
while [ "$n" -gt 0 ]; do
|
||||
if [ -f "$MOCK_DIR/response_$n.json" ]; then
|
||||
cat "$MOCK_DIR/response_$n.json"
|
||||
exit 0
|
||||
fi
|
||||
n=$(( n - 1 ))
|
||||
done
|
||||
echo "api outage" >&2
|
||||
exit 1
|
||||
;;
|
||||
"repos/o/r/pulls/42")
|
||||
if [ -f "$MOCK_DIR/pr.json" ]; then
|
||||
cat "$MOCK_DIR/pr.json"
|
||||
else
|
||||
echo '{"labels": [{"name": "ready"}]}'
|
||||
fi
|
||||
;;
|
||||
*)
|
||||
echo "unexpected gh endpoint: $2" >> "$MOCK_DIR/endpoint_error"
|
||||
exit 2
|
||||
;;
|
||||
esac
|
||||
EOF
|
||||
chmod +x "$tmp/gh"
|
||||
|
||||
PC_OK='{"name": "pre-commit", "id": 1, "status": "completed", "conclusion": "success"}'
|
||||
PC_BAD='{"name": "pre-commit", "id": 1, "status": "completed", "conclusion": "failure"}'
|
||||
PC_PENDING='{"name": "pre-commit", "id": 1, "status": "in_progress", "conclusion": null}'
|
||||
DOCS_OK='{"name": "Deploy Documentation", "id": 2, "status": "completed", "conclusion": "success"}'
|
||||
DOCS_BAD='{"name": "Deploy Documentation", "id": 2, "status": "completed", "conclusion": "failure"}'
|
||||
DOCS_CANCELLED='{"name": "Deploy Documentation", "id": 2, "status": "completed", "conclusion": "cancelled"}'
|
||||
OTHER='{"name": "Trigger Merge Gate", "id": 3, "status": "in_progress", "conclusion": null}'
|
||||
NULL_NAME='{"name": null, "id": 4, "status": "completed", "conclusion": "failure"}'
|
||||
PC_OK_RERUN='{"name": "pre-commit", "id": 5, "status": "completed", "conclusion": "success"}'
|
||||
|
||||
fails=0
|
||||
want_log="" # optional: expect() also greps out.log for this regex, then resets
|
||||
pr_json="" # optional: served for the pulls (label re-check) endpoint, then resets
|
||||
raw_body="" # optional: serve responses verbatim instead of wrapping in workflow_runs
|
||||
expect() { # <name> <expected-exit> <response json>...
|
||||
local name=$1 want=$2 dir i=1
|
||||
shift 2
|
||||
dir=$(mktemp -d "$tmp/test_XXXXXX")
|
||||
for body in "$@"; do
|
||||
if [ -n "$raw_body" ]; then
|
||||
printf '%s' "$body" > "$dir/response_$i.json"
|
||||
else
|
||||
printf '{"workflow_runs": [%s]}' "$body" > "$dir/response_$i.json"
|
||||
fi
|
||||
i=$(( i + 1 ))
|
||||
done
|
||||
[ -n "$pr_json" ] && printf '%s' "$pr_json" > "$dir/pr.json"
|
||||
( export PATH="$tmp:$PATH" MOCK_DIR="$dir" PR_SHA=deadbeef PR_NUMBER=42 \
|
||||
GITHUB_REPOSITORY=o/r POLL_SECS=0 GRACE_SECS=1 MAX_WAIT_SECS=3
|
||||
bash "$here/gate_full_suite.sh" > "$dir/out.log" 2>&1 )
|
||||
local rc=$?
|
||||
if [ "$rc" -ne "$want" ]; then
|
||||
echo "FAIL: $name (exit $rc, want $want)"
|
||||
cat "$dir/out.log"
|
||||
fails=1
|
||||
elif [ -f "$dir/endpoint_error" ]; then
|
||||
echo "FAIL: $name (mock gh got an unexpected call)"
|
||||
cat "$dir/endpoint_error"
|
||||
fails=1
|
||||
elif [ -n "$want_log" ] && ! grep -Eq "$want_log" "$dir/out.log"; then
|
||||
echo "FAIL: $name (log does not match: $want_log)"
|
||||
cat "$dir/out.log"
|
||||
fails=1
|
||||
else
|
||||
echo "ok: $name"
|
||||
fi
|
||||
want_log="" pr_json="" raw_body=""
|
||||
}
|
||||
|
||||
expect "both green -> proceed" 0 "$PC_OK, $DOCS_OK, $OTHER, $NULL_NAME"
|
||||
expect "docs build failed -> blocked" 1 "$PC_OK, $DOCS_BAD"
|
||||
expect "pre-commit failed -> blocked" 1 "$PC_BAD"
|
||||
expect "pending then green -> proceed" 0 "$PC_PENDING" "$PC_OK, $DOCS_OK"
|
||||
want_log="never appeared.*Deploy Documentation"
|
||||
expect "docs run absent (path-filtered) -> proceed after grace" 0 "$PC_OK"
|
||||
expect "API outage -> fail open" 0
|
||||
want_log="FAILING OPEN"
|
||||
expect "pending past MAX_WAIT -> fail open" 0 "$PC_PENDING"
|
||||
want_log="FAILING OPEN"
|
||||
expect "unrelated runs only -> no grace, fail open at MAX_WAIT" 0 "$OTHER"
|
||||
expect "cancelled docs then green -> proceed" 0 \
|
||||
"$PC_OK, $DOCS_CANCELLED" "$PC_OK, $DOCS_OK"
|
||||
want_log="FAILING OPEN"
|
||||
expect "cancelled docs forever -> fail open at MAX_WAIT" 0 "$PC_OK, $DOCS_CANCELLED"
|
||||
want_log="FAILING OPEN"
|
||||
expect "pre-commit absent -> no grace, fail open at MAX_WAIT" 0 "$DOCS_OK"
|
||||
expect "duplicate run names -> latest wins" 0 "$PC_BAD, $PC_OK_RERUN, $DOCS_OK"
|
||||
raw_body=1
|
||||
expect "garbage response body -> fail open" 0 "this is not json"
|
||||
pr_json='{"labels": [{"name": "other"}]}'
|
||||
expect "ready label removed mid-gate -> blocked" 1 "$PC_OK, $DOCS_OK"
|
||||
|
||||
exit "$fails"
|
||||
@@ -0,0 +1,201 @@
|
||||
name: Build Image Template
|
||||
|
||||
on:
|
||||
workflow_call:
|
||||
inputs:
|
||||
python_version:
|
||||
required: true
|
||||
type: string
|
||||
dockerfile_path:
|
||||
required: true
|
||||
type: string
|
||||
tag_suffix:
|
||||
required: true
|
||||
type: string
|
||||
image_name:
|
||||
required: false
|
||||
type: string
|
||||
default: fastvideo-dev
|
||||
build_args:
|
||||
required: false
|
||||
type: string
|
||||
default: ''
|
||||
include_latest_tags:
|
||||
required: false
|
||||
type: boolean
|
||||
default: true
|
||||
mark_as_latest:
|
||||
required: false
|
||||
type: boolean
|
||||
default: false
|
||||
runner:
|
||||
required: false
|
||||
type: string
|
||||
default: ubuntu-latest
|
||||
architecture:
|
||||
required: false
|
||||
type: string
|
||||
default: amd64
|
||||
push_by_digest:
|
||||
required: false
|
||||
type: boolean
|
||||
default: false
|
||||
digest_artifact_name:
|
||||
required: false
|
||||
type: string
|
||||
default: ''
|
||||
|
||||
jobs:
|
||||
build-and-push:
|
||||
runs-on: ${{ inputs.runner }}
|
||||
permissions:
|
||||
contents: read
|
||||
packages: write
|
||||
|
||||
steps:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@v4
|
||||
# The Docker context intentionally includes .git so the kernel build can
|
||||
# initialize its pinned submodules. Do not copy the checkout token with it.
|
||||
with:
|
||||
persist-credentials: false
|
||||
|
||||
- name: Free up disk space
|
||||
run: |
|
||||
# Display initial space
|
||||
echo "Initial disk space:"
|
||||
df -h
|
||||
|
||||
# Remove large directories directly
|
||||
sudo rm -rf /usr/share/dotnet
|
||||
sudo rm -rf /usr/local/lib/android
|
||||
sudo rm -rf /opt/ghc
|
||||
sudo rm -rf /usr/local/share/boost
|
||||
sudo rm -rf /usr/share/swift
|
||||
sudo rm -rf /usr/local/lib/node_modules
|
||||
sudo rm -rf /usr/local/share/powershell
|
||||
sudo rm -rf /usr/share/rust
|
||||
sudo rm -rf /usr/local/.ghcup
|
||||
|
||||
# Remove cached files
|
||||
sudo rm -rf /var/lib/apt/lists/*
|
||||
sudo rm -rf /var/cache/apt/archives/*
|
||||
|
||||
# Clean Docker
|
||||
docker system prune -af --volumes
|
||||
|
||||
# Display available space after cleanup
|
||||
echo "Disk space after cleanup:"
|
||||
df -h
|
||||
|
||||
- name: Set up Docker Buildx
|
||||
uses: docker/setup-buildx-action@v3
|
||||
|
||||
- name: Login to GitHub Container Registry
|
||||
uses: docker/login-action@v3
|
||||
with:
|
||||
registry: ghcr.io
|
||||
username: ${{ github.repository_owner }}
|
||||
password: ${{ secrets.GITHUB_TOKEN }}
|
||||
|
||||
- name: Normalize image reference
|
||||
id: image
|
||||
env:
|
||||
IMAGE: ghcr.io/${{ github.repository }}/${{ inputs.image_name }}
|
||||
run: echo "name=${IMAGE,,}" >> "${GITHUB_OUTPUT}"
|
||||
|
||||
- name: Prepare tags
|
||||
id: prepare-tags
|
||||
run: |
|
||||
SHORT_SHA=$(echo ${{ github.sha }} | cut -c1-7)
|
||||
|
||||
TAGS="type=raw,value=${{ inputs.tag_suffix }}-sha-${SHORT_SHA}"
|
||||
|
||||
if [[ "${{ inputs.include_latest_tags }}" == "true" ]]; then
|
||||
TAGS="type=raw,value=${{ inputs.tag_suffix }}-latest\n${TAGS}"
|
||||
fi
|
||||
|
||||
# Tag the designated default variant as the global `latest` image
|
||||
if [[ "${{ inputs.include_latest_tags }}" == "true" && "${{ inputs.mark_as_latest }}" == "true" ]]; then
|
||||
TAGS="${TAGS}\ntype=raw,value=latest"
|
||||
fi
|
||||
|
||||
{
|
||||
echo "tags<<EOF"
|
||||
echo -e "$TAGS"
|
||||
echo "EOF"
|
||||
} >> $GITHUB_OUTPUT
|
||||
|
||||
- name: Extract metadata for Docker
|
||||
id: meta
|
||||
uses: docker/metadata-action@v5
|
||||
with:
|
||||
images: ${{ steps.image.outputs.name }}
|
||||
tags: ${{ steps.prepare-tags.outputs.tags }}
|
||||
|
||||
- name: Build and push Docker image
|
||||
if: ${{ !inputs.push_by_digest }}
|
||||
id: build-push
|
||||
uses: docker/build-push-action@v6
|
||||
with:
|
||||
context: .
|
||||
file: ${{ inputs.dockerfile_path }}
|
||||
platforms: linux/${{ inputs.architecture }}
|
||||
push: true
|
||||
tags: ${{ steps.meta.outputs.tags }}
|
||||
labels: ${{ steps.meta.outputs.labels }}
|
||||
build-args: ${{ inputs.build_args }}
|
||||
cache-from: type=gha,scope=${{ inputs.image_name }}-${{ inputs.tag_suffix }}-${{ inputs.architecture }}
|
||||
cache-to: type=gha,mode=max,scope=${{ inputs.image_name }}-${{ inputs.tag_suffix }}-${{ inputs.architecture }}
|
||||
|
||||
# Multi-architecture callers publish immutable manifests by digest here,
|
||||
# then create the shared user-facing tags in a single downstream job.
|
||||
# This prevents architecture jobs from racing to replace the same tag.
|
||||
- name: Build and push Docker image by digest
|
||||
if: ${{ inputs.push_by_digest }}
|
||||
id: build-push-digest
|
||||
uses: docker/build-push-action@v6
|
||||
with:
|
||||
context: .
|
||||
file: ${{ inputs.dockerfile_path }}
|
||||
platforms: linux/${{ inputs.architecture }}
|
||||
labels: ${{ steps.meta.outputs.labels }}
|
||||
build-args: ${{ inputs.build_args }}
|
||||
outputs: type=image,name=${{ steps.image.outputs.name }},push-by-digest=true,name-canonical=true,push=true
|
||||
cache-from: type=gha,scope=${{ inputs.image_name }}-${{ inputs.tag_suffix }}-${{ inputs.architecture }}
|
||||
cache-to: type=gha,mode=max,scope=${{ inputs.image_name }}-${{ inputs.tag_suffix }}-${{ inputs.architecture }}
|
||||
|
||||
- name: Export digest
|
||||
if: ${{ inputs.push_by_digest }}
|
||||
env:
|
||||
DIGEST: ${{ steps.build-push-digest.outputs.digest }}
|
||||
run: |
|
||||
if [[ -z "${{ inputs.digest_artifact_name }}" ]]; then
|
||||
echo "digest_artifact_name is required when push_by_digest is true" >&2
|
||||
exit 1
|
||||
fi
|
||||
mkdir -p /tmp/digests
|
||||
touch "/tmp/digests/${DIGEST#sha256:}"
|
||||
|
||||
- name: Upload digest
|
||||
if: ${{ inputs.push_by_digest }}
|
||||
uses: actions/upload-artifact@v4
|
||||
with:
|
||||
name: ${{ inputs.digest_artifact_name }}
|
||||
path: /tmp/digests/*
|
||||
if-no-files-found: error
|
||||
retention-days: 1
|
||||
|
||||
- name: Success message
|
||||
if: ${{ !inputs.push_by_digest }}
|
||||
run: |
|
||||
echo "✅ Python ${{ inputs.python_version }} image successfully built and pushed to ${{ steps.image.outputs.name }}:${{ inputs.tag_suffix }}-sha-${GITHUB_SHA::7}"
|
||||
echo "Digest: ${{ steps.build-push.outputs.digest }}"
|
||||
echo "To run tests with this image, manually trigger the 'Run Tests' workflow."
|
||||
|
||||
- name: Digest success message
|
||||
if: ${{ inputs.push_by_digest }}
|
||||
env:
|
||||
DIGEST: ${{ steps.build-push-digest.outputs.digest }}
|
||||
run: |
|
||||
echo "✅ Python ${{ inputs.python_version }} linux/${{ inputs.architecture }} image pushed as ${DIGEST}"
|
||||
@@ -1,106 +0,0 @@
|
||||
name: Build Image Template
|
||||
|
||||
on:
|
||||
workflow_call:
|
||||
inputs:
|
||||
python_version:
|
||||
required: true
|
||||
type: string
|
||||
dockerfile_path:
|
||||
required: true
|
||||
type: string
|
||||
tag_suffix:
|
||||
required: true
|
||||
type: string
|
||||
|
||||
jobs:
|
||||
build-and-push:
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
contents: read
|
||||
packages: write
|
||||
|
||||
steps:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Free up disk space
|
||||
run: |
|
||||
# Display initial space
|
||||
echo "Initial disk space:"
|
||||
df -h
|
||||
|
||||
# Remove large directories directly
|
||||
sudo rm -rf /usr/share/dotnet
|
||||
sudo rm -rf /usr/local/lib/android
|
||||
sudo rm -rf /opt/ghc
|
||||
sudo rm -rf /usr/local/share/boost
|
||||
sudo rm -rf /usr/share/swift
|
||||
sudo rm -rf /usr/local/lib/node_modules
|
||||
sudo rm -rf /usr/local/share/powershell
|
||||
sudo rm -rf /usr/share/rust
|
||||
sudo rm -rf /usr/local/.ghcup
|
||||
|
||||
# Remove cached files
|
||||
sudo rm -rf /var/lib/apt/lists/*
|
||||
sudo rm -rf /var/cache/apt/archives/*
|
||||
|
||||
# Clean Docker
|
||||
docker system prune -af --volumes
|
||||
|
||||
# Display available space after cleanup
|
||||
echo "Disk space after cleanup:"
|
||||
df -h
|
||||
|
||||
- name: Set up Docker Buildx
|
||||
uses: docker/setup-buildx-action@v3
|
||||
|
||||
- name: Login to GitHub Container Registry
|
||||
uses: docker/login-action@v3
|
||||
with:
|
||||
registry: ghcr.io
|
||||
username: ${{ github.repository_owner }}
|
||||
password: ${{ secrets.GITHUB_TOKEN }}
|
||||
|
||||
- name: Prepare tags
|
||||
id: prepare-tags
|
||||
run: |
|
||||
SHORT_SHA=$(echo ${{ github.sha }} | cut -c1-7)
|
||||
|
||||
TAGS="type=raw,value=${{ inputs.tag_suffix }}-latest"
|
||||
TAGS="${TAGS}\ntype=raw,value=${{ inputs.tag_suffix }}-sha-${SHORT_SHA}"
|
||||
|
||||
# Set Python 3.10 as the default image
|
||||
if [[ "${{ inputs.python_version }}" == "3.10" ]]; then
|
||||
TAGS="${TAGS}\ntype=raw,value=latest"
|
||||
fi
|
||||
|
||||
{
|
||||
echo "tags<<EOF"
|
||||
echo -e "$TAGS"
|
||||
echo "EOF"
|
||||
} >> $GITHUB_OUTPUT
|
||||
|
||||
- name: Extract metadata for Docker
|
||||
id: meta
|
||||
uses: docker/metadata-action@v5
|
||||
with:
|
||||
images: ghcr.io/${{ github.repository }}/fastvideo-dev
|
||||
tags: ${{ steps.prepare-tags.outputs.tags }}
|
||||
|
||||
- name: Build and push Docker image
|
||||
id: build-push
|
||||
uses: docker/build-push-action@v6
|
||||
with:
|
||||
context: .
|
||||
file: ${{ inputs.dockerfile_path }}
|
||||
push: true
|
||||
tags: ${{ steps.meta.outputs.tags }}
|
||||
labels: ${{ steps.meta.outputs.labels }}
|
||||
cache-from: type=gha
|
||||
cache-to: type=gha,mode=max
|
||||
|
||||
- name: Success message
|
||||
run: |
|
||||
echo "✅ Python ${{ inputs.python_version }} image successfully built and pushed to ghcr.io/${{ github.repository }}/fastvideo-dev:${{ inputs.tag_suffix }}-latest"
|
||||
echo "To run tests with this image, manually trigger the 'Run Tests' workflow."
|
||||
@@ -1,52 +0,0 @@
|
||||
name: Build and Push Docker Images
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
python_3_10:
|
||||
description: 'Build Python 3.10 image'
|
||||
required: false
|
||||
default: false
|
||||
type: boolean
|
||||
python_3_11:
|
||||
description: 'Build Python 3.11 image'
|
||||
required: false
|
||||
default: false
|
||||
type: boolean
|
||||
python_3_12:
|
||||
description: 'Build Python 3.12 image'
|
||||
required: false
|
||||
default: false
|
||||
type: boolean
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
packages: write
|
||||
|
||||
jobs:
|
||||
build-python-3-10:
|
||||
if: ${{ github.event.inputs.python_3_10 == 'true' }}
|
||||
uses: ./.github/workflows/build-image-template.yml
|
||||
with:
|
||||
python_version: '3.10'
|
||||
dockerfile_path: docker/Dockerfile.python3.10
|
||||
tag_suffix: py3.10
|
||||
secrets: inherit
|
||||
|
||||
build-python-3-11:
|
||||
if: ${{ github.event.inputs.python_3_11 == 'true' }}
|
||||
uses: ./.github/workflows/build-image-template.yml
|
||||
with:
|
||||
python_version: '3.11'
|
||||
dockerfile_path: docker/Dockerfile.python3.11
|
||||
tag_suffix: py3.11
|
||||
secrets: inherit
|
||||
|
||||
build-python-3-12:
|
||||
if: ${{ github.event.inputs.python_3_12 == 'true' }}
|
||||
uses: ./.github/workflows/build-image-template.yml
|
||||
with:
|
||||
python_version: '3.12'
|
||||
dockerfile_path: docker/Dockerfile.python3.12
|
||||
tag_suffix: py3.12
|
||||
secrets: inherit
|
||||
@@ -0,0 +1,100 @@
|
||||
name: Aggregate Test Status
|
||||
|
||||
on:
|
||||
status:
|
||||
|
||||
permissions:
|
||||
statuses: write
|
||||
|
||||
jobs:
|
||||
aggregate:
|
||||
if: >-
|
||||
github.event.context == 'direct-test-completed'
|
||||
&& github.event.state == 'success'
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Check and update aggregate status
|
||||
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7.0.1
|
||||
with:
|
||||
script: |
|
||||
const sha = context.payload.sha;
|
||||
|
||||
const { data } = await github.rest.repos.getCombinedStatusForRef({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
ref: sha,
|
||||
per_page: 100,
|
||||
});
|
||||
|
||||
// Buildkite derives the GitHub context prefix from the label emoji.
|
||||
// Keep hard Full Suite lanes in test-tube/bar-chart namespaces and
|
||||
// Fastcheck lanes in microscope so targeted reruns cannot clear the
|
||||
// wrong aggregate status. Automatic PR jobs use pr-fastcheck while
|
||||
// slash-command and Full Suite jobs use ci; normalize the suffix
|
||||
// and keep the newest status for each logical lane.
|
||||
const FASTCHECK_PREFIXES = [
|
||||
'buildkite/pr-fastcheck/microscope-',
|
||||
'buildkite/ci/microscope-',
|
||||
];
|
||||
const FULL_SUITE_PREFIXES = [
|
||||
'buildkite/ci/test-tube-',
|
||||
'buildkite/ci/bar-chart-',
|
||||
];
|
||||
|
||||
function newestByLane(prefixes) {
|
||||
const statuses = new Map();
|
||||
for (const status of data.statuses) {
|
||||
const prefix = prefixes.find(p => status.context.startsWith(p));
|
||||
if (!prefix) continue;
|
||||
const lane = status.context.slice(prefix.length);
|
||||
const previous = statuses.get(lane);
|
||||
if (!previous || Date.parse(status.updated_at) > Date.parse(previous.updated_at)) {
|
||||
statuses.set(lane, status);
|
||||
}
|
||||
}
|
||||
return statuses;
|
||||
}
|
||||
|
||||
const fastcheck = newestByLane(FASTCHECK_PREFIXES);
|
||||
const fullSuiteOnly = newestByLane(FULL_SUITE_PREFIXES);
|
||||
const fastcheckPassed =
|
||||
fastcheck.size === 6
|
||||
&& [...fastcheck.values()].every(s => s.state === 'success');
|
||||
const fullSuitePassed =
|
||||
fastcheckPassed
|
||||
&& fullSuiteOnly.size === 14
|
||||
&& [...fullSuiteOnly.values()].every(s => s.state === 'success');
|
||||
|
||||
// Direct reruns may repair a failed suite, never create a gate for
|
||||
// a suite that did not run.
|
||||
const failedAggregate = context => data.statuses.some(
|
||||
s => s.context === context && s.state === 'failure'
|
||||
);
|
||||
|
||||
if (failedAggregate('fastcheck-passed') && fastcheckPassed) {
|
||||
core.info(
|
||||
`All ${fastcheck.size} fastcheck tests passed — updating fastcheck-passed`
|
||||
);
|
||||
await github.rest.repos.createCommitStatus({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
sha,
|
||||
state: 'success',
|
||||
context: 'fastcheck-passed',
|
||||
description: `All ${fastcheck.size} fastcheck tests passed`,
|
||||
});
|
||||
}
|
||||
|
||||
if (failedAggregate('full-suite-passed') && fullSuitePassed) {
|
||||
core.info(
|
||||
'All 20 full suite tests passed — updating full-suite-passed'
|
||||
);
|
||||
await github.rest.repos.createCommitStatus({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
sha,
|
||||
state: 'success',
|
||||
context: 'full-suite-passed',
|
||||
description: 'All 20 full suite tests passed',
|
||||
});
|
||||
}
|
||||
@@ -0,0 +1,170 @@
|
||||
name: macOS MLX Smoke
|
||||
|
||||
on:
|
||||
pull_request:
|
||||
branches: [main]
|
||||
paths:
|
||||
- ".github/workflows/ci-macos-mlx.yml"
|
||||
- "fastvideo/mlx_runtime/**"
|
||||
- "fastvideo/tests/mlx/**"
|
||||
- "fastvideo/tests/platforms/test_mps_vsa_error.py"
|
||||
- "fastvideo/tests/platforms/test_cpu_sdpa.py"
|
||||
- "fastvideo/platforms/cpu.py"
|
||||
- "fastvideo/platforms/mps.py"
|
||||
- "fastvideo/platforms/__init__.py"
|
||||
- "fastvideo/__init__.py"
|
||||
- "examples/inference/basic/mlx_*.py"
|
||||
- "fastvideo/benchmarks/mlx_*.py"
|
||||
- "pyproject.toml"
|
||||
workflow_dispatch:
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: macos-mlx-${{ github.ref }}
|
||||
cancel-in-progress: true
|
||||
|
||||
jobs:
|
||||
mlx-smoke:
|
||||
if: github.event_name == 'workflow_dispatch' || github.event.pull_request.draft != true
|
||||
runs-on: macos-15
|
||||
timeout-minutes: 25
|
||||
env:
|
||||
FASTVIDEO_ATTENTION_BACKEND: TORCH_SDPA
|
||||
TOKENIZERS_PARALLELISM: "false"
|
||||
MASTER_ADDR: "127.0.0.1"
|
||||
MASTER_PORT: "29513"
|
||||
GLOO_SOCKET_IFNAME: lo0
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- uses: actions/setup-python@v5
|
||||
with:
|
||||
python-version: "3.12"
|
||||
|
||||
- uses: astral-sh/setup-uv@v3
|
||||
|
||||
- name: Install lightweight MLX smoke dependencies
|
||||
run: |
|
||||
uv pip install --system \
|
||||
--index-url https://download.pytorch.org/whl/cpu \
|
||||
torch==2.12.0 torchvision torchaudio
|
||||
uv pip install --system \
|
||||
pytest pytest-timeout numpy scipy pillow imageio einops cloudpickle filelock \
|
||||
PyYAML diffusers huggingface_hub remote-pdb safetensors loguru mlx \
|
||||
"ftfy>=6.3.1" "opencv-python>=4.10.0.84" psutil "transformers>=5.0.0"
|
||||
|
||||
- name: Show Apple runtime
|
||||
run: |
|
||||
python - <<'PY'
|
||||
import platform
|
||||
import mlx.core as mx
|
||||
import torch
|
||||
|
||||
print("machine:", platform.machine())
|
||||
print("processor:", platform.processor())
|
||||
print("mlx default device:", mx.default_device())
|
||||
device_info = mx.metal.device_info() if mx.metal.is_available() else "metal unavailable"
|
||||
print("mlx device_info:", device_info)
|
||||
print("torch:", torch.__version__)
|
||||
print("torch mps available:", torch.backends.mps.is_available())
|
||||
PY
|
||||
|
||||
- name: Run MLX smoke tests
|
||||
run: |
|
||||
python -m pytest \
|
||||
fastvideo/mlx_runtime/tests/ \
|
||||
fastvideo/tests/mlx/test_dmd_sampling.py \
|
||||
fastvideo/tests/mlx/test_memory_limits.py \
|
||||
fastvideo/tests/mlx/test_quant_capability.py \
|
||||
fastvideo/tests/mlx/test_mlx_dit_parity.py \
|
||||
fastvideo/tests/mlx/test_mlx_compile_parity.py \
|
||||
fastvideo/tests/mlx/test_mlx_checkpoint.py \
|
||||
fastvideo/tests/mlx/test_mlx_checkpoint_compat.py \
|
||||
fastvideo/tests/mlx/test_mlx_affine_dq_gemm.py \
|
||||
fastvideo/tests/mlx/test_mlx_minimax_h3_parity.py \
|
||||
fastvideo/tests/mlx/test_mlx_minimax_h3_vsa.py \
|
||||
fastvideo/tests/mlx/test_mlx_minimax_h3_vsa_regressions.py \
|
||||
fastvideo/tests/mlx/test_mlx_minimax_h3_fast_mode.py \
|
||||
fastvideo/tests/mlx/test_mlx_minimax_h3_fast_spatial.py \
|
||||
fastvideo/tests/mlx/test_mlx_fastwan_benchmark.py \
|
||||
fastvideo/tests/mlx/test_taehv_decode.py \
|
||||
fastvideo/tests/mlx/test_frame_upsample.py \
|
||||
fastvideo/tests/mlx/test_mlx_fast_spatial.py \
|
||||
fastvideo/tests/mlx/test_mlx_refine.py \
|
||||
fastvideo/tests/mlx/test_mlx_prompt_enhance.py \
|
||||
fastvideo/tests/mlx/test_mlx_prompt_to_video_decode.py \
|
||||
fastvideo/tests/mlx/test_mlx_wan22_prompt_cache_fingerprint.py \
|
||||
fastvideo/tests/mlx/test_wan22_sample.py \
|
||||
fastvideo/tests/mlx/test_windowed_attention.py \
|
||||
fastvideo/tests/mlx/test_mlx_rife_interpolation.py::test_rife_download_unavailable_has_specific_error \
|
||||
fastvideo/tests/mlx/test_mlx_rife_interpolation.py::test_rife_backend_regression_is_not_skip_eligible \
|
||||
fastvideo/tests/platforms/test_mps_vsa_error.py \
|
||||
fastvideo/tests/platforms/test_cpu_sdpa.py \
|
||||
-v -s --timeout=120 -o faulthandler_timeout=120
|
||||
|
||||
# Same tests on MLX's CPU backend. Hosted macOS runners are scarce and
|
||||
# slower to schedule; this Linux job gives fast PR signal on the identical
|
||||
# graph (the parity tests were designed to be backend-agnostic), while the
|
||||
# macOS job above stays the source of truth for Metal behavior.
|
||||
mlx-smoke-linux-cpu:
|
||||
if: github.event_name == 'workflow_dispatch' || github.event.pull_request.draft != true
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 20
|
||||
env:
|
||||
FASTVIDEO_ATTENTION_BACKEND: TORCH_SDPA
|
||||
TOKENIZERS_PARALLELISM: "false"
|
||||
MASTER_ADDR: localhost
|
||||
MASTER_PORT: "29513"
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- uses: actions/setup-python@v5
|
||||
with:
|
||||
python-version: "3.12"
|
||||
|
||||
- uses: astral-sh/setup-uv@v3
|
||||
|
||||
- name: Install lightweight MLX smoke dependencies (CPU backend)
|
||||
run: |
|
||||
uv pip install --system \
|
||||
--index-url https://download.pytorch.org/whl/cpu \
|
||||
torch==2.12.0 torchvision torchaudio
|
||||
uv pip install --system \
|
||||
pytest pytest-timeout numpy scipy pillow imageio einops cloudpickle filelock \
|
||||
PyYAML diffusers huggingface_hub remote-pdb safetensors loguru "mlx[cpu]" \
|
||||
"ftfy>=6.3.1" "opencv-python>=4.10.0.84" psutil "transformers>=5.0.0"
|
||||
|
||||
- name: Run MLX smoke tests (CPU backend)
|
||||
run: |
|
||||
python -m pytest \
|
||||
fastvideo/mlx_runtime/tests/ \
|
||||
fastvideo/tests/mlx/test_dmd_sampling.py \
|
||||
fastvideo/tests/mlx/test_memory_limits.py \
|
||||
fastvideo/tests/mlx/test_quant_capability.py \
|
||||
fastvideo/tests/mlx/test_mlx_dit_parity.py \
|
||||
fastvideo/tests/mlx/test_mlx_compile_parity.py \
|
||||
fastvideo/tests/mlx/test_mlx_checkpoint.py \
|
||||
fastvideo/tests/mlx/test_mlx_checkpoint_compat.py \
|
||||
fastvideo/tests/mlx/test_mlx_affine_dq_gemm.py \
|
||||
fastvideo/tests/mlx/test_mlx_minimax_h3_parity.py \
|
||||
fastvideo/tests/mlx/test_mlx_minimax_h3_vsa.py \
|
||||
fastvideo/tests/mlx/test_mlx_minimax_h3_vsa_regressions.py \
|
||||
fastvideo/tests/mlx/test_mlx_minimax_h3_fast_mode.py \
|
||||
fastvideo/tests/mlx/test_mlx_minimax_h3_fast_spatial.py \
|
||||
fastvideo/tests/mlx/test_mlx_fastwan_benchmark.py \
|
||||
fastvideo/tests/mlx/test_taehv_decode.py \
|
||||
fastvideo/tests/mlx/test_frame_upsample.py \
|
||||
fastvideo/tests/mlx/test_mlx_fast_spatial.py \
|
||||
fastvideo/tests/mlx/test_mlx_refine.py \
|
||||
fastvideo/tests/mlx/test_mlx_prompt_enhance.py \
|
||||
fastvideo/tests/mlx/test_mlx_prompt_to_video_decode.py \
|
||||
fastvideo/tests/mlx/test_mlx_wan22_prompt_cache_fingerprint.py \
|
||||
fastvideo/tests/mlx/test_wan22_sample.py \
|
||||
fastvideo/tests/mlx/test_windowed_attention.py \
|
||||
fastvideo/tests/mlx/test_mlx_rife_interpolation.py::test_rife_download_unavailable_has_specific_error \
|
||||
fastvideo/tests/mlx/test_mlx_rife_interpolation.py::test_rife_backend_regression_is_not_skip_eligible \
|
||||
fastvideo/tests/platforms/test_mps_vsa_error.py \
|
||||
fastvideo/tests/platforms/test_cpu_sdpa.py \
|
||||
-v -s --timeout=120 -o faulthandler_timeout=120
|
||||
@@ -0,0 +1,65 @@
|
||||
name: pre-commit
|
||||
|
||||
on:
|
||||
# pull_request_target instead of pull_request: the workflow definition and
|
||||
# the hook config are always taken from the BASE branch, so fork /
|
||||
# first-time-contributor PRs run immediately without a maintainer clicking
|
||||
# "Approve and run". The PR head is checked out as data only.
|
||||
pull_request_target:
|
||||
branches: [main]
|
||||
workflow_call:
|
||||
inputs:
|
||||
ref:
|
||||
description: 'Git ref to checkout (defaults to github.ref)'
|
||||
required: false
|
||||
type: string
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
pre-commit:
|
||||
if: github.event.pull_request.draft != true
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
with:
|
||||
ref: ${{ inputs.ref || '' }}
|
||||
# For PR events, lint the PR head — but keep the hook definitions from
|
||||
# the base branch so an untrusted PR cannot alter what gets executed.
|
||||
# The gate scripts are saved too: the self-test step below executes them,
|
||||
# so it must run the base-branch copies, not the PR head's.
|
||||
- name: Save trusted hook config and gate scripts
|
||||
if: github.event_name == 'pull_request_target'
|
||||
run: |
|
||||
cp .pre-commit-config.yaml "$RUNNER_TEMP/trusted-pre-commit-config.yaml"
|
||||
cp -a .github/scripts "$RUNNER_TEMP/trusted-scripts"
|
||||
echo "GATE_SCRIPTS_DIR=$RUNNER_TEMP/trusted-scripts" >> "$GITHUB_ENV"
|
||||
# allow-unsafe-pr-checkout acknowledges checkout's pull_request_target
|
||||
# guard: the head is data for the trusted hooks to lint; nothing from it
|
||||
# is executed (config and gate scripts are pinned to the base branch
|
||||
# above) and credentials are not persisted. SHA-pinned to v4.4.0 because
|
||||
# actionlint's action schema does not know the new input yet.
|
||||
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
|
||||
if: github.event_name == 'pull_request_target'
|
||||
with:
|
||||
ref: ${{ github.event.pull_request.head.sha }}
|
||||
persist-credentials: false
|
||||
allow-unsafe-pr-checkout: true
|
||||
- name: Restore trusted hook config
|
||||
if: github.event_name == 'pull_request_target'
|
||||
run: cp "$RUNNER_TEMP/trusted-pre-commit-config.yaml" .pre-commit-config.yaml
|
||||
- uses: actions/setup-python@v5
|
||||
with:
|
||||
python-version: "3.12"
|
||||
- run: echo "::add-matcher::.github/workflows/matchers/actionlint.json"
|
||||
- run: echo "::add-matcher::.github/workflows/matchers/mypy.json"
|
||||
- run: echo "::add-matcher::.github/workflows/matchers/ruff.json"
|
||||
- uses: pre-commit/action@v3.0.1
|
||||
with:
|
||||
extra_args: --all-files --hook-stage manual
|
||||
# After pre-commit so a self-test failure cannot mask lint failures.
|
||||
# GATE_SCRIPTS_DIR points at the base-branch copy on fork PRs (set above);
|
||||
# push / workflow_call runs use the checked-out tree directly.
|
||||
- name: Full-suite gate self-test
|
||||
run: bash "${GATE_SCRIPTS_DIR:-.github/scripts}/test_gate_full_suite.sh"
|
||||
@@ -0,0 +1,44 @@
|
||||
name: Scheduled Full SSIM
|
||||
|
||||
on:
|
||||
schedule:
|
||||
- cron: "0 5 * * 0"
|
||||
workflow_dispatch:
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
trigger:
|
||||
if: github.repository == 'hao-ai-lab/FastVideo'
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Trigger weekly full SSIM on Slinky Slurm
|
||||
env:
|
||||
BUILDKITE_API_TOKEN: ${{ secrets.BUILDKITE_API_TOKEN }}
|
||||
SOURCE_SHA: ${{ github.sha }}
|
||||
SOURCE_BRANCH: ${{ github.event.repository.default_branch }}
|
||||
BK_ORG: ${{ vars.BUILDKITE_ORG_SLUG }}
|
||||
BK_PIPELINE: ${{ vars.BUILDKITE_PIPELINE_SLUG }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
curl -sS --fail-with-body -X POST \
|
||||
"https://api.buildkite.com/v2/organizations/${BK_ORG}/pipelines/${BK_PIPELINE}/builds" \
|
||||
-H "Authorization: Bearer $BUILDKITE_API_TOKEN" \
|
||||
-H "Content-Type: application/json" \
|
||||
--data-raw "$(jq -n \
|
||||
--arg commit "$SOURCE_SHA" \
|
||||
--arg branch "$SOURCE_BRANCH" \
|
||||
'{
|
||||
commit: $commit,
|
||||
branch: $branch,
|
||||
message: "Weekly full SSIM on Slinky Slurm",
|
||||
ignore_pipeline_branch_filters: true,
|
||||
env: {
|
||||
TEST_SCOPE: "scheduled",
|
||||
FULL_SUITE: "false",
|
||||
TEST_TYPE: "ssim",
|
||||
PR_NUMBER: "false",
|
||||
PR_TITLE: "Scheduled full SSIM"
|
||||
}
|
||||
}')"
|
||||
@@ -0,0 +1,268 @@
|
||||
name: Slash Commands
|
||||
|
||||
on:
|
||||
issue_comment:
|
||||
types: [created]
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
pull-requests: write
|
||||
statuses: write
|
||||
|
||||
jobs:
|
||||
handle-merge:
|
||||
if: >-
|
||||
github.event.issue.pull_request != null
|
||||
&& startsWith(github.event.comment.body, '/merge')
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Check write permission
|
||||
id: perm
|
||||
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7.0.1
|
||||
with:
|
||||
script: |
|
||||
const { data: perm } = await github.rest.repos.getCollaboratorPermissionLevel({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
username: context.payload.comment.user.login,
|
||||
});
|
||||
const hasWrite = ['admin', 'write'].includes(perm.permission);
|
||||
if (!hasWrite) {
|
||||
core.setFailed(`User ${context.payload.comment.user.login} lacks write permission (has: ${perm.permission}).`);
|
||||
}
|
||||
core.setOutput('has_write', String(hasWrite));
|
||||
|
||||
- name: Add ready label
|
||||
if: steps.perm.outputs.has_write == 'true'
|
||||
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7.0.1
|
||||
with:
|
||||
script: |
|
||||
const owner = context.repo.owner;
|
||||
const repo = context.repo.repo;
|
||||
const prNumber = context.payload.issue.number;
|
||||
await github.rest.issues.addLabels({ owner, repo, issue_number: prNumber, labels: ['ready'] });
|
||||
|
||||
- name: React to comment
|
||||
if: steps.perm.outputs.has_write == 'true'
|
||||
continue-on-error: true
|
||||
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7.0.1
|
||||
with:
|
||||
script: |
|
||||
await github.rest.reactions.createForIssueComment({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
comment_id: context.payload.comment.id,
|
||||
content: 'rocket',
|
||||
});
|
||||
|
||||
trigger-merge-gate:
|
||||
needs: handle-merge
|
||||
if: needs.handle-merge.result == 'success'
|
||||
permissions:
|
||||
actions: read
|
||||
contents: read
|
||||
pull-requests: read
|
||||
uses: ./.github/workflows/ci-trigger-full-suite.yml
|
||||
with:
|
||||
pr_number: ${{ github.event.issue.number }}
|
||||
secrets:
|
||||
BUILDKITE_API_TOKEN: ${{ secrets.BUILDKITE_API_TOKEN }}
|
||||
|
||||
parse-command:
|
||||
if: >-
|
||||
github.event.issue.pull_request != null
|
||||
&& startsWith(github.event.comment.body, '/test')
|
||||
runs-on: ubuntu-latest
|
||||
outputs:
|
||||
test_type: ${{ steps.parse.outputs.test_type }}
|
||||
test_scope: ${{ steps.parse.outputs.test_scope }}
|
||||
full_suite: ${{ steps.parse.outputs.full_suite }}
|
||||
pr_sha: ${{ steps.pr.outputs.sha }}
|
||||
pr_branch: ${{ steps.pr.outputs.branch }}
|
||||
has_write: ${{ steps.perm.outputs.has_write }}
|
||||
steps:
|
||||
- name: Check write permission
|
||||
id: perm
|
||||
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7.0.1
|
||||
with:
|
||||
script: |
|
||||
const { data: perm } = await github.rest.repos.getCollaboratorPermissionLevel({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
username: context.payload.comment.user.login,
|
||||
});
|
||||
const hasWrite = ['admin', 'write'].includes(perm.permission);
|
||||
core.setOutput('has_write', String(hasWrite));
|
||||
if (!hasWrite) {
|
||||
core.info(`User ${context.payload.comment.user.login} lacks write permission — ignoring.`);
|
||||
}
|
||||
|
||||
- name: Parse /test command
|
||||
id: parse
|
||||
if: steps.perm.outputs.has_write == 'true'
|
||||
shell: bash
|
||||
env:
|
||||
COMMENT: ${{ github.event.comment.body }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
TEST_NAME=$(echo "$COMMENT" | grep -oP '(?<=/test\s)\S+' | head -1 || true)
|
||||
|
||||
VALID="encoder vae transformer kernel unit dreamverse ssim golden-gate training lora-inference lora-training lora-extraction distillation self-forcing vsa vmoba performance api train-framework eval unit-ci kernel-ci dreamverse-ci ssim-ci golden-gate-ci encoder-ci vae-ci transformer-ci lora-inference-ci lora-training-ci lora-extraction-ci training-ci distillation-ci self-forcing-ci vsa-ci vmoba-ci performance-ci api-ci train-framework-ci eval-ci full fastcheck pre-commit"
|
||||
if [ -z "$TEST_NAME" ] || ! echo "$VALID" | grep -qw "$TEST_NAME"; then
|
||||
echo "Unknown test: '$TEST_NAME'. Valid: $VALID"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
declare -A MAP=(
|
||||
[encoder]=encoder [vae]=vae [transformer]=transformer
|
||||
[kernel]=kernel_tests [unit]=unit_test [unit-ci]=unit_test_ci
|
||||
[kernel-ci]=kernel_tests_ci [dreamverse-ci]=dreamverse_app_ci
|
||||
[ssim-ci]=ssim_ci [vmoba-ci]=inference_vmoba_ci
|
||||
[golden-gate-ci]=golden_gate_ci [training-ci]=training_ci
|
||||
[encoder-ci]=encoder_ci [vae-ci]=vae_ci [transformer-ci]=transformer_ci
|
||||
[lora-inference-ci]=inference_lora_ci [lora-training-ci]=training_lora_ci
|
||||
[lora-extraction-ci]=lora_extraction_ci [distillation-ci]=distillation_dmd_ci
|
||||
[self-forcing-ci]=self_forcing_ci [vsa-ci]=training_vsa_ci
|
||||
[performance-ci]=performance_ci [api-ci]=api_server_ci
|
||||
[train-framework-ci]=train_framework_ci [eval-ci]=eval_ci
|
||||
[dreamverse]=dreamverse_app
|
||||
[ssim]=ssim [golden-gate]=golden_gate [training]=training
|
||||
[lora-inference]=inference_lora [lora-training]=training_lora
|
||||
[lora-extraction]=lora_extraction
|
||||
[distillation]=distillation_dmd [self-forcing]=self_forcing
|
||||
[vsa]=training_vsa [vmoba]=inference_vmoba
|
||||
[performance]=performance [api]=api_server
|
||||
[train-framework]=train_framework [eval]=eval
|
||||
)
|
||||
|
||||
if [ "$TEST_NAME" = "full" ]; then
|
||||
{
|
||||
echo "test_type=all"
|
||||
echo "test_scope=full"
|
||||
echo "full_suite=true"
|
||||
} >> "$GITHUB_OUTPUT"
|
||||
elif [ "$TEST_NAME" = "fastcheck" ]; then
|
||||
{
|
||||
echo "test_type=fastcheck"
|
||||
echo "test_scope=fastcheck"
|
||||
echo "full_suite=false"
|
||||
} >> "$GITHUB_OUTPUT"
|
||||
elif [ "$TEST_NAME" = "pre-commit" ]; then
|
||||
{
|
||||
echo "test_type="
|
||||
echo "test_scope=precommit"
|
||||
echo "full_suite=false"
|
||||
} >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
{
|
||||
echo "test_type=${MAP[$TEST_NAME]}"
|
||||
echo "test_scope=direct"
|
||||
echo "full_suite=false"
|
||||
} >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
|
||||
- name: Get PR details
|
||||
id: pr
|
||||
if: steps.perm.outputs.has_write == 'true'
|
||||
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7.0.1
|
||||
with:
|
||||
script: |
|
||||
const { data: pr } = await github.rest.pulls.get({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
pull_number: context.payload.issue.number,
|
||||
});
|
||||
core.setOutput('sha', pr.head.sha);
|
||||
core.setOutput('branch', pr.head.ref);
|
||||
|
||||
- name: React to comment
|
||||
if: steps.perm.outputs.has_write == 'true'
|
||||
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7.0.1
|
||||
with:
|
||||
script: |
|
||||
await github.rest.reactions.createForIssueComment({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
comment_id: context.payload.comment.id,
|
||||
content: 'rocket',
|
||||
});
|
||||
|
||||
pre-commit:
|
||||
needs: parse-command
|
||||
if: >-
|
||||
needs.parse-command.outputs.has_write == 'true'
|
||||
&& needs.parse-command.outputs.test_scope == 'precommit'
|
||||
uses: ./.github/workflows/ci-precommit.yml
|
||||
with:
|
||||
ref: refs/pull/${{ github.event.issue.number }}/merge
|
||||
|
||||
post-precommit-status:
|
||||
needs: [parse-command, pre-commit]
|
||||
if: always() && needs.parse-command.outputs.test_scope == 'precommit'
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7.0.1
|
||||
env:
|
||||
PR_SHA: ${{ needs.parse-command.outputs.pr_sha }}
|
||||
RESULT: ${{ needs.pre-commit.result }}
|
||||
with:
|
||||
script: |
|
||||
const state = process.env.RESULT === 'success' ? 'success' : 'failure';
|
||||
await github.rest.repos.createCommitStatus({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
sha: process.env.PR_SHA,
|
||||
state,
|
||||
context: 'pre-commit',
|
||||
description: `Triggered via /test pre-commit (${state})`,
|
||||
});
|
||||
|
||||
trigger-buildkite:
|
||||
needs: parse-command
|
||||
if: >-
|
||||
needs.parse-command.outputs.has_write == 'true'
|
||||
&& needs.parse-command.outputs.test_type != ''
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Trigger Buildkite
|
||||
env:
|
||||
BUILDKITE_API_TOKEN: ${{ secrets.BUILDKITE_API_TOKEN }}
|
||||
PR_SHA: ${{ needs.parse-command.outputs.pr_sha }}
|
||||
PR_BRANCH: ${{ needs.parse-command.outputs.pr_branch }}
|
||||
PR_NUMBER: ${{ github.event.issue.number }}
|
||||
TEST_SCOPE: ${{ needs.parse-command.outputs.test_scope }}
|
||||
FULL_SUITE: ${{ needs.parse-command.outputs.full_suite }}
|
||||
TEST_TYPE: ${{ needs.parse-command.outputs.test_type }}
|
||||
PR_TITLE: ${{ github.event.issue.title }}
|
||||
BK_ORG: ${{ vars.BUILDKITE_ORG_SLUG }}
|
||||
BK_PIPELINE: ${{ vars.BUILDKITE_PIPELINE_SLUG }}
|
||||
run: |
|
||||
curl -sS --fail-with-body -X POST \
|
||||
"https://api.buildkite.com/v2/organizations/${BK_ORG}/pipelines/${BK_PIPELINE}/builds" \
|
||||
-H "Authorization: Bearer $BUILDKITE_API_TOKEN" \
|
||||
-H "Content-Type: application/json" \
|
||||
--data-raw "$(jq -n \
|
||||
--arg commit "$PR_SHA" \
|
||||
--arg branch "$PR_BRANCH" \
|
||||
--arg message "/test ${TEST_TYPE} on PR #${PR_NUMBER}" \
|
||||
--argjson pr_id "$PR_NUMBER" \
|
||||
--arg test_scope "$TEST_SCOPE" \
|
||||
--arg full_suite "$FULL_SUITE" \
|
||||
--arg test_type "$TEST_TYPE" \
|
||||
--arg pr_number "$PR_NUMBER" \
|
||||
--arg pr_title "$PR_TITLE" \
|
||||
'{
|
||||
commit: $commit,
|
||||
branch: $branch,
|
||||
message: $message,
|
||||
ignore_pipeline_branch_filters: true,
|
||||
pull_request_id: $pr_id,
|
||||
pull_request_base_branch: "main",
|
||||
env: {
|
||||
TEST_SCOPE: $test_scope,
|
||||
FULL_SUITE: $full_suite,
|
||||
TEST_TYPE: $test_type,
|
||||
PR_NUMBER: $pr_number,
|
||||
PR_TITLE: $pr_title
|
||||
}
|
||||
}')"
|
||||
@@ -0,0 +1,260 @@
|
||||
name: Trigger Merge Gate
|
||||
|
||||
on:
|
||||
pull_request_target:
|
||||
types: [labeled, synchronize]
|
||||
workflow_call:
|
||||
inputs:
|
||||
pr_number:
|
||||
description: Pull request number to enter into the merge gate
|
||||
required: true
|
||||
type: number
|
||||
secrets:
|
||||
BUILDKITE_API_TOKEN:
|
||||
required: true
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
pull-requests: read
|
||||
actions: read
|
||||
|
||||
jobs:
|
||||
trigger:
|
||||
if: >-
|
||||
inputs.pr_number > 0
|
||||
|| (github.event.action == 'labeled' && github.event.label.name == 'ready')
|
||||
|| github.event.action == 'synchronize'
|
||||
runs-on: ubuntu-latest
|
||||
# Job-level concurrency: only this guarded job acquires the group, so an
|
||||
# unrelated `labeled` event (which skips the job) cannot cancel an in-flight
|
||||
# gate and then skip its replacement. The newest real trigger (`ready`,
|
||||
# push, or `/merge`) supersedes the in-flight run, whose Buildkite build the
|
||||
# cancel step below replaces.
|
||||
concurrency:
|
||||
group: merge-gate-${{ inputs.pr_number || github.event.pull_request.number }}
|
||||
cancel-in-progress: true
|
||||
# Gate below may wait for cheap checks (up to MAX_WAIT_SECS = 25 min).
|
||||
timeout-minutes: 35
|
||||
steps:
|
||||
- name: Check ready label
|
||||
id: check
|
||||
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7.0.1
|
||||
env:
|
||||
CALLED_PR_NUMBER: ${{ inputs.pr_number }}
|
||||
with:
|
||||
script: |
|
||||
const eventPrNumber = context.payload.pull_request?.number;
|
||||
const calledPrNumber = Number(process.env.CALLED_PR_NUMBER);
|
||||
const prNumber = eventPrNumber ?? calledPrNumber;
|
||||
if (!Number.isSafeInteger(prNumber) || prNumber <= 0) {
|
||||
core.setFailed(`Invalid pull request number: ${process.env.CALLED_PR_NUMBER}`);
|
||||
return;
|
||||
}
|
||||
const { data: pr } = await github.rest.pulls.get({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
pull_number: prNumber,
|
||||
});
|
||||
if (pr.state !== 'open') {
|
||||
core.setFailed(`PR #${prNumber} is not open.`);
|
||||
return;
|
||||
}
|
||||
if (pr.base.repo.full_name !== context.payload.repository.full_name
|
||||
|| pr.base.ref !== context.payload.repository.default_branch) {
|
||||
core.setFailed(`PR #${prNumber} does not target this repository's default branch.`);
|
||||
return;
|
||||
}
|
||||
const hasReady = pr.labels.some(l => l.name === 'ready');
|
||||
core.setOutput('has_ready', String(hasReady));
|
||||
core.setOutput('changed_files', String(pr.changed_files));
|
||||
core.setOutput('pr_number', String(pr.number));
|
||||
core.setOutput('head_sha', pr.head.sha);
|
||||
core.setOutput('head_ref', pr.head.ref);
|
||||
core.setOutput('base_sha', pr.base.sha);
|
||||
core.setOutput('title', pr.title);
|
||||
if (!hasReady) core.info('No ready label — skipping merge-gate trigger.');
|
||||
|
||||
- name: Cancel previous Buildkite builds
|
||||
# Cancelling stale builds only saves agent time. If it cannot run, the
|
||||
# merge gate must still be triggered by the steps below, so a failure
|
||||
# here is reported and stepped over rather than ending the job.
|
||||
continue-on-error: true
|
||||
timeout-minutes: 3
|
||||
if: steps.check.outputs.has_ready == 'true'
|
||||
env:
|
||||
BK_ORG: ${{ vars.BUILDKITE_ORG_SLUG }}
|
||||
BK_PIPELINE: ${{ vars.BUILDKITE_PIPELINE_SLUG }}
|
||||
BUILDKITE_API_TOKEN: ${{ secrets.BUILDKITE_API_TOKEN }}
|
||||
PR_BRANCH: ${{ steps.check.outputs.head_ref }}
|
||||
PR_NUMBER: ${{ steps.check.outputs.pr_number }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
response_file=$(mktemp)
|
||||
builds_file=$(mktemp)
|
||||
trap 'rm -f "$response_file" "$builds_file"' EXIT
|
||||
|
||||
if [[ ! "$PR_NUMBER" =~ ^[1-9][0-9]*$ ]]; then
|
||||
echo "::warning::Invalid pull request number; stale Buildkite builds may continue."
|
||||
exit 1
|
||||
fi
|
||||
|
||||
if ! curl -sS --fail-with-body --connect-timeout 5 --max-time 20 --get \
|
||||
-H "Authorization: Bearer $BUILDKITE_API_TOKEN" \
|
||||
--data-urlencode "branch=$PR_BRANCH" \
|
||||
--data-urlencode "state[]=running" \
|
||||
--data-urlencode "state[]=scheduled" \
|
||||
--data-urlencode "state[]=failing" \
|
||||
--data-urlencode "exclude_jobs=true" \
|
||||
--data-urlencode "exclude_pipeline=true" \
|
||||
--output "$response_file" \
|
||||
"https://api.buildkite.com/v2/organizations/${BK_ORG}/pipelines/${BK_PIPELINE}/builds"; then
|
||||
echo "::warning::Could not list Buildkite builds; stale merge-gate builds may continue."
|
||||
exit 1
|
||||
fi
|
||||
|
||||
if ! jq -e '
|
||||
if type != "array" then false
|
||||
else all(.[];
|
||||
if type != "object" then false
|
||||
else
|
||||
(.number | if type == "number" then . > 0 and floor == . else false end)
|
||||
and (
|
||||
(.env? | if . == null then {} else . end) as $env
|
||||
| if ($env | type) != "object" then false
|
||||
else
|
||||
($env.TEST_SCOPE? | . == null or type == "string")
|
||||
and ($env.PR_NUMBER? | . == null or type == "string")
|
||||
end
|
||||
)
|
||||
end
|
||||
)
|
||||
end
|
||||
' "$response_file" >/dev/null 2>&1; then
|
||||
# Do not print the response body: it is remote data and may contain
|
||||
# multiline values that would be interpreted as workflow commands.
|
||||
echo "::warning::Buildkite returned an invalid build list; stale merge-gate builds may continue."
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Match both branch and PR number: forks can reuse the same branch name.
|
||||
if ! jq -r --arg pr_number "$PR_NUMBER" '
|
||||
.[]
|
||||
| select((.env.TEST_SCOPE? == "merge") and (.env.PR_NUMBER? == $pr_number))
|
||||
| .number
|
||||
' "$response_file" > "$builds_file"; then
|
||||
echo "::warning::Could not select stale Buildkite builds; stale merge-gate builds may continue."
|
||||
exit 1
|
||||
fi
|
||||
|
||||
cancellation_failed=0
|
||||
while IFS= read -r build_num; do
|
||||
echo "Cancelling Buildkite build #$build_num"
|
||||
if ! curl -sS --fail-with-body --connect-timeout 5 --max-time 20 -o /dev/null -X PUT \
|
||||
-H "Authorization: Bearer $BUILDKITE_API_TOKEN" \
|
||||
"https://api.buildkite.com/v2/organizations/${BK_ORG}/pipelines/${BK_PIPELINE}/builds/${build_num}/cancel"; then
|
||||
echo "::warning::Could not cancel Buildkite build #$build_num; trying remaining builds."
|
||||
cancellation_failed=1
|
||||
fi
|
||||
done < "$builds_file"
|
||||
|
||||
if (( cancellation_failed != 0 )); then
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Check out the immutable BASE SHA: neither pull_request_target nor the
|
||||
# privileged slash-command call may run code from the untrusted PR head.
|
||||
- name: Checkout trusted merge planner
|
||||
if: steps.check.outputs.has_ready == 'true'
|
||||
uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683 # v4.2.2
|
||||
with:
|
||||
ref: ${{ steps.check.outputs.base_sha }}
|
||||
persist-credentials: false
|
||||
|
||||
- name: Collect changed paths
|
||||
if: steps.check.outputs.has_ready == 'true'
|
||||
env:
|
||||
GH_TOKEN: ${{ github.token }}
|
||||
PR_NUMBER: ${{ steps.check.outputs.pr_number }}
|
||||
EXPECTED_CHANGED_FILES: ${{ steps.check.outputs.changed_files }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
changed_json="$RUNNER_TEMP/merge-changed-files.json"
|
||||
changed_paths="$RUNNER_TEMP/merge-changed-paths.txt"
|
||||
if gh api --paginate --slurp \
|
||||
"repos/${GITHUB_REPOSITORY}/pulls/${PR_NUMBER}/files?per_page=100" \
|
||||
> "$changed_json"; then
|
||||
observed=$(jq '[.[][] | .filename] | unique | length' "$changed_json")
|
||||
if [ "$observed" = "$EXPECTED_CHANGED_FILES" ]; then
|
||||
jq -r '.[][] | .filename, (.previous_filename // empty)' "$changed_json" \
|
||||
| sort -u > "$changed_paths"
|
||||
else
|
||||
echo "::warning::Changed-file API returned $observed of $EXPECTED_CHANGED_FILES paths; selecting all merge lanes."
|
||||
echo '__FASTVIDEO_CI_PLAN_ALL__' > "$changed_paths"
|
||||
fi
|
||||
else
|
||||
echo "::warning::Changed-file API failed; selecting all merge lanes."
|
||||
echo '__FASTVIDEO_CI_PLAN_ALL__' > "$changed_paths"
|
||||
fi
|
||||
|
||||
- name: Select minimal merge tests
|
||||
id: plan
|
||||
if: steps.check.outputs.has_ready == 'true'
|
||||
run: |
|
||||
python3 .github/scripts/plan_merge_ci.py \
|
||||
--paths-file "$RUNNER_TEMP/merge-changed-paths.txt" \
|
||||
--github-output "$GITHUB_OUTPUT" \
|
||||
--summary-file "$GITHUB_STEP_SUMMARY"
|
||||
|
||||
- name: Wait for pre-commit and docs build
|
||||
if: steps.check.outputs.has_ready == 'true'
|
||||
env:
|
||||
GH_TOKEN: ${{ github.token }}
|
||||
PR_SHA: ${{ steps.check.outputs.head_sha }}
|
||||
PR_NUMBER: ${{ steps.check.outputs.pr_number }}
|
||||
run: bash .github/scripts/gate_full_suite.sh
|
||||
|
||||
- name: Trigger Buildkite merge gate
|
||||
if: steps.check.outputs.has_ready == 'true'
|
||||
env:
|
||||
BUILDKITE_API_TOKEN: ${{ secrets.BUILDKITE_API_TOKEN }}
|
||||
PR_SHA: ${{ steps.check.outputs.head_sha }}
|
||||
PR_BRANCH: ${{ steps.check.outputs.head_ref }}
|
||||
PR_NUMBER: ${{ steps.check.outputs.pr_number }}
|
||||
PR_TITLE: ${{ steps.check.outputs.title }}
|
||||
BK_ORG: ${{ vars.BUILDKITE_ORG_SLUG }}
|
||||
BK_PIPELINE: ${{ vars.BUILDKITE_PIPELINE_SLUG }}
|
||||
MERGE_TEST_PLAN: ${{ steps.plan.outputs.merge_test_plan }}
|
||||
MERGE_GOLDEN_TESTS: ${{ steps.plan.outputs.merge_golden_tests }}
|
||||
MERGE_SSIM_TESTS: ${{ steps.plan.outputs.merge_ssim_tests }}
|
||||
MERGE_PLAN_LABEL: ${{ steps.plan.outputs.merge_plan_label }}
|
||||
run: |
|
||||
curl -sS --fail-with-body -X POST \
|
||||
"https://api.buildkite.com/v2/organizations/${BK_ORG}/pipelines/${BK_PIPELINE}/builds" \
|
||||
-H "Authorization: Bearer $BUILDKITE_API_TOKEN" \
|
||||
-H "Content-Type: application/json" \
|
||||
--data-raw "$(jq -n \
|
||||
--arg commit "$PR_SHA" \
|
||||
--arg branch "$PR_BRANCH" \
|
||||
--arg message "Merge gate [${MERGE_PLAN_LABEL}] for PR #${PR_NUMBER}" \
|
||||
--arg pr_title "$PR_TITLE" \
|
||||
--arg merge_test_plan "$MERGE_TEST_PLAN" \
|
||||
--arg merge_golden_tests "$MERGE_GOLDEN_TESTS" \
|
||||
--arg merge_ssim_tests "$MERGE_SSIM_TESTS" \
|
||||
--argjson pr_id "$PR_NUMBER" \
|
||||
'{
|
||||
commit: $commit,
|
||||
branch: $branch,
|
||||
message: $message,
|
||||
ignore_pipeline_branch_filters: true,
|
||||
pull_request_id: $pr_id,
|
||||
pull_request_base_branch: "main",
|
||||
env: {
|
||||
TEST_SCOPE: "merge",
|
||||
FULL_SUITE: "true",
|
||||
MERGE_TEST_PLAN: $merge_test_plan,
|
||||
MERGE_GOLDEN_TESTS: $merge_golden_tests,
|
||||
MERGE_SSIM_TESTS: $merge_ssim_tests,
|
||||
PR_NUMBER: ($pr_id | tostring),
|
||||
PR_TITLE: $pr_title
|
||||
}
|
||||
}')"
|
||||
@@ -0,0 +1,65 @@
|
||||
name: Auto-Label Issues
|
||||
|
||||
on:
|
||||
issues:
|
||||
types: [opened, edited]
|
||||
|
||||
permissions:
|
||||
issues: write
|
||||
|
||||
jobs:
|
||||
label-issues:
|
||||
if: github.repository == 'hao-ai-lab/FastVideo'
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Label by keywords
|
||||
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7.0.1
|
||||
with:
|
||||
script: |
|
||||
const title = context.payload.issue.title.toLowerCase();
|
||||
const body = (context.payload.issue.body || '').toLowerCase();
|
||||
const text = title + ' ' + body;
|
||||
const labels = [];
|
||||
|
||||
const rules = [
|
||||
// scope labels (shared with PR labeling via Mergify)
|
||||
// Mapping: label → repo directories
|
||||
// scope: training → fastvideo/train/, fastvideo/training/, fastvideo/distillation/
|
||||
// scope: inference → fastvideo/pipelines/, fastvideo/entrypoints/, fastvideo/worker/
|
||||
// scope: attention → fastvideo/attention/
|
||||
// scope: kernel → fastvideo-kernel/, csrc/
|
||||
// scope: model → fastvideo/models/, fastvideo/layers/, fastvideo/configs/models/
|
||||
// scope: data → fastvideo/dataset/, fastvideo/pipelines/preprocess/
|
||||
// scope: distributed → fastvideo/distributed/
|
||||
// scope: docs → docs/
|
||||
{ keywords: ['training', 'finetune', 'fine-tune', 'lora', 'fsdp', 'distill'], label: 'scope: training' },
|
||||
{ keywords: ['inference', 'generate', 'pipeline', 'slow', 'latency'], label: 'scope: inference' },
|
||||
{ keywords: ['attention', 'vsa', 'flash', 'sta', 'vmoba', 'sparse attn'], label: 'scope: attention' },
|
||||
{ keywords: ['kernel', 'csrc', 'cuda kernel', 'thunderkittens'], label: 'scope: kernel' },
|
||||
{ keywords: ['wan', 'hunyuan', 'mochi', 'ltx', 'cogvideo', 'flux', 'sd3', 'cosmos'], label: 'scope: model' },
|
||||
{ keywords: ['dataset', 'dataloader', 'preprocessing', 'preprocess'], label: 'scope: data' },
|
||||
{ keywords: ['distributed', 'sequence parallel', 'fsdp', 'tensor parallel', 'multi-node', 'multi-gpu'], label: 'scope: distributed' },
|
||||
{ keywords: ['docs', 'documentation', 'tutorial', 'example'], label: 'scope: docs' },
|
||||
// issue-only labels (cross-module, no single repo directory)
|
||||
{ keywords: ['install', 'setup', 'pip', 'cuda', 'uv ', 'import error', 'modulenotfound'], label: 'installation' },
|
||||
{ keywords: ['memory', 'oom', 'out of memory', 'gpu memory', 'vram'], label: 'performance' },
|
||||
{ keywords: ['windows', 'macos', 'mac os', 'apple', 'mps', 'rocm', 'amd', 'npu'], label: 'platform' },
|
||||
];
|
||||
|
||||
for (const rule of rules) {
|
||||
if (rule.keywords.some(kw => text.includes(kw))) {
|
||||
labels.push(rule.label);
|
||||
}
|
||||
}
|
||||
|
||||
if (labels.length > 0) {
|
||||
await github.rest.issues.addLabels({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
issue_number: context.payload.issue.number,
|
||||
labels: labels,
|
||||
});
|
||||
console.log(`Added labels: ${labels.join(', ')}`);
|
||||
} else {
|
||||
console.log('No keyword matches found');
|
||||
}
|
||||
@@ -0,0 +1,51 @@
|
||||
name: Close Stale Issues and PRs
|
||||
|
||||
on:
|
||||
schedule:
|
||||
# Daily at 1:30 AM UTC
|
||||
- cron: '30 1 * * *'
|
||||
|
||||
jobs:
|
||||
stale:
|
||||
if: github.repository == 'hao-ai-lab/FastVideo'
|
||||
permissions:
|
||||
issues: write
|
||||
pull-requests: write
|
||||
actions: write
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/stale@997185467fa4f803885201cee163a9f38240193d # v10.1.1
|
||||
with:
|
||||
operations-per-run: 500
|
||||
|
||||
exempt-draft-pr: true
|
||||
exempt-issue-labels: 'keep-open,pinned,security,Bug,RFC'
|
||||
exempt-pr-labels: 'keep-open,pinned'
|
||||
|
||||
labels-to-add-when-unstale: 'unstale'
|
||||
labels-to-remove-when-stale: 'unstale'
|
||||
|
||||
days-before-issue-stale: 90
|
||||
days-before-issue-close: 30
|
||||
stale-issue-label: 'stale'
|
||||
stale-issue-message: >
|
||||
This issue has been automatically marked as stale because it has not
|
||||
had any activity within 90 days. It will be automatically closed if
|
||||
no further activity occurs within 30 days. Leave a comment if you
|
||||
feel this issue should remain open. Thank you!
|
||||
close-issue-message: >
|
||||
This issue has been automatically closed due to inactivity. Please
|
||||
feel free to reopen if you feel it is still relevant. Thank you!
|
||||
|
||||
days-before-pr-stale: 60
|
||||
days-before-pr-close: 14
|
||||
stale-pr-label: 'stale'
|
||||
stale-pr-message: >
|
||||
This pull request has been automatically marked as stale because it
|
||||
has not had any activity within 60 days. It will be automatically
|
||||
closed if no further activity occurs within 14 days. Leave a comment
|
||||
if you feel this pull request should remain open. Thank you!
|
||||
close-pr-message: >
|
||||
This pull request has been automatically closed due to inactivity.
|
||||
Please feel free to reopen if you intend to continue working on it.
|
||||
Thank you!
|
||||
@@ -0,0 +1,56 @@
|
||||
name: Welcome First-Time Contributors
|
||||
|
||||
on:
|
||||
issues:
|
||||
types: [opened]
|
||||
pull_request_target:
|
||||
types: [opened]
|
||||
|
||||
permissions:
|
||||
issues: write
|
||||
pull-requests: write
|
||||
|
||||
jobs:
|
||||
welcome:
|
||||
if: github.repository == 'hao-ai-lab/FastVideo'
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/first-interaction@34f15e814fe48ac9312ccf29db4e74fa767cbab7 # v1.3.0
|
||||
with:
|
||||
repo-token: ${{ secrets.GITHUB_TOKEN }}
|
||||
issue-message: |
|
||||
Welcome to FastVideo! Thanks for opening your first issue.
|
||||
|
||||
To help us investigate, please include:
|
||||
- **FastVideo version**: `pip show fastvideo`
|
||||
- **GPU**: `nvidia-smi` output (GPU model, driver, CUDA version)
|
||||
- **Python version**: `python --version`
|
||||
- **OS**: e.g., Ubuntu 22.04
|
||||
|
||||
If this is a bug, a minimal reproduction script helps us fix it faster.
|
||||
|
||||
Useful links:
|
||||
- [Documentation](https://hao-ai-lab.github.io/FastVideo)
|
||||
- [Contributing Guide](https://hao-ai-lab.github.io/FastVideo/contributing/overview/)
|
||||
- [Slack](https://join.slack.com/t/fastvideo/shared_invite/zt-3f4lao1uq-u~Ipx6Lt4J27AlD2y~IdLQ)
|
||||
pr-message: |
|
||||
Welcome to FastVideo! Thanks for your first pull request.
|
||||
|
||||
**How our CI works:**
|
||||
|
||||
PRs run a three-tier CI system:
|
||||
1. **Pre-commit** — formatting (yapf), linting (ruff), type checking (mypy). Runs immediately on every PR.
|
||||
2. **Fastcheck** — six core GPU lanes run automatically via Buildkite (~10-15 min).
|
||||
3. **Merge gate** — a reviewer adds `ready`; changed paths select only the relevant integration, training, golden, or SSIM coverage.
|
||||
|
||||
**Before your PR is reviewed:**
|
||||
- [ ] `pre-commit run --all-files` passes locally
|
||||
- [ ] You've added or updated tests for your changes
|
||||
- [ ] The PR description explains what and why
|
||||
|
||||
If pre-commit fails, a bot comment will explain how to fix it. Fastcheck and merge-gate results appear in the Checks section below.
|
||||
|
||||
**Useful links:**
|
||||
- [Contributing Guide](https://hao-ai-lab.github.io/FastVideo/contributing/overview/)
|
||||
- [Development Roadmap](https://github.com/hao-ai-lab/FastVideo/issues/899)
|
||||
- [Slack](https://join.slack.com/t/fastvideo/shared_invite/zt-3f4lao1uq-u~Ipx6Lt4J27AlD2y~IdLQ)
|
||||
@@ -1,83 +0,0 @@
|
||||
# Sample workflow for building and deploying a Hugo site to GitHub Pages
|
||||
name: Deploy FastVideo Docs to Pages
|
||||
|
||||
on:
|
||||
# Runs on pushes targeting the default branch
|
||||
push:
|
||||
branches:
|
||||
- main
|
||||
paths:
|
||||
- "docs/**/*.md"
|
||||
- "fastvideo/v1/examples/**/*.py"
|
||||
pull_request:
|
||||
branches:
|
||||
- main
|
||||
types: [opened, ready_for_review, synchronize, reopened]
|
||||
paths:
|
||||
- "docs/**/*.md"
|
||||
- "fastvideo/v1/examples/**/*.py"
|
||||
|
||||
# Allows you to run this workflow manually from the Actions tab
|
||||
workflow_dispatch:
|
||||
|
||||
# Sets permissions of the GITHUB_TOKEN to allow deployment to GitHub Pages
|
||||
permissions:
|
||||
contents: read
|
||||
pages: write
|
||||
id-token: write
|
||||
|
||||
# Allow only one concurrent deployment, skipping runs queued between the run in-progress and latest queued.
|
||||
# However, do NOT cancel in-progress runs as we want to allow these production deployments to complete.
|
||||
concurrency:
|
||||
group: "pages"
|
||||
cancel-in-progress: false
|
||||
|
||||
# Default to bash
|
||||
defaults:
|
||||
run:
|
||||
shell: bash
|
||||
|
||||
jobs:
|
||||
pre-commit:
|
||||
uses: ./.github/workflows/pre-commit.yml
|
||||
|
||||
# Build job
|
||||
build:
|
||||
runs-on: ubuntu-latest
|
||||
needs: pre-commit
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
- name: Setup Pages
|
||||
id: pages
|
||||
uses: actions/configure-pages@v5
|
||||
- name: Set up Python
|
||||
uses: actions/setup-python@v5
|
||||
with:
|
||||
python-version: "3.10"
|
||||
- name: Install dependencies
|
||||
run: |
|
||||
cd docs
|
||||
pip install -r requirements-docs.txt
|
||||
- name: Build docs
|
||||
run: |
|
||||
cd docs
|
||||
make clean
|
||||
make html
|
||||
- name: Upload artifact
|
||||
uses: actions/upload-pages-artifact@v3
|
||||
with:
|
||||
path: ./docs/build/html
|
||||
|
||||
# Deployment job
|
||||
deploy:
|
||||
environment:
|
||||
name: github-pages
|
||||
url: ${{ steps.deployment.outputs.page_url }}
|
||||
if: ${{ github.event_name == 'push' }}
|
||||
runs-on: ubuntu-latest
|
||||
needs: build
|
||||
steps:
|
||||
- name: Deploy to GitHub Pages
|
||||
id: deployment
|
||||
uses: actions/deploy-pages@v4
|
||||
@@ -1,71 +0,0 @@
|
||||
name: Publish FastVideo to PyPI on Version Change
|
||||
|
||||
on:
|
||||
push:
|
||||
branches:
|
||||
- main
|
||||
paths:
|
||||
- 'pyproject.toml' # Trigger when pyproject.toml changes
|
||||
workflow_dispatch:
|
||||
|
||||
jobs:
|
||||
check-version-change:
|
||||
runs-on: ubuntu-latest
|
||||
outputs:
|
||||
version-changed: ${{ steps.check-version.outputs.changed }}
|
||||
new-version: ${{ steps.check-version.outputs.new-version }}
|
||||
steps:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@v4
|
||||
with:
|
||||
fetch-depth: 2
|
||||
|
||||
- name: Check if version changed
|
||||
id: check-version
|
||||
run: |
|
||||
# Get current commit's version
|
||||
NEW_VERSION=$(grep -oP "version\\s*=\\s*\"\\K[^\"]+\"" pyproject.toml)
|
||||
echo "New version: $NEW_VERSION"
|
||||
|
||||
# Get previous version from git history
|
||||
OLD_VERSION=$(git show HEAD~1:./pyproject.toml | grep -oP "version\\s*=\\s*\"\\K[^\"]+\"" || echo "0.0.0")
|
||||
echo "Old version: $OLD_VERSION"
|
||||
|
||||
if [ "$NEW_VERSION" != "$OLD_VERSION" ]; then
|
||||
echo "Version changed from $OLD_VERSION to $NEW_VERSION"
|
||||
echo "changed=true" >> $GITHUB_OUTPUT
|
||||
echo "new-version=$NEW_VERSION" >> $GITHUB_OUTPUT
|
||||
else
|
||||
echo "Version did not change"
|
||||
echo "changed=false" >> $GITHUB_OUTPUT
|
||||
fi
|
||||
|
||||
build-publish-main:
|
||||
needs: check-version-change
|
||||
if: ${{ needs.check-version-change.outputs.version-changed == 'true' || github.event_name == 'workflow_dispatch' }}
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
id-token: write # Needed for OIDC Trusted Publishing
|
||||
|
||||
steps:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Set up Python
|
||||
uses: actions/setup-python@v5
|
||||
with:
|
||||
python-version: '3.10'
|
||||
|
||||
- name: Install build dependencies
|
||||
run: |
|
||||
python -m pip install --upgrade pip
|
||||
pip install build twine wheel
|
||||
|
||||
- name: Build package
|
||||
run: |
|
||||
python -m build
|
||||
|
||||
- name: Publish release distributions to PyPI
|
||||
uses: pypa/gh-action-pypi-publish@release/v1
|
||||
with:
|
||||
packages-dir: dist/
|
||||
@@ -0,0 +1,257 @@
|
||||
name: Build and Push Docker Images
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
build_cuda_matrix:
|
||||
description: 'Build multi-arch CUDA images (12.6.3 default; 13.0.0 explicitly tagged)'
|
||||
required: false
|
||||
default: true
|
||||
type: boolean
|
||||
build_dreamverse_matrix:
|
||||
description: 'Build the amd64 Dreamverse matrix (backend + UI x CUDA 12.6.3/13.0.0)'
|
||||
required: false
|
||||
default: false
|
||||
type: boolean
|
||||
build_ci_runner_image:
|
||||
description: 'Build the ARM64 CUDA 13 CI runner image (sm_100)'
|
||||
required: false
|
||||
default: false
|
||||
type: boolean
|
||||
# Auto-rebuild the CUDA images when a repository-controlled image input
|
||||
# changes on main. This includes the trusted SM89 kernel artifact's source,
|
||||
# metadata/key helper, ABI dependency metadata, and build orchestration.
|
||||
# Dreamverse (apps/dreamverse/docker/Dockerfile) and the ROCm Dockerfile stay
|
||||
# manual-dispatch only.
|
||||
push:
|
||||
branches: [main]
|
||||
paths:
|
||||
- '.dockerignore'
|
||||
- '.github/workflows/_template-build-image.yml'
|
||||
- '.github/workflows/infra-build-image.yml'
|
||||
- '.gitmodules'
|
||||
- 'docker/Dockerfile'
|
||||
- 'docker/uv-excludes'
|
||||
- 'fastvideo-kernel/**'
|
||||
- 'fastvideo/tests/modal/kernel_build_cache.py'
|
||||
- 'pyproject.toml'
|
||||
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
packages: write
|
||||
|
||||
# One static group, no cancellation: every run of this workflow writes the same
|
||||
# mutable registry tags (latest, py3.12-latest, ...), so runs must serialize —
|
||||
# concurrent push/dispatch runs would race on those tags, and cancelling a run
|
||||
# mid-publish can strand the cu126/cu130 tag families at different commits. An
|
||||
# in-flight superseded build wastes its runner time, but its tags are then
|
||||
# overwritten by the newer queued run. GitHub keeps a single pending run per
|
||||
# group: the newest queued run replaces any older queued one.
|
||||
concurrency:
|
||||
group: infra-build-image
|
||||
cancel-in-progress: false
|
||||
|
||||
jobs:
|
||||
# CUDA matrix: Python 3.12 x {12.6.3, 13.0.0} x {amd64, arm64}. Each architecture
|
||||
# builds natively and pushes only by digest; publish-cuda-manifests is the sole
|
||||
# owner of the shared tags. CUDA 12.6 arm64 targets Hopper (sm_90a / GH200 class),
|
||||
# because CUDA 12.6 cannot compile sm_121. CUDA 13 arm64 targets GB10 (sm_121).
|
||||
# 12.6.3/cu126 owns the default `py3.12`/`latest` tags and keeps its versioned
|
||||
# aliases; 13.0.0/cu130 is published under explicit versioned tags. Flash-attn
|
||||
# 2.8.3 comes from the architecture-specific prebuilt releases.
|
||||
build-cuda-images:
|
||||
# Runs on a manual dispatch when build_cuda_matrix is set, or automatically
|
||||
# on an in-scope main push (inputs are null on push). The
|
||||
# repository guard keeps fork syncs from auto-building; manual dispatch
|
||||
# still works in forks.
|
||||
if: ${{ (github.event_name == 'push' && github.repository == 'hao-ai-lab/FastVideo') || github.event.inputs.build_cuda_matrix == 'true' }}
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
cuda:
|
||||
- version: '12.6.3'
|
||||
suffix: '-cuda12.6.3'
|
||||
torch_backend: 'cu126'
|
||||
fa_tag: 'cu126torch2.12'
|
||||
cmake_build_parallel_level: '4'
|
||||
torch_cuda_arch_list:
|
||||
amd64: '9.0a'
|
||||
arm64: '9.0a'
|
||||
- version: '13.0.0'
|
||||
suffix: '-cuda13.0.0'
|
||||
torch_backend: 'cu130'
|
||||
fa_tag: 'cu130torch2.12'
|
||||
cmake_build_parallel_level: '1'
|
||||
torch_cuda_arch_list:
|
||||
amd64: '9.0a'
|
||||
arm64: '12.1'
|
||||
architecture:
|
||||
- name: amd64
|
||||
runner: ubuntu-latest
|
||||
- name: arm64
|
||||
runner: ubuntu-24.04-arm
|
||||
uses: ./.github/workflows/_template-build-image.yml
|
||||
with:
|
||||
python_version: '3.12'
|
||||
dockerfile_path: docker/Dockerfile
|
||||
tag_suffix: py3.12${{ matrix.cuda.suffix }}
|
||||
runner: ${{ matrix.architecture.runner }}
|
||||
architecture: ${{ matrix.architecture.name }}
|
||||
push_by_digest: true
|
||||
digest_artifact_name: fastvideo-dev-cuda${{ matrix.cuda.version }}-${{ matrix.architecture.name }}
|
||||
build_args: |
|
||||
PYTHON_VERSION=3.12
|
||||
CUDA_VERSION=${{ matrix.cuda.version }}
|
||||
UV_TORCH_BACKEND=${{ matrix.cuda.torch_backend }}
|
||||
TORCH_CUDA_ARCH_LIST=${{ matrix.cuda.torch_cuda_arch_list[matrix.architecture.name] }}
|
||||
CMAKE_BUILD_PARALLEL_LEVEL=${{ matrix.cuda.cmake_build_parallel_level }}
|
||||
FLASH_ATTN_WHEEL_TAG=${{ matrix.cuda.fa_tag }}
|
||||
FLASH_ATTN_WHEEL_RELEASE=https://github.com/mjun0812/flash-attention-prebuild-wheels/releases/download/v0.9.17
|
||||
FLASH_ATTN_WHEEL_RELEASE_ARM64=https://github.com/mjun0812/flash-attention-prebuild-wheels/releases/download/v0.9.22
|
||||
secrets: inherit
|
||||
|
||||
publish-cuda-manifests:
|
||||
# !cancelled(): publish lanes whose digests exist even if a sibling build
|
||||
# leg failed (the digest-count check fails incomplete lanes); it also
|
||||
# bypasses skipped-needs propagation, hence the explicit skipped check.
|
||||
if: ${{ !cancelled() && needs.build-cuda-images.result != 'skipped' }}
|
||||
needs: build-cuda-images
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
contents: read
|
||||
packages: write
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
cuda:
|
||||
- version: '12.6.3'
|
||||
suffix: '-cuda12.6.3'
|
||||
is_default: true
|
||||
- version: '13.0.0'
|
||||
suffix: '-cuda13.0.0'
|
||||
is_default: false
|
||||
steps:
|
||||
- name: Download architecture digests
|
||||
uses: actions/download-artifact@v4
|
||||
with:
|
||||
path: /tmp/digests
|
||||
pattern: fastvideo-dev-cuda${{ matrix.cuda.version }}-*
|
||||
merge-multiple: true
|
||||
|
||||
- name: Set up Docker Buildx
|
||||
uses: docker/setup-buildx-action@v3
|
||||
|
||||
- name: Login to GitHub Container Registry
|
||||
uses: docker/login-action@v3
|
||||
with:
|
||||
registry: ghcr.io
|
||||
username: ${{ github.repository_owner }}
|
||||
password: ${{ secrets.GITHUB_TOKEN }}
|
||||
|
||||
- name: Normalize image reference
|
||||
id: image
|
||||
env:
|
||||
IMAGE: ghcr.io/${{ github.repository }}/fastvideo-dev
|
||||
run: echo "name=${IMAGE,,}" >> "${GITHUB_OUTPUT}"
|
||||
|
||||
- name: Prepare tags
|
||||
id: prepare-tags
|
||||
env:
|
||||
VERSIONED_TAG_PREFIX: py3.12${{ matrix.cuda.suffix }}
|
||||
PUBLISH_DEFAULT_TAGS: ${{ matrix.cuda.is_default }}
|
||||
run: |
|
||||
SHORT_SHA="${GITHUB_SHA::7}"
|
||||
TAGS="type=raw,value=${VERSIONED_TAG_PREFIX}-latest\ntype=raw,value=${VERSIONED_TAG_PREFIX}-sha-${SHORT_SHA}"
|
||||
if [[ "${PUBLISH_DEFAULT_TAGS}" == "true" ]]; then
|
||||
TAGS="type=raw,value=latest\ntype=raw,value=py3.12-latest\ntype=raw,value=py3.12-sha-${SHORT_SHA}\n${TAGS}"
|
||||
fi
|
||||
{
|
||||
echo "tags<<EOF"
|
||||
echo -e "${TAGS}"
|
||||
echo "EOF"
|
||||
} >> "${GITHUB_OUTPUT}"
|
||||
|
||||
- name: Extract metadata for Docker
|
||||
id: meta
|
||||
uses: docker/metadata-action@v5
|
||||
with:
|
||||
images: ${{ steps.image.outputs.name }}
|
||||
tags: ${{ steps.prepare-tags.outputs.tags }}
|
||||
|
||||
- name: Create multi-architecture manifests
|
||||
env:
|
||||
IMAGE: ${{ steps.image.outputs.name }}
|
||||
run: |
|
||||
mapfile -t DIGESTS < <(find /tmp/digests -maxdepth 1 -type f -printf '%f\n' | sort)
|
||||
if [[ "${#DIGESTS[@]}" -ne 2 ]]; then
|
||||
echo "Expected exactly two architecture digests, found ${#DIGESTS[@]}" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
mapfile -t TAGS < <(jq -r '.tags[]' <<< "${DOCKER_METADATA_OUTPUT_JSON}")
|
||||
TAG_ARGS=()
|
||||
for TAG in "${TAGS[@]}"; do
|
||||
TAG_ARGS+=(--tag "${TAG}")
|
||||
done
|
||||
|
||||
IMAGE_REFS=()
|
||||
for DIGEST in "${DIGESTS[@]}"; do
|
||||
IMAGE_REFS+=("${IMAGE}@sha256:${DIGEST}")
|
||||
done
|
||||
|
||||
docker buildx imagetools create "${TAG_ARGS[@]}" "${IMAGE_REFS[@]}"
|
||||
docker buildx imagetools inspect "${TAGS[0]}"
|
||||
|
||||
# The CI runner is ARM64 like DGX Spark, but targets sm_100 rather than sm_121.
|
||||
# Publish a single-architecture variant so the self-hosted CI runner can reuse
|
||||
# the exact prebuilt kernel instead of compiling it in every job.
|
||||
build-ci-runner-image:
|
||||
if: ${{ (github.event_name == 'push' && github.repository == 'hao-ai-lab/FastVideo') || github.event.inputs.build_ci_runner_image == 'true' }}
|
||||
uses: ./.github/workflows/_template-build-image.yml
|
||||
with:
|
||||
python_version: '3.12'
|
||||
dockerfile_path: docker/Dockerfile
|
||||
tag_suffix: py3.12-cuda13.0.0-sm100
|
||||
runner: ubuntu-24.04-arm
|
||||
architecture: arm64
|
||||
build_args: |
|
||||
PYTHON_VERSION=3.12
|
||||
CUDA_VERSION=13.0.0
|
||||
UV_TORCH_BACKEND=cu130
|
||||
TORCH_CUDA_ARCH_LIST=10.0
|
||||
CMAKE_BUILD_PARALLEL_LEVEL=1
|
||||
FLASH_ATTN_WHEEL_TAG=cu130torch2.12
|
||||
FLASH_ATTN_WHEEL_RELEASE_ARM64=https://github.com/mjun0812/flash-attention-prebuild-wheels/releases/download/v0.9.22
|
||||
secrets: inherit
|
||||
|
||||
# Dreamverse matrix: {backend, UI} x {12.6.3, 13.0.0}, Python 3.12. Torch backend
|
||||
# matches the base CUDA (cu126 / cu130). Keep these images amd64-only until the
|
||||
# required FA4 dependency stack is available and validated on arm64.
|
||||
build-dreamverse:
|
||||
if: ${{ github.event.inputs.build_dreamverse_matrix == 'true' }}
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
cuda:
|
||||
- version: '12.6.3'
|
||||
torch_backend: 'cu126'
|
||||
- version: '13.0.0'
|
||||
torch_backend: 'cu130'
|
||||
variant:
|
||||
- name: backend
|
||||
ui: '0'
|
||||
- name: ui
|
||||
ui: '1'
|
||||
uses: ./.github/workflows/_template-build-image.yml
|
||||
with:
|
||||
python_version: '3.12'
|
||||
dockerfile_path: apps/dreamverse/docker/Dockerfile
|
||||
tag_suffix: dreamverse-${{ matrix.variant.name }}-cuda${{ matrix.cuda.version }}
|
||||
image_name: dreamverse
|
||||
build_args: |
|
||||
CUDA_VERSION=${{ matrix.cuda.version }}
|
||||
UV_TORCH_BACKEND=${{ matrix.cuda.torch_backend }}
|
||||
BUILD_DREAMVERSE_UI=${{ matrix.variant.ui }}
|
||||
include_latest_tags: false
|
||||
secrets: inherit
|
||||
@@ -0,0 +1,80 @@
|
||||
name: Deploy Documentation
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [ main ]
|
||||
paths:
|
||||
- 'docs/**'
|
||||
- 'examples/**'
|
||||
- 'scripts/inference/**'
|
||||
- 'mkdocs.yml'
|
||||
- 'requirements-mkdocs.in'
|
||||
- 'requirements-mkdocs.txt'
|
||||
- 'scripts/check_docs_links.py'
|
||||
- '.github/workflows/infra-docs.yml'
|
||||
pull_request:
|
||||
branches: [ main ]
|
||||
paths:
|
||||
- 'docs/**'
|
||||
- 'examples/**'
|
||||
- 'scripts/inference/**'
|
||||
- 'mkdocs.yml'
|
||||
- 'requirements-mkdocs.in'
|
||||
- 'requirements-mkdocs.txt'
|
||||
- 'scripts/check_docs_links.py'
|
||||
- '.github/workflows/infra-docs.yml'
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
pages: write
|
||||
id-token: write
|
||||
|
||||
concurrency:
|
||||
group: "pages"
|
||||
cancel-in-progress: false
|
||||
|
||||
jobs:
|
||||
build:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
with:
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Setup Python
|
||||
uses: actions/setup-python@v5
|
||||
with:
|
||||
python-version: '3.12'
|
||||
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@v3
|
||||
|
||||
- name: Install dependencies
|
||||
run: uv pip install --system -r requirements-mkdocs.txt
|
||||
|
||||
- name: Setup Pages
|
||||
uses: actions/configure-pages@v4
|
||||
|
||||
- name: Build documentation
|
||||
run: mkdocs build
|
||||
|
||||
- name: Check docs links
|
||||
run: python scripts/check_docs_links.py
|
||||
|
||||
- name: Upload artifact
|
||||
uses: actions/upload-pages-artifact@v3
|
||||
with:
|
||||
path: ./site
|
||||
|
||||
deploy:
|
||||
environment:
|
||||
name: github-pages
|
||||
url: ${{ steps.deployment.outputs.page_url }}
|
||||
runs-on: ubuntu-latest
|
||||
needs: build
|
||||
if: github.ref == 'refs/heads/main'
|
||||
steps:
|
||||
- name: Deploy to GitHub Pages
|
||||
id: deployment
|
||||
uses: actions/deploy-pages@v4
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user