Compare commits
643
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
0403c6f47e | ||
|
|
a55b1cdde4 | ||
|
|
aec9a20a16 | ||
|
|
b4458e5bae | ||
|
|
a790153705 | ||
|
|
0df1445d0d | ||
|
|
038da6e02b | ||
|
|
3fb2fbe1a2 | ||
|
|
acea9d23e1 | ||
|
|
b7a448cf5b | ||
|
|
04e32991c6 | ||
|
|
424c643b6e | ||
|
|
de803cb250 | ||
|
|
effc1d3492 | ||
|
|
b1ddb1ba33 | ||
|
|
9ff65c83b6 | ||
|
|
15473f3c77 | ||
|
|
490641cf47 | ||
|
|
4ad6880ca6 | ||
|
|
9d0af28307 | ||
|
|
d6dfe95466 | ||
|
|
636d3b743e | ||
|
|
e3a5c6954f | ||
|
|
f633e30ebb | ||
|
|
5ce4947ac2 | ||
|
|
323d74c0d2 | ||
|
|
d98aeafc86 | ||
|
|
f6396fb8c6 | ||
|
|
6300329cd5 | ||
|
|
c17d33bf33 | ||
|
|
2aaeee2ab8 | ||
|
|
eb3a394224 | ||
|
|
f673423b51 | ||
|
|
eb0a41528a | ||
|
|
140bd1a6cf | ||
|
|
11f5a8e582 | ||
|
|
71b3cb8c34 | ||
|
|
c85f6a477f | ||
|
|
40d4930d73 | ||
|
|
f9be085243 | ||
|
|
36b53ff350 | ||
|
|
9801037c3d | ||
|
|
74d09b0efd | ||
|
|
38dc8820ac | ||
|
|
c77a76c6af | ||
|
|
d14d5aadea | ||
|
|
4c915b7742 | ||
|
|
9a8bbe18fa | ||
|
|
ea25441ef0 | ||
|
|
48957fcde1 | ||
|
|
7b872cc41e | ||
|
|
37418946c8 | ||
|
|
95fd29e0cb | ||
|
|
e17cd2633c | ||
|
|
e0dc5f2b0c | ||
|
|
70ee5d230c | ||
|
|
24ced500f5 | ||
|
|
4ddcdf541f | ||
|
|
0e3529869c | ||
|
|
e1e0d91c00 | ||
|
|
145a3f166b | ||
|
|
88a5a933ab | ||
|
|
c591d6d2a6 | ||
|
|
65dff806a8 | ||
|
|
b85f0f4c2a | ||
|
|
76c62d7a00 | ||
|
|
f6e65ff668 | ||
|
|
c220aa8000 | ||
|
|
4713fc17ed | ||
|
|
5789955bbe | ||
|
|
2ad84a3b78 | ||
|
|
12d699cd78 | ||
|
|
34f14ded21 | ||
|
|
71d1ab411f | ||
|
|
805e487773 | ||
|
|
8803b4547e | ||
|
|
3b3806b3f6 | ||
|
|
38d962e89d | ||
|
|
3966a365d0 | ||
|
|
d73fd14af0 | ||
|
|
a87cc89916 | ||
|
|
81fd80c8ee | ||
|
|
ff22439f28 | ||
|
|
de0de04212 | ||
|
|
7f2c3e1f64 | ||
|
|
ab55e57c22 | ||
|
|
46f6b43a53 | ||
|
|
833a33b663 | ||
|
|
9ea1307cd4 | ||
|
|
be35003cb1 | ||
|
|
26bd4db253 | ||
|
|
e294ca011c | ||
|
|
2085a4fc4a | ||
|
|
c0c8e39c04 | ||
|
|
f72618dafb | ||
|
|
b3edfacdd8 | ||
|
|
30129a3350 | ||
|
|
4d49f7b0aa | ||
|
|
71bfc13d75 | ||
|
|
74db6e18d1 | ||
|
|
7d263c6a36 | ||
|
|
454c32d1d1 | ||
|
|
d1240b9238 | ||
|
|
4105094fa5 | ||
|
|
f036469d3d | ||
|
|
14261bc98c | ||
|
|
d92858659d | ||
|
|
1a383f3f66 | ||
|
|
bc27a032c5 | ||
|
|
2b13e117f0 | ||
|
|
99c166c381 | ||
|
|
95066245db | ||
|
|
6dcaac768b | ||
|
|
02c1c49b75 | ||
|
|
cd1b7cf139 | ||
|
|
e63b7d8ac4 | ||
|
|
5190c1bb1e | ||
|
|
2cb3bba658 | ||
|
|
f9e1c46c3c | ||
|
|
e1eda47589 | ||
|
|
d902967208 | ||
|
|
fea556269b | ||
|
|
69dd3c68f6 | ||
|
|
5433f6e80b | ||
|
|
e315657066 | ||
|
|
f8d9a0c57f | ||
|
|
fa6d276925 | ||
|
|
fc80d95d7e | ||
|
|
37cab18780 | ||
|
|
128d0b7fc5 | ||
|
|
8092f02e6d | ||
|
|
03d9ce2edb | ||
|
|
10fc92dba5 | ||
|
|
6736dc06a5 | ||
|
|
8c002c62af | ||
|
|
7061313d04 | ||
|
|
76d3ba69e0 | ||
|
|
8e39ce38c9 | ||
|
|
d4bd8bf2c0 | ||
|
|
e83d7bc50c | ||
|
|
ff3d5aff75 | ||
|
|
959dbcc8a2 | ||
|
|
36bf37e9ba | ||
|
|
7a83e0e6fc | ||
|
|
8be1313b86 | ||
|
|
d925ad05f3 | ||
|
|
7f795600c8 | ||
|
|
ec16b6b01d | ||
|
|
31c0f1b341 | ||
|
|
4bee0fa199 | ||
|
|
530e6b8363 | ||
|
|
9ab2725db1 | ||
|
|
0aff68f51d | ||
|
|
ad58f802f3 | ||
|
|
04fa356ee3 | ||
|
|
f9c076fe2b | ||
|
|
f76efe798e | ||
|
|
09f455233e | ||
|
|
aea300f690 | ||
|
|
b92219f6a6 | ||
|
|
a321b95a8a | ||
|
|
98308db7e0 | ||
|
|
c1e18f6722 | ||
|
|
75e193a2c9 | ||
|
|
d6e0a7d0dd | ||
|
|
7fc5f241da | ||
|
|
aae48a7e90 | ||
|
|
88f38eb0f4 | ||
|
|
d750b463dc | ||
|
|
e10b26a3d8 | ||
|
|
74636ba246 | ||
|
|
7e2f3f14e7 | ||
|
|
caa1c402ba | ||
|
|
38a6bd93d3 | ||
|
|
b867ef7e7c | ||
|
|
3ae58c277a | ||
|
|
0c6862ca55 | ||
|
|
e8c854bcf1 | ||
|
|
06860e96fe | ||
|
|
1b503554d1 | ||
|
|
351ceb7c59 | ||
|
|
10875e0d7b | ||
|
|
1eaae8a10b | ||
|
|
59e00f6164 | ||
|
|
745cc05b10 | ||
|
|
c5dc244871 | ||
|
|
dbf3917bf4 | ||
|
|
050f189c95 | ||
|
|
029216029f | ||
|
|
31f44110b5 | ||
|
|
21f3ce6577 | ||
|
|
785d123e36 | ||
|
|
d58c551c11 | ||
|
|
560628709c | ||
|
|
0f53b51e6c | ||
|
|
06093a9c4e | ||
|
|
dbddfab6d2 | ||
|
|
7188170277 | ||
|
|
b7f69c2c1d | ||
|
|
23a4531491 | ||
|
|
7d52ad0118 | ||
|
|
4d7bf35fa3 | ||
|
|
a6a9c9ca07 | ||
|
|
f4704847c2 | ||
|
|
d9c996310b | ||
|
|
d6651afd2e | ||
|
|
cf67618cad | ||
|
|
2f0a2b3c57 | ||
|
|
e7748d9952 | ||
|
|
8eb3140b2f | ||
|
|
d6ddcea682 | ||
|
|
3559ba2377 | ||
|
|
61e63ea0d7 | ||
|
|
4ce4ac4734 | ||
|
|
e7f6db9bd1 | ||
|
|
d83f45a6a0 | ||
|
|
dd91542cd1 | ||
|
|
581e8115fe | ||
|
|
dea69cf651 | ||
|
|
60ac6537df | ||
|
|
5285116e73 | ||
|
|
40ce2d72f5 | ||
|
|
704bc9aaf9 | ||
|
|
de264fcc99 | ||
|
|
7b952e4673 | ||
|
|
551b2d2048 | ||
|
|
7bfaf82fd7 | ||
|
|
9cd6a86b95 | ||
|
|
16e9552778 | ||
|
|
87f8a2782d | ||
|
|
cbbb09d7b8 | ||
|
|
2f6230abcf | ||
|
|
f8bfc76015 | ||
|
|
8f1e6c3336 | ||
|
|
8e7d2e7879 | ||
|
|
6ab2870942 | ||
|
|
e0ad145152 | ||
|
|
da04d08426 | ||
|
|
1f70032af5 | ||
|
|
7f71994653 | ||
|
|
2bb3349da1 | ||
|
|
8fe1689968 | ||
|
|
e53730f324 | ||
|
|
7a4fe9086a | ||
|
|
d277361aae | ||
|
|
734a54e7a9 | ||
|
|
91364982df | ||
|
|
50145e4fcb | ||
|
|
4112507e99 | ||
|
|
424fc2b4ae | ||
|
|
e6066223e6 | ||
|
|
b6fa3d24d8 | ||
|
|
55c2e7cd76 | ||
|
|
5a549af823 | ||
|
|
92fb660c2e | ||
|
|
3ff640b2e6 | ||
|
|
c722429ab5 | ||
|
|
e04a192de6 | ||
|
|
c9ca6d1298 | ||
|
|
754292c419 | ||
|
|
0082bc66fc | ||
|
|
8b1937422e | ||
|
|
fb6cbf23e6 | ||
|
|
c8fdd5ed7b | ||
|
|
1c19a6a00c | ||
|
|
d44409c704 | ||
|
|
5d1c7852b7 | ||
|
|
77a211d006 | ||
|
|
bef8169bb1 | ||
|
|
681f1583f9 | ||
|
|
e3b4564d5a | ||
|
|
c0d03fc43d | ||
|
|
404ee8538e | ||
|
|
e57ac59462 | ||
|
|
8c55fdaf7e | ||
|
|
c30779184f | ||
|
|
9d188c0b6c | ||
|
|
9dd7c54221 | ||
|
|
62b95d8287 | ||
|
|
fdf21702f5 | ||
|
|
2972fc9449 | ||
|
|
8f5712629f | ||
|
|
436c701b9f | ||
|
|
543fea88e3 | ||
|
|
bdec816b31 | ||
|
|
2cd2e57d2e | ||
|
|
9370234294 | ||
|
|
50da62e722 | ||
|
|
4f3e8751db | ||
|
|
f4c58894d9 | ||
|
|
01c94ef385 | ||
|
|
2415226d25 | ||
|
|
404314d00f | ||
|
|
87489f0872 | ||
|
|
9ce7c8039e | ||
|
|
e1e25e95f9 | ||
|
|
490bde90e1 | ||
|
|
dc7596b973 | ||
|
|
335afa4457 | ||
|
|
3f77a6805a | ||
|
|
13d0aae706 | ||
|
|
404cbf4f3c | ||
|
|
958ffec844 | ||
|
|
31f000d1cc | ||
|
|
cd32b3e02f | ||
|
|
bf27908095 | ||
|
|
c5f9ea53b2 | ||
|
|
d32a7184da | ||
|
|
2930abe456 | ||
|
|
b93ef4289d | ||
|
|
401bdbd316 | ||
|
|
1048d79cf8 | ||
|
|
1e8406162d | ||
|
|
03edd35c83 | ||
|
|
ac11127397 | ||
|
|
e028dcc7c0 | ||
|
|
076f45c1ee | ||
|
|
85eb7265db | ||
|
|
d3ceb67e66 | ||
|
|
7ac153a5ca | ||
|
|
d1e7aa0abd | ||
|
|
2d846c55a1 | ||
|
|
b318063c0a | ||
|
|
4aa307be55 | ||
|
|
055e52e5ea | ||
|
|
7d2069596b | ||
|
|
c45009c9a4 | ||
|
|
b91020b407 | ||
|
|
2dcc5ea4f6 | ||
|
|
359151d9a0 | ||
|
|
ce67cd3729 | ||
|
|
7c554e5da8 | ||
|
|
663ea33ff1 | ||
|
|
3ef04f1654 | ||
|
|
0eced76a41 | ||
|
|
3ab6470d1a | ||
|
|
989a03532c | ||
|
|
fa15369a02 | ||
|
|
78a9cb88d8 | ||
|
|
a0bff12746 | ||
|
|
98f2af94e5 | ||
|
|
46f7b6d574 | ||
|
|
911a6a6a35 | ||
|
|
38c7949d5c | ||
|
|
7e7a0dba9d | ||
|
|
f62e210ae6 | ||
|
|
6ceb4942a0 | ||
|
|
8cae5e4708 | ||
|
|
2a773fa34e | ||
|
|
5357f63327 | ||
|
|
60f61c8101 | ||
|
|
3d75ba8251 | ||
|
|
6c6bcd914d | ||
|
|
f79b08de81 | ||
|
|
f2bc037fff | ||
|
|
86604a684b | ||
|
|
47bd1e0178 | ||
|
|
c41305ad18 | ||
|
|
98ce9034f0 | ||
|
|
0ceff110da | ||
|
|
1d018acb3e | ||
|
|
7d8cf38dbe | ||
|
|
8d483fe4aa | ||
|
|
c1191250bf | ||
|
|
4b7266349a | ||
|
|
22f9b7681f | ||
|
|
589d32cc39 | ||
|
|
89199837db | ||
|
|
d6ebaf1b49 | ||
|
|
fac927777c | ||
|
|
ecbd697dae | ||
|
|
7d4acef64d | ||
|
|
9f0ce517cf | ||
|
|
c718e56b0d | ||
|
|
b65f0316d1 | ||
|
|
8d8bcb76b0 | ||
|
|
5f42748ed1 | ||
|
|
c9005045dc | ||
|
|
6c81befc87 | ||
|
|
dfe0b288e1 | ||
|
|
31200fbb83 | ||
|
|
9185978c55 | ||
|
|
fcba463553 | ||
|
|
2c53d3eecf | ||
|
|
516ecd374a | ||
|
|
3b1b54a74d | ||
|
|
6914e7c904 | ||
|
|
5452369749 | ||
|
|
a113311e77 | ||
|
|
44da97da92 | ||
|
|
f759980a58 | ||
|
|
37e0f8c236 | ||
|
|
51711d5906 | ||
|
|
6375223b16 | ||
|
|
4cb046768d | ||
|
|
3322542444 | ||
|
|
65f707354b | ||
|
|
109e2e7e9d | ||
|
|
cbc3a6bb9d | ||
|
|
2fa8d4ae6d | ||
|
|
7b6c8aee99 | ||
|
|
6284eaa363 | ||
|
|
636524e87f | ||
|
|
202b2f3972 | ||
|
|
247fe273d8 | ||
|
|
cb320dfa3a | ||
|
|
d8bb5abc46 | ||
|
|
cc703eca51 | ||
|
|
81c9df629c | ||
|
|
d3c0c52208 | ||
|
|
744e0555c0 | ||
|
|
3a38f7dfdc | ||
|
|
f572319bd9 | ||
|
|
48528f468c | ||
|
|
4264a80ca9 | ||
|
|
8573d4f05e | ||
|
|
210a733515 | ||
|
|
0aef0e6f63 | ||
|
|
dd022ad9be | ||
|
|
832ad61e5b | ||
|
|
9419c04ee3 | ||
|
|
a37b39d83c | ||
|
|
bb8c769c8e | ||
|
|
576c214f28 | ||
|
|
b79d1fc15b | ||
|
|
eb66e1c18d | ||
|
|
616d43c1cf | ||
|
|
7244a4b27f | ||
|
|
7e5ebb4582 | ||
|
|
65ed588570 | ||
|
|
14adfe2edc | ||
|
|
6198c6a640 | ||
|
|
e6b71b531b | ||
|
|
ae1d112c6a | ||
|
|
bf4de1f38f | ||
|
|
66fdcc8e76 | ||
|
|
ed1e8d6bad | ||
|
|
b9423ca3f8 | ||
|
|
ad16289871 | ||
|
|
2a41da1e6b | ||
|
|
32133171da | ||
|
|
19674c6f29 | ||
|
|
508afb7002 | ||
|
|
288ea88105 | ||
|
|
eb0f1318f3 | ||
|
|
ce9b5910cc | ||
|
|
d0e5a6214a | ||
|
|
834562b2db | ||
|
|
060cc7b9ba | ||
|
|
6c58a5ba62 | ||
|
|
48d9f61f86 | ||
|
|
5f938b5844 | ||
|
|
74da2a7370 | ||
|
|
580d6dfe1f | ||
|
|
344e43006a | ||
|
|
c5155b256e | ||
|
|
e005c7f3ac | ||
|
|
ff5a79ef60 | ||
|
|
ab01dc4ba5 | ||
|
|
285a950c1b | ||
|
|
46a0a85d85 | ||
|
|
4aeabbc629 | ||
|
|
949bb5c835 | ||
|
|
aab74c1271 | ||
|
|
f89d86944f | ||
|
|
8741d204a5 | ||
|
|
cdc85f58a8 | ||
|
|
0262d2f089 | ||
|
|
62c0343465 | ||
|
|
1e1a023fb0 | ||
|
|
1d2517ad8e | ||
|
|
d41186cb4a | ||
|
|
78e0c7eec9 | ||
|
|
1c41a94b62 | ||
|
|
2e66aafe20 | ||
|
|
55074bda76 | ||
|
|
de65bec2b7 | ||
|
|
7664dd0de3 | ||
|
|
019a88ced4 | ||
|
|
72de11abcc | ||
|
|
d71a4ebffc | ||
|
|
1089ab43bf | ||
|
|
97d4b984c9 | ||
|
|
2a8953d74d | ||
|
|
8801b10da7 | ||
|
|
6b413f2ec4 | ||
|
|
28b72694aa | ||
|
|
4afb0cfe4f | ||
|
|
3eec1281cf | ||
|
|
0660489e38 | ||
|
|
dd871a17bf | ||
|
|
dc11529862 | ||
|
|
ffabf85e31 | ||
|
|
c0026ca5ba | ||
|
|
0f2bbe71ac | ||
|
|
2a46902ecb | ||
|
|
66012d3a4c | ||
|
|
f666b9de41 | ||
|
|
7e3c073b55 | ||
|
|
a6aa21bd07 | ||
|
|
6519b57aab | ||
|
|
675aea6ece | ||
|
|
46e7a15e0d | ||
|
|
e4f702d7ec | ||
|
|
bb68fcc809 | ||
|
|
b392e6a874 | ||
|
|
0991003905 | ||
|
|
8f8ce6d9e1 | ||
|
|
e3d0cbe185 | ||
|
|
d5ec468d43 | ||
|
|
e55fa6e5dc | ||
|
|
61b6ddeee1 | ||
|
|
a9a000f45d | ||
|
|
66b8b8561e | ||
|
|
6684872616 | ||
|
|
7f654e3332 | ||
|
|
8631c1b806 | ||
|
|
5357e12b5a | ||
|
|
bdfdf1dfee | ||
|
|
d156461785 | ||
|
|
6edf113838 | ||
|
|
dcf7738cbc | ||
|
|
b2ebaaf865 | ||
|
|
7768bb80f6 | ||
|
|
a335811869 | ||
|
|
357b0533fe | ||
|
|
2ec3732758 | ||
|
|
a004408a93 | ||
|
|
007e237e69 | ||
|
|
8e18dc9f71 | ||
|
|
7ab32539af | ||
|
|
6ef8fcb61d | ||
|
|
016e24da63 | ||
|
|
85b8717545 | ||
|
|
657fd745e1 | ||
|
|
12647457a7 | ||
|
|
298f74f956 | ||
|
|
ee8babb298 | ||
|
|
60295cc03f | ||
|
|
1572e13b6e | ||
|
|
a157275b4c | ||
|
|
c4dbe7dac3 | ||
|
|
d39591108e | ||
|
|
ace6e971e5 | ||
|
|
b4f6758253 | ||
|
|
535d29b392 | ||
|
|
b4255517e0 | ||
|
|
6eeb60613f | ||
|
|
53d2c7791f | ||
|
|
53cb693dca | ||
|
|
6f72d24876 | ||
|
|
d1459e9976 | ||
|
|
59ab481eb1 | ||
|
|
0cf001986a | ||
|
|
51956369a5 | ||
|
|
4b0970cbbf | ||
|
|
94bf47a572 | ||
|
|
1a3ac9074b | ||
|
|
fb0581d5b0 | ||
|
|
6c74ab4132 | ||
|
|
9a91021c56 | ||
|
|
51c94d6a73 | ||
|
|
dba38dbc03 | ||
|
|
2034cc3c4f | ||
|
|
c69afce2f6 | ||
|
|
b08e758eb3 | ||
|
|
3f3462d7ce | ||
|
|
c9c47dd89c | ||
|
|
9c4ef7c2f1 | ||
|
|
f25eb4b905 | ||
|
|
048d55ccbb | ||
|
|
a271c55fe4 | ||
|
|
f663ae0d8a | ||
|
|
c0911aa3dd | ||
|
|
5f59687ae7 | ||
|
|
5adbc81cdc | ||
|
|
f26d5c37c1 | ||
|
|
6a4ef42378 | ||
|
|
f1098c77dc | ||
|
|
0405b618f8 | ||
|
|
eac79b753f | ||
|
|
4d58cf20d0 | ||
|
|
52c93ecc9d | ||
|
|
42d63166ac | ||
|
|
6db20345a2 | ||
|
|
ad27ea596c | ||
|
|
9aadb4bf8c | ||
|
|
bd941df271 | ||
|
|
8a73876d3b | ||
|
|
1483a1138a | ||
|
|
5e243d8292 | ||
|
|
b0c66d3200 | ||
|
|
c86da2c736 | ||
|
|
bae2a19dcf | ||
|
|
057686f59d | ||
|
|
67da56628b | ||
|
|
2325adffa2 | ||
|
|
20cf836ef1 | ||
|
|
2bf69b6f92 | ||
|
|
13583f5ffb | ||
|
|
008ee2099a | ||
|
|
137f61f2fe | ||
|
|
ccb262974e | ||
|
|
30966e3bc9 | ||
|
|
7b4272d6b7 | ||
|
|
15553f7706 | ||
|
|
60eeea50bb | ||
|
|
927b3a40b9 | ||
|
|
c64f826ae2 | ||
|
|
55c1040f0b | ||
|
|
8a3e7aa761 | ||
|
|
4324c1c21d | ||
|
|
2c342ee37f | ||
|
|
708201f531 | ||
|
|
1fee098f10 | ||
|
|
8a77cf22c9 | ||
|
|
d869d90d12 | ||
|
|
554ee17de5 | ||
|
|
0be4fc62c9 | ||
|
|
1e08893546 | ||
|
|
09ab452610 | ||
|
|
e768b5ec5b | ||
|
|
59ec42f40e | ||
|
|
5ae5b247b3 | ||
|
|
ead6c62be4 | ||
|
|
6805eaa06c | ||
|
|
c39a15551c | ||
|
|
e6dda263b0 | ||
|
|
f9482d113c | ||
|
|
a3ec969397 | ||
|
|
76a12cc8a1 | ||
|
|
ac490399c6 | ||
|
|
9ea39cee57 | ||
|
|
52e6e612a2 | ||
|
|
9aebc4ada1 | ||
|
|
b53cf7425c | ||
|
|
d9ce056901 | ||
|
|
218449c54d | ||
|
|
221958bcde | ||
|
|
4a1f1e35bb | ||
|
|
e0e05f97f2 | ||
|
|
dd75ee8509 | ||
|
|
0aed1868df |
@@ -0,0 +1,46 @@
|
||||
# Exploration Logs
|
||||
|
||||
This directory holds draft procedures and investigation notes for tasks that
|
||||
don't yet have a standardized skill or SOP. Each exploration should follow this
|
||||
template.
|
||||
|
||||
## When to Create an Exploration Log
|
||||
|
||||
- You are working on a task with no existing skill or workflow.
|
||||
- You are experimenting with a new metric, training technique, or tool.
|
||||
- You want to document findings before they are promoted to a standard.
|
||||
|
||||
## File Naming
|
||||
|
||||
`<topic-slug>.md` — e.g., `fvd-metric-investigation.md`
|
||||
|
||||
## Template
|
||||
|
||||
```markdown
|
||||
# Exploration Log: <Topic>
|
||||
|
||||
## Status: draft | under_review | promoted | abandoned
|
||||
|
||||
## Context
|
||||
<Why this exploration is needed — link to experiment or task if applicable.>
|
||||
|
||||
## Progress
|
||||
- [ ] Step 1: ...
|
||||
- [ ] Step 2: ...
|
||||
|
||||
## Findings
|
||||
<What you have learned so far.>
|
||||
|
||||
## Mistakes / Dead Ends
|
||||
<What didn't work and why — these become lessons.>
|
||||
|
||||
## Proposed Standardization
|
||||
<If this works, describe the skill/SOP/workflow to create.>
|
||||
```
|
||||
|
||||
## Lifecycle
|
||||
|
||||
1. **Create** during exploration mode.
|
||||
2. **Update** as you make progress.
|
||||
3. **Promote**: If findings are solid, create a skill in `.agents/skills/` or an SOP in `.agents/workflows/`.
|
||||
4. **Archive mistakes**: Move failures into `.agents/lessons/`.
|
||||
@@ -0,0 +1,73 @@
|
||||
---
|
||||
date: 2026-05-07
|
||||
experiment: PR #1280 (daVinci-MagiHuman port), distill DiT parity bring-up
|
||||
category: porting
|
||||
severity: important
|
||||
---
|
||||
|
||||
# Conversion `--cast-bf16` Needs an FP32-Keep Suffix Allowlist
|
||||
|
||||
## What Happened
|
||||
|
||||
`scripts/checkpoint_conversion/convert_magi_human_to_diffusers.py --cast-bf16`
|
||||
produced a converted distill DiT checkpoint that loaded cleanly, ran end-to-
|
||||
end, and emitted reasonable output — but `test_magi_human_distill_parity`
|
||||
showed `diff_mean=0.114` against the upstream reference. The base DiT was
|
||||
bit-exact with the same conversion script. Only the distill variant
|
||||
regressed.
|
||||
|
||||
The error was small enough that visual quality looked normal, but large
|
||||
enough to fail bit-exact parity. The MagiHuman base + distill DiTs share
|
||||
most of their architecture, so a difference that affected only distill was
|
||||
counterintuitive.
|
||||
|
||||
## Root Cause
|
||||
|
||||
`--cast-bf16` was downcasting **all** fp32 tensors to bf16 indiscriminately.
|
||||
The base checkpoint and the FastVideo `final_linear` / adapter modules
|
||||
require eight specific tensors to remain in fp32:
|
||||
|
||||
- LayerNorm `gamma` / `beta` weights for the final residual exit
|
||||
- Adapter projection biases
|
||||
- A handful of scale parameters in the output projection chain
|
||||
|
||||
These tensors participate in chains where bf16 precision causes accumulation
|
||||
error large enough to drift the parity check. The base DiT happened to not
|
||||
hit those specific chains in the path the test exercised (different
|
||||
attention mask shape, different audio interleave); the distill variant did.
|
||||
|
||||
## Fix / Workaround
|
||||
|
||||
Added `_FP32_KEEP_SUFFIXES` allowlist to
|
||||
`convert_magi_human_to_diffusers.py` (commit `829f70d3`) and gated `--cast-
|
||||
bf16` on it. Tensors whose state-dict key ends with any allowlisted suffix
|
||||
keep their original fp32 dtype regardless of the flag.
|
||||
|
||||
Distill DiT parity went from `diff_mean=0.114` (silently wrong) to bit-exact
|
||||
in one commit.
|
||||
|
||||
## Prevention
|
||||
|
||||
1. **Treat `--cast-bf16` as opinionated, not blanket.** Any conversion
|
||||
script that supports a global dtype downcast flag MUST own an explicit
|
||||
allowlist of fp32-keep tensors, documented at the top of the file.
|
||||
|
||||
2. **The `add-model-conversion` skill** should enforce two checks for any
|
||||
converter that ships a `--cast-bf16`-style flag:
|
||||
- Run the parity test for **every** variant of the model (base, distill,
|
||||
SR, etc.), not just the headline variant. Different variants exercise
|
||||
different code paths.
|
||||
- Diff the converted checkpoint's dtype map against the upstream
|
||||
reference and assert the allowlist covers every fp32 tensor in the
|
||||
reference.
|
||||
|
||||
3. **For MagiHuman specifically**: if you add or rename DiT modules that
|
||||
touch `final_linear`, the adapter, or any LayerNorm in the residual exit
|
||||
path, **check that any fp32-required tensors are covered by
|
||||
`_FP32_KEEP_SUFFIXES`** in the conversion script and re-run
|
||||
`test_magi_human_distill_parity` (it's the canary).
|
||||
|
||||
4. The lesson generalizes beyond MagiHuman: any DiT that uses bf16 mixed
|
||||
precision but keeps specific tensors in fp32 (a common pattern with
|
||||
flash-attn-style backends) needs this allowlist for any conversion that
|
||||
downcasts.
|
||||
@@ -0,0 +1,77 @@
|
||||
---
|
||||
date: 2026-05-07
|
||||
experiment: PR #1280 (daVinci-MagiHuman port), DiT parity bring-up
|
||||
category: porting
|
||||
severity: important
|
||||
---
|
||||
|
||||
# DiT Dtype Boundary Alignment with Flash-Attn-Style Backends
|
||||
|
||||
## What Happened
|
||||
|
||||
DiT bit-exact parity for daVinci-MagiHuman against the upstream reference
|
||||
sat at `diff_max=0.5` after the architecture port was complete and weight
|
||||
loading was correct. The error grew with depth (later layers diverged more
|
||||
than earlier ones), suggesting an accumulating numerical drift rather than
|
||||
a structural mismatch. None of the obvious culprits (RoPE, GQA expansion,
|
||||
attention mask handling) accounted for the pattern.
|
||||
|
||||
## Root Cause
|
||||
|
||||
Four cumulative dtype-boundary mismatches, each individually small but
|
||||
together pushing parity from `diff_max=0.5` to bit-exact (`diff_max=0.0`):
|
||||
|
||||
1. **SDPA inputs were not cast to bf16.** Upstream's `flash_attn_with_cp`
|
||||
internally casts Q/K/V to bf16 at `dit_module.py:508` before the kernel.
|
||||
FastVideo was passing fp32 tensors through, getting numerically different
|
||||
intermediates even though the kernel accepts both.
|
||||
|
||||
2. **Post-attention output was kept in bf16 across the per-head gating
|
||||
multiply.** Upstream upcasts to fp32 before the gating, FastVideo did the
|
||||
gate in bf16 then upcast.
|
||||
|
||||
3. **A residual-stream cast at the block boundary.** FastVideo had a
|
||||
`.to(bf16)` then `.to(fp32)` at the start of each block. Upstream keeps
|
||||
the residual stream **continuously in fp32** across all 40 layers; only
|
||||
the inputs to specific kernels are temporarily downcast.
|
||||
|
||||
4. **Parity test scheduler used a double-shift.** A separate per-block fix
|
||||
(Wave 11 production migration) — single-shift schedule is what upstream
|
||||
uses; the parity test was double-shifting.
|
||||
|
||||
## Fix / Workaround
|
||||
|
||||
Four cumulative changes in `fastvideo/models/dits/magi_human.py` (commit
|
||||
`3a4816cb`), each with a comment at the call site explaining the upstream
|
||||
parity rationale:
|
||||
|
||||
- Cast SDPA inputs to bf16 right before the attention call.
|
||||
- Upcast attention output to fp32 before the per-head gate multiply.
|
||||
- Drop the residual-stream `.to(bf16)`/`.to(fp32)` wrapper at the block
|
||||
boundary; let the residual stay fp32 throughout.
|
||||
- Single-shift schedule in the parity test fixture (matches upstream Wave 11).
|
||||
|
||||
## Prevention
|
||||
|
||||
1. **For any DiT port with a flash-attn-style backend**, treat the dtype of
|
||||
the residual stream as a load-bearing invariant, not a performance knob.
|
||||
Document it in the model's per-pipeline AGENTS.md. MagiHuman's invariant:
|
||||
*residual stream stays fp32 across all blocks; only kernel inputs are
|
||||
temporarily bf16*.
|
||||
|
||||
2. **Use layer-by-layer activation hooks** when DiT parity is close-but-not-
|
||||
bit-exact and the gap grows with depth. The
|
||||
`fastvideo/hooks/activation_trace.py` infra exists exactly for this case
|
||||
(`add-model-trace` skill). In MagiHuman's case it would have localized the
|
||||
first divergence point in one pass.
|
||||
|
||||
3. **The `add-model-port-dit` skill** should explicitly call out:
|
||||
- SDPA input dtype must match the upstream kernel's internal cast.
|
||||
- Post-attention upcast happens **before** any per-head gate, not after.
|
||||
- Residual stream dtype across block boundaries is a parity invariant.
|
||||
These rules apply to any DiT port whose upstream uses a flash-attn-style
|
||||
backend (`flash_attn_with_cp`, `flex_flash_attn_func`, etc.).
|
||||
|
||||
4. **Add an "intermediate-layer parity" test** for new DiT ports — comparing
|
||||
activations at layer 5, 10, 20, 30 — not just the final output. A growing-
|
||||
with-depth pattern is otherwise indistinguishable from "almost right".
|
||||
@@ -0,0 +1,69 @@
|
||||
---
|
||||
date: 2026-05-07
|
||||
experiment: PR #1280 (daVinci-MagiHuman port), Wave 14
|
||||
category: porting
|
||||
severity: critical
|
||||
---
|
||||
|
||||
# Silent Channel-Major Token-Packing Bugs
|
||||
|
||||
## What Happened
|
||||
|
||||
While porting daVinci-MagiHuman (`fastvideo/pipelines/basic/magi_human/`),
|
||||
the pipeline-parity test passed bit-exactly but the E2E user-visible output
|
||||
was **pure static noise**. Latent tensors compared identically against the
|
||||
upstream reference at every checkpointed boundary, yet decoded videos showed
|
||||
no recognizable content. The discrepancy reproduced on every variant
|
||||
(base / distill / SR-540p / SR-1080p) with the same noise profile.
|
||||
|
||||
## Root Cause
|
||||
|
||||
Video tokens were being packed **spatial-major** instead of **channel-major**:
|
||||
|
||||
```python
|
||||
# What we had (spatial-major, WRONG)
|
||||
einops.rearrange(x, "b c (T pT) (H pH) (W pW) -> b (T H W) (pT pH pW C)", ...)
|
||||
|
||||
# What upstream's UnfoldNd produces (channel-major, CORRECT)
|
||||
einops.rearrange(x, "b c (T pT) (H pH) (W pW) -> b (T H W) (C pT pH pW)", ...)
|
||||
```
|
||||
|
||||
A single-character einops reorder. The pipeline-parity test used FastVideo's
|
||||
own packer on **both** sides of the comparison, so the bug was invisible there
|
||||
— both sides agreed on the wrong layout. The DiT consumed those tokens
|
||||
without complaint because the channel dimension only matters at decode time,
|
||||
when the VAE's first conv expects channel-major input. By that point the test
|
||||
boundary was already passed.
|
||||
|
||||
The bug was load-bearing for any token-packed format that downstream feeds
|
||||
into a `UnfoldNd`-shaped consumer. Wave 14 of the port took multiple bug-hunt
|
||||
iterations and an Oracle consultation to localize.
|
||||
|
||||
## Fix / Workaround
|
||||
|
||||
Single-character einops change in `stages/latent_preparation.py:_img2tokens`
|
||||
(commit `6d190693` of the original PR). After the fix, all four variants
|
||||
produced expected E2E output and the pipeline-parity tests still passed
|
||||
because both sides of the parity check are now correct.
|
||||
|
||||
## Prevention
|
||||
|
||||
1. **Never use the FastVideo-side packer on both sides of a parity test.**
|
||||
At least one parity boundary must compare against an upstream tensor
|
||||
produced by the upstream packer. For MagiHuman this means a separate
|
||||
`_img2tokens` parity test that feeds upstream `UnfoldNd` output as the
|
||||
reference, not FastVideo's reformatted equivalent.
|
||||
|
||||
2. **Add an E2E hash check** alongside latent-parity. The mp4 SHA was the
|
||||
first signal that something was wrong; if it had been part of the standard
|
||||
parity battery, the bug would have surfaced in Wave 1, not Wave 14. See
|
||||
`fastvideo/tests/ssim/test_magi_human_similarity.py` for the CI version.
|
||||
|
||||
3. **For any new model port that involves explicit tensor reshaping into
|
||||
tokens**, document the expected packing order (`(C pT pH pW)` vs
|
||||
`(pT pH pW C)`) at the call site and assert the layout matches the
|
||||
downstream consumer's expectation.
|
||||
|
||||
4. The `add-model-port-dit` skill's parity gate should require an E2E hash
|
||||
check for any DiT that does video token packing, not just latent
|
||||
bit-exactness.
|
||||
@@ -0,0 +1,48 @@
|
||||
# Lessons Learned Database
|
||||
|
||||
This directory stores documented mistakes, unexpected behaviors, and their fixes.
|
||||
Each lesson is a permanent record that helps agents and humans avoid repeating
|
||||
past errors.
|
||||
|
||||
## When to Create a Lesson
|
||||
|
||||
- An experiment failed for a non-obvious reason.
|
||||
- A configuration or hyperparameter choice led to wasted compute.
|
||||
- A porting, data, or infrastructure issue was discovered and resolved.
|
||||
- A workaround was needed for a known framework/library bug.
|
||||
|
||||
## File Naming
|
||||
|
||||
`<YYYY-MM-DD>_<short-slug>.md` — e.g., `2026-03-02_lr-too-high-for-lora.md`
|
||||
|
||||
## Template
|
||||
|
||||
```markdown
|
||||
---
|
||||
date: <ISO-8601>
|
||||
experiment: <reference to experiment_journal.md entry, if applicable>
|
||||
category: hyperparameter | data | infrastructure | evaluation | porting | other
|
||||
severity: critical | important | minor
|
||||
---
|
||||
|
||||
# <Short Descriptive Title>
|
||||
|
||||
## What Happened
|
||||
<Description of the problem and its symptoms.>
|
||||
|
||||
## Root Cause
|
||||
<Analysis of why it happened.>
|
||||
|
||||
## Fix / Workaround
|
||||
<What resolved the issue.>
|
||||
|
||||
## Prevention
|
||||
<How to avoid this in the future — updated skills, SOPs, or checks.>
|
||||
```
|
||||
|
||||
## Usage
|
||||
|
||||
- Before starting a task, **search this directory** for relevant lessons.
|
||||
- After completing or failing a task, **check if a new lesson should be created**.
|
||||
- Periodically review lessons for **patterns** — recurring themes may warrant
|
||||
a new skill, SOP, or codebase fix.
|
||||
@@ -0,0 +1,130 @@
|
||||
# FastVideo-WorldModel — Codebase Map
|
||||
|
||||
High-level structural index for agent orientation. Updated 2026-03-08.
|
||||
|
||||
## Repository Layout
|
||||
|
||||
```
|
||||
FastVideo-WorldModel/
|
||||
├── fastvideo/ # Core Python package
|
||||
│ ├── models/ # Model implementations
|
||||
│ │ ├── dits/ # DiT transformers (wanvideo, ltx2, ...)
|
||||
│ │ ├── vaes/ # VAE models
|
||||
│ │ ├── encoders/ # Text/image encoders (T5, CLIP)
|
||||
│ │ ├── schedulers/ # Noise schedulers
|
||||
│ │ ├── upsamplers/ # Super-resolution models
|
||||
│ │ ├── audio/ # Audio models
|
||||
│ │ └── loader/ # Component loaders for HF repos
|
||||
│ ├── configs/ # Configuration system
|
||||
│ │ ├── models/ # Arch configs + param_names_mapping
|
||||
│ │ ├── pipelines/ # Pipeline wiring
|
||||
│ │ └── sample/ # Default sampling parameters
|
||||
│ ├── pipelines/ # End-to-end pipelines
|
||||
│ │ ├── basic/ # Per-model pipelines (wan/, ltx2/, ...)
|
||||
│ │ └── stages/ # Reusable pipeline stages
|
||||
│ ├── train/ # Refactored training framework (YAML-driven, preferred)
|
||||
│ │ ├── trainer.py # Main training loop coordinator
|
||||
│ │ ├── entrypoint/ # Training entrypoint (train.py) + checkpoint conversion
|
||||
│ │ ├── methods/ # Training algorithms (FineTune, DFSFT, DMD2, SelfForcing)
|
||||
│ │ │ ├── base.py # TrainingMethod ABC
|
||||
│ │ │ ├── fine_tuning/ # FineTuneMethod, DiffusionForcingSFTMethod
|
||||
│ │ │ └── distribution_matching/ # DMD2Method, SelfForcingMethod
|
||||
│ │ ├── models/ # Per-role model wrappers (ModelBase, CausalModelBase)
|
||||
│ │ │ ├── wan/ # WanModel, WanCausalModel
|
||||
│ │ │ └── matrixgame/ # MatrixGameModel, MatrixGameCausalModel
|
||||
│ │ ├── callbacks/ # Composable hooks (grad_clip, ema, validation)
|
||||
│ │ └── utils/ # Config, builder, checkpoint, optimizer, tracking
|
||||
│ ├── training/ # Legacy training infrastructure (being phased out)
|
||||
│ │ ├── trackers.py # W&B tracker (BaseTracker → WandbTracker)
|
||||
│ │ ├── training_utils.py # Checkpointing, grad clipping, state dicts
|
||||
│ │ ├── training_pipeline.py # Base training pipeline
|
||||
│ │ ├── wan_training_pipeline.py # Wan T2V training
|
||||
│ │ ├── wan_i2v_training_pipeline.py # Wan I2V training
|
||||
│ │ ├── distillation_pipeline.py # Distillation base
|
||||
│ │ ├── wan_distillation_pipeline.py # Wan distillation
|
||||
│ │ ├── self_forcing_distillation_pipeline.py # Self-forcing distill
|
||||
│ │ ├── ltx2_training_pipeline.py # LTX-2 training
|
||||
│ │ └── matrixgame_training_pipeline.py # MatrixGame training
|
||||
│ ├── attention/ # Attention backends
|
||||
│ ├── distributed/ # Sequence/tensor parallel utilities
|
||||
│ ├── layers/ # Tensor-parallel layers
|
||||
│ ├── tests/ # Package-level tests
|
||||
│ │ ├── training/ # Training regression tests (W&B summary comparison)
|
||||
│ │ ├── ssim/ # SSIM visual regression tests
|
||||
│ │ ├── encoders/ # Encoder parity tests
|
||||
│ │ └── modal/ # Modal CI test runner
|
||||
│ └── registry.py # Unified config registry
|
||||
├── fastvideo-kernel/ # CUDA/custom kernels (separate build: ./build.sh)
|
||||
├── scripts/ # Utility scripts
|
||||
│ ├── distill/ # Distillation launch scripts
|
||||
│ ├── inference/ # Inference scripts
|
||||
│ ├── checkpoint_conversion/ # Weight conversion tools
|
||||
│ ├── finetune/ # Finetune scripts
|
||||
│ └── preprocess/ # Data preprocessing
|
||||
├── examples/ # Ready-to-run examples
|
||||
│ ├── training/ # Training examples (finetune/, consistency_finetune/)
|
||||
│ ├── distill/ # Distillation examples
|
||||
│ ├── inference/ # Inference examples
|
||||
│ └── dataset/ # Dataset examples
|
||||
├── docs/ # MkDocs documentation source
|
||||
│ ├── design/overview.md # Architecture overview
|
||||
│ ├── training/ # Training guides
|
||||
│ └── contributing/ # Contributor guides + coding_agents.md
|
||||
├── tests/ # Top-level tests (local_tests/)
|
||||
├── AGENTS.md # Agent coding guidelines
|
||||
└── .agents/ # Agent infrastructure (you are here)
|
||||
```
|
||||
|
||||
## Key Training Entrypoints
|
||||
|
||||
### New framework (`fastvideo/train/`) — preferred
|
||||
|
||||
| Method | Config Example | Launch Pattern |
|
||||
|--------|---------------|----------------|
|
||||
| FineTune (Wan) | `examples/train/finetune_wan2.1_t2v_1.3B_vsa_*.yaml` | `torchrun -m fastvideo.train.entrypoint.train --config <yaml>` |
|
||||
| DFSFT (Wan causal) | `examples/train/dfsft_wan_causal_t2v_1.3B.yaml` | `torchrun -m fastvideo.train.entrypoint.train --config <yaml>` |
|
||||
| DMD2 distillation | `examples/train/distill_wan2.1_t2v_1.3B_dmd2.yaml` | `torchrun -m fastvideo.train.entrypoint.train --config <yaml>` |
|
||||
| Self-Forcing | `examples/train/self_forcing_wan_causal_t2v_1.3B.yaml` | `torchrun -m fastvideo.train.entrypoint.train --config <yaml>` |
|
||||
|
||||
### Legacy pipelines (`fastvideo/training/`) — being phased out
|
||||
|
||||
| Pipeline | Entrypoint | Launch Pattern |
|
||||
|----------|-----------|----------------|
|
||||
| Wan T2V finetune | `fastvideo/training/wan_training_pipeline.py` | `torchrun --nproc_per_node N` |
|
||||
| Wan I2V finetune | `fastvideo/training/wan_i2v_training_pipeline.py` | `torchrun --nproc_per_node N` |
|
||||
| Wan distillation (DMD) | `fastvideo/training/wan_distillation_pipeline.py` | `torchrun --nproc_per_node N` |
|
||||
| Self-forcing distill | `fastvideo/training/wan_self_forcing_distillation_pipeline.py` | `torchrun --nproc_per_node N` |
|
||||
| LTX-2 finetune | `fastvideo/training/ltx2_training_pipeline.py` | `torchrun --nproc_per_node N` |
|
||||
| MatrixGame | `fastvideo/training/matrixgame_training_pipeline.py` | `torchrun --nproc_per_node N` |
|
||||
|
||||
## W&B Integration
|
||||
|
||||
- **Tracker classes**: `fastvideo/training/trackers.py`
|
||||
- `WandbTracker` — logs metrics, videos, timing
|
||||
- `SequentialTracker` — fan-out to multiple trackers
|
||||
- `DummyTracker` — no-op for offline/test
|
||||
- **Run summary location**: `<output_dir>/tracker/wandb/latest-run/files/wandb-summary.json`
|
||||
- **Reference summaries**: `fastvideo/tests/training/*/` (e.g., `a40_reference_wandb_summary.json`)
|
||||
- **Environment**: `WANDB_API_KEY`, `WANDB_BASE_URL`, `WANDB_MODE`
|
||||
|
||||
## Critical Environment Variables
|
||||
|
||||
| Variable | Purpose |
|
||||
|----------|---------|
|
||||
| `WANDB_API_KEY` | W&B authentication |
|
||||
| `WANDB_MODE` | `online` / `offline` |
|
||||
| `FASTVIDEO_ATTENTION_BACKEND` | `FLASH_ATTN` / `TORCH_SDPA` |
|
||||
| `TOKENIZERS_PARALLELISM` | Set `false` to avoid fork warnings |
|
||||
| `HF_HOME` | HuggingFace cache directory |
|
||||
|
||||
## Build & Test Commands
|
||||
|
||||
```bash
|
||||
uv pip install -e ".[dev]" # Editable install
|
||||
pre-commit run --all-files # Lint/format/spell
|
||||
pytest tests/ # Top-level tests
|
||||
pytest fastvideo/tests/ -v # Package tests
|
||||
pytest fastvideo/tests/training/Vanilla -srP # Training loss regression
|
||||
pytest fastvideo/tests/ssim/ -vs # SSIM visual regression
|
||||
cd fastvideo-kernel && ./build.sh # Build kernels
|
||||
```
|
||||
@@ -0,0 +1,201 @@
|
||||
# Dreamverse Integration — Memory Index
|
||||
|
||||
Living knowledge base for the FastVideo ↔ Dreamverse ↔ Dynamo integration.
|
||||
Tracks the public API refactor (PRs 0-17), the LTX-2 streaming server
|
||||
upstream, the Dreamverse switch from `FastVideo-internal` to public
|
||||
`FastVideo`, and the NVFP4 quantization landing.
|
||||
|
||||
**Last reconciled:** 2026-05-06 (**D-26** EXECUTED — rebased
|
||||
`will/dreamverse-monorepo` directly onto `origin/main` (`c17d33bf`)
|
||||
via `git rebase origin/main`; 66 commits cleanly replayed; force-
|
||||
pushed via `--force-with-lease`. New tip `83829c5e`. Local backup
|
||||
branch `will/dreamverse-monorepo-pre-main-rebase-backup-20260506`
|
||||
preserved at the pre-rebase tip `2ee839a3`. PR #1288 on
|
||||
`will/ltx2_sr_port` is untouched. The branch is now ready to open
|
||||
as a single PR against main.
|
||||
|
||||
**Earlier — D-21** EXECUTED — chunk-stutter root cause
|
||||
analysis + NVENC build path + opt-in `--nvenc` flag + benchmark regression
|
||||
test. 3 parallel explore agents confirmed apps/dreamverse matches
|
||||
FastVideo-internal byte-for-byte on NVFP4 + torch.compile coverage; stutter
|
||||
is NOT a regression. Software libx264 encoding consumes ~22% of segment
|
||||
wall-time. Built ffmpeg with NVENC support; B200 silicon doesn't have NVENC
|
||||
encoder hardware so the path is currently moot on this dev host but works
|
||||
on RTX 50-series / T4 / A10 deploys. Benchmark captured libx264 ultrafast
|
||||
at 611ms median (8.25x realtime in isolation). Open follow-ups D-22/D-23/D-24.
|
||||
|
||||
**Earlier — D-20** EXECUTED — segment-2 BrokenPipe
|
||||
root cause was a TWO-direction silent drop of LTX-2 audio kwargs in
|
||||
public `VideoGenerator` (inbound `SamplingParam.update()` rejected
|
||||
`audio_num_frames`/`ltx2_audio_clean_latent`/etc. as unknown fields and
|
||||
`logger.error`'d, outbound result dict didn't surface
|
||||
`ltx2_audio_latents` from `output_batch.extra`). Ported the
|
||||
FastVideo-internal extra-overrides routing block + made `update()`
|
||||
strict + added regression test (7 tests, all pass) + landed 4 commits
|
||||
on `will/dreamverse-monorepo` @ `5eaf0a13` (11 commits ahead of
|
||||
`fbd823df`). End-to-end verified on GPU4: `Cached audio latents shape
|
||||
=(1, 8, 126, 16) for segment 2`, `Segment 2: relayed av chunks=22,
|
||||
bytes=3.8MB`, no BrokenPipeError. Public-API fix needs cherry-pick to
|
||||
`will/ltx2_sr_port` for PR #1288 — see open-threads.md item D-20-CP.
|
||||
**D-19** EXECUTED previously: Dreamverse migration landed on
|
||||
`will/dreamverse-monorepo` @ `c1fe5d4c` (5 commits ahead of
|
||||
`will/ltx2_sr_port` HEAD `fbd823df`). 164 files, 53,294 LOC, 31,725
|
||||
files under `apps/dreamverse/`. e2e PASSES against migrated code (8/8
|
||||
Playwright in 5.1s, `/proc/$PID/cwd` verified). Significant deviation
|
||||
from [integration-plan.md](integration-plan.md): the plan's "DELETE
|
||||
generic-merged from Dreamverse, import public substitutes" assumption
|
||||
was invalid (public APIs aren't drop-ins) — generic-merged files now
|
||||
carried PRODUCT-LOCAL inside `apps/dreamverse/server/`. Public
|
||||
`fastvideo.entrypoints.streaming.*` reverts to `fbd823df` state. See
|
||||
[decisions-log.md D-19](decisions-log.md#d-19) +
|
||||
[D-20](decisions-log.md#d-20) for full context.).
|
||||
FastVideo `will/ltx2_sr_port` @ HEAD (post-D-17 STACK.md removal +
|
||||
integration-review.md addition + integration-plan.md addition + D-18
|
||||
reconciliation). Dreamverse `will/integrate-public-fastvideo` @ `ec8ef92`.
|
||||
PRs #1257 / #1258 / #1284 / #1286 MERGED to main. **PR #1287 CLOSED
|
||||
(in favor of consolidation); PR #1288 OPEN as the single mega-PR
|
||||
landing the entire `will/ltx2_sr_port` chain at once** (LTX-2 SR
|
||||
runtime + NVFP4 + `generate_async`/Dynamo contract + agents memory dir).
|
||||
Split branches kept as historical bookmarks; STACK.md model **abandoned** —
|
||||
see [decisions-log.md D-17](decisions-log.md#d-17). Local backup
|
||||
`will/ltx2_sr_port-pre-1286-rebase` @ `1baa60bb` preserves the
|
||||
pre-rebase chain.
|
||||
|
||||
## Fresh-context onboarding (read in order)
|
||||
|
||||
If you're an agent picking up this work for the first time, do these
|
||||
**5 things in this order**. Once done, you have full context to continue
|
||||
any open thread, commit correctly, push, and propagate to the open PR.
|
||||
|
||||
1. **Confirm worktree state** — run the "First 60 seconds" block in
|
||||
[runbook.md](runbook.md). Tells you the branch is right, services
|
||||
are up, and PR #1286's head matches what this dir claims.
|
||||
|
||||
2. **Read [state.md](state.md)** — single-page snapshot of branch tips,
|
||||
live services, test status, pre-existing failures, "do not pop"
|
||||
stashes.
|
||||
|
||||
3. **Read [pr-roadmap.md](pr-roadmap.md)** — what PRs landed, what's in
|
||||
flight, what's planned. Identifies the active open PR (currently
|
||||
#1286) and where it sits in the dependency chain.
|
||||
|
||||
4. **Read [open-threads.md](open-threads.md)** — prioritized work items
|
||||
with effort estimates and dependencies. The "Recommended pull order"
|
||||
section is a ready-made TODO list if you need one.
|
||||
|
||||
5. **Skim [runbook.md](runbook.md) end-to-end** — operational how-to:
|
||||
verify, commit (with co-author trailers), push, propagate to PR
|
||||
#1286, maintain the memory dir, and the "Common pitfalls" section
|
||||
that catches the recurring traps.
|
||||
|
||||
Skip the deep-context docs (design / streaming-server / cross-repo /
|
||||
quantization / decisions-log) until you need them — they're indexed in
|
||||
the "Deep-dive reading guide" below.
|
||||
|
||||
Final check: run the "Self-test" block at the bottom of
|
||||
[runbook.md](runbook.md). If you can answer all 8 questions from this
|
||||
dir alone, you're ready. If you can't, the gap is a memory-dir bug —
|
||||
file it in [open-threads.md](open-threads.md) before continuing.
|
||||
|
||||
## Deep-dive reading guide
|
||||
|
||||
| Question / task | File |
|
||||
|---|---|
|
||||
| "What's running right now? What just landed?" | [state.md](state.md) |
|
||||
| "How do I commit / push / propagate to PR #1286?" | [runbook.md](runbook.md) |
|
||||
| "Why is the schema typed this way? What's the philosophy?" | [design.md](design.md) |
|
||||
| "What PRs landed? In flight? Planned?" | [pr-roadmap.md](pr-roadmap.md) |
|
||||
| "Streaming server, `generate_async`, `build_app` routes?" | [streaming-server.md](streaming-server.md) |
|
||||
| "How does Dreamverse use FastVideo? What about Dynamo?" | [cross-repo-surfaces.md](cross-repo-surfaces.md) |
|
||||
| "NVFP4? Layer profiles? `LinearBase` fallback? AbsMaxFP8?" | [quantization.md](quantization.md) |
|
||||
| "Why was decision X made? What's resolved vs. open?" | [decisions-log.md](decisions-log.md) |
|
||||
| "What should I work on next? Priority order?" | [open-threads.md](open-threads.md) |
|
||||
| "Who should be co-authored on commits in this scope?" | [authors.md](authors.md) |
|
||||
| "How do we execute the Dreamverse → FastVideo monorepo merge?" | [integration-plan.md](integration-plan.md) ← **CURRENT** |
|
||||
| "Historical drift audit + Option-D evaluation (deprecated by D-18)" | [integration-review.md](integration-review.md) (DEPRECATED) |
|
||||
|
||||
## Repo + worktree paths
|
||||
|
||||
| Repo | Path | Active branch |
|
||||
|---|---|---|
|
||||
| FastVideo (public) | `/home/william5lin/FastVideo` | `will/ltx2_sr_port` |
|
||||
| Dreamverse | `/home/william5lin/Dreamverse` | `will/integrate-public-fastvideo` |
|
||||
| FastVideo-internal (read-only ref) | `/home/william5lin/FastVideo-internal` | their `main` |
|
||||
| Dynamo (read-only ref) | `/home/william5lin/dynamo` | upstream |
|
||||
|
||||
## Glossary
|
||||
|
||||
- **NVFP4**: NVIDIA's specific block-scaled FP4 (e2m1 mantissa, fp32 alpha,
|
||||
`layout_128x4` scale layout, group size 16). Distinct from MX-FP4 / OCP-FP4.
|
||||
- **`GeneratorConfig`**: typed init-time public config (model_path, engine,
|
||||
pipeline). Replaces flat `from_pretrained(**kwargs)`.
|
||||
- **`GenerationRequest`**: typed per-call request (prompt, inputs, sampling,
|
||||
runtime, output, stage_overrides, state, plan, extensions). Replaces flat
|
||||
`generate_video(**kwargs)`.
|
||||
- **`ServeConfig`** / **`RunConfig`**: top-level YAML envelopes. ServeConfig
|
||||
for `fastvideo serve`; RunConfig for offline `fastvideo generate`.
|
||||
- **`InferencePreset`**: model-owned named preset (e.g. `ltx2_two_stage`)
|
||||
defining stage topology + per-stage defaults + valid override types.
|
||||
- **`ContinuationState`**: opaque round-trip state envelope `{kind, payload}`.
|
||||
Hybrid: server-held for streaming WS, client-round-trip for stateless HTTP.
|
||||
- **`generate_async`**: future canonical async exec API (PR 7.10) yielding
|
||||
`VideoProgressEvent` / `VideoPartialEvent` / `VideoFinalEvent`. Substrate
|
||||
for streaming server, OpenAI server, AND Dynamo backend.
|
||||
- **`build_app`**: FastAPI app factory in
|
||||
`fastvideo.entrypoints.streaming.server`. Currently exposes only
|
||||
`/health` + `/v1/stream`. FE-required `/healthz`+`/readyz`+`/status`
|
||||
migration is open follow-up #1.
|
||||
- **`LLMProvider`**: protocol abstraction for prompt enhancer providers
|
||||
(cerebras, cerebras_ifm, groq). Public schema currently restricts to
|
||||
`Literal["cerebras", "groq"]`; `cerebras_ifm` is internal-only.
|
||||
- **`compat.py`**: legacy kwargs translation layer (~370 lines). Scheduled
|
||||
for death across PRs 14-17.
|
||||
- **`prepare_for_compile`**: duck-type protocol method called via
|
||||
`getattr(module, "prepare_for_compile", None)` before `torch.compile`.
|
||||
Currently only Gemma3 implements it.
|
||||
- **`SubprocessGpuPool`**: PR 7.6 public replacement for the internal
|
||||
`realtime/local_runtime.GPUPool`. Per-GPU subprocess workers, typed
|
||||
`GeneratorConfig` boundary.
|
||||
- **PR 5.5**: streaming server subpackage skeleton — adds
|
||||
`fastvideo/entrypoints/streaming/` parallel to `openai/`.
|
||||
- **PR 7.10**: the unlock PR. Closes Q-5 (audio re-encode), Q-9 (Dynamo
|
||||
progress), and PR 7.5's mid-segment cancellation TODO simultaneously.
|
||||
|
||||
## Live process map (as of 2026-05-03)
|
||||
|
||||
| Port | Service | Source |
|
||||
|---|---|---|
|
||||
| 8009 | `dreamverse-server` | running, `/readyz` 200, 1 warmed GPU worker |
|
||||
| 5274 | `next-server` (dev) | running |
|
||||
| 8000 | unknown FastAPI | not in handoff — verify before launching new BE |
|
||||
|
||||
## How this directory is maintained
|
||||
|
||||
- Source of truth for the integration story. Update when state changes.
|
||||
- Each file has a "Last updated" header; bump when you edit.
|
||||
- Cross-reference siblings via relative links; do NOT duplicate content.
|
||||
- New entries: register in `../index.jsonl`.
|
||||
- These files supersede the untracked source docs in the repo root and
|
||||
`.agents/exploration/` — see [state.md](state.md) "Untracked but
|
||||
present" section for disposition.
|
||||
|
||||
## Source documents (archived 2026-05-03)
|
||||
|
||||
The 7 source docs that this directory consolidates have been moved into
|
||||
[`source-archive/`](source-archive/). They remain available for agents
|
||||
who want the full unsynthesized rationale, but the synthesized memory
|
||||
files in this dir are the canonical source of truth.
|
||||
|
||||
| Source doc | Lines | Synthesized into |
|
||||
|---|---|---|
|
||||
| [`source-archive/apirefactor.md`](source-archive/apirefactor.md) | 838 | [design.md](design.md) |
|
||||
| [`source-archive/PR-plan.md`](source-archive/PR-plan.md) | 1145 | [pr-roadmap.md](pr-roadmap.md) |
|
||||
| [`source-archive/dreamverse_review.md`](source-archive/dreamverse_review.md) | 390 | [state.md](state.md) + [decisions-log.md](decisions-log.md) |
|
||||
| [`source-archive/handoff-nvfp4-launch-demo.md`](source-archive/handoff-nvfp4-launch-demo.md) | 518 | [state.md](state.md) + [quantization.md](quantization.md) + [open-threads.md](open-threads.md) |
|
||||
| [`source-archive/streaming-server-upstream-plan.md`](source-archive/streaming-server-upstream-plan.md) | 539 | [streaming-server.md](streaming-server.md) + [decisions-log.md](decisions-log.md) |
|
||||
| [`source-archive/dreamverse_integration.md`](source-archive/dreamverse_integration.md) | 285 | [cross-repo-surfaces.md](cross-repo-surfaces.md) |
|
||||
| [`source-archive/video-generator-config-api-design.md`](source-archive/video-generator-config-api-design.md) | 93 | [design.md](design.md) (early-draft material) |
|
||||
| `.agents/exploration/pr-link-review.md` | 29 | already promoted to `.agents/skills/review-pr-link/` (kept in exploration dir) |
|
||||
|
||||
See [`source-archive/README.md`](source-archive/README.md) for the
|
||||
archive policy.
|
||||
@@ -0,0 +1,149 @@
|
||||
# Authors — Dreamverse Integration
|
||||
|
||||
**Status:** PERMANENT — keep around as the source of truth for who collaborated
|
||||
on the dreamverse-integration work, even after every PR in the integration
|
||||
scope has merged.
|
||||
**Last updated:** 2026-05-05 (strategy reversal — single mega-PR #1288 on `will/ltx2_sr_port` replaces planned 6-PR split; #1287 closed; per [decisions-log.md D-17](decisions-log.md#d-17))
|
||||
|
||||
This file documents the human co-authors credited on every commit in the
|
||||
dreamverse-integration scope (FastVideo public-API refactor, streaming server
|
||||
upstream, GPU pool, prompt enhancer, NVFP4 wire-up, LTX-2 SR port). The
|
||||
4 collaborators below worked on the FastVideo-internal precursor of this code
|
||||
and are credited as co-authors on every public-side upstream commit via Git's
|
||||
standard
|
||||
[`Co-authored-by`](https://docs.github.com/en/pull-requests/committing-changes-to-your-project/creating-and-editing-commits/creating-a-commit-with-multiple-authors)
|
||||
trailer convention.
|
||||
|
||||
Scope-wise this is the dreamverse-integration-flavored mirror of
|
||||
[co-authors.md](co-authors.md), which is scoped to the broader
|
||||
`will/ltx2_sr_port` 10-PR stack. The roster is identical; both files now live
|
||||
in this memory dir so the provenance docs stay self-contained and discoverable.
|
||||
|
||||
## Co-author roster
|
||||
|
||||
| GitHub user | Real name | GitHub ID | Trailer email |
|
||||
|---|---|---|---|
|
||||
| [`@Davids048`](https://github.com/Davids048) | Junda (David) Su | 90978028 | `90978028+Davids048@users.noreply.github.com` |
|
||||
| [`@RandNMR73`](https://github.com/RandNMR73) | Matthew Noto | 99706358 | `99706358+RandNMR73@users.noreply.github.com` |
|
||||
| [`@XOR-op`](https://github.com/XOR-op) | (unset) | 17672363 | `17672363+XOR-op@users.noreply.github.com` |
|
||||
| [`@jzhang38`](https://github.com/jzhang38) | Zhang Peiyuan | 42993249 | `42993249+jzhang38@users.noreply.github.com` |
|
||||
|
||||
## Verification — where these trailers appear
|
||||
|
||||
Verified via `gh pr view <PR> --json commits --jq '.commits[].messageBody'`
|
||||
across every PR in the integration scope:
|
||||
|
||||
| PR | Branch | Status | Trailers present on every commit |
|
||||
|---|---|---|---|
|
||||
| #1257 | `will/api_7.6` (GPU pool upstream) | ✅ merged 2026-05-04 | yes (4/4) |
|
||||
| #1258 | `will/api_7.7` (prompt enhancer + LLMProvider) | ✅ merged 2026-05-04 | yes (3/3) |
|
||||
| #1284 | `will/api_7.8` (streaming auxiliaries) | ✅ merged 2026-05-04 | yes (2/2) |
|
||||
| #1286 | `will/api_7.9` (streaming router) | ✅ merged 2026-05-05 at `2aaeee2a` (squash) | yes on commits 1-3; commit `a152cb77` (`[fix] streaming: router polish`) was missing trailers but got squashed into the merge commit, so the merge commit on main inherits the trailers from the other 3. The trailerless cherry-pick partner (`40e265b8` on `will/ltx2_sr_port`) was dropped by the post-#1286 rebase — gap permanently resolved. |
|
||||
| #1287 | `will/api_7.10` (`generate_async` + `VideoEvent`) | ❌ CLOSED 2026-05-05 — superseded by #1288 per [D-17](decisions-log.md#d-17) | yes on all 3 commits (now part of #1288's chain) |
|
||||
| **#1288** | **`will/ltx2_sr_port`** (mega-PR — full stack: SR runtime + NVFP4 + generate_async + Dynamo contract + agents memory + integration-review) | 🟢 OPEN, MERGEABLE at `b36bdbc9`, 36 commits / 70 files / ~+13.0k LOC (post STACK.md removal) | yes on all 36 commits |
|
||||
|
||||
Aggregate count across `will/ltx2_sr_port` (top of stack) at the time of
|
||||
writing: 32-33 commits per co-author, matching the 32 commits in the stack
|
||||
on top of base `cfccd292`. Numbers stay consistent because the rebase
|
||||
command (see "How the trailers were applied" below) walks every commit.
|
||||
|
||||
## Trailer block (copy-paste ready)
|
||||
|
||||
The trailers added to every commit on `will/ltx2_sr_port` and every
|
||||
dreamverse-integration PR:
|
||||
|
||||
```
|
||||
Co-authored-by: Junda (David) Su <90978028+Davids048@users.noreply.github.com>
|
||||
Co-authored-by: Matthew Noto <99706358+RandNMR73@users.noreply.github.com>
|
||||
Co-authored-by: XOR-op <17672363+XOR-op@users.noreply.github.com>
|
||||
Co-authored-by: Zhang Peiyuan <42993249+jzhang38@users.noreply.github.com>
|
||||
```
|
||||
|
||||
For one-off `git commit -m` invocations, use `--trailer` flags:
|
||||
|
||||
```bash
|
||||
git commit -m "..." \
|
||||
--trailer "Co-authored-by: Junda (David) Su <90978028+Davids048@users.noreply.github.com>" \
|
||||
--trailer "Co-authored-by: Matthew Noto <99706358+RandNMR73@users.noreply.github.com>" \
|
||||
--trailer "Co-authored-by: XOR-op <17672363+XOR-op@users.noreply.github.com>" \
|
||||
--trailer "Co-authored-by: Zhang Peiyuan <42993249+jzhang38@users.noreply.github.com>"
|
||||
```
|
||||
|
||||
`--trailer` is idempotent (dedupes by full `key: value`) so re-running is safe.
|
||||
|
||||
## Why no-reply emails
|
||||
|
||||
GitHub's `<id>+<username>@users.noreply.github.com` form is the most reliable
|
||||
way to link a `Co-authored-by` trailer to a GitHub account. It:
|
||||
|
||||
- Always works regardless of whether the user has a public verified email
|
||||
- Survives the user changing their primary email
|
||||
- Doesn't expose anyone's personal email to git history
|
||||
- Is the format GitHub itself produces when you click "Add co-author" in the
|
||||
web UI
|
||||
|
||||
(All 4 collaborators have this email already used in `FastVideo-internal`
|
||||
git history, verified via `git log --all` on that repo.)
|
||||
|
||||
## How the trailers were applied (bulk rebase)
|
||||
|
||||
```bash
|
||||
git rebase --exec '
|
||||
git commit --amend --no-edit \
|
||||
--trailer "Co-authored-by: Junda (David) Su <90978028+Davids048@users.noreply.github.com>" \
|
||||
--trailer "Co-authored-by: Matthew Noto <99706358+RandNMR73@users.noreply.github.com>" \
|
||||
--trailer "Co-authored-by: XOR-op <17672363+XOR-op@users.noreply.github.com>" \
|
||||
--trailer "Co-authored-by: Zhang Peiyuan <42993249+jzhang38@users.noreply.github.com>"
|
||||
' origin/main will/ltx2_sr_port
|
||||
```
|
||||
|
||||
After running, re-slice all 10 split branches per [`STACK.md`](../../../STACK.md)
|
||||
and force-push the published branches (`will/api_7.9`, `will/ltx2_sr_port`).
|
||||
|
||||
## How to add a new co-author later
|
||||
|
||||
1. Add the user to the roster table above (and [co-authors.md](co-authors.md)
|
||||
— keep them in sync).
|
||||
2. Append their `Co-authored-by` line to the trailer block above.
|
||||
3. Re-run the bulk rebase command on `will/ltx2_sr_port` — git's trailer
|
||||
dedupe handles the existing 4; the new one gets appended.
|
||||
4. Re-slice all split branches per [`STACK.md`](../../../STACK.md).
|
||||
5. Force-push the published branches.
|
||||
|
||||
## What we do NOT add
|
||||
|
||||
Per the repo's top-level [`AGENTS.md`](../../../AGENTS.md):
|
||||
|
||||
> Never add any coding agent or models such as Claude (or Claude Code), GPT,
|
||||
> Codex or others as a co-author in commits or PRs. Do not include
|
||||
> `Co-Authored-By: Claude ...` trailers or "Generated with Claude Code" and
|
||||
> other such lines.
|
||||
|
||||
So no `Co-authored-by: Claude <noreply@anthropic.com>`, no
|
||||
`Generated with Claude Code` footer, no `Cursor <cursoragent@cursor.com>`
|
||||
trailer (one such commit exists on `will/ltx2_sr_port` from a pre-policy
|
||||
external contribution and stays grandfathered; new commits MUST NOT introduce
|
||||
the pattern). Only human collaborators.
|
||||
|
||||
## Known gaps
|
||||
|
||||
**Resolved 2026-05-05 by the post-#1286 rebase.** The two trailerless
|
||||
commits (`a152cb77` on `will/api_7.9` and `40e265b8` on
|
||||
`will/ltx2_sr_port`) are no longer reachable from any active branch:
|
||||
|
||||
- `a152cb77` was absorbed into squash merge `2aaeee2a` on main, which
|
||||
inherits the trailers from the other 3 commits in the squash.
|
||||
- `40e265b8` was dropped by the post-#1286 rebase of
|
||||
`will/ltx2_sr_port`.
|
||||
|
||||
Both still exist on the local backup `will/ltx2_sr_port-pre-1286-rebase`
|
||||
for archeological reference. No further action needed.
|
||||
|
||||
## See also
|
||||
|
||||
- [co-authors.md](co-authors.md) — stack-scoped co-authors file (same roster,
|
||||
broader scope)
|
||||
- [`../../../STACK.md`](../../../STACK.md) — 10-PR split layout for
|
||||
`will/ltx2_sr_port` (re-slice commands live here)
|
||||
- [`pr-roadmap.md`](pr-roadmap.md) — per-PR status within the
|
||||
dreamverse-integration scope
|
||||
@@ -0,0 +1,182 @@
|
||||
# `.agents/` Cleanup Log — Phase 1 (Deletes Only)
|
||||
|
||||
**Status:** TEMPORARY — delete this file after the cleanup is reviewed/committed.
|
||||
**Date:** 2026-05-04
|
||||
**Branch:** `will/ltx2_sr_port`
|
||||
**Scope:** Phase 1 of the `.agents/` cleanup plan (deletes only; no rewrites or additions).
|
||||
|
||||
For the full multi-phase plan, see the prior session analysis. This file tracks
|
||||
exactly what got deleted, why, and what cross-references still point at deleted
|
||||
content (to fix in a future phase).
|
||||
|
||||
---
|
||||
|
||||
## Deletions executed
|
||||
|
||||
### Files deleted
|
||||
|
||||
| Path | Size | Reason |
|
||||
|---|---|---|
|
||||
| `.agents/STATUS.md` | 3.85 KB | Stale dashboard, last synced 2026-03-02. Counts wrong (claimed 8 skills/4 workflows/4 memory; actual 9/5/5). References old snake_case filenames (`codebase_map.md`/`experiment_journal.md`) that don't exist. Hand-maintained derivative of `.agents/{memory,skills}/index.jsonl` — strictly redundant. |
|
||||
| `.agents/exploration/pr-link-review.md` | 1.11 KB | Status: "promoted" to `.agents/skills/review-pr-link/`. Per `.agents/exploration/README.md` lifecycle, promoted exploration logs should not linger after the skill exists. |
|
||||
| `.agents/workflows/sync-dashboard.md` | 1.87 KB | SOP for maintaining `STATUS.md` (which is also deleted). Contained obsolete file paths (`.agents/skills/launch-experiment.md` flat layout vs. actual `<skill>/SKILL.md` per-dir layout). Has never been run successfully (judging by stale dates everywhere). |
|
||||
|
||||
### Skill directories deleted
|
||||
|
||||
| Path | Size | Reason |
|
||||
|---|---|---|
|
||||
| `.agents/skills/index-related-work/` | 2.18 KB | Vapor-skill operating on the empty `.agents/memory/related-work/` registry. Never used (the registry has zero entries despite ~6 weeks since skill creation). Re-add when the related-work catalog gains entries. |
|
||||
| `.agents/skills/search-related-work/` | 1.91 KB | Same: vapor-skill against empty registry. The skill description literally requires "The related work index has entries" as a prerequisite, and there are none. |
|
||||
|
||||
**Total deleted: 5 items, ~10.9 KB.**
|
||||
|
||||
### Registry updates
|
||||
|
||||
| File | Change |
|
||||
|---|---|
|
||||
| `.agents/skills/index.jsonl` | Removed entries for `index-related-work` and `search-related-work`. Was 9 entries; now 7. |
|
||||
|
||||
### Symlink hygiene
|
||||
|
||||
`.agents/scripts/sync-skills.sh` was run to prune now-stale symlinks under
|
||||
`.claude/skills/` that pointed at the deleted skill directories. Output captured
|
||||
in the run log.
|
||||
|
||||
---
|
||||
|
||||
## What was KEPT (despite being candidates)
|
||||
|
||||
| Path | Why kept |
|
||||
|---|---|
|
||||
| `.agents/scripts/sync-skills.sh` | User explicitly requested keep. **Verified**: this script is INDEPENDENT of STATUS.md / sync-dashboard.md. It mirrors `.agents/skills/` → `.claude/skills/` via symlinks for Claude Code skill discovery. Self-contained, useful, prunes its own stale symlinks. |
|
||||
| `.agents/memory/related-work/README.md` | Empty placeholder, but the schema/template is reusable. Kept for when first related-work entry is added. |
|
||||
| `.agents/memory/experiment-journal/README.md` | Same: empty placeholder with template; kept for when journaling begins. |
|
||||
| `.agents/lessons/README.md` | Same: empty placeholder, reusable schema. |
|
||||
| `.agents/exploration/README.md` | Active template for new exploration logs. Kept. |
|
||||
|
||||
---
|
||||
|
||||
## Remaining broken cross-references (FOLLOW-UP NEEDED)
|
||||
|
||||
These files still reference deleted content. **NOT fixed in Phase 1** — track for
|
||||
the next pass (Phase 2: rewrites/dedupe).
|
||||
|
||||
### References to deleted `STATUS.md`
|
||||
|
||||
| Referencing file | Action needed |
|
||||
|---|---|
|
||||
| `.agents/onboarding/README.md` | Quick-reference tree (line ~65) lists `STATUS.md ← dashboard: completeness & trust of all components`. Remove that line + the `ONBOARDING.md` typo (file is `README.md`). |
|
||||
|
||||
### References to deleted `pr-link-review.md`
|
||||
|
||||
| Referencing file | Action needed |
|
||||
|---|---|
|
||||
| `.agents/memory/dreamverse-integration/state.md` | "Untracked but present" / "Source docs (archived)" sections still mention `pr-link-review.md` as kept. Update to reflect deletion. |
|
||||
| `.agents/memory/dreamverse-integration/README.md` | Same — table row for `pr-link-review.md` says "kept in exploration dir". Update or remove the row. |
|
||||
|
||||
### References to deleted skills (`index-related-work`, `search-related-work`)
|
||||
|
||||
| Referencing file | Action needed |
|
||||
|---|---|
|
||||
| `.agents/memory/related-work/README.md` | Says "Use the `index-related-work` skill". Either remove that hint or note "skill removed; re-add when registry has entries". |
|
||||
| `.agents/workflows/evaluation-development.md` | Step 1 says "Search `.agents/memory/related-work/` for existing evaluation approaches" — that's still valid (manual search). No change needed. |
|
||||
|
||||
### References to deleted `sync-dashboard.md`
|
||||
|
||||
| Referencing file | Action needed |
|
||||
|---|---|
|
||||
| `.agents/memory/evaluation-registry/README.md` | Doesn't reference sync-dashboard directly. No change. |
|
||||
| `.agents/STATUS.md` | Already being deleted. |
|
||||
|
||||
---
|
||||
|
||||
## Other registry inconsistencies discovered (NOT FIXED in Phase 1)
|
||||
|
||||
While editing `.agents/skills/index.jsonl`, two skill directories were found
|
||||
that exist on disk but **are not registered** in `index.jsonl`:
|
||||
|
||||
| Skill dir | Status | Why missing from index |
|
||||
|---|---|---|
|
||||
| `.agents/skills/diagnose-ssim-failure/` | Untracked locally; NOT on `origin/main`. 12.3 KB SKILL.md + `scripts/compare_latent_pt.py`. Recent mtime (2026-05-01). | Created in a prior session but the registration step was skipped. |
|
||||
| `.agents/skills/review-pr-link/` | Untracked locally; NOT on `origin/main`. 2.9 KB SKILL.md + `scripts/prepare_pr_review.py` + `agents/openai.yaml`. The promotion target of the deleted `pr-link-review.md` exploration log. | Skipped registration when promoted from exploration log. |
|
||||
|
||||
Both skills are functional and exposed via `sync-skills.sh` symlinks (just verified in
|
||||
`.claude/skills/`), but agents reading `index.jsonl` to discover skills will miss them.
|
||||
|
||||
**Action for Phase 2**: Add entries to `.agents/skills/index.jsonl` for both,
|
||||
likely with `trust: medium` since they have working scripts and recent use.
|
||||
|
||||
---
|
||||
|
||||
## Skill registry parity check
|
||||
|
||||
After Phase 1, `.agents/skills/` contains 9 directories but `index.jsonl` lists 7:
|
||||
|
||||
| In `index.jsonl` | On disk |
|
||||
|---|---|
|
||||
| ✓ launch-experiment | ✓ launch-experiment/ |
|
||||
| ✓ monitor-experiment | ✓ monitor-experiment/ |
|
||||
| ✓ summarize-run | ✓ summarize-run/ |
|
||||
| ✓ log-experiment | ✓ log-experiment/ |
|
||||
| ✓ evaluate-video-quality | ✓ evaluate-video-quality/ |
|
||||
| ✓ seed-ssim-references | ✓ seed-ssim-references/ |
|
||||
| ✓ reseed-ssim-references | ✓ reseed-ssim-references/ |
|
||||
| ❌ (missing) | ⚠ diagnose-ssim-failure/ |
|
||||
| ❌ (missing) | ⚠ review-pr-link/ |
|
||||
|
||||
`.claude/skills/` symlinks (the runtime-discoverable surface) include all 9 ✓.
|
||||
|
||||
---
|
||||
|
||||
## Phase 2+ items (NOT executed in this session)
|
||||
|
||||
For future cleanup sessions, the prior plan identified:
|
||||
|
||||
**Phase 2 (rewrites)**:
|
||||
- Rewrite `.agents/onboarding/worldmodel-training/README.md` to drop ~50% structural duplication with `codebase-map/README.md`
|
||||
- Refresh `.agents/memory/codebase-map/README.md` (last updated 2026-03-08; missing `fastvideo/api/`, `fastvideo/entrypoints/streaming/`, etc.)
|
||||
- Refresh `.agents/memory/evaluation-registry/README.md` (last updated 2026-03-02; references old `evaluation_registry.md` filename)
|
||||
- Merge `.agents/workflows/experiment-journaling.md` into `experiment-lifecycle.md` (one SOP per workflow)
|
||||
- Fix the broken cross-references listed above
|
||||
|
||||
**Phase 3 (additions)**:
|
||||
- `fastvideo/api/AGENTS.md`
|
||||
- `fastvideo/entrypoints/AGENTS.md`
|
||||
- `tests/AGENTS.md` (top-level, distinct from `fastvideo/tests/AGENTS.md`)
|
||||
- `fastvideo/distributed/AGENTS.md`
|
||||
- `examples/AGENTS.md`
|
||||
- `docs/AGENTS.md`
|
||||
- `benchmarks/AGENTS.md`
|
||||
|
||||
**Phase 4 (registry)**:
|
||||
- Add `.agents/workflows/index.jsonl`
|
||||
- Standardize all three index.jsonl schemas
|
||||
|
||||
**Phase 5 (skills quality)**:
|
||||
- Promote tested skills (`seed-ssim-references`, `reseed-ssim-references`, `diagnose-ssim-failure`, `review-pr-link`) from `trust: low` to `trust: medium`
|
||||
- Mark untested skills (`launch-experiment`, `monitor-experiment`, `summarize-run`, `log-experiment`, `evaluate-video-quality`) explicitly with their gating prerequisite (e.g. "operates on empty registry")
|
||||
|
||||
---
|
||||
|
||||
## Recovery
|
||||
|
||||
All deletions are local (`will/ltx2_sr_port`, not committed). To restore any
|
||||
deleted file:
|
||||
|
||||
```bash
|
||||
git restore --source=HEAD .agents/STATUS.md
|
||||
git restore --source=HEAD .agents/exploration/pr-link-review.md
|
||||
git restore --source=HEAD .agents/workflows/sync-dashboard.md
|
||||
git restore --source=HEAD .agents/skills/index-related-work/SKILL.md
|
||||
git restore --source=HEAD .agents/skills/search-related-work/SKILL.md
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## When to delete THIS file
|
||||
|
||||
Once:
|
||||
1. The Phase 1 deletions are committed (or merged), AND
|
||||
2. Phase 2 (broken cross-reference cleanup) is also committed,
|
||||
|
||||
remove this file. Its purpose is transient bookkeeping for a multi-phase cleanup.
|
||||
@@ -0,0 +1,88 @@
|
||||
# Co-Authors — `will/ltx2_sr_port` Stack
|
||||
|
||||
**Status:** PERMANENT — keep around as the source of truth for who collaborated on this work, even after the stack merges.
|
||||
**Last updated:** 2026-05-04
|
||||
|
||||
This file documents the human co-authors credited on every commit in the
|
||||
`will/ltx2_sr_port` stack and its 10 split PRs. The 4 collaborators below
|
||||
worked on the FastVideo-internal precursor of this code (LTX-2 streaming
|
||||
server, NVFP4 wire-up, GPU pool, prompt enhancer, etc.) and are credited as
|
||||
co-authors on the public-side upstream commits via Git's standard
|
||||
[`Co-authored-by`](https://docs.github.com/en/pull-requests/committing-changes-to-your-project/creating-and-editing-commits/creating-a-commit-with-multiple-authors)
|
||||
trailer convention.
|
||||
|
||||
The trailers are added to every commit on `will/ltx2_sr_port` (see
|
||||
[`STACK.md`](STACK.md)), which means GitHub will:
|
||||
|
||||
- Show the 4 co-authors on every commit detail page
|
||||
- Show them on the merge commit / squash commit summary
|
||||
- Display their avatars in the PR's "Contributors" sidebar
|
||||
- Surface them in [`/contributors`](https://github.com/hao-ai-lab/FastVideo/contributors) once the stack lands
|
||||
|
||||
## Co-author roster
|
||||
|
||||
| GitHub user | Real name | GitHub ID | Trailer email |
|
||||
|---|---|---|---|
|
||||
| [`@Davids048`](https://github.com/Davids048) | Junda (David) Su | 90978028 | `90978028+Davids048@users.noreply.github.com` |
|
||||
| [`@RandNMR73`](https://github.com/RandNMR73) | Matthew Noto | 99706358 | `99706358+RandNMR73@users.noreply.github.com` |
|
||||
| [`@XOR-op`](https://github.com/XOR-op) | (unset) | 17672363 | `17672363+XOR-op@users.noreply.github.com` |
|
||||
| [`@jzhang38`](https://github.com/jzhang38) | Zhang Peiyuan | 42993249 | `42993249+jzhang38@users.noreply.github.com` |
|
||||
|
||||
## Why no-reply emails
|
||||
|
||||
GitHub's `<id>+<username>@users.noreply.github.com` form is the most reliable
|
||||
way to link a `Co-authored-by` trailer to a GitHub account. It:
|
||||
|
||||
- Always works regardless of whether the user has a public verified email
|
||||
- Survives the user changing their primary email
|
||||
- Doesn't expose anyone's personal email to git history
|
||||
- Is the format GitHub itself produces when you click "Add co-author" in the
|
||||
web UI
|
||||
|
||||
(All 4 collaborators have this email already used in `FastVideo-internal`
|
||||
git history, verified via `git log --all` on that repo.)
|
||||
|
||||
## Trailer block (copy-paste ready)
|
||||
|
||||
The trailers added to every commit on `will/ltx2_sr_port`:
|
||||
|
||||
```
|
||||
Co-authored-by: Junda (David) Su <90978028+Davids048@users.noreply.github.com>
|
||||
Co-authored-by: Matthew Noto <99706358+RandNMR73@users.noreply.github.com>
|
||||
Co-authored-by: XOR-op <17672363+XOR-op@users.noreply.github.com>
|
||||
Co-authored-by: Zhang Peiyuan <42993249+jzhang38@users.noreply.github.com>
|
||||
```
|
||||
|
||||
## How the trailers were applied
|
||||
|
||||
```bash
|
||||
git rebase --exec '
|
||||
git commit --amend --no-edit \
|
||||
--trailer "Co-authored-by: Junda (David) Su <90978028+Davids048@users.noreply.github.com>" \
|
||||
--trailer "Co-authored-by: Matthew Noto <99706358+RandNMR73@users.noreply.github.com>" \
|
||||
--trailer "Co-authored-by: XOR-op <17672363+XOR-op@users.noreply.github.com>" \
|
||||
--trailer "Co-authored-by: Zhang Peiyuan <42993249+jzhang38@users.noreply.github.com>"
|
||||
' origin/main will/ltx2_sr_port
|
||||
```
|
||||
|
||||
Git's `--trailer` flag is idempotent (it dedupes by the full `key: value`
|
||||
string), so re-running the rebase is safe and won't add duplicates.
|
||||
|
||||
## How to add a new co-author later
|
||||
|
||||
1. Add the user to the roster table above.
|
||||
2. Append their `Co-authored-by` line to the trailer block.
|
||||
3. Re-run the rebase command above on `will/ltx2_sr_port` — git's
|
||||
trailer dedupe handles the existing 4; the new one gets appended.
|
||||
4. Re-slice all 10 split branches per [`STACK.md`](STACK.md).
|
||||
5. Force-push `will/api_7.6`, `will/api_7.7`, and `will/ltx2_sr_port`.
|
||||
|
||||
## What we do NOT add
|
||||
|
||||
Per [`AGENTS.md`](AGENTS.md):
|
||||
|
||||
> Never add any coding agent or models such as Claude (or Claude Code), GPT,
|
||||
> Codex or others as a co-author in commits or PRs.
|
||||
|
||||
So no `Co-authored-by: Claude <noreply@anthropic.com>` or similar. Only
|
||||
human collaborators.
|
||||
@@ -0,0 +1,277 @@
|
||||
# Cross-Repo Surfaces — Dreamverse + Dynamo
|
||||
|
||||
How Dreamverse consumes FastVideo today, what's already shared, what's
|
||||
ad hoc, and what migrations land alongside each PR. Plus the Dynamo
|
||||
backend contract.
|
||||
|
||||
For the streaming-server side see [streaming-server.md](streaming-server.md).
|
||||
For the API design see [design.md](design.md). For PR sequence see
|
||||
[pr-roadmap.md](pr-roadmap.md).
|
||||
|
||||
**Last updated:** 2026-05-03.
|
||||
|
||||
## The three surfaces
|
||||
|
||||
Dreamverse depends on FastVideo across three surfaces (in order of
|
||||
stability):
|
||||
|
||||
1. **Pipeline construction** (stable)
|
||||
2. **Realtime runtime** (in flight: PRs 7.5/7.6)
|
||||
3. **Continuation state** (PR 7 typed; PR 7.6 wires server-held)
|
||||
|
||||
## Surface 1: Pipeline construction (stable)
|
||||
|
||||
`Dreamverse/server/video_generation.py:VideoGenerationWorker` calls
|
||||
`VideoGenerator.from_pretrained(...)`.
|
||||
|
||||
After PR 6 the typed `GeneratorConfig` path exists; **as of `d80c2a8`
|
||||
(May 2)** Dreamverse migrated to the typed path:
|
||||
|
||||
| Dreamverse usage | FastVideo public surface (post-PR 6) |
|
||||
|---|---|
|
||||
| `VideoGenerator.from_pretrained(model_path, ltx2_refine_enabled=…, …)` | `VideoGenerator.from_pretrained(config=GeneratorConfig(...))` |
|
||||
| Flat `torch_compile_kwargs={…}` dict | `engine.compile.{backend,fullgraph,mode,dynamic,extras}` |
|
||||
| `ltx2_vae_tiling=True` | `pipeline.vae_tiling=True` |
|
||||
| `ltx2_refine_*` family | `pipeline.preset_overrides.refine.*` + `pipeline.components.upsampler_weights` |
|
||||
| `enable_torch_compile_text_encoder` | `engine.compile.text_encoder_enabled` |
|
||||
|
||||
Refine knobs moved from `ltx2_refine_*` flat kwargs into
|
||||
`preset_overrides["refine"]`. **The in-memory `pipeline_config` pin**
|
||||
(`dit_config.quant_config = NVFP4Config()`) keeps using the legacy
|
||||
`experimental["pipeline_config"]` carrier because typed
|
||||
`transformer_quant: "NVFP4"` doesn't yet support setting
|
||||
`layer_profile` (see [open-threads.md](open-threads.md) follow-up #4 +
|
||||
[quantization.md](quantization.md)).
|
||||
|
||||
Legacy flat-kwarg path stays supported via `compat.py`; migration is
|
||||
opt-in. PR 13's deprecation warnings are the eventual nudge.
|
||||
|
||||
## Surface 2: Realtime runtime (in flight: PRs 7.5–7.6)
|
||||
|
||||
`Dreamverse/server/runtime/factory.py` selects a runtime backend at
|
||||
process start:
|
||||
|
||||
```python
|
||||
def create_runtime_pool() -> RuntimePool:
|
||||
if os.getenv("FASTVIDEO_REALTIME_BASE_URL"):
|
||||
return FastVideoRealtimePool(base_url=..., ws_url=..., default_model_id=...)
|
||||
return GPUPool(get_available_gpus()) # in-process, wraps
|
||||
# fastvideo.entrypoints.realtime.local_runtime
|
||||
```
|
||||
|
||||
Both backends speak the same `RuntimePool` / `RuntimeSlot` Protocol
|
||||
(`server/runtime/interfaces.py`):
|
||||
|
||||
- `acquire(client_id, websocket=None) -> (gpu_id, RuntimeSlot)`
|
||||
- `release(client_id)`
|
||||
- `RuntimeSlot.{join_user, user_step, leave_user, register_stream_queue, ...}`
|
||||
|
||||
Today both impls reach into FastVideo-internal's
|
||||
`fastvideo.entrypoints.realtime.local_runtime` (which exposes
|
||||
`RealtimeRuntimeConfig`, `GPUPool`, `GPUSlot`). The remote backend talks
|
||||
HTTP+WS to a separately-deployed runtime of the same shape.
|
||||
|
||||
**Contract that PR 7.5/7.6 must preserve:**
|
||||
|
||||
- `RealtimeRuntimeConfig` accepts `model_registry`, `default_model_id`,
|
||||
`default_height/width/num_frames/fps/num_inference_steps/guidance_scale/seed/negative_prompt`,
|
||||
`default_ltx2_image_crf`, `startup_warmup_{enabled,prompt,timeout_seconds}`.
|
||||
- `GPUPool(gpu_ids: list[int], config: RealtimeRuntimeConfig)` constructor.
|
||||
- `pool.initialize() / shutdown() / acquire() / release() / get_status()`.
|
||||
- HTTP endpoints on the remote variant: `GET /healthz`, `GET /readyz`,
|
||||
`GET /status`, `WS /ws`. (Already match what
|
||||
`Dreamverse/server/routes/health.py` consumes.)
|
||||
|
||||
**These three health routes still need to migrate into FastVideo's
|
||||
`build_app` to make `BE_FLAVOR=fastvideo` FE-compatible** — see
|
||||
[streaming-server.md](streaming-server.md) "build_app route contract" +
|
||||
[open-threads.md](open-threads.md) follow-up #1.
|
||||
|
||||
When PR 7.6 lands the upstream of `fastvideo/entrypoints/realtime/`,
|
||||
Dreamverse should not need any code change unless the import path
|
||||
renames. Decided: keep `streaming/` (post-PR-5.5 public name); ship
|
||||
`realtime/__init__.py` as a re-export with `DeprecationWarning` for one
|
||||
release cycle.
|
||||
|
||||
### Note on `default_ltx2_image_crf`
|
||||
|
||||
Dreamverse's `RealtimeRuntimeConfig` includes `default_ltx2_image_crf`.
|
||||
The April 26 Dreamverse review (D-8) showed this getting passed to
|
||||
`SamplingParam(...)` and **silently dropped** by the public schema. Post
|
||||
`d80c2a8` (May 2 typed-config refactor), the migration target is
|
||||
`request.stage_overrides.refine.image_crf` (per
|
||||
[design.md](design.md) compatibility mapping table).
|
||||
|
||||
**Whether `d80c2a8` actually wired this through, or it's still latent,
|
||||
is unverified.** See [open-threads.md](open-threads.md) item D-8.
|
||||
|
||||
## Surface 3: Continuation state (PR 7)
|
||||
|
||||
`Dreamverse/server/video_generation.py:89 ContinuationState` is
|
||||
Dreamverse's hand-rolled per-session state holder. PR 7 introduced the
|
||||
typed equivalent at
|
||||
[`fastvideo/pipelines/basic/ltx2/continuation.py`](file:///home/william5lin/FastVideo/fastvideo/pipelines/basic/ltx2/continuation.py).
|
||||
|
||||
### Field mapping
|
||||
|
||||
| Dreamverse | PR 7 `LTX2ContinuationState` | Notes |
|
||||
|---|---|---|
|
||||
| `video_images: list[PIL.Image]` | `video_frames: list[np.ndarray]` (uint8 H×W×3) | numpy is leaner; Dreamverse already round-trips PIL→numpy→PIL just to add noise |
|
||||
| `audio_latents: torch.Tensor` `[B, C, T, mel]` | `audio_latents: torch.Tensor` (safetensors-serialized; bf16-safe) | unchanged shape; safetensors preserves bf16 |
|
||||
| `LTX2_VIDEO_CONDITIONING_FRAME_IDX` (env) | `video_conditioning_frame_idx: int` | env constant → per-state field |
|
||||
| `LTX2_VIDEO_CONDITIONING_STRENGTH` (env) | `video_conditioning_strength: float` | env constant → per-state field |
|
||||
| `AUDIO_CONDITIONING_NUM_FRAMES` (env) | `audio_conditioning_num_frames: int` | env constant → per-state field |
|
||||
| `AUDIO_CONDITIONING_STRENGTH` (env) | `audio_conditioning_strength: float` | env constant → per-state field |
|
||||
| `audio_lps` (passed into `apply_audio`) | `audio_sample_rate: int \| None` | analogous; rename worth confirming with audio team |
|
||||
| Computed `prefix_sec` per segment | `video_position_offset_sec: float` | **see open question below** |
|
||||
| `segment_idx` (param to `apply_*`) | `segment_index: int` | per-state field |
|
||||
| `VIDEO_CONTEXT_NOISE`, `AUDIO_CONTEXT_NOISE`, `ENABLE_AUDIO_COND` | not on state | runtime policy / regularization knobs, not portable session data |
|
||||
| `apply_video / apply_audio / save_video / save_audio_latents / clear` | not on PR-7 state class | state is a pure data carrier; runtime owns lifecycle policy |
|
||||
|
||||
PR 7 is a strict superset of Dreamverse's data model **plus** lifts
|
||||
several env globals into per-session typed fields.
|
||||
|
||||
### Lifecycle mapping
|
||||
|
||||
| Dreamverse pattern | `SessionStore` API |
|
||||
|---|---|
|
||||
| `self.continuation = ContinuationState()` per session | `state = session_store.snapshot(sid) or LTX2ContinuationState()` |
|
||||
| `apply_video(req_kwargs, segment_idx)` + `apply_audio(req_kwargs, segment_idx, audio_lps)` | `state = session_store.snapshot(sid)`; runtime builds request from `state.video_frames` / `state.audio_latents` |
|
||||
| `save_video(frames)` + `save_audio_latents(latents)` | runtime constructs new `LTX2ContinuationState`, `session_store.store(sid, ...)` |
|
||||
| `clear()` at end of session | `session_store.drop(sid)` |
|
||||
|
||||
`SessionStore` and `BlobStore` ABCs ship with thread-safe in-memory
|
||||
defaults (`InMemorySessionStore`, `InMemoryBlobStore`). Dreamverse can
|
||||
adopt them as-is for the local runtime; remote runtimes can plug in
|
||||
redis-backed implementations later.
|
||||
|
||||
### Wire format (HTTP/WS round-trip)
|
||||
|
||||
Dreamverse's `FastVideoRealtimePool` already speaks the realtime
|
||||
runtime's HTTP+WS protocol. When PR 7.5/7.6 land state emission on the
|
||||
server side, the on-the-wire payload is the public envelope:
|
||||
|
||||
```json
|
||||
{
|
||||
"kind": "ltx2.v1",
|
||||
"payload": {
|
||||
"schema_version": 1,
|
||||
"segment_index": 3,
|
||||
"video_conditioning_frame_idx": 9,
|
||||
"video_conditioning_strength": 0.75,
|
||||
"audio_sample_rate": 24000,
|
||||
"audio_conditioning_num_frames": 5,
|
||||
"audio_conditioning_strength": 0.5,
|
||||
"video_position_offset_sec": 0.2,
|
||||
"video": {"frames_b64": ["..."]},
|
||||
"audio": {"safetensors_b64": "..."},
|
||||
"metadata": {}
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
JSON-serializable end-to-end; safetensors blob preserves audio dtype
|
||||
(incl. bf16). For payloads above the inline threshold a `BlobStore`
|
||||
indirection replaces the b64-encoded body with `{"blob_id": "..."}`;
|
||||
the blob itself stays inside the runtime that produced it.
|
||||
|
||||
## Migration plan per PR
|
||||
|
||||
| PR | Dreamverse action |
|
||||
|---|---|
|
||||
| PR 6 (landed) | Typed `GeneratorConfig` available; flat-kwarg path still works via compat. Optional migration. |
|
||||
| PR 7 (landed) | Typed `LTX2ContinuationState` available. ~50-line Dreamverse PR: replace `server/video_generation.py:89` import; move `apply_*`/`save_*`/`clear` off the state class onto `VideoGenerationWorker`; read knobs from typed state instead of env globals; swap `list[PIL.Image]` → `list[np.ndarray]`. |
|
||||
| PR 7.5 (open) | Streaming server skeleton — Dreamverse's `runtime/factory.py` either keeps building `GPUPool` from `RealtimeRuntimeConfig` (current path), or migrates to `ServeConfig.streaming` shape and invokes `fastvideo serve --config realtime.yaml`. Dreamverse's `RuntimePool`/`RuntimeSlot` Protocol can stay in place. |
|
||||
| PR 7.6 (branch ready) | GPU pool upstream — `local_runtime.py` import becomes a public import with same symbols (`RealtimeRuntimeConfig`, `GPUPool`, `get_available_gpus`). Per-GPU continuation state inside the worker becomes a `SessionStore` reference (Dreamverse doesn't see this). `request.state` / `result.state` round-trip starts working end-to-end on the local runtime. |
|
||||
| PR 7.10 (planned) | `generate_async` is canonical. Dreamverse's per-segment `user_step` flow can migrate from sync `generate_video(..., **kwargs)` to consuming the typed event stream. Optional; sync wrapper stays. |
|
||||
|
||||
## Dynamo backend contract
|
||||
|
||||
**FastVideo does not host any Dynamo code.** The backend package
|
||||
(`args.py`, `main.py`, `backend.py`, `register.py`, `health_check.py`,
|
||||
adapter, Dockerfile) lives entirely in the Dynamo repo at
|
||||
`components/src/dynamo/fastvideo/`, modeled on
|
||||
`components/src/dynamo/sglang/`.
|
||||
|
||||
FastVideo's only obligation is to expose a stable, typed Python API
|
||||
that Dynamo's backend package imports.
|
||||
|
||||
### Contract surface
|
||||
|
||||
| Surface | Exposed as |
|
||||
|---|---|
|
||||
| Construction | `VideoGenerator.from_pretrained(model_path, **typed_kwargs)` (typed_kwargs = a stable subset from `GeneratorConfig`; no flat LTX2 legacy) |
|
||||
| Sync execution | `generator.generate_video(request: GenerationRequest) -> VideoResult` |
|
||||
| Async execution | `generator.generate_async(request: GenerationRequest) -> AsyncGenerator[VideoEvent, None]` (PR 7.10) |
|
||||
| Typed request | `fastvideo.api.GenerationRequest`, `SamplingConfig`, `InputConfig` |
|
||||
| Typed result | `VideoResult` with `video_bytes` or tensor frames + optional `ContinuationState` |
|
||||
| Continuation | `ContinuationState(kind, payload)` — schema-versioned payloads |
|
||||
| Health-check input | `VideoGenerator.default_health_check_request() -> GenerationRequest` (256x256 / 8 frames / 1 step) |
|
||||
| Config dump | `GeneratorConfig.to_dict()` / `ServeConfig.to_dict()` |
|
||||
|
||||
### Request/response mapping (Dynamo ↔ FastVideo)
|
||||
|
||||
```
|
||||
NvCreateVideoRequest -> fastvideo.api.GenerationRequest
|
||||
prompt -> sampling.prompt
|
||||
size="WxH" -> sampling.width, sampling.height
|
||||
seconds -> (seconds * nvext.fps) -> sampling.num_frames
|
||||
input_reference -> input.image_path / input.video_path
|
||||
nvext.fps -> sampling.fps
|
||||
nvext.num_frames -> sampling.num_frames (overrides seconds*fps)
|
||||
nvext.num_inference_steps -> sampling.num_inference_steps
|
||||
nvext.guidance_scale -> sampling.guidance_scale
|
||||
nvext.seed -> sampling.seed
|
||||
nvext.negative_prompt -> sampling.negative_prompt
|
||||
response_format -> (handled by adapter at output)
|
||||
|
||||
VideoFinalEvent -> NvVideosResponse
|
||||
video_bytes -> data[0].b64_json (if response_format=b64_json)
|
||||
video_url (after upload) -> data[0].url (if response_format=url)
|
||||
metadata.inference_time_s -> inference_time_s
|
||||
```
|
||||
|
||||
All fields exist on FastVideo's typed schema after PR 6 expansion (typed
|
||||
LTX2 kwargs) + PR 7.10 (`generate_async` + health check).
|
||||
|
||||
### Reference: PR ai-dynamo/dynamo#7544
|
||||
|
||||
Closed draft establishing the Dynamo backend shape. Two frictions
|
||||
identified:
|
||||
|
||||
1. Flat legacy LTX2 kwargs — solved by PR 6.
|
||||
2. Sync-only generation — solved by PR 7.10's `generate_async`.
|
||||
|
||||
Next iteration of this PR (or its successor) will reopen against PR 8's
|
||||
docs reference and land cleanly.
|
||||
|
||||
## Open questions across surfaces
|
||||
|
||||
| # | Question | Source | Status |
|
||||
|---|---|---|---|
|
||||
| Q-1 | Multi-model GPU pool | dreamverse_review D-1 | Deferred (production single-model) |
|
||||
| Q-2 | LTX-2 prompt orchestration promotion to public | dreamverse_review D-2 | Open; consumer-side until 2nd consumer |
|
||||
| Q-3 | Race-based provider fallback | dreamverse_review D-3 | Open; sequential is current public |
|
||||
| Q-4 | Router upstream skip on Dreamverse | dreamverse_review D-4 | Resolved (PR 7.9 lands publicly, Dreamverse doesn't consume) |
|
||||
| Q-5 / D-5 | `generate_async` cutover (audio re-encode) | dreamverse_review | **Blocked on PR 7.10** |
|
||||
| D-6 | Don't upstream `realtime/local_runtime.py` | dreamverse_review | Resolved (Dreamverse switches to `streaming.gpu_pool.SubprocessGpuPool`) |
|
||||
| D-7 / Q-6 | FP4Config public colocation | dreamverse_review | **Resolved May 2** — public NVFP4 landed with lazy flashinfer |
|
||||
| D-8 | `ltx2_image_crf` silently dropped | dreamverse_review | **Unverified post-`d80c2a8`** — see [open-threads.md](open-threads.md) |
|
||||
| D-9 | `aarch64-conda-linux-gnu-cc` triton compile failure | dreamverse_review | Operational; `ENABLE_TORCH_COMPILE=0` workaround |
|
||||
| D-10 | Warmup OOM on shared GPU | dreamverse_review | Operational; idle-GPU pre-warm probe |
|
||||
| D-11 | ffmpeg `Broken pipe` on disconnect | dreamverse_review | Cosmetic logging cleanup |
|
||||
| — | `video_position_offset_sec` semantics (persistent vs per-segment) | dreamverse_integration | **Open — needs decision before PR 7.6 emits state** |
|
||||
| — | `SessionStore` / `BlobStore` lifecycle (TTL/eviction/blob-drop) | dreamverse_integration | Open — defer to PR 7.5 design pass |
|
||||
|
||||
See [decisions-log.md](decisions-log.md) for full rationale per
|
||||
decision.
|
||||
|
||||
## Don't / Cautions
|
||||
|
||||
- **Don't pop the Dreamverse stash on this branch.** It's 3867 lines of
|
||||
orphan modular refactor with broken absolute imports.
|
||||
- **Don't change `RealtimeRuntimeConfig` shape without coordinating
|
||||
with Dreamverse `runtime/factory.py`.**
|
||||
- **Don't promise public compatibility for private Dreamverse-only
|
||||
field aliases.** Those belong in the private adapter layer per design
|
||||
spec.
|
||||
@@ -0,0 +1,978 @@
|
||||
# Decisions Log — D + Q Resolutions
|
||||
|
||||
Cross-doc consolidated decision log. Each entry: ID, source doc,
|
||||
question/decision, rationale, current status.
|
||||
|
||||
For implementation status see [pr-roadmap.md](pr-roadmap.md). For
|
||||
follow-up actions see [open-threads.md](open-threads.md).
|
||||
|
||||
**Last updated:** 2026-05-06 (added D-21 — chunk-stutter root cause is software libx264 encoding consuming ~22% of segment wall-time, NOT a migration regression; verified by 3 parallel explore agents that NVFP4 + torch.compile coverage matches FastVideo-internal exactly; landed opt-in NVENC build path in install_native_ffmpeg.sh + `--nvenc`/`--no-nvenc` flag in dreamverse-deploy.sh + `apps/dreamverse/server/benchmarks/benchmark_av_streaming.py` regression test + memory dir update; default codec stays `libx264` for backward compat, opt-in via `--nvenc`. Earlier: added D-12 — GpuPool layer separation, Oracle review post-#1257-merge; added D-13 — prompt enhancer / LLMProvider abstraction shape, Oracle review pre-#1258-merge; added D-14 — streaming auxiliaries cohesion, Oracle review during #1284 review cycle; added D-15 — streaming router placement + sticky/active-active deferral, Oracle review during #1286 review cycle; added D-16 — streaming router polish round 2, second-pass review on top of D-15 covering bridge cancellation hygiene, registry state machine, httpx hard-fail, replica YAML parsing, and `websockets` dep; added D-17 — strategy reversal: abandon 6-PR split in favor of single mega-PR #1288 on `will/ltx2_sr_port`; added D-18 — Option B+ chosen: Dreamverse FE+product-server move into FastVideo as `apps/dreamverse/` subfolder while generic backend stays at `fastvideo.entrypoints.streaming.*`; integration-review.md deprecated, integration-plan.md is the executable migration plan; added D-19 — D-18 executed: 5 commits land on `will/dreamverse-monorepo`, fix-up commits corrected the integration-plan's invalid "delete generic-merged, import public substitutes" assumption — generic-merged files carried product-local instead, e2e passes against migrated code with /proc-verified evidence; added D-20 — segment-2 BrokenPipe root cause was a TWO-direction silent drop of LTX-2 audio kwargs in public `VideoGenerator`).
|
||||
|
||||
## Status legend
|
||||
|
||||
- ✅ **Resolved** — decision made and implementation complete (or no implementation needed)
|
||||
- 🟡 **Deferred** — decision made, implementation deferred to a known PR
|
||||
- 🔴 **Open** — needs decision
|
||||
|
||||
## Post-merge architecture decisions
|
||||
|
||||
### D-26: Rebase `will/dreamverse-monorepo` directly onto `origin/main` (PR base flip)
|
||||
|
||||
**Status:** ✅ Resolved 2026-05-06 (UTC; 2026-05-07 local). 66 commits cleanly replayed via `git rebase origin/main`; force-pushed via `--force-with-lease`. Local backup branch `will/dreamverse-monorepo-pre-main-rebase-backup-20260506` preserved at the pre-rebase tip `2ee839a3`. PR #1288 on `will/ltx2_sr_port` is untouched.
|
||||
**Source:** User directive: "update this branch will/dreamverse-monorepo to be directly against origin/main for PR purposes instead of ltx_sr_port (don't open new PRs)".
|
||||
|
||||
**Question:** `will/dreamverse-monorepo` was forked from `will/ltx2_sr_port` (PR #1288's head). Should it stay stacked on top of `will/ltx2_sr_port`, or be rebased so its PR base becomes `origin/main` directly?
|
||||
|
||||
**Decision:** Rebase `will/dreamverse-monorepo` onto `origin/main` directly. This makes the branch openable as a single PR against main containing the entire 66-commit chain (40 from the legacy `will/ltx2_sr_port` work + 26 dreamverse-monorepo-specific commits including D-19/D-20/D-21/D-22).
|
||||
|
||||
**Pre-rebase state:**
|
||||
|
||||
```
|
||||
origin/main (c17d33bf) ← 1 new commit since 2aaeee2a (PR #1253 SSIM cosine)
|
||||
│
|
||||
└─ 2aaeee2a (PR #1286 squash merge; old shared base)
|
||||
├─ 40 commits → will/ltx2_sr_port @ fbd823df (PR #1288)
|
||||
└─ 40 + 26 = 66 commits → will/dreamverse-monorepo @ 2ee839a3
|
||||
```
|
||||
|
||||
**Post-rebase state:**
|
||||
|
||||
```
|
||||
origin/main (c17d33bf)
|
||||
│
|
||||
└─ 66 commits → will/dreamverse-monorepo @ 83829c5e (NEW SHAs, same content)
|
||||
|
||||
origin/main (c17d33bf) -- still in main lineage
|
||||
│
|
||||
└─ 2aaeee2a
|
||||
└─ 40 commits → will/ltx2_sr_port @ fbd823df (PR #1288, UNCHANGED)
|
||||
```
|
||||
|
||||
**Operation:**
|
||||
|
||||
1. **Backup**: `git branch will/dreamverse-monorepo-pre-main-rebase-backup-20260506 will/dreamverse-monorepo` (local-only safety net pinning `2ee839a3`).
|
||||
2. **Rebase**: `git rebase origin/main` while on `will/dreamverse-monorepo`. All 66 commits replayed conflict-free in ~30s. The single new origin/main commit `c17d33bf` (#1253 SSIM cosine regression) touches only `fastvideo/tests/ssim/*`, `fastvideo/tests/modal/ssim_test.py`, `.agents/skills/seed-ssim-references/SKILL.md`, `fastvideo/pipelines/basic/stable_audio/stages/decoding.py`, and a 56-line block in `fastvideo/entrypoints/video_generator.py`. None of our 66 commits touch the SSIM files; the `video_generator.py` overlap auto-merged because our D-20 changes (`_BATCH_EXTRA_PASSTHROUGH_KEYS`, result-dict surface) and #1253's changes are in different parts of the file.
|
||||
3. **Verification (file-level, before push)**:
|
||||
- Commit count: 66 (PASS)
|
||||
- Co-author trailer count: 231 across 66 commits — within historical norm (some legacy `will/ltx2_sr_port` commits had partial trailers per [`authors.md`](authors.md) "Known gaps").
|
||||
- Tree-diff vs backup: only the 845-line forward delta from `c17d33bf` (expected). All D-20/D-21/D-22-introduced content (`_BATCH_EXTRA_PASSTHROUGH_KEYS` x2, "unknown field" x1, `av_chunk_interval_ms` x4, `av_chunk_publish_ms` x2, `enable-nvenc` x2, `NVENC_OVERRIDE` x4) survives intact.
|
||||
- `pre-commit run --files` on the 7 most-edited files: PASS (yapf/ruff/codespell/mypy/spaces).
|
||||
- Runtime pytest hung mid-import on this shared dev box (pre-existing CUDA/torch init issue affecting other commands too) — NOT a rebase regression. The same `fastvideo/tests/api/` suite passed 185/185 pre-rebase.
|
||||
4. **Force-push**: `git push --force-with-lease=will/dreamverse-monorepo:2ee839a3... origin will/dreamverse-monorepo`. Updated `2ee839a3...83829c5e (forced update)`.
|
||||
|
||||
**Implications:**
|
||||
|
||||
- The branch can now be opened as a PR against `origin/main` containing all the LTX-2 SR port + NVFP4 + Dreamverse monorepo migration + audio kwarg fix + warmup + NVENC build + benchmarks + integration memory dir, in a single review unit.
|
||||
- PR #1288 on `will/ltx2_sr_port` is untouched and continues to track its own subset of the work. If PR #1288 lands first, the duplicated commits on `will/dreamverse-monorepo` will be reconciled at the next rebase by `git rebase origin/main` dropping commits whose content is now in main (same mechanism as the post-#1286 rebase recorded in [state.md](state.md) "Post-#1286 rebase summary").
|
||||
- Local backup `will/dreamverse-monorepo-pre-main-rebase-backup-20260506 @ 2ee839a3` keeps the old chain available; recommend deleting after the new branch state is verified by running on a non-stuck dev box (or after the PR merges).
|
||||
|
||||
**Watch-outs:**
|
||||
|
||||
- Anyone with a checkout of the OLD `will/dreamverse-monorepo` (pre-rebase) needs to `git fetch origin will/dreamverse-monorepo --force` and discard local commits, OR rebase their local commits onto the new `83829c5e`. None observed in the runbook's worktree-sharing model.
|
||||
- The trailer-count 231 (vs expected 264 for full coverage) is NOT introduced by this rebase — it's the historical gap from [`authors.md`](authors.md). Verifiable by counting trailers on the backup branch (same 231).
|
||||
|
||||
### D-21: AV chunk stutter root cause — software libx264 encoding consumes ~22% of segment wall time; opt-in NVENC fix
|
||||
|
||||
**Status:** 🟡 Deferred — fix landed (NVENC build + `--nvenc` flag), default unchanged so no behavior regression for operators who haven't rebuilt ffmpeg yet. Switch default to `h264_nvenc` once benchmark numbers are captured + a release notes entry is published.
|
||||
**Source:** Live debug session 2026-05-06 prompted by user observation: "stuttering still between chunks" after D-20 + warmup r3 fix. 3 explore agents fanned out (NVFP4 config, torch.compile config, AV chunk pacing).
|
||||
|
||||
**Question:** With D-20 (audio kwarg routing fix) + r3 warmup pass landed and warmup_success=true, why is the live deploy still stuttering between chunks? Is the migration missing some quantization or compile coverage that the FastVideo-internal reference has?
|
||||
|
||||
**Investigation (3 parallel explore agents):**
|
||||
|
||||
1. **NVFP4 config diff** (`bg_0629273d`): No divergence. apps/dreamverse, original Dreamverse, and FastVideo-internal ALL use `layer_profile="refine"` with the same 48-block `fp4_layers` superset, e2m1 mantissa, fp32 alpha, layout_128x4 scale, group size 16. Stage gating (`base` vs `refine`) and `transformer_refine_quant` carrier match. The only naming difference is the public surface rename `FP4Config` → `NVFP4Config`.
|
||||
2. **torch.compile config diff** (`bg_97081cc5`): No divergence. ALL THREE repos use `inductor / fullgraph=True / max-autotune-no-cudagraphs / dynamic=False` and compile only the transformer + text_encoder. **VAE / audio_vae / vocoder are eager bf16 in all three** — so the migrated repo is not missing a compile pass that the original had. Only behavioral difference: FastVideo-internal calls `target.eval()` on its audio encoder; the migrated worker does not (separate D-22 follow-up).
|
||||
3. **AV chunk pacing** (`bg_574e831e`): `stream_fmp4` has only one timer (`av_encode_stream_ms`) — no per-chunk instrumentation. Between `worker.generate_step()` returning and `stream_fmp4()` returning there is no significant work other than ffmpeg's own encode + chunk emission. The `main_user_step − worker_e2e ≈ 1300ms` gap is therefore mostly ffmpeg + controller relay loop, NOT IPC. Per-chunk emission cadence is driven by stdout read size (1 MiB), not fragment boundary, so non-uniform pacing is plausible.
|
||||
|
||||
**Root cause synthesis:**
|
||||
|
||||
| Phase | Wall-time | Compute | Compiled? | NVFP4? |
|
||||
|---|---|---|---|---|
|
||||
| transformer denoise (gen) | ~4500 ms | GPU | YES | YES |
|
||||
| save_conditioning | ~100 ms | GPU/CPU | n/a | n/a |
|
||||
| **stream_fmp4 (libx264 software encode)** | **~1300 ms** | **CPU** | **NO** | **NO** |
|
||||
| TOTAL wall | ~6000 ms | producing 4.67 s playable | | |
|
||||
|
||||
Realtime ratio = 4.67 s playable / 6.0 s wall = **0.78x**. FE buffer drains 1.3 s per segment until empty → inter-segment stutter. **The cause is not a migration regression** (FastVideo-internal exhibits the same architectural limit); it is the choice to use `libx264 ultrafast` software encoding on the segment-streaming hot path on machines that have idle hardware encoders.
|
||||
|
||||
**Build-time finding:** `apps/dreamverse/scripts/install_native_ffmpeg.sh` was building ffmpeg with only `--enable-libx264 --enable-lto`. No `--enable-nvenc`, no nv-codec-headers prereq. So even setting `FASTVIDEO_VIDEO_CODEC=h264_nvenc` would have failed at runtime with "encoder not found".
|
||||
|
||||
**Resolution (this round, default-preserving):**
|
||||
|
||||
1. **`apps/dreamverse/scripts/install_native_ffmpeg.sh`** now:
|
||||
- Clones `nv-codec-headers` (NVENC/NVDEC API headers; no CUDA libs) into `$INSTALL_PREFIX` so `ffnvcodec.pc` is on `pkg-config`'s search path.
|
||||
- Configures ffmpeg with `--enable-cuda --enable-nvenc --enable-cuvid --enable-nvdec` plus `--extra-cflags=-I$CUDA_PREFIX/include` and `--extra-ldflags=-L$CUDA_PREFIX/lib64`.
|
||||
- Drops `--enable-cuda-nvcc` and `--enable-libnpp` so the build does NOT require `--enable-nonfree` (we only need hardware encode/decode, not GPU-side filters).
|
||||
- New env knobs: `ENABLE_NVENC=0|1` (default 1), `CUDA_PREFIX` (default `/usr/local/cuda`), `NV_CODEC_REF`.
|
||||
- Sanity check verifies the resulting binary exposes `h264_nvenc` / `hevc_nvenc` encoders before exiting 0.
|
||||
- **Emits `FASTVIDEO_VIDEO_CODEC=libx264` in the env file by default** so existing deploys are runtime-unchanged.
|
||||
|
||||
2. **`.agents/skills/dreamverse-deploy/scripts/dreamverse-deploy.sh`** now:
|
||||
- Accepts `--nvenc` / `--no-nvenc` flag (defaults to off, matches existing behavior). When `--nvenc`, exports `FASTVIDEO_VIDEO_CODEC=h264_nvenc` into the backend setsid block.
|
||||
- Validates the binary actually has `h264_nvenc` encoder via `ffmpeg -encoders | grep h264_nvenc`. Fails fast if `--nvenc` is requested against a libx264-only ffmpeg with a clear hint to rerun the install script.
|
||||
- `DREAMVERSE_NVENC` env var as the env-var counterpart (flag overrides).
|
||||
- Banner now prints `nvenc=true|false` alongside `warmup` and `torch_compile`.
|
||||
|
||||
3. **`apps/dreamverse/server/benchmarks/benchmark_av_streaming.py`** is the regression test:
|
||||
- Drives `stream_fmp4` directly with synthetic 121-frame 1920x1088 video + 5-second 24kHz stereo audio (production shape).
|
||||
- Sweeps `libx264` and `h264_nvenc` (auto-skips encoders not present in the active ffmpeg).
|
||||
- Reports per-codec: wall_ms_min/median/p95/max, bytes_median, chunks_median, and **realtime_ratio_median** (playable / wall, must be ≥ 1.0 to avoid buffer drain).
|
||||
- Exit code 1 if any codec produces realtime_ratio < 1.0; useful as a CI gate.
|
||||
|
||||
**Empirical findings (this dev host, B200):**
|
||||
|
||||
- `libx264 ultrafast` benchmark on production-shape input (121 frames, 1920x1088, 5s 24kHz stereo audio): wall_median=611ms, wall_p95=644ms, 8.25x realtime ratio in isolation. Captured via `apps/dreamverse/server/benchmarks/benchmark_av_streaming.py --runs 5 --codecs libx264`.
|
||||
- **B200 has NO NVENC silicon.** Direct ffmpeg probe (`-c:v h264_nvenc` against a 64x64 0.2s color frame) fails with `OpenEncodeSessionEx failed: unsupported device (2): No capable devices found`. This is a hardware omission — datacenter Blackwell (B200) and some H100 SKUs prioritize compute density and ship without NVENC. Only consumer Blackwell (RTX 50-series) and select datacenter SKUs (T4, A10, A100 PCIe) have NVENC.
|
||||
- The deploy script's `--nvenc` path now does TWO checks: (a) `h264_nvenc` is in the encoder list (build-time); (b) a 64x64 ffmpeg probe actually succeeds (runtime-time). On a B200, the runtime probe fails fast with a clear "no NVENC silicon on this host" error pointing at SKU-level alternatives.
|
||||
- **Stutter on B200 is NOT primarily ffmpeg encoding.** The benchmark shows libx264 ultrafast at 611ms, but production logs show a 1300ms gap between `worker_e2e` and `main_user_step`. The other ~700ms is IPC + controller WebSocket relay (not measured per-chunk in current `stream_fmp4`).
|
||||
|
||||
**Open / deferred:**
|
||||
|
||||
- ~~D-22~~ ✅ **Resolved 2026-05-06** in `bade2c0a` — `stream_fmp4` is now fully instrumented with per-phase + per-chunk timings; gpu_pool prints a per-segment summary line. The controller-side WS-send slice is tracked separately as D-22-CTL.
|
||||
- D-22-CTL: per-chunk timing in the controller's AV relay loop in `session/controller.py` (between media event arrival and `ws_send_bytes`). Worker side is captured by D-22; the controller side is the still-unmeasured remainder of the 700ms IPC+relay gap.
|
||||
- D-23: B200 + RTX 5090 split — for B200-class deploys without NVENC, the stutter fix needs a different approach (pipeline gen N+1 with encode N, or larger initial FE buffer pre-fill). For RTX 5090 / T4 / A10 deploys, `--nvenc` is the answer once tested. Make the deploy default conditional: probe NVENC at boot, default `--nvenc` if available.
|
||||
- D-24: re-evaluate flipping the deploy default from `libx264` to `h264_nvenc` after a real-NVENC host (RTX 5090 or H100 PCIe) benchmark is run. Tradeoff: NVENC has ~5-10% lower compression at same quality but is hardware-accelerated. For real-time streaming the latency win dominates IF the host has NVENC.
|
||||
- D-25: pipeline benchmark via Python SDK (`apps/dreamverse/server/benchmarks/benchmark_pipeline.py`, landed in `f98811e0`) — captures per-stage timings via `FASTVIDEO_STAGE_LOGGING=1`. Cold-run baseline on B200 NVFP4 (no compile, no warmup): 6.99s for 5.04s playable = 0.72x realtime. Per-stage profile dominated by LTX2RefineLoRAStage (40% on cold, includes one-time LoRA conversion + adapter load), LTX2DenoisingStage (20% — base DiT 5-step denoise), LTX2TextEncodingStage (11%), DecodingStage (4%), audio_decoding/upsample/init each <1%. Future work: rerun under `compile_warm` scenario for steady-state numbers.
|
||||
- The `target.eval()` call on the audio encoder that FastVideo-internal does and the migrated Dreamverse does not — surface as a separate decision once observed empirically (probably a no-op in inference path but worth the symmetry).
|
||||
|
||||
**Cross-references:**
|
||||
|
||||
- [D-20](decisions-log.md#d-20) (audio kwarg routing fix) made segment 2 generate cleanly. D-21 makes the segment-to-segment chain stutter-free in real-time.
|
||||
- [`apps/dreamverse/server/av_streaming.py::stream_fmp4`](file:///home/william5lin/FastVideo/apps/dreamverse/server/av_streaming.py) is the hot path; the cmd builder there respects `FASTVIDEO_VIDEO_CODEC` and branches on `*_nvenc` codecs to use NVENC presets (`p1`/`p2`/...) and `-rc constqp -qp 28` instead of libx264 presets (`ultrafast`/...).
|
||||
- The benchmark cross-references D-21 in its module docstring so a future contributor reading the script alone has the context.
|
||||
|
||||
### D-20: Segment-2 BrokenPipe root cause — two-direction silent drop of LTX-2 audio kwargs in public `VideoGenerator`
|
||||
|
||||
**Status:** ✅ Resolved 2026-05-05. 4 commits on `will/dreamverse-monorepo` (`1b686f4e`..`5eaf0a13`); pushed to origin. End-to-end verified on GPU4 (`/proc/$BE_PID/environ` + `Cached audio latents shape=(1, 8, 126, 16) for segment 2` log line + `Segment 2: relayed av chunks=22, bytes=3.8MB`).
|
||||
**Source:** Live debugging session triggered by recurring `RuntimeError: ffmpeg frame writer failed: [Errno 32] Broken pipe` on segment 2 in `/tmp/opencode/dreamverse-deploy/backend-gpu4.log`.
|
||||
|
||||
**Question:** Why does segment 2 of every Dreamverse session crash with a writer-side EPIPE, while segment 1 streams fine?
|
||||
|
||||
**Symptom chain decoded:**
|
||||
|
||||
1. ffmpeg writer thread in `apps/dreamverse/server/av_streaming.py:307-317` raises `BrokenPipeError` mid-frame.
|
||||
2. ffmpeg's exit code is `0` — the `if rc != 0` branch of `stream_fmp4` never fires; only `if writer_error[0] is not None` does.
|
||||
3. `rc=0 + BrokenPipeError` means ffmpeg exited cleanly *before* the writer finished pushing all frames → ffmpeg closed stdin early due to `-shortest` + audio shorter than video.
|
||||
4. Manual repro confirmed: with audio 71240 samples (~2.97s) and video 112 frames (~4.67s), ffmpeg `-shortest` + 1MB pipe + 1920x1088x3 frames produces exactly this signature (`rc=0 out_bytes=751028 writer_error=BrokenPipeError(32)`).
|
||||
5. The 1.7s audio undershoot is suspicious because Dreamverse's `apply_audio()` in `apps/dreamverse/server/video_generation.py:132-186` explicitly extends `audio_num_frames = NUM_FRAMES + audio_extra` (=`121 + 40 = 161`) for continuation segments. Audio should be 6.71s, not 5.01s. The kwarg was being silently dropped.
|
||||
|
||||
**Root cause (TWO directions):**
|
||||
|
||||
| Direction | Where | What was missing | What it broke |
|
||||
|---|---|---|---|
|
||||
| **Inbound** (kwargs → `batch.extra`) | `fastvideo/entrypoints/video_generator.py::_generate_video_impl` | The 5-key extraction block (`ltx2_audio_latents`, `ltx2_audio_clean_latent`, `ltx2_audio_denoise_mask`, `audio_num_frames`, `video_position_offset_sec`) that FastVideo-internal has at lines 168-183. Without it, `apply_audio`'s kwargs landed in `sampling_param.update(kwargs)` which `logger.error`'d "%s has no field %s" and dropped them. | `audio_num_frames` never reached `batch.extra` → `ltx2_denoising.py:325` fell back to `batch.num_frames=121` → audio generated for 5.01s instead of 6.71s. After `head_trim_audio_frames=49` removed the leading 2.04s, only 2.97s of audio remained for 4.67s of video. |
|
||||
| **Outbound** (`batch.extra` → result dict) | `fastvideo/entrypoints/video_generator.py::_generate_single_video` | `"ltx2_audio_latents": output_batch.extra.get("ltx2_audio_latents")` in the result dict. Internal exposes it at line 556; public didn't expose it at all. | `Dreamverse._derive_next_audio_latents()` always saw `None` → `self.continuation.audio_latents` was never set → segment 2's `apply_audio` short-circuited (`if not (… and self.audio_latents is not None): return`) → no audio continuation, no `audio_num_frames` extension, no `ltx2_audio_clean_latent` carry-over. |
|
||||
|
||||
The two directions hid each other: even after the inbound block was ported, segment 2 still broke until the outbound surface was added. Both must be present for audio continuation to round-trip.
|
||||
|
||||
**Resolution — 4 commits on `will/dreamverse-monorepo`:**
|
||||
|
||||
| SHA | Subject |
|
||||
|---|---|
|
||||
| `1b686f4e` | `[fix] dreamverse: pin ffmpeg native build toolchain by uname -m` |
|
||||
| `265ce1a6` | `[fix] api: route LTX-2 audio kwargs through batch.extra; strict update` |
|
||||
| `dab9499c` | `[feat] dreamverse-deploy: native ffmpeg + compile-off defaults` |
|
||||
| `5eaf0a13` | `[feat] dreamverse-deploy: --warmup / --torch-compile CLI flags` |
|
||||
|
||||
`265ce1a6` is the substantive public-API fix: ports `_BATCH_EXTRA_PASSTHROUGH_KEYS` extraction in `_generate_video_impl`, ports `_extra_overrides` consumption in `_generate_single_video` (writes into `batch.extra`), surfaces `ltx2_audio_latents` in the result dict, and converts `SamplingParam.update()` from `logger.error`-and-drop to `raise ValueError` on unknown keys so the next contributor who adds an unrecognized kwarg hits a loud failure pointing at `_BATCH_EXTRA_PASSTHROUGH_KEYS` instead of debugging a broken-pipe-shaped symptom hours later. Adds `fastvideo/tests/api/test_extra_overrides_routing.py` (7 tests pinning the contract).
|
||||
|
||||
**Why the strict-update is non-negotiable:** the `logger.error`-and-continue pattern is the exact mechanism that hid this bug for the entire Dreamverse-monorepo migration window. It silently traded loud-failure-now for silent-corruption-later. Strict raise + a clear "route via `_BATCH_EXTRA_PASSTHROUGH_KEYS`" hint converts the next regression of this shape from "broken pipe in production" to "ValueError at first call".
|
||||
|
||||
**Cross-references:**
|
||||
|
||||
- [D-11](decisions-log.md#d-11) (Apr 26) noted "ffmpeg fragment write Broken pipe" but framed it as cosmetic (client-disconnect race). Today's was a different code path: server-side EPIPE caused by a server-internal A/V duration mismatch. Both are now resolved; D-11's fix domain (swallow on intentional disconnect) remains unchanged but lower priority.
|
||||
|
||||
- The `[fix] api: route LTX-2 audio kwargs ...` commit (`265ce1a6`) is on `will/dreamverse-monorepo` only. Cherry-picking onto `will/ltx2_sr_port` so it lands as part of PR #1288's mega-PR is a deferred follow-up — see open-threads.md "Cherry-pick API audio routing fix to PR #1288".
|
||||
|
||||
**Side effects of the fix:**
|
||||
|
||||
- `[feat] dreamverse-deploy: native ffmpeg + compile-off defaults` (`dab9499c`) wires the native LTO+libx264+native-arch ffmpeg the team playbook mandates into every deploy via `FASTVIDEO_FFMPEG_BIN=$HOME/opt/ffmpeg-native/bin/ffmpeg` and disables `torch.compile` by default (`ENABLE_TORCH_COMPILE=0`) so segment-1 cold start drops from ~3-4min (max-autotune) to ~45s (pure inference) — required for any iterative debugging cycle to fit inside the 300s session timeout.
|
||||
- `[fix] dreamverse: pin ffmpeg native build toolchain by uname -m` (`1b686f4e`) makes `apps/dreamverse/scripts/install_native_ffmpeg.sh` immune to conda envs that activate both `gcc_linux-64` and `gcc_linux-aarch64` (the aarch64 activation script sorts later and wins, so the inherited `CC=aarch64-conda-linux-gnu-cc` defeated the script's `: "${CC:=...}"` deferred default and tripped x264's compiler probe with "unknown value 'native' for '-march'").
|
||||
- `[feat] dreamverse-deploy: --warmup / --torch-compile CLI flags` (`5eaf0a13`) adds `--warmup`/`--no-warmup` and `--torch-compile`/`--no-torch-compile` flags that override the env-var defaults and accept any position relative to the positional GPU/port args (verified across 13 parser permutations).
|
||||
|
||||
**Watch-outs for downstream contributors:**
|
||||
|
||||
- Adding a new pipeline-specific kwarg now requires either making it a `SamplingParam` field OR adding it to `_BATCH_EXTRA_PASSTHROUGH_KEYS` in `fastvideo/entrypoints/video_generator.py`. Strict `update()` will raise `ValueError` on unknown kwargs that aren't routed through one of those paths.
|
||||
- The `[fix] api:` commit (`265ce1a6`) is BACKWARD-INCOMPATIBLE for any caller that was relying on `SamplingParam.update()` to silently swallow unknown keys. If pre-existing CI breaks elsewhere on this surface, the fix is to add the legitimate kwarg to `SamplingParam` or `_BATCH_EXTRA_PASSTHROUGH_KEYS` (not to revert the strict mode).
|
||||
|
||||
### D-19: D-18 executed — Dreamverse migration on `will/dreamverse-monorepo`, generic-merged files carried product-local
|
||||
|
||||
**Status:** ✅ Resolved 2026-05-05. 5 commits on top of `will/ltx2_sr_port` HEAD `fbd823df`. Branch pushed to origin at `c1fe5d4c`. e2e definitively passes.
|
||||
**Source:** Execution of [integration-plan.md](integration-plan.md) (D-18 plan), with one significant deviation surfaced by Oracle review.
|
||||
|
||||
**Commits:**
|
||||
|
||||
| SHA | Message |
|
||||
|---|---|
|
||||
| `08828d96` | `[feat] dreamverse-monorepo: Phase 1 — skeleton + tooling` |
|
||||
| `f3a863ba` | `[feat] dreamverse-monorepo: Phase 2 — backend move + import rewires` |
|
||||
| `876f7eb3` | `[feat] dreamverse-monorepo: Phase 3 — frontend + public assets` |
|
||||
| `1d47ede6` | `[fix] dreamverse-monorepo: carry generic-merged files product-local` |
|
||||
| `c1fe5d4c` | `[fix] dreamverse-monorepo: entrypoint + audio re-encode + verification` |
|
||||
|
||||
Final stats: 164 files changed, +53294/-2 LOC, 31725 files under `apps/dreamverse/`.
|
||||
|
||||
**Question:** [integration-plan.md](integration-plan.md) Phase 2 instructed "DELETE generic-merged files (`gpu_pool.py`, `av_streaming.py`, `worker_ipc.py`, `mock_server.py`, `session_init_image.py`, `session_logger.py`) from the Dreamverse copy and rewire imports to public `fastvideo.entrypoints.streaming.*` substitutes". The first execution attempt followed this instruction. Was that the right call?
|
||||
|
||||
**Decision (forced by Oracle finding):** No. The "delete + import substitute" strategy assumed the public modules were API-compatible drop-ins for the Dreamverse-product modules. They are NOT. Public `fastvideo.entrypoints.streaming.GpuPool` is an abstract base class with `acquire/run/release/shutdown/health` semantics designed for a future Phase 4 streaming-runtime API. Dreamverse's product `GPUPool(gpu_ids).initialize()` has totally different shape (per-GPU subprocess workers with `slot.user_step / register_stream_queue / acquire(client_id, websocket) -> (gpu_id, slot)` semantics). Substituting one for the other made `apps/dreamverse/server/main.py:63` fail at boot with `TypeError: GpuPool() takes no arguments`.
|
||||
|
||||
**Fix:** Carry all 7 generic-merged files (gpu_pool, av_streaming, worker_ipc, mock_server, session_init_image, session_logger, server_entry) into `apps/dreamverse/server/` as PRODUCT-LOCAL — same status as the rest of the Dreamverse server tree. Imports stay flat (`from gpu_pool import GPUPool`, etc.), matching Dreamverse's existing sys-path-injection convention. Public `fastvideo/entrypoints/streaming/*` reverts cleanly to its `fbd823df` state — `git diff fbd823df..c1fe5d4c -- fastvideo/entrypoints/streaming/` is empty.
|
||||
|
||||
The "import public substitutes" promise becomes a future Phase 4 task: actually harmonize the APIs so Dreamverse can drop its product-local copies. Not in scope for this migration.
|
||||
|
||||
**Why the e2e initially gave a false positive:**
|
||||
|
||||
The first run of all 8 Playwright tests passed (5.1s) — Oracle caught that this was misleading. The `dreamverse-server` console script in `/home/william5lin/miniconda3/envs/fv-main/bin/dreamverse-server` was installed by a prior `pip install -e /home/william5lin/Dreamverse`, so `from server_entry import cli` resolved to `/home/william5lin/Dreamverse/server/...` (the canonical install), not `apps/dreamverse/server/...` (the migrated tree). The migrated code was never actually exercised.
|
||||
|
||||
**Fix to entrypoint resolution** (in `c1fe5d4c`): wrapper scripts at `apps/dreamverse/scripts/dreamverse-server` (and `dreamverse-mock-server`) explicitly do:
|
||||
|
||||
```bash
|
||||
#!/usr/bin/env bash
|
||||
REPO_ROOT="$(cd "$(dirname "$0")/../../.." && pwd)"
|
||||
cd "${REPO_ROOT}/apps/dreamverse/server"
|
||||
exec "${REPO_ROOT}/.venv/bin/python" main.py "$@"
|
||||
```
|
||||
|
||||
This guarantees the migrated tree is what runs. Verified end-to-end via `/proc/$PID/cwd = /home/william5lin/FastVideo/apps/dreamverse/server` during the second e2e run. All 8 Playwright tests pass against this confirmed-migrated backend.
|
||||
|
||||
**Phase 0 environment prereqs (validated 2026-05-05):**
|
||||
|
||||
- `flashinfer-python` in FastVideo `.venv` — required for NVFP4 path. Without it, model load fails with `ImportError: NVFP4 quantization requires flashinfer`.
|
||||
- `cerebras-cloud-sdk` and `openai` in `.venv` — required by the migrated prompt enhancer.
|
||||
- For B200 / sm_100a + gcc-15 conda toolchain: nvcc rejects host compiler. Workaround:
|
||||
```bash
|
||||
CUDAHOSTCXX=/usr/bin/g++-13
|
||||
NVCC_PREPEND_FLAGS="-ccbin /usr/bin/gcc-13 -allow-unsupported-compiler"
|
||||
```
|
||||
Without these, flashinfer JIT compilation fails with `error: #error -- unsupported GNU version! gcc versions later than 14 are not supported!`.
|
||||
|
||||
These are **operator-side prerequisites**, not migration code defects. Documented in `apps/dreamverse/README.md` and `docs/contributing/dreamverse-development.md`.
|
||||
|
||||
**E2E evidence (live, post-fix-up):**
|
||||
|
||||
```
|
||||
PID 179112 cwd: /home/william5lin/FastVideo/apps/dreamverse/server
|
||||
/healthz → 200 {"status":"ok","service":"ltx2-streaming-backend",...}
|
||||
/readyz → 200 {"status":"ready","ready_gpu_workers":1,"total_gpus":1,...}
|
||||
GPU4 mem: 50.9 GiB (NVFP4 model loaded)
|
||||
|
||||
Playwright (8/8 PASS in 5.1s):
|
||||
✓ backend-health/healthz returns ok via the next.js rewrite (32ms)
|
||||
✓ backend-health/readyz reports gpu pool state (11ms)
|
||||
✓ backend-health/status endpoint exposes gpu pool snapshot (12ms)
|
||||
✓ backend-health/prompt-system-config exposes operator-tunable prompts (15ms)
|
||||
✓ backend-health/curated presets endpoint serves a non-empty list (devtools only) (19ms)
|
||||
✓ frontend-shell/main page loads and exposes the FastVideo brand chip (1.4s)
|
||||
✓ frontend-shell/composer hydrates with curated preset cards (1.3s)
|
||||
✓ preset-prompt-generation/generates the first segment from a curated preset prompt (1.8s)
|
||||
```
|
||||
|
||||
**Implications:**
|
||||
|
||||
- The integration-plan.md is **partially superseded** by D-19 outcome:
|
||||
- Phase 2's "DELETE generic-merged" prescription is invalid; replace with "carry product-local".
|
||||
- Phase 0 prereqs need the gcc-13 / `NVCC_PREPEND_FLAGS` workaround documented for B200 hosts.
|
||||
- Phase 4 (in a future PR) is now responsible for actually harmonizing public-vs-Dreamverse pool APIs so the carried product-local modules can be deleted.
|
||||
- `dreamverse-mock-server` script under `apps/dreamverse/scripts/` is the only canonical launcher. The conda env's legacy `/home/william5lin/miniconda3/envs/fv-main/bin/dreamverse-server` should NOT be used (it points at canonical Dreamverse repo).
|
||||
- The audio re-encode handling in `apps/dreamverse/server/video_generation.py` was carried per `c1fe5d4c`; verify the exact shape (restored from source vs deferred) when reading the commit.
|
||||
- Once this branch lands as a PR + merges, the Dreamverse repo can be archived per [integration-plan.md](integration-plan.md) Phase 7.
|
||||
|
||||
**Open follow-ups:**
|
||||
|
||||
- Open a PR for `will/dreamverse-monorepo` (target main, base on `will/ltx2_sr_port` until #1288 merges).
|
||||
- Re-Oracle the post-fix-up state to confirm the prior FAIL is now PASS.
|
||||
- Eventually move audio re-encode into a public module (Phase 4) so the carried product file can be slimmed.
|
||||
- Eventually do real Phase 4 API harmonization between Dreamverse pool and public `fastvideo.entrypoints.streaming.GpuPool` so the 7 carried product files can be deleted.
|
||||
|
||||
### D-18: Option B+ — Dreamverse becomes `apps/dreamverse/` subfolder under FastVideo
|
||||
|
||||
**Status:** ✅ Resolved 2026-05-05. [integration-plan.md](integration-plan.md) is the executable migration plan; [integration-review.md](integration-review.md) is deprecated but kept for drift audit + OSS precedents.
|
||||
**Source:** User decision after reviewing [integration-review.md](integration-review.md)'s Option D recommendation.
|
||||
|
||||
**Question:** [integration-review.md](integration-review.md) recommended **Option D** — Dreamverse stays a separate repo, generic backend (streaming runtime, GPU pool, prompt enhancer, router) merges into `fastvideo.entrypoints.streaming.*`. The user reviewed this and chose a different shape: keep the generic-backend principle from Option D but ALSO move the Dreamverse FE + product server into FastVideo as a subfolder (`apps/dreamverse/`). Combination is "Option B+" (Option B layout with Option D's backend principle).
|
||||
|
||||
**Decision:** Option B+. Concrete shape:
|
||||
|
||||
- **One repo**: `hao-ai-lab/FastVideo`. Dreamverse repo gets archived after migration completes.
|
||||
- **Python ML library** stays at root: `fastvideo/`, `fastvideo-kernel/`.
|
||||
- **Generic backend** stays at `fastvideo.entrypoints.streaming.*` (already there per #1257/#1258/#1284/#1286/#1288).
|
||||
- **Dreamverse product** moves into `apps/dreamverse/{server,web,prompts,serve_configs,scripts}/`.
|
||||
- **Tooling**: uv workspace for Python (`[tool.uv.workspace] members = ["apps/dreamverse/server"]`), standalone pnpm for the FE (no root `package.json`), split CI workflows with path-filter triggers.
|
||||
|
||||
**Rationale:**
|
||||
|
||||
- Drops the cross-repo coordination overhead identified in the post-#1286 rebase cycle (D-17 handled by consolidating into mega-PR; D-18 prevents the next round of cross-repo coordination from happening).
|
||||
- Keeps the architectural separation Option D recommended (FastVideo owns reusable runtime; product owns product). The boundary is now `apps/dreamverse/` directory rather than two repos.
|
||||
- Single repo means atomic cross-cutting refactors (e.g. GpuPool API change + Dreamverse adoption) ship as one PR.
|
||||
- OSS precedents support the shape (chainlit uv-workspace + pnpm; open-webui Python + Svelte with paths-ignore CI). The librarian explicitly noted no precedent for "Python ML library + Next.js product merged into library namespace" — but this isn't that pattern. Dreamverse goes into a sibling directory, NOT into `fastvideo.entrypoints.dreamverse.*`. Library namespace stays clean.
|
||||
|
||||
**Why not Option D (separate repos):**
|
||||
|
||||
- Each upstream merge into FastVideo invalidates Dreamverse's lockfile/imports; the post-#1286 rebase showed this requires coordination overhead that scales with feature velocity.
|
||||
- Cross-repo contract tests catch shape drift but not behavior drift.
|
||||
- Two repos means two `AGENTS.md`, two CI configs, two release stories, two Dependabot dashboards.
|
||||
|
||||
**Why not Option C (full merge into `fastvideo.entrypoints.dreamverse.*`):**
|
||||
|
||||
- Forces FastVideo to ship Tailwind config + curated preset JSON + Next.js build artifacts.
|
||||
- Locks Dreamverse product cadence to FastVideo PyPI releases.
|
||||
- Librarian: "no 1:1 precedent for Python ML library + Next.js product merged into library namespace" — argues against this.
|
||||
|
||||
**Why not Option B (subfolder, but generic backend folded into `apps/dreamverse/server/`):**
|
||||
|
||||
- Other consumers (Dynamo, future streaming clients) need the backend without the Dreamverse product. Folding the backend under `apps/dreamverse/server/` would force Dynamo to either depend on `apps/` paths (ugly) or carry a fork.
|
||||
|
||||
**Implications:**
|
||||
|
||||
- [integration-review.md](integration-review.md) is **deprecated** (banner header + reading-guide demotion). Kept in tree for drift audit + OSS precedent reference.
|
||||
- [integration-plan.md](integration-plan.md) is the **canonical executable plan** with 7 phases (Phase 0: land #1288; Phase 1: skeleton + tooling; Phase 2: backend move; Phase 3: FE move; Phase 4: promote generic-pending; Phase 5: prompt enhancer fork retirement; Phase 6: CI/release cutover; Phase 7: archive Dreamverse repo).
|
||||
- Dreamverse repo will be **archived** at end of Phase 7 — not before.
|
||||
- Dreamverse history does NOT migrate cross-repo via `git mv` (technical limitation); original history stays in archived Dreamverse repo, and Phase 2 PR body records the source SHA(s).
|
||||
- New top-level `apps/` directory created — must be excluded from FastVideo PyPI wheel via `[tool.setuptools.packages.find] exclude = ["apps*", ...]`.
|
||||
- Drift items from [integration-review.md](integration-review.md) get folded into specific phases of [integration-plan.md](integration-plan.md) (e.g. health routes → Phase 4, DR-1 → Phase 5).
|
||||
|
||||
**Open questions deferred to phase planning:**
|
||||
|
||||
- DR-2 (`cerebras_ifm`): public Literal vs Dreamverse-side custom provider — decide before Phase 5.
|
||||
- VPO (`video_position_offset_sec` semantics): persistent vs per-segment — decide in Phase 4.
|
||||
- Cross-repo history: fresh import vs `git subtree` import — decide before Phase 2.
|
||||
- CORS / write-endpoint security policy: dev-only vs auth vs firewall — decide before Phase 6.
|
||||
|
||||
### D-17: Abandon 6-PR split — land everything as single mega-PR #1288
|
||||
|
||||
**Status:** ✅ Resolved 2026-05-05. PR #1287 closed; PR #1288 opened on `will/ltx2_sr_port` covering the full chain.
|
||||
**Source:** User decision after observing the post-#1286 rebase + re-slice cycle.
|
||||
|
||||
**Question:** The original plan ([STACK.md](../../../STACK.md), [pr-roadmap.md](pr-roadmap.md)) called for the remaining `will/ltx2_sr_port` content (after PRs 7.5/7.6/7.7/7.8/7.9 landed) to ship as 6 stacked PRs: 7.10 (#1287, generate_async), 8 (server contract docs), LTX-2 SR runtime, NVFP4, post-fixes, agents-cleanup. PR #1287 was opened on 2026-05-05 as the first slice. Should the remaining 5 slices be opened sequentially as planned, or should everything be consolidated into one PR?
|
||||
|
||||
**Decision:** Consolidate. Close #1287; open one mega-PR (#1288) on `will/ltx2_sr_port` covering all 34 commits / 71 files / +13,074 LOC at once.
|
||||
|
||||
**Rationale:**
|
||||
|
||||
- The post-#1286 rebase + re-slice cycle exposed real overhead: backup branch, interactive rebase with manual `drop` directives, force-push, re-slice 6 bookmarks, push next slice as new remote, open new PR, update memory dir. Repeating that 6 more times for the remaining slices accumulates substantial review-coordination overhead with diminishing structural benefit.
|
||||
- The 6 layers are not independent in the way that landed PRs 7.5-7.9 were. PR 7.10 (`generate_async`) is the only API-shape change; PR 8 is docs+tests on top; LTX-2 SR / NVFP4 / post-fixes / agents-cleanup are feature/fix/docs work that doesn't shape the public API. Reviewing them as one ordered diff is at least as easy as reviewing 6 stacked PRs whose dependencies must be tracked manually.
|
||||
- Single PR keeps CI / merge queue simpler and avoids the 6-PR cascade where every upstream merge invalidates the chain below it.
|
||||
|
||||
**Implications:**
|
||||
|
||||
- [STACK.md](../../../STACK.md) (top-level, 10-PR split tracker) is **deprecated**. Kept in tree as a historical artifact with the merged half (PRs 1-4 of the 10) accurate. Safe to delete in a follow-up.
|
||||
- [authors.md](authors.md), [co-authors.md](co-authors.md) — co-author roster is unchanged; trailers still apply per-commit on every commit in the consolidated PR.
|
||||
- [runbook.md](runbook.md) — "After a PR merges (re-slice protocol)" section replaced by a simpler "After PR #1288 merges" section.
|
||||
- Local split bookmarks (`will/api_7.10`, `will/api_8`, `will/ltx2_sr_runtime`, `will/ltx2_nvfp4`, `will/ltx2_post_fixes`, `will/agents_cleanup`) are no longer maintained; safe to delete locally.
|
||||
- `origin/will/api_7.10` — pushed during the #1287 cycle; can be deleted on origin once #1287 close-cleanup completes.
|
||||
|
||||
**Watch outs:**
|
||||
|
||||
- The PR is large (71 files, +13,074 LOC). Reviewers will need commit-by-commit review; the PR body structures the layers in commit order to make this tractable.
|
||||
- If #1288 becomes too large to merge cleanly later (e.g. main moves significantly underneath it), the fallback is to re-split — but the current expectation is to land it as-is.
|
||||
|
||||
### D-12: `GpuPool` layer separation — keep distinct from `VideoGenerator`
|
||||
|
||||
**Status:** ✅ Resolved (interim) + 🟡 Deferred long-term shape to PR 7.10.
|
||||
**Source:** Oracle review on 2026-05-04, post-PR-#1257 merge.
|
||||
|
||||
**Question:** Should `fastvideo.entrypoints.streaming.GpuPool` (PR #1257) be
|
||||
folded into `fastvideo.entrypoints.video_generator.VideoGenerator`, or kept
|
||||
separate? Three alternatives were evaluated:
|
||||
|
||||
| Alt | Approach | Verdict |
|
||||
|---|---|---|
|
||||
| A | Status quo — `VideoGenerator` (single inference call) and `GpuPool` (multi-session orchestration) stay separate | ✅ Correct as **interim** |
|
||||
| B | `VideoGenerator` absorbs the pool's role (`from_pretrained_pool`, `acquire/release/run`) | ❌ **Wrong layer.** Conflates execution with serving scheduler. |
|
||||
| C | `GpuPool` becomes a thin **session-aware async executor** over PR 7.10's `generate_async` | ✅ Correct **long-term destination** |
|
||||
|
||||
**Decision:** Alt A as interim; evolve toward Alt C once PR 7.10 lands
|
||||
`generate_async`. Do NOT pursue Alt B.
|
||||
|
||||
**Rationale:**
|
||||
|
||||
- `VideoGenerator` is a library handle — "execute one request, possibly
|
||||
across ranks via `MultiprocExecutor`/`RayDistributedExecutor`."
|
||||
- `GpuPool` is serving infrastructure — "schedule N concurrent sessions
|
||||
across N independent replicas, with sticky session-to-GPU affinity for
|
||||
cache locality."
|
||||
- These are different layers driven by different consumers (a Python
|
||||
script doing `gen.generate(req)` vs. a WebSocket server with sticky
|
||||
sessions). Folding them muddies both surfaces.
|
||||
|
||||
**Key finding — `MultiprocExecutor` and `SubprocessGpuPool` are orthogonal,
|
||||
not redundant:**
|
||||
|
||||
| Layer | Job | Granularity |
|
||||
|---|---|---|
|
||||
| `MultiprocExecutor` (`fastvideo/worker/`) | TP/SP shard ONE inference call across N GPU ranks | per-call |
|
||||
| `streaming_generator.py` (existing real-time path) | Per-frame streaming via `MultiprocExecutor.submit_step`/`get_result` | per-step within one generator |
|
||||
| `SubprocessGpuPool` (`entrypoints/streaming/`, PR #1257) | Serve N concurrent sessions on N replicas, sticky-bound | per-session |
|
||||
|
||||
Both spawn subprocesses because **CUDA contexts demand process boundaries**,
|
||||
not because they solve the same problem. Sharing low-level lifecycle
|
||||
utilities (process spawn, queue plumbing, shutdown) is a future refactor;
|
||||
unifying the abstractions is wrong.
|
||||
|
||||
**Sticky binding stays in the pool, NOT in `VideoGenerator`:** sticky
|
||||
session-to-GPU affinity is a serving policy driven by LTX-2's per-GPU
|
||||
continuation cache (last-9-decoded-frames + audio-latents). Different
|
||||
consumers want different policies — stateless OpenAI HTTP wants
|
||||
per-request leasing; LTX-2 streaming wants sticky affinity; per-frame
|
||||
real-time streaming wants a continuous queue. Keeping policy in the pool
|
||||
keeps `VideoGenerator` policy-free.
|
||||
|
||||
**Specific risks flagged in PR #1257 (already merged):**
|
||||
|
||||
| Risk | Mitigation (when relevant) |
|
||||
|---|---|
|
||||
| `GpuPool.run() -> Any` is sync — fine for whole-segment dispatch, blocks on cancellation | Replace with `run_async() -> AsyncIterator[VideoEvent]` in PR 7.10 cycle (`generate_async` makes this trivial) |
|
||||
| `PoolAssignment.gpu_id: int` assumes one-GPU-per-worker | Don't lock as public API. Future may need `device_ids: list[int]` for topology-aware pooling (one worker = group of GPUs running internal `MultiprocExecutor`) |
|
||||
| `GpuPool` could be documented as the canonical FastVideo serving API | Mark as **experimental / server-internal** in docstring until PR 7.10 lands. Don't include in user-facing API docs yet |
|
||||
| Memory: N processes = N model replicas (~10-50 GB each) | Expected for concurrent serving with crash isolation. CUDA IPC weight sharing loses isolation; CPU-shared-memory loading helps host RAM not device. Real scalable path is topology-aware pooling later. |
|
||||
|
||||
**Action items (carried into post-7.10 cycle):**
|
||||
|
||||
- [ ] Update `GpuPool` ABC docstring to note "API may change post-PR-7.10"
|
||||
- [ ] Plan to replace `run()` with `run_async() -> AsyncIterator[VideoEvent]` in PR 7.10 cycle
|
||||
- [ ] Don't promote `gpu_id: int` to public API; revisit shape post-7.10
|
||||
- [ ] Consider clarifying field naming (e.g. `worker_id` is the stable identifier; `gpu_id` is current-impl detail)
|
||||
- [ ] When opening 7.10's PR, have it consume `generate_async` from `GpuPool.run_async` end-to-end
|
||||
|
||||
**Open thread it touches:** PR 7.10 (`open-threads.md` item D — generate_async)
|
||||
unblocks Alt C and is the natural place to land the API shape change.
|
||||
|
||||
### D-15: Streaming router (PR #1286) — keep in-repo, defer sticky / active-active
|
||||
|
||||
**Status:** ✅ Resolved (interim). Pre-merge polishes applied. Three follow-up
|
||||
items tracked.
|
||||
**Source:** Oracle review on 2026-05-05, during PR #1286 review cycle.
|
||||
|
||||
**Question:** Where should the multi-replica WebSocket router live? Should it
|
||||
ship at all (vs. delegating to nginx/envoy)? Should sticky session routing
|
||||
or weighted/round-robin balancing be in the initial PR?
|
||||
|
||||
| Alt | Approach | Verdict |
|
||||
|---|---|---|
|
||||
| A | Status quo — `fastvideo/entrypoints/streaming/router/`, FastAPI-based, single-primary failover, lazy `httpx`/`websockets` imports | ✅ **Keep** |
|
||||
| B | Move to separate package `fastvideo-router/` | ❌ **Premature** — adds packaging/release/compat overhead before evidence of independent adoption |
|
||||
| C | Fold router into the streaming server itself (one app, mode flag) | ❌ Conflates router/generator lifecycles, mode-dependent config, drags inference deps into routing deployments |
|
||||
| D | Replace with reverse proxy (nginx/envoy/HAProxy) recipes | ❌ Not as the SOLE answer — mature proxies don't naturally emit FastVideo typed `gpu_unavailable` frames or evolve with FastVideo session semantics. Recommend external proxies as a complement at high scale. |
|
||||
| E | Add sticky session routing now | ❌ **Defer** — implementing correctly depends on where `session_id` is available (URL/header is easy, first JSON frame is invasive). Reconnects are rare today. |
|
||||
| F | Add weighted / round-robin now | ❌ **Defer** — active-active without sticky routing is worse for LTX-2 continuation locality than active-passive failover |
|
||||
|
||||
**Decision:** Alt A — keep current shape. Apply pre-merge polishes; preserve
|
||||
forward-compat for sticky routing.
|
||||
|
||||
**Rationale:**
|
||||
|
||||
- Python router is justified as a FastVideo-aware control-plane component,
|
||||
not a replacement for Envoy/HAProxy. It can emit typed
|
||||
`gpu_unavailable` frames, evolve with FastVideo session semantics,
|
||||
and ship local/dev deployment without ceremony.
|
||||
- The current abstraction is small + testable: `RouterConfig`,
|
||||
`ReplicaRegistry`, `ReplicaStatus`, `HttpProbe` (Protocol/structural alias).
|
||||
Adding strategy registries / telemetry interfaces / active-active policies
|
||||
now would be over-engineering.
|
||||
- Active-passive (single primary) is the right MVP for LTX-2 streaming —
|
||||
preserves continuation cache locality (D-12 sticky binding rationale)
|
||||
better than naive active-active.
|
||||
- The biggest architectural risk isn't placement; it's accidentally baking
|
||||
in unstated semantics. Define single-primary behavior + config validation
|
||||
now so future active-active or sticky routing becomes additive.
|
||||
|
||||
**Pre-merge polishes applied (per gemini + Oracle review):**
|
||||
|
||||
| # | What | Why |
|
||||
|---|---|---|
|
||||
| 1 | `ReplicaRegistry.select()` docstring rewrite | gemini flagged "round-robin via insertion order" claim was misleading — implementation always returns `[0]`. Replaced with explicit "first healthy primary, else first healthy non-primary; this MVP picks first match within tier; round-robin/weighted deferred". |
|
||||
| 2 | Refactored `run_health_check_loop` to share single `httpx.AsyncClient` across the loop's lifetime via `_build_default_probe()` async context manager | gemini flagged per-probe client instantiation as inefficient. With ~1 probe/second default polling, TCP/TLS handshake overhead is non-trivial; now reuses connection. Tests inject probes directly so the path stays bypassable. |
|
||||
| 3 | Probe all replicas concurrently per cycle via `asyncio.gather(..., return_exceptions=True)` | gemini flagged sequential probes risk falling behind `health_check_interval_seconds` if replicas time out. Now per-cycle wall time = max(probe latencies), not sum. |
|
||||
| 4 | `RouterConfig.__post_init__` validation | Oracle recommended: empty replicas, non-positive intervals/timeouts, thresholds < 1, non-`http(s)://` URLs, and >1 primary all `raise ValueError`. Surfaces misconfiguration at config-load instead of confusing runtime failures. |
|
||||
| 5 | Migrated `@app.on_event("startup"/"shutdown")` to `@contextlib.asynccontextmanager`-based `_lifespan()` | Pre-merge — FastAPI deprecated the old API. Was tracked as the 7.9 caveat in pr-roadmap.md. |
|
||||
|
||||
**One review comment intentionally not implemented:**
|
||||
|
||||
| Comment | Decision |
|
||||
|---|---|
|
||||
| gemini medium: `_load_router_config` duplicates `fastvideo.api.parser.parse_config` logic | Kept manual flat-from-nested mapping. The YAML schema has nested `health_check:` block but `RouterConfig` is flat; using `parse_config` directly would require either restructuring `RouterConfig` to have a nested `HealthCheckConfig` (schema change beyond this PR's scope) or accepting incomplete parsing. Manual mapping is intentional and well-typed. |
|
||||
|
||||
All 4 review threads marked resolved on the GitHub PR.
|
||||
|
||||
**Action items (deferred):**
|
||||
|
||||
- [ ] Track sticky session routing extensibility — when needed, add
|
||||
`ReplicaRegistry.select(routing_key: str | None = None)` so registry
|
||||
evolution is additive; document upfront where `session_id` should
|
||||
appear (URL/header preferred over first JSON frame to avoid
|
||||
buffering/peeking)
|
||||
- [ ] Track `_bridge_session()` backpressure note — fine for MVP because
|
||||
`websockets` library provides basic transport backpressure, but at
|
||||
high scale add max_size/timeouts or recommend Envoy/HAProxy in front
|
||||
- [ ] If active-active multi-primary becomes a requirement, define
|
||||
behavior (round-robin within healthy primaries, weighted, sticky-by-key)
|
||||
rather than letting `select()` silently pick `[0]`
|
||||
|
||||
**Watch outs:**
|
||||
|
||||
- `session_id` in WebSocket URL/headers is the cleanest sticky-routing
|
||||
hook. If it ends up only in the first JSON message, sticky routing
|
||||
later will require buffering/peeking before backend selection.
|
||||
- Multi-primary configs are now explicitly rejected by validation;
|
||||
documented + enforced.
|
||||
- `_bridge_session()` is fine for MVP (the libraries provide basic
|
||||
backpressure), but not production-grade for edge load. Document the
|
||||
limit.
|
||||
|
||||
**Open thread it touches:** open-threads.md items #13 (sticky routing),
|
||||
#14 (bridge backpressure), #15 (multi-primary semantics).
|
||||
|
||||
### D-16: Streaming router polish round 2 — second-pass fixes on top of D-15
|
||||
|
||||
**Status:** ✅ Resolved. Applied as `[fix] streaming: router polish — bridge
|
||||
cancel + state machine + deps` (`a152cb77` on `will/api_7.9`, `40e265b8` on
|
||||
`will/ltx2_sr_port`).
|
||||
**Source:** Second-pass review on PR #1286, 2026-05-05, after D-15's pre-merge
|
||||
polishes landed.
|
||||
|
||||
**Question:** D-15 closed the structural review (placement, sticky/active-active
|
||||
deferral, basic `__post_init__` validation). On a second pass through the same
|
||||
files, five latent issues surfaced that weren't covered by gemini's first pass
|
||||
or Oracle's structural review. Apply them on top of the merged D-15 polishes,
|
||||
or queue for a follow-up PR?
|
||||
|
||||
**Decision:** Apply on top of `will/api_7.9` directly. All five are bug-class
|
||||
or DX-class — none are scope-expanding architecture changes — so folding them
|
||||
into PR #1286 keeps the router landing in one reviewable unit instead of
|
||||
shipping a router PR plus an immediate follow-up fix PR.
|
||||
|
||||
**Fixes applied:**
|
||||
|
||||
| # | File | What | Why |
|
||||
|---|---|---|---|
|
||||
| 1 | `router/main.py::_bridge_session` | Replaced `asyncio.gather()` with `wait(FIRST_COMPLETED)` + explicit `cancel()`/drain + `_is_normal_disconnect()` classifier | `gather` waited for both directions; on client disconnect, the backend-reader task leaked and stayed pending. Backend `ConnectionClosed` also surfaced as an unhandled exception in server logs. New shape: first task to finish triggers explicit cancel of the other, both are drained, and only non-routine exceptions re-raise. |
|
||||
| 2 | `router/registry.py::record_success` | Split state transitions: `UNKNOWN -> HEALTHY` is now immediate on first successful probe; only `UNHEALTHY -> HEALTHY` remains gated by `recovery_threshold` | Previously a fresh registry needed `recovery_threshold` consecutive successes before any replica was selectable. With default `recovery_threshold=2` and `health_check_interval=1s`, that meant 2-3s of `gpu_unavailable` rejections at startup. Now the first probe promotes immediately; recovery gating still protects against flapping replicas. |
|
||||
| 3 | `router/registry.py::_build_default_probe` | Missing `httpx` now raises `RuntimeError` with install hint instead of yielding a "disabled" probe stub | Previous behavior: silently returned `(0.0, "httpx not installed; ...")` for every probe, which `record_failure` then folded into `UNHEALTHY` after `failure_threshold` cycles. Operators saw replicas drop UNHEALTHY with a confusing reason and no clear remediation. Hard-fail at startup is the right surface. |
|
||||
| 4 | `router/config.py::__post_init__` | Extended D-15 polish #4 with: rejects `urlparse(url).path not in ("", "/")`, rejects `query`/`fragment`, rejects duplicate URLs across replicas | D-15's validation rejected non-`http(s)://` URLs and >1 primary; it didn't catch `http://host/api` (the router appends `/health` and `/v1/stream` itself, so a base-URL with path yields malformed routes) or `[{url: x}, {url: x}]` (replica registry keys by URL — duplicates would silently collapse to one entry, masking the misconfiguration). |
|
||||
| 5 | `cli/router_serve.py::_load_router_config` | Replaced silent list-comprehension filter (`for r in replicas_raw if isinstance(r, dict) and r.get("url")`) with per-index `raise ValueError` | Original parser silently dropped malformed YAML entries. A single typo in `replicas[2].url` would yield 2 replicas instead of 3 with no log line. New shape: explicit per-index error message ("missing required key 'url'", "must be a mapping"). |
|
||||
| 6 | `pyproject.toml::[streaming]` extra | Added `websockets` as explicit dep | `router/main.py::_bridge_session` does `import websockets` lazily and raises `RuntimeError` if missing. The `[streaming]` extra was an implicit transitive — anyone installing only `[streaming]` (and not the broader requirements) hit the runtime error. Now explicit. |
|
||||
|
||||
**Tests added (7 cases in `fastvideo/tests/entrypoints/streaming/test_router.py`):**
|
||||
|
||||
- `TestUnknownToHealthyImmediate.test_first_success_promotes_unknown` — first probe success transitions `UNKNOWN -> HEALTHY` regardless of `recovery_threshold`
|
||||
- `TestUnknownToHealthyImmediate.test_unhealthy_recovery_still_gated_by_threshold` — `UNHEALTHY -> HEALTHY` still requires `recovery_threshold` successes
|
||||
- `TestConfigValidation.test_rejects_path_in_url` / `test_rejects_query_in_url` / `test_rejects_fragment_in_url` / `test_rejects_duplicate_urls` / `test_accepts_trailing_slash` — `__post_init__` URL validation matrix
|
||||
|
||||
**Verification:** 17/17 router tests pass on both branches. `pre-commit run`
|
||||
clean (yapf / ruff / codespell / mypy). `lsp_diagnostics` clean on changed
|
||||
regions; the one pre-existing `Task` generic-type warning at `main.py:37` is
|
||||
unrelated and predates this commit.
|
||||
|
||||
**In-flight pre-commit corrections (not part of the 6 fixes themselves):**
|
||||
|
||||
- yapf auto-reformatted 4 files (kept verbatim).
|
||||
- ruff `UP038`: rewrote `isinstance(exc, (CancelledError, WebSocketDisconnect))`
|
||||
to `isinstance(exc, CancelledError | WebSocketDisconnect)`.
|
||||
- mypy `[misc]`: renamed loop var `exc` (inside `for task in done`) to
|
||||
`task_exc` to avoid name collision with the outer
|
||||
`except ImportError as exc` binding.
|
||||
|
||||
**Open thread it touches:** None new. Item #14 (bridge backpressure) and
|
||||
item #13 (sticky routing) from D-15 remain deferred — this round addressed
|
||||
**cancellation/disconnect** semantics on the bridge, which is distinct from
|
||||
**throughput backpressure**. Item #14 still applies: at higher load, add
|
||||
`_bridge_session()` max-size + timeout limits or recommend Envoy/HAProxy
|
||||
in front.
|
||||
|
||||
### D-14: Streaming auxiliaries (PR #1284) — cohesion + concrete-vs-Protocol scoping
|
||||
|
||||
**Status:** ✅ Resolved (interim). Two polish items applied during review; one
|
||||
operational caveat tracked.
|
||||
**Source:** Oracle review on 2026-05-04, during PR #1284 review cycle.
|
||||
|
||||
**Question:** Is PR #1284's bundle of 4 streaming-server auxiliary modules
|
||||
(`prompt/safety.py`, `prompt/rewrite.py`, `session_logger.py`,
|
||||
`mock_server.py`) correctly scoped? Should `mock_server` live in production
|
||||
module path? Should `PromptSafetyFilter` be a Protocol? Should the bundle
|
||||
have been split into 4 PRs?
|
||||
|
||||
| Alt | Approach | Verdict |
|
||||
|---|---|---|
|
||||
| A | Status quo — single PR, 4 modules under `streaming/`, mock_server in production path, concrete safety filter | ✅ **Keep** |
|
||||
| B | Split into 4 separate PRs | ❌ Process overhead, not architectural improvement |
|
||||
| C | Move `mock_server.py` into `tests/` | ❌ Would reduce discoverability + install-time usability of `python -m fastvideo.entrypoints.streaming.mock_server` |
|
||||
| D | Move `session_logger.py` to `streaming/observability/` (or top-level `fastvideo/observability/`) | ❌ Premature — currently session-shaped + streaming-specific; promote when a non-streaming consumer appears |
|
||||
| E | Convert `PromptSafetyFilter` to Protocol (like `LLMProvider`) | ❌ Premature abstraction — only one classifier exists; small duck-typed surface preserves future Protocol introduction without breaking the concrete |
|
||||
| F | Convert `MockGenerator` to Protocol | ❌ Same — small duck-typed surface; no second mock generator exists |
|
||||
|
||||
**Decision:** Alt A — keep current shape. Apply two polish items from
|
||||
Oracle's review before merge.
|
||||
|
||||
**Rationale:**
|
||||
|
||||
- "Streaming-server auxiliaries" is cohesive enough at 730 LOC with
|
||||
isolated modules + tests. Each module has independent code path but
|
||||
shared deployment context (the streaming server boots them all).
|
||||
- `mock_server.py` in production path is a strength: reuses
|
||||
`build_app()` for protocol parity. Hiding it under `tests/` would lose
|
||||
`python -m fastvideo.entrypoints.streaming.mock_server` CLI access for
|
||||
FE devs.
|
||||
- Concrete `PromptSafetyFilter` matches "ship what we have, abstract
|
||||
later" pattern. Internal had multi-classifier composition; public
|
||||
ships single + leaves chaining as a Dreamverse-side concern (per D-2).
|
||||
- Same pattern for `MockGenerator`: small duck-typed `_GeneratorLike`
|
||||
surface lets a second mock implementation drop in without inheritance.
|
||||
- `threading.Lock` (not `asyncio.Lock`) in `session_logger.py` is
|
||||
correct — writes come from real encoder/control threads via
|
||||
`run_in_executor`, not from coroutines directly. `asyncio.Lock` would
|
||||
be the wrong primitive for cross-thread concurrency.
|
||||
|
||||
**Pre-merge polishes applied (per Oracle):**
|
||||
|
||||
| Polish | What | Why |
|
||||
|---|---|---|
|
||||
| 1 | Removed `RewriteOptions.user_system_prompt_override` | Inert public field — was declared but never threaded through to `enhancer.rewrite()`. Shipping unused public options is more likely to bite than any structural choice. Re-add when actually wired through. |
|
||||
| 2 | Sanitized `session_id` filename in `session_logger.SessionLogger._get_file()` | Defense-in-depth: today session_id is server-generated UUID, but a future code path that accepts client-supplied ids would otherwise allow path traversal via `../`. Added `_FILENAME_SANITIZE_RE = re.compile(r"[^A-Za-z0-9._-]")` + sub before `os.path.join`. |
|
||||
|
||||
**Operational caveat tracked (not a code change):**
|
||||
|
||||
- `SafetyDecision.UNAVAILABLE` is treated as `ALLOW` by callers — a
|
||||
policy choice that's correct for an opt-in safety filter, but
|
||||
callers should log loudly so operators know the filter is degraded.
|
||||
Tracked as open-threads.md item #12.
|
||||
|
||||
**Pre-merge review feedback (4 of 4 resolved on the GitHub PR):**
|
||||
|
||||
| # | File:Line | Severity | Issue | Fix applied |
|
||||
|---|---|---|---|---|
|
||||
| 1 | `session_logger.py:57` | High | `log()` race vs `close()` — `KeyError` on `_locks[session_id]` | Atomic capture in `_get_file()`; master `_registry_lock`; `with lock, contextlib.suppress(ValueError):` |
|
||||
| 2 | `rewrite.py:71` | Medium | `re.compile()` in hot path | Module-level `_LEADING_MARKER_RE`, top-level `import re` |
|
||||
| 3 | `safety.py:105` | Medium | `_ensure_loaded()` race on concurrent fastText load | `_load_lock = threading.Lock()` + double-check pattern |
|
||||
| 4 | `pyproject.toml:145` | Medium | `streaming` extra missing `prompt-safety` | Added to aggregator |
|
||||
|
||||
All 4 review threads marked resolved via GraphQL `resolveReviewThread`.
|
||||
|
||||
**Action items (deferred):**
|
||||
|
||||
- [ ] Track `SafetyDecision.UNAVAILABLE` log loudness in
|
||||
open-threads.md item #12 — when streaming server starts using the
|
||||
safety filter, ensure operator-visible logging on `UNAVAILABLE`
|
||||
results
|
||||
- [ ] If a second safety classifier appears (Perspective API, Detoxify,
|
||||
custom rules), promote `PromptSafetyFilter` to a Protocol — same
|
||||
pattern as `LLMProvider` per D-13
|
||||
- [ ] If a second mock generator appears (different frame patterns,
|
||||
different latency models), promote `MockGenerator` to a Protocol
|
||||
|
||||
**Open thread it touches:** PR #1284 itself; future safety-classifier
|
||||
Protocol promotion; future observability module extraction.
|
||||
|
||||
### D-13: Prompt enhancer / `LLMProvider` abstraction shape — keep streaming-scoped
|
||||
|
||||
**Status:** ✅ Resolved (interim) + 🟡 Three deferred polishes after metrics or 2nd consumer.
|
||||
**Source:** Oracle review on 2026-05-04, pre-PR-#1258-merge.
|
||||
|
||||
**Question:** Is PR #1258's `fastvideo.entrypoints.streaming.prompt.*` module
|
||||
correctly designed? Should it be (a) Protocol-based vs ABC, (b) under
|
||||
`streaming/` vs top-level `fastvideo.prompt.*`, (c) closed 3-op enum vs
|
||||
open `complete()` API?
|
||||
|
||||
| Alt | Approach | Verdict |
|
||||
|---|---|---|
|
||||
| A | Status quo — `streaming/prompt/*`, Protocol provider, fixed 3 ops, lazy `httpx`, per-call `AsyncClient` | ✅ **Keep** |
|
||||
| B | Move to top-level `fastvideo.prompt.*` (decouple from streaming) | ❌ **Premature.** No second consumer exists yet. |
|
||||
| C | Convert `LLMProvider` Protocol → ABC with default impls + retry classification | ❌ **Wrong direction.** Biases extension toward OpenAI shape; `_openai_compat.py` already factors that as helper not inheritance. |
|
||||
|
||||
**Decision:** Alt A as interim. Promote to Alt B only when a second
|
||||
non-streaming consumer (OpenAI server, batch generation, tooling) actually
|
||||
needs the prompt enhancer. Don't pursue Alt C.
|
||||
|
||||
**Rationale:**
|
||||
|
||||
- Public contract is tiny — `name: str` + `async complete(LLMRequest) -> LLMResponse`. ABC adds zero value.
|
||||
- `_openai_compat.py` is the right place for shared logic — helper, not base class. Anthropic / local / custom providers stay first-class.
|
||||
- The 3 ops (enhance / auto_extend / rewrite) are LTX-2 streaming concepts. `auto_extend` (continue prompt sequence) and `rewrite` (multi-line alternatives) come directly from session UX. Calling this "the FastVideo prompt API" misrepresents that.
|
||||
|
||||
**Specific risks flagged in PR #1258 (already merged-pending review):**
|
||||
|
||||
| Risk | Mitigation (when relevant) |
|
||||
|---|---|
|
||||
| API publicity — calling this "the FastVideo prompt API" before a second consumer exists | Document module as "streaming-server prompt enhancement" in user-facing docs; keep it nested under `entrypoints/streaming/` |
|
||||
| `httpx.AsyncClient` per-call (no connection pooling) | Acceptable for ~6-10 calls per LTX-2 session; LLM latency dominates. Add optional `client_factory` parameter LATER if metrics show connect overhead is meaningful. |
|
||||
| 3 fixed operations could constrain future generic use | Closed enum is right for application-level orchestration. Future generic consumers should either call `provider.complete()` directly, or get a thin separate enhancer that shares the provider/fallback machinery. |
|
||||
| `register_provider(priority=-1)` semantics rely on Python's negative-index `list.insert` | Cosmetic concern; docstring is clear. Could be tightened to explicit branch later. |
|
||||
| `runtime_checkable` Protocol with `name: str` instance attribute — static type checkers may miss missing `name` | Acceptable; runtime check via `isinstance(p, LLMProvider)` works for plugin discovery. |
|
||||
|
||||
**Action items (deferred):**
|
||||
|
||||
- [ ] Document `fastvideo.entrypoints.streaming.prompt.*` as streaming-scoped in user-facing docs (PR 12 docs migration); avoid promoting as framework-level
|
||||
- [ ] Add optional `client_factory` parameter to providers when metrics justify pooling
|
||||
- [ ] Plan future move to `fastvideo.prompt.*` (with import shim) when second non-streaming consumer materializes
|
||||
- [ ] Track Q-2 reactivation: promote LTX-2 prompt orchestration (locked segments, segment-prompts JSON parsing) to public `fastvideo.entrypoints.streaming.prompt.ltx2_orchestration` when a second LTX-2-style consumer appears
|
||||
|
||||
**Open thread it touches:** Dreamverse migration (open-threads.md DR-1)
|
||||
will be the first real test of the public surface. Lessons learned there
|
||||
inform whether Alt B becomes feasible.
|
||||
|
||||
## D-decisions (from `dreamverse_review.md`, Apr 26)
|
||||
|
||||
### D-1: Realtime runtime → streaming GpuPool migration shape
|
||||
|
||||
**Status:** ✅ Resolved.
|
||||
|
||||
Internal `RealtimeRuntimeConfig` had a multi-model registry +
|
||||
flattened sampling defaults. Public `SubprocessGpuPool` is single-model
|
||||
+ uses per-request `SamplingConfig`.
|
||||
|
||||
**Decision:** Drop multi-model registry on integration branch (not used
|
||||
in production). Construct `GeneratorConfig` for chosen model and pass to
|
||||
`SubprocessGpuPool`. Move sampling defaults to a server-side
|
||||
`default_request: GenerationRequest` template.
|
||||
|
||||
**Risk:** Migration branch surfaces missing-model errors if a flow
|
||||
silently relied on registry to swap models per-session. Integration
|
||||
tests exercise at least one segment per supported model id before
|
||||
merging.
|
||||
|
||||
### D-2: PR 7.7 prompt enhancer API surface narrower than internal
|
||||
|
||||
**Status:** ✅ Resolved.
|
||||
|
||||
Public `PromptEnhancer.enhance/auto_extend/rewrite` returns
|
||||
`LLMResponse(content, provider, model, latency_ms, fallback_used)`.
|
||||
Internal returns `EnhanceResult(prompt, fallback_used, error, ...)` /
|
||||
`RewriteResult(prompts, ..., rollout_id, rollout_label, ...)`.
|
||||
|
||||
**Decision:** Adapt at the call site via
|
||||
`Dreamverse/server/prompting/_internal_compat.py` shim. Locked-segment /
|
||||
next-segment-index plumbing stays Dreamverse-side. Public stays minimal
|
||||
and provider-agnostic.
|
||||
|
||||
**Open question (Q-2):** Promote LTX-2-specific orchestration into
|
||||
`fastvideo.entrypoints.streaming.prompt.ltx2_orchestration` once a
|
||||
second consumer appears. Logged for future review.
|
||||
|
||||
### D-3: Multi-stage provider race vs. sequential fallback
|
||||
|
||||
**Status:** ✅ Resolved (public stays sequential).
|
||||
|
||||
Internal enhancer runs all providers in a stage in parallel
|
||||
(`_run_provider_race`). Public enhancer runs sequentially with
|
||||
retryable-error fallback.
|
||||
|
||||
**Decision:** Public stays sequential for PR 7.7. Race is a
|
||||
Dreamverse-specific tail-latency optimization that depends on parallel
|
||||
API budgets.
|
||||
|
||||
**Risk / Q-3:** First-segment latency on Dreamverse may regress
|
||||
slightly when Cerebras has a bad minute (sequential waits 20s before
|
||||
trying Groq). If real production concern, add public
|
||||
`concurrency: int = 1` knob behind a race path — but only after measuring.
|
||||
|
||||
### D-4: Skip PR 7.9 router for the integration branch
|
||||
|
||||
**Status:** ✅ Resolved.
|
||||
|
||||
Internal stack ships `router/main.py` for multi-replica load balancing.
|
||||
Dreamverse deployment uses single replica per region.
|
||||
|
||||
**Decision:** Land PR 7.9 publicly (upstream the surface). Skip wiring
|
||||
into Dreamverse integration branch. Dreamverse's `server/main.py` does
|
||||
not import from `router/`.
|
||||
|
||||
### D-5: Audio re-encode (PR 7.10) needed for streaming, deferred
|
||||
|
||||
**Status:** 🟡 Deferred to PR 7.10.
|
||||
|
||||
Internal streaming server's per-step path runs `_re_encode_audio` inside
|
||||
`_stream_av_fmp4_events` so each fMP4 segment ships with
|
||||
continuation-conditioning audio. Whole-segment `pool.run()` path doesn't
|
||||
need this.
|
||||
|
||||
**Decision:** Land PR 7.10's `generate_async` publicly. Dreamverse
|
||||
integration branch initially keeps using `pool.run()` (whole segment, no
|
||||
re-encode). Follow-up branch swaps to `generate_async` + audio re-encode.
|
||||
|
||||
**Open question (Q-5):** Acceptable for first switch, or does
|
||||
Dreamverse audio quality regress vs. internal until 7.10 wires in?
|
||||
|
||||
### D-6: `realtime/local_runtime.py` is NOT upstreamed
|
||||
|
||||
**Status:** ✅ Resolved.
|
||||
|
||||
It was the FastVideo-internal precursor to `streaming.gpu_pool`.
|
||||
Upstreaming both would create two GPU pool implementations in public.
|
||||
|
||||
**Decision:** Don't upstream `realtime/local_runtime.py`. Dreamverse
|
||||
switches to `streaming.gpu_pool.SubprocessGpuPool` on integration
|
||||
branch. Internal module can be deleted at follow-up.
|
||||
|
||||
### D-7 / Q-6: `FP4Config` is private-only
|
||||
|
||||
**Status:** ✅ **Resolved May 2.**
|
||||
|
||||
April 26: `Dreamverse/server/video_generation.py:271` imported
|
||||
`fastvideo.layers.quantization.fp4_config.FP4Config` from
|
||||
FastVideo-internal only. The 411-line module hard-imported `flashinfer`.
|
||||
|
||||
**Two options at the time:**
|
||||
|
||||
1. Colocate publicly with `flashinfer` as optional extra
|
||||
`pip install fastvideo[fp4]`; refactor `FP4QuantizeMethod` to take
|
||||
layer-prefix list from a pipeline-config field instead of hardcoding
|
||||
ltx2 paths.
|
||||
2. Keep private — Dreamverse imports from internal via thin shim.
|
||||
|
||||
**Recommendation at the time:** option 1 once API refactor settles.
|
||||
|
||||
**Resolution:** May 2 work chose option 1.
|
||||
- `365a66c7` upstreamed FP4Config with lazy `flashinfer` import in
|
||||
loader helper (no public hard-dep)
|
||||
- `94c983a2` renamed FP4 → NVFP4 to disambiguate from MX-FP4 / OCP-FP4
|
||||
- `42b30bf9` wired through `fastvideo.layers.quantization`
|
||||
|
||||
See [quantization.md](quantization.md) for full details.
|
||||
|
||||
### D-8: `ltx2_image_crf` silently dropped by public schema
|
||||
|
||||
**Status:** 🔴 **Unverified post-`d80c2a8`.**
|
||||
|
||||
April 26: Dreamverse's `server/video_generation.py:406` passed
|
||||
`ltx2_image_crf=0.0` to `SamplingParam(...)`. Public
|
||||
`fastvideo.api.sampling_param.SamplingParam` did NOT have this field;
|
||||
the BE logged ERROR and silently dropped the kwarg.
|
||||
|
||||
**Migration target** (per [design.md](design.md) compatibility map):
|
||||
`request.stage_overrides.refine.image_crf`.
|
||||
|
||||
**Resolution status:** `d80c2a8` (May 2) refactored
|
||||
`server/video_generation.py` to use typed `GeneratorConfig` +
|
||||
`preset_overrides["refine"]`. Whether this PR routed `image_crf`
|
||||
through the typed `stage_overrides` path or left it silently dropped is
|
||||
unverified. See [open-threads.md](open-threads.md).
|
||||
|
||||
### D-9: `aarch64-conda-linux-gnu-cc` triton compile failure
|
||||
|
||||
**Status:** ✅ Resolved (operational).
|
||||
|
||||
Conda env injected an ARM cross-compiler ahead of `gcc` on `$PATH`, so
|
||||
`torch._inductor`'s triton launcher failed compilation. Setting
|
||||
`ENABLE_TORCH_COMPILE=0` bypasses it.
|
||||
|
||||
**Long-term fix:** clean conda env's compiler shadowing or add
|
||||
`CC=gcc` override in Dreamverse's worker bootstrap.
|
||||
|
||||
### D-10: Warmup OOM on shared GPU
|
||||
|
||||
**Status:** ✅ Resolved (operational).
|
||||
|
||||
When `CUDA_VISIBLE_DEVICES` lands on a GPU another tenant uses, LTX-2
|
||||
warmup fails with OOM. Picking an idle GPU (4-7 in test setup) is a
|
||||
manual step.
|
||||
|
||||
**Improvement:** pre-warm probe that checks free memory before booting
|
||||
the pool would prevent this.
|
||||
|
||||
### D-11: ffmpeg fragment write `Broken pipe`
|
||||
|
||||
**Status:** ✅ Resolved (cosmetic).
|
||||
|
||||
When WS client closes before backend finishes streaming first segment,
|
||||
ffmpeg hits `[Errno 32] Broken pipe`. Currently propagates to
|
||||
"User step failed". Cosmetic — swallowing pipe-broken on intentional
|
||||
disconnect would clean up logs.
|
||||
|
||||
## Q-questions (from `streaming-server-upstream-plan.md`, Apr 17)
|
||||
|
||||
### Q-1: Router placement (in-repo or separate package)
|
||||
|
||||
**Status:** ✅ Resolved (in-tree).
|
||||
|
||||
**Recommendation at the time:** separate package `fastvideo-router/` or
|
||||
`fastvideo/contrib/router/`; defer final call to PR 7.9.
|
||||
|
||||
**Resolution:** PR 7.9 implementation places router in-tree at
|
||||
`fastvideo/entrypoints/streaming/router/`.
|
||||
|
||||
### Q-2: Session ID authority
|
||||
|
||||
**Status:** ✅ Resolved (server-generated).
|
||||
|
||||
**Recommendation:** server-generated UUID; accept externally provided
|
||||
session ID only for resume flows.
|
||||
|
||||
### Q-3: Torch compile kwargs typing (opaque vs full vs hybrid)
|
||||
|
||||
**Status:** ✅ Resolved (hybrid).
|
||||
|
||||
**Recommendation:** hybrid — type the common four (`backend`,
|
||||
`fullgraph`, `mode`, `dynamic`) + allow `extras: dict[str, Any]`.
|
||||
|
||||
**Resolution:** PR 6 + NVFP4 `221cb20a` shipped exactly this hybrid.
|
||||
|
||||
### Q-4: Prompt safety / fasttext dependency
|
||||
|
||||
**Status:** ✅ Resolved (optional extra).
|
||||
|
||||
**Recommendation:** ship as optional extra `pip install fastvideo[prompt-safety]`.
|
||||
|
||||
**Resolution:** PR 7.8 implements as optional extra.
|
||||
|
||||
### Q-5: Audio-specific tensor payloads in continuation
|
||||
|
||||
**Status:** ✅ Resolved (typed `LTX2ContinuationState`).
|
||||
|
||||
`ltx2_audio_clean_latent`, `ltx2_audio_denoise_mask`,
|
||||
`ltx2_audio_latents` not in pre-refactor public schema.
|
||||
|
||||
**Recommendation:** classify as opaque fields inside
|
||||
`LTX2ContinuationState.payload`, not top-level sampling fields.
|
||||
|
||||
**Resolution:** PR 7's typed `LTX2ContinuationState` lifts these into
|
||||
typed fields (see [cross-repo-surfaces.md](cross-repo-surfaces.md)
|
||||
field mapping table).
|
||||
|
||||
### Q-6: Dynamo subpackage home
|
||||
|
||||
**Status:** ✅ Resolved (lives in Dynamo repo).
|
||||
|
||||
**Resolution:** No Dynamo code in FastVideo. Full backend package
|
||||
(handler, adapter, registration, health check) owned by Dynamo repo at
|
||||
`components/src/dynamo/fastvideo/`, same pattern as vllm/sglang.
|
||||
FastVideo only guarantees the public API contract.
|
||||
|
||||
### Q-7 (was Q-6 in dreamverse_review): How to land FP4Config publicly
|
||||
|
||||
**Status:** ✅ Resolved May 2 — option 1 (colocate publicly).
|
||||
|
||||
See D-7 above.
|
||||
|
||||
### Q-8: Disaggregation readiness contract test
|
||||
|
||||
**Status:** 🟡 Recommended; not yet shipped.
|
||||
|
||||
PR ai-dynamo/dynamo#7544 is aggregated-only. `ContinuationState` hybrid
|
||||
already supports future prefill/decode split.
|
||||
|
||||
**Recommendation:** PR 7.10 explicitly validate `ContinuationState`
|
||||
survives round-trip through Dynamo-style RPC (pickle or JSON), even
|
||||
though Dynamo isn't using it today. Cheap regression guard.
|
||||
|
||||
### Q-9: Dynamo progress/status passthrough
|
||||
|
||||
**Status:** 🟡 Deferred until Dynamo clarifies.
|
||||
|
||||
`NvVideosResponse` has `status` and `progress` fields.
|
||||
|
||||
**Recommendation:** PR 7.10 stays aggregated-final-only to match PR
|
||||
#7544 shape; revisit after Dynamo clarifies their streaming/progress
|
||||
semantics.
|
||||
|
||||
## Cross-doc questions still 🔴 OPEN
|
||||
|
||||
These need decisions; tracked also in [open-threads.md](open-threads.md):
|
||||
|
||||
| ID | Question | Source | Why it matters |
|
||||
|---|---|---|---|
|
||||
| **D-8** | Did `d80c2a8` route `ltx2_image_crf` correctly, or is it still silently dropped? | dreamverse_review | Latent silent-drop bug; FP4-disabled paths may degrade |
|
||||
| **VPO** | `video_position_offset_sec` — persistent accumulation (a) vs per-segment hint (b) | dreamverse_integration | Needs decision before PR 7.6 emits state |
|
||||
| **SBS** | `SessionStore` / `BlobStore` lifecycle (TTL/eviction/blob-drop on state replacement) | dreamverse_integration | Needs decision in PR 7.5 design pass |
|
||||
| **#1** | Migrate `/healthz`+`/readyz`+`/status` into FastVideo `build_app` | streaming-upstream-plan + handoff | Closes BE_FLAVOR=fastvideo FE-compatibility |
|
||||
| **#3** | Add `cerebras_ifm` to public `PromptEnhancerConfig.provider` Literal | handoff | Internal supports it; public schema doesn't |
|
||||
| **#4** | Expose `layer_profile` on typed `engine.quantization` | handoff | Removes Dreamverse's `experimental["pipeline_config"]` dodge |
|
||||
| **#5** | Typed `dit_config.quant_config` carrier (design TBD) | handoff | Eliminates the `experimental["pipeline_config"]` escape hatch entirely |
|
||||
@@ -0,0 +1,332 @@
|
||||
# Design — Typed Public Inference API
|
||||
|
||||
Synthesis of the FastVideo public inference API refactor design philosophy.
|
||||
For PR-by-PR execution see [pr-roadmap.md](pr-roadmap.md). For the streaming
|
||||
extension see [streaming-server.md](streaming-server.md).
|
||||
|
||||
**Last updated:** 2026-05-03.
|
||||
|
||||
## Why the refactor
|
||||
|
||||
The pre-refactor public boundary mixed three concerns through `**kwargs`:
|
||||
|
||||
- `VideoGenerator.from_pretrained(..., **kwargs)` mixed engine/runtime,
|
||||
pipeline init, and component overrides.
|
||||
- `VideoGenerator.generate_video(..., **kwargs)` mixed prompt+inputs,
|
||||
sampling, output, and model-specific workflow knobs.
|
||||
- Unknown keys silently filtered or merely logged → API drift hard to detect.
|
||||
- Multi-stage models (LTX-2 two-stage, Hunyuan15 SR, LongCat distill+refine)
|
||||
exposed via ad hoc top-level flags.
|
||||
|
||||
This was already painful for LTX2/Dreamverse and would worsen as more
|
||||
multi-stage pipelines came in.
|
||||
|
||||
## Core decision
|
||||
|
||||
FastVideo has:
|
||||
|
||||
1. **Typed nested public schema** — `RunConfig`, `ServeConfig`,
|
||||
`GeneratorConfig`, `GenerationRequest`, `ContinuationState`.
|
||||
2. **Model-owned named pipeline presets** — `ltx2_two_stage`,
|
||||
`longcat_distill_refine`, `hunyuan15_sr_1080p`, etc. All 13 model families
|
||||
landed presets in PR 4.
|
||||
3. **Semantic stage overrides by stage name** —
|
||||
`request.stage_overrides["refine"] = LTX2RefineStageOverride(...)`.
|
||||
4. **Optional advanced explicit plans** for power users — `GenerationPlan`
|
||||
(escape hatch only; not the canonical surface).
|
||||
5. **YAML-first CLI** with dotted overrides —
|
||||
`fastvideo generate --config run.yaml --request.sampling.seed 42`.
|
||||
|
||||
The canonical user experience: choose a model → choose a preset → override
|
||||
a few typed fields → generate. Dicts/YAML/JSON are supported as
|
||||
serialization, but parse immediately into typed objects with strict
|
||||
unknown-key validation.
|
||||
|
||||
## Schema surface
|
||||
|
||||
Implemented in [`fastvideo/api/`](file:///home/william5lin/FastVideo/fastvideo/api/):
|
||||
|
||||
| Type | Role |
|
||||
|---|---|
|
||||
| `RunConfig` | Offline envelope: `generator` + `request` |
|
||||
| `ServeConfig` | Serving envelope: `generator` + `server` + `default_request` + optional `streaming` |
|
||||
| `GeneratorConfig` | `model_path`, `revision`, `trust_remote_code`, `engine`, `pipeline` |
|
||||
| `EngineConfig` | parallelism / offload / compile / quantization / flags |
|
||||
| `PipelineSelection` | `workload_type`, `preset`, `preset_version`, `components`, `preset_overrides`, `experimental` |
|
||||
| `GenerationRequest` | `prompt`, `negative_prompt`, `inputs`, `sampling`, `runtime`, `output`, `stage_overrides`, `state`, `plan`, `extensions` |
|
||||
| `ContinuationState` | Opaque envelope `{kind: str, payload: dict[str, Any]}` |
|
||||
| `GenerationPlan` | Advanced/escape-hatch only; `{stages: list[PlannedStage], final_stage: str|None}` |
|
||||
|
||||
Files:
|
||||
|
||||
| File | Role |
|
||||
|---|---|
|
||||
| [`schema.py`](file:///home/william5lin/FastVideo/fastvideo/api/schema.py) | All public dataclasses |
|
||||
| [`parser.py`](file:///home/william5lin/FastVideo/fastvideo/api/parser.py) | `from_dict`, `to_dict`, `load_yaml`, `load_json`, validation |
|
||||
| [`overrides.py`](file:///home/william5lin/FastVideo/fastvideo/api/overrides.py) | Dotted override application |
|
||||
| [`compat.py`](file:///home/william5lin/FastVideo/fastvideo/api/compat.py) | Legacy kwargs translation (~370 lines, scheduled for death PRs 14-17) |
|
||||
| [`presets.py`](file:///home/william5lin/FastVideo/fastvideo/api/presets.py) | Preset registry |
|
||||
| [`sampling_param.py`](file:///home/william5lin/FastVideo/fastvideo/api/sampling_param.py) | Internal `SamplingParam` adapter (canonical home since PR 4) |
|
||||
| [`results.py`](file:///home/william5lin/FastVideo/fastvideo/api/results.py) | `GenerationResult` / `VideoResult` |
|
||||
| [`errors.py`](file:///home/william5lin/FastVideo/fastvideo/api/errors.py) | Path-aware validation errors |
|
||||
|
||||
## Boundary normalization rule
|
||||
|
||||
Every public inference entrypoint normalizes into typed config objects
|
||||
before touching legacy internals (`FastVideoArgs`, `SamplingParam`).
|
||||
Includes Python constructors, `generate*` calls, CLI `generate`, CLI
|
||||
`serve`, OpenAI server request translation, streaming server request
|
||||
translation.
|
||||
|
||||
Legacy internals (`FastVideoArgs`, `SamplingParam`) may remain temporarily,
|
||||
but only behind a typed normalization boundary.
|
||||
|
||||
## Strict-by-default validation
|
||||
|
||||
All structured inputs are strict:
|
||||
|
||||
- Unknown keys → error
|
||||
- Wrong types → error
|
||||
- Invalid stage names → error
|
||||
- Incompatible state/preset combinations → error
|
||||
|
||||
The only intentional escape hatches:
|
||||
|
||||
- `generator.pipeline.experimental` — for in-flight features without typed home
|
||||
- `request.extensions` — same, request-side
|
||||
|
||||
These bypass validation by design. Intent: shrink as presets absorb
|
||||
model-specific fields. New fields should not land in `experimental` /
|
||||
`extensions` without a plan to either promote them to typed fields or
|
||||
remove them within two PR cycles.
|
||||
|
||||
Error format includes nested path:
|
||||
|
||||
```
|
||||
Invalid field: request.stage_overrides.refine.num_inference_steps
|
||||
Expected int, got "two"
|
||||
Preset: ltx2_two_stage
|
||||
Stage: refine
|
||||
```
|
||||
|
||||
## Request mutation tracking
|
||||
|
||||
When a `GenerationRequest` is parsed from raw dict (YAML/JSON/Python),
|
||||
FastVideo tracks which fields the user explicitly provided vs. which got
|
||||
schema defaults. Matters for `request_to_sampling_param()` — explicit
|
||||
values override model defaults; schema defaults do NOT.
|
||||
|
||||
Mechanics:
|
||||
|
||||
- At parse time, original raw dict + baseline snapshot stored on the request.
|
||||
- Dataclass field mutations (e.g. `request.sampling.seed = 7`) captured via
|
||||
lightweight `__setattr__` dirty-path recording.
|
||||
- Dict-typed field mutations (e.g. `del request.stage_overrides["refine"]`)
|
||||
detected at access time by diffing current dict vs. baseline.
|
||||
- Setting a field to its schema default value IS captured as explicit, so
|
||||
it overrides model defaults.
|
||||
- Raw dict reconciled lazily when `normalize_generation_request()` is called.
|
||||
|
||||
## Schema purity (model-specific fields still in shared schema)
|
||||
|
||||
Remain for back-compat during initial migration; targeted for migration
|
||||
into preset-owned typed override classes:
|
||||
|
||||
| Field | Owner | Migration target |
|
||||
|---|---|---|
|
||||
| `SamplingConfig.height_sr` / `width_sr` / `num_inference_steps_sr` | Hunyuan15 SR | `HunyuanSRStageOverride` (PR 10) |
|
||||
| `SamplingConfig.guidance_scale_2`, `boundary_ratio` | Wan2.2, LingBotWorld | preset-owned (per-family PR) |
|
||||
| `InputConfig.mouse_cond`, `keyboard_cond`, `grid_sizes` | MatrixGame | `request.extensions` or typed input config |
|
||||
| `InputConfig.c2ws_plucker_emb` | LingBotWorld | `request.extensions` or typed input config |
|
||||
| `InputConfig.refine_from`, `stage1_video` | LongCat | `LongCatRefineStageOverride` inputs (PR 9) |
|
||||
|
||||
LTX-2 multi-modal CFG knobs (`ltx2_modality_scale_video/_audio`,
|
||||
`ltx2_rescale_scale`, `ltx2_stg_scale_video/_audio`,
|
||||
`ltx2_stg_blocks_video/_audio`) still leak into shared `SamplingParam` but
|
||||
only LTX-2 reads them today. Migration to typed `LTX2SamplingOverride` is
|
||||
deferred to per-model migration sweep.
|
||||
|
||||
**LTX-2 CFG-force fix landed in PR 6**: defaults moved from `3.0/7.0` to
|
||||
`1.0/1.0` to stop force-enabling CFG for non-LTX-2 families.
|
||||
`ltx2_base` preset still sets `3.0/7.0` explicitly. Regression guard:
|
||||
`test_presets.py::TestPresetDefaultTypes::test_ltx2_cfg_defaults_are_off`.
|
||||
|
||||
## Continuation state
|
||||
|
||||
Public surface:
|
||||
|
||||
```python
|
||||
@dataclass
|
||||
class ContinuationState:
|
||||
kind: str # e.g. "ltx2.v1"
|
||||
payload: dict[str, Any]
|
||||
```
|
||||
|
||||
Internally, model-specific typed subclasses (e.g. `LTX2ContinuationState`
|
||||
at [`fastvideo/pipelines/basic/ltx2/continuation.py`](file:///home/william5lin/FastVideo/fastvideo/pipelines/basic/ltx2/continuation.py)).
|
||||
|
||||
Payload must be JSON-serializable or use opaque blob-ID indirection for
|
||||
large tensors — supports both stateless OpenAI client round-trip AND
|
||||
future Dynamo prefill/decode disaggregation.
|
||||
|
||||
Hybrid model: server-held for streaming WS, client-round-trip for
|
||||
stateless HTTP. See [streaming-server.md](streaming-server.md) D-1.
|
||||
|
||||
## Pipeline package structure (target)
|
||||
|
||||
Per-family colocation under `pipelines/basic/<family>/`:
|
||||
|
||||
```
|
||||
fastvideo/pipelines/basic/<family>/
|
||||
├── <family>_pipeline.py # pipeline implementation(s)
|
||||
├── presets.py # user-facing presets (DONE in PR 4)
|
||||
├── pipeline_configs.py # engine/arch config (from configs/pipelines/)
|
||||
└── stages/ # model-specific stages (optional, if >2 files)
|
||||
```
|
||||
|
||||
What stays shared:
|
||||
|
||||
- `configs/pipelines/base.py` — `PipelineConfig` base class
|
||||
- `configs/models/` — architecture defs (dits/, vaes/, encoders/)
|
||||
- `pipelines/stages/` — shared stages only (denoising, encoding, decoding,
|
||||
text_encoding, timestep_preparation, ...)
|
||||
|
||||
What's gone (PR 4):
|
||||
|
||||
- `fastvideo/configs/sample/` — directory removed entirely; defaults
|
||||
absorbed into per-family `presets.py`.
|
||||
- All 12 `*_SamplingParam` subclass files — `SamplingParam` lives at
|
||||
`fastvideo/api/sampling_param.py`; defaults flow through
|
||||
`SamplingParam.from_pretrained()` → `_from_preset()`.
|
||||
|
||||
What's pending: `configs/pipelines/<family>.py` colocation, optional
|
||||
`pipelines/stages/<family>_*.py` colocation. Per-model migration PRs
|
||||
(6/9/10) include the colocation step for that family.
|
||||
|
||||
## YAML examples
|
||||
|
||||
### Run config
|
||||
|
||||
```yaml
|
||||
generator:
|
||||
model_path: /models/ltx2
|
||||
engine:
|
||||
num_gpus: 1
|
||||
parallelism: {tp_size: -1, sp_size: -1}
|
||||
offload: {dit: false, text_encoder: false, vae: false, pin_cpu_memory: true}
|
||||
pipeline:
|
||||
workload_type: t2v
|
||||
preset: ltx2_two_stage
|
||||
components:
|
||||
config_root: /models/ltx2-config
|
||||
upsampler_weights: /models/ltx2-refine
|
||||
lora_path: /models/ltx2-refine-lora
|
||||
preset_overrides:
|
||||
refine: {enabled: true, add_noise: true}
|
||||
|
||||
request:
|
||||
prompt: "a fox running through snow"
|
||||
sampling: {num_frames: 121, height: 1024, width: 1536, num_inference_steps: 8, seed: 42}
|
||||
output: {save_video: true, return_state: true}
|
||||
stage_overrides:
|
||||
refine: {num_inference_steps: 2, guidance_scale: 1.0}
|
||||
```
|
||||
|
||||
### Serve config
|
||||
|
||||
See [`Dreamverse/serve_configs/streaming_demo.yaml`](file:///home/william5lin/Dreamverse/serve_configs/streaming_demo.yaml)
|
||||
for a canonical example matching internal/ui defaults (LTX-2 distilled,
|
||||
NVFP4, 121 frames @ 1088×1920 24fps, 5 inference steps, 2-step refine).
|
||||
|
||||
## Compatibility mapping (legacy → typed)
|
||||
|
||||
| Legacy field | New path |
|
||||
|---|---|
|
||||
| `model_path` | `generator.model_path` |
|
||||
| `num_gpus` | `generator.engine.num_gpus` |
|
||||
| `tp_size` / `sp_size` | `generator.engine.parallelism.{tp_size,sp_size}` |
|
||||
| `dit_cpu_offload` | `generator.engine.offload.dit` |
|
||||
| `enable_torch_compile` | `generator.engine.compile.enabled` |
|
||||
| `torch_compile_kwargs` | split: `generator.engine.compile.{backend,fullgraph,mode,dynamic}` + `.extras` |
|
||||
| `enable_torch_compile_text_encoder` | `generator.engine.compile.text_encoder_enabled` |
|
||||
| `prompt_txt` | `request.inputs.prompt_path` |
|
||||
| `image_path` / `video_path` | `request.inputs.{image_path,video_path}` |
|
||||
| `output_path` / `save_video` / `return_frames` | `request.output.*` |
|
||||
| `seed` / `num_frames` / `height` / `width` / `fps` / `num_inference_steps` / `guidance_scale` | `request.sampling.*` |
|
||||
| `enable_teacache` / `return_trajectory_*` | `request.runtime.*` |
|
||||
|
||||
LTX-2 specific (private adapter, NOT public compat promise):
|
||||
|
||||
| Legacy LTX-2 field | New path |
|
||||
|---|---|
|
||||
| `config_model_path` | `generator.pipeline.components.config_root` |
|
||||
| `ltx2_refine_enabled` | `generator.pipeline.preset_overrides.refine.enabled` |
|
||||
| `ltx2_refine_upsampler_path` | `generator.pipeline.components.upsampler_weights` |
|
||||
| `ltx2_refine_lora_path` | `generator.pipeline.components.lora_path` |
|
||||
| `ltx2_refine_num_inference_steps` | `request.stage_overrides.refine.num_inference_steps` |
|
||||
| `ltx2_refine_guidance_scale` | `request.stage_overrides.refine.guidance_scale` |
|
||||
| `ltx2_refine_add_noise` | `generator.pipeline.preset_overrides.refine.add_noise` |
|
||||
| `ltx2_image_crf` | `request.stage_overrides.refine.image_crf` |
|
||||
| `return_continuation_state` | `request.output.return_state` |
|
||||
|
||||
LongCat:
|
||||
|
||||
| Legacy | New |
|
||||
|---|---|
|
||||
| `refine_from` / `stage1_video` | `request.inputs.{refine_from,stage1_video}` |
|
||||
| `t_thresh` / `spatial_refine_only` / `num_cond_frames` | `request.stage_overrides.refine.*` |
|
||||
|
||||
## External inspirations (and limits)
|
||||
|
||||
| Source | Useful idea | Don't copy |
|
||||
|---|---|---|
|
||||
| Ray | YAML-first config interchange | Ray's package layout |
|
||||
| SGL `multimodal_gen` | Split instance/request config; dict input parsed into typed objects; merge user overrides on model defaults | `SamplingParams._adjust(ServerArgs)` (request depending on engine config); broad weakly-typed request bags |
|
||||
| vLLM-Omni | Model-owned pipeline presets; explicit stage topology; per-stage default sampling | Positional `sampling_params_list`; serving-engine stage-index semantics in primary Python API |
|
||||
|
||||
## Naming guidance
|
||||
|
||||
- Public schema names namespaced under `fastvideo.api`
|
||||
- Don't export from top-level `fastvideo/__init__.py` until migration further along
|
||||
- `RunConfig` / `ServeConfig` get sufficient disambiguation from training
|
||||
config via the namespace
|
||||
- Future rename to `EngineQuantizationConfig` reserved if a collision
|
||||
arises (deferred)
|
||||
|
||||
## Public Python API (canonical form)
|
||||
|
||||
```python
|
||||
from fastvideo import VideoGenerator
|
||||
from fastvideo.api import (
|
||||
GeneratorConfig, GenerationRequest,
|
||||
EngineConfig, OutputConfig,
|
||||
PipelineSelection, SamplingConfig,
|
||||
)
|
||||
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
config=GeneratorConfig(
|
||||
model_path="/models/ltx2",
|
||||
engine=EngineConfig(num_gpus=1),
|
||||
pipeline=PipelineSelection(workload_type="t2v", preset="ltx2_two_stage"),
|
||||
)
|
||||
)
|
||||
|
||||
result = generator.generate(
|
||||
GenerationRequest(
|
||||
prompt="a fox running through snow",
|
||||
sampling=SamplingConfig(num_frames=121, height=1024, width=1536,
|
||||
num_inference_steps=8, seed=42),
|
||||
output=OutputConfig(save_video=True, return_state=True),
|
||||
)
|
||||
)
|
||||
```
|
||||
|
||||
Accepted constructor forms:
|
||||
|
||||
```python
|
||||
VideoGenerator.from_pretrained(config=GeneratorConfig(...))
|
||||
VideoGenerator.from_config(GeneratorConfig(...))
|
||||
VideoGenerator.from_file("run.yaml")
|
||||
VideoGenerator.from_pretrained("model-id", num_gpus=2, ...) # stable convenience
|
||||
VideoGenerator.from_pretrained(model_path, **legacy_kwargs) # compat (deprecated PR 13)
|
||||
```
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,888 @@
|
||||
# Integration Review — Drift Audit + Path Forward
|
||||
|
||||
> # ⚠️ DEPRECATED — superseded by [integration-plan.md](integration-plan.md)
|
||||
>
|
||||
> This document recommended **Option D** (Dreamverse stays a separate repo;
|
||||
> generic backend merges into FastVideo). On 2026-05-05 the team chose
|
||||
> **Option B+** instead (Dreamverse FE + product server move into FastVideo
|
||||
> as `apps/dreamverse/`; generic backend stays at
|
||||
> `fastvideo.entrypoints.streaming.*` per Option D's principle).
|
||||
> See [decisions-log.md D-18](decisions-log.md#d-18) for the strategy
|
||||
> reversal rationale and [integration-plan.md](integration-plan.md) for the
|
||||
> executable migration plan.
|
||||
>
|
||||
> **What's still authoritative in this file:**
|
||||
> - **Part 1 — Drift audit** (the 17-row drift summary table). The drift
|
||||
> findings remain valid; the migration plan in `integration-plan.md`
|
||||
> folds them into specific phases.
|
||||
> - **OSS precedent citations** (vLLM, BentoML, Ray Serve, TGI+ChatUI,
|
||||
> Transformers.js, ComfyUI, AUTOMATIC1111). Reused in `integration-plan.md`.
|
||||
>
|
||||
> **What's superseded:**
|
||||
> - **Part 2 — Recommendation (Option D)**. Replaced by Option B+ in the
|
||||
> new plan. Read `integration-plan.md` for the current decision.
|
||||
> - **Part 3 — Action items**. Replaced by the phased migration plan.
|
||||
>
|
||||
> Kept in tree for historical reference and audit trail. Do not delete.
|
||||
|
||||
**Last updated:** 2026-05-05 (deprecated header added).
|
||||
|
||||
**Scope:** FastVideo public `will/ltx2_sr_port` at the requested audit
|
||||
anchor `b36bdbc9`; Dreamverse `will/integrate-public-fastvideo` at
|
||||
`ec8ef92`; FastVideo-internal `will/rebase-nbv` as read-only comparison.
|
||||
|
||||
**Memory-dir context:** the current integration memory snapshot tracks the
|
||||
same public mega-PR lineage as `will/ltx2_sr_port`, with PRs #1257,
|
||||
#1258, #1284, and #1286 already merged, #1287 closed, and #1288 open as
|
||||
the consolidated landing vehicle for LTX-2 SR runtime, NVFP4,
|
||||
`generate_async`, Dynamo contract, and memory-dir cleanup. Source:
|
||||
[memory index](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/README.md#L8-L19)
|
||||
and [D-17](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/decisions-log.md#L19-L45).
|
||||
|
||||
**Bottom line:** Zero core typed API drift — typed construction, typed
|
||||
continuation state, NVFP4 wiring, and Dynamo-facing async events are either
|
||||
already public or in #1288. **Real drift remains on the realtime-runtime
|
||||
contract surface (`/healthz` / `/readyz` / `/status` routes per
|
||||
[cross-repo-surfaces.md](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/cross-repo-surfaces.md#L74-L88))
|
||||
and on operational/product edges**: stale Dreamverse docs/scripts, a
|
||||
1933-LOC Dreamverse prompt-enhancer fork, two unresolved per-session
|
||||
fields (`ltx2_image_crf` D-8, `video_position_offset_sec` VPO), one
|
||||
missing example config, and two internal-only utilities whose product
|
||||
relevance is not yet proven.
|
||||
|
||||
---
|
||||
|
||||
## Part 1 — Drift audit
|
||||
|
||||
### Methodology
|
||||
|
||||
1. **Compared three repositories and branches.**
|
||||
- FastVideo public: `/home/william5lin/FastVideo`, branch
|
||||
`will/ltx2_sr_port`.
|
||||
- Dreamverse: `/home/william5lin/Dreamverse`, branch
|
||||
`will/integrate-public-fastvideo`.
|
||||
- FastVideo-internal: `/home/william5lin/FastVideo-internal`, branch
|
||||
`will/rebase-nbv`.
|
||||
- Canonical repo paths are listed in the integration memory index:
|
||||
[repo paths](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/README.md#L72-L79).
|
||||
|
||||
2. **Scoped the audit to the ultimate integration goal.**
|
||||
- Dreamverse should depend on public `fastvideo`, not
|
||||
`FastVideo-internal`.
|
||||
- FastVideo should own the reusable backend subset that Dreamverse
|
||||
currently needs from internal: streaming runtime, GPU pool, router,
|
||||
prompt enhancer, NVFP4, continuation state, and typed generation.
|
||||
- Dynamo should consume FastVideo through typed public Python APIs, not
|
||||
through private modules.
|
||||
- The three Dreamverse surfaces are documented as pipeline construction,
|
||||
realtime runtime, and continuation state:
|
||||
[cross-repo surfaces](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/cross-repo-surfaces.md#L13-L20).
|
||||
|
||||
3. **Separated intentional refactor from drift.**
|
||||
- A path rename is not drift if the public branch contains the same
|
||||
responsibility under the typed design.
|
||||
- A deleted file is not drift if the public design intentionally
|
||||
consolidated it.
|
||||
- A private alias is not drift if the public schema exposes a typed
|
||||
replacement with contract tests.
|
||||
- This matches the typed-public-boundary rule in
|
||||
[design.md](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/design.md#L24-L43).
|
||||
|
||||
4. **Used memory docs for rationale and worktree files for concrete proof.**
|
||||
- API schema and public exports:
|
||||
[schema](file:///home/william5lin/FastVideo/fastvideo/api/schema.py#L68-L85),
|
||||
[api exports](file:///home/william5lin/FastVideo/fastvideo/api/__init__.py#L49-L109).
|
||||
- Streaming server current routes:
|
||||
[build_app](file:///home/william5lin/FastVideo/fastvideo/entrypoints/streaming/server.py#L88-L160).
|
||||
- Dreamverse dependency state:
|
||||
[pyproject server extra](file:///home/william5lin/Dreamverse/pyproject.toml#L17-L22),
|
||||
[uv lock editable source](file:///home/william5lin/Dreamverse/uv.lock#L716-L722).
|
||||
- Contract tests:
|
||||
[Dreamverse shape](file:///home/william5lin/FastVideo/fastvideo/tests/contract/test_dreamverse_shape.py#L1-L26),
|
||||
[Dynamo shape](file:///home/william5lin/FastVideo/fastvideo/tests/contract/test_dynamo_shape.py#L1-L19),
|
||||
[generate_async](file:///home/william5lin/FastVideo/fastvideo/tests/contract/test_generate_async.py#L1-L7).
|
||||
|
||||
5. **Did not treat product-only Dreamverse behavior as FastVideo drift.**
|
||||
- Dreamverse keeps a local product server and Next.js UI today:
|
||||
[README baseline](file:///home/william5lin/Dreamverse/README.md#L5-L16).
|
||||
- Product-only routes, curated presets, devtools, and frontend-specific
|
||||
behavior belong in Dreamverse unless a second non-Dreamverse consumer
|
||||
needs them.
|
||||
|
||||
6. **Risk scale used below.**
|
||||
- **P0:** blocks Dreamverse from running without FastVideo-internal.
|
||||
- **P1:** blocks clean `BE_FLAVOR=fastvideo` or Dynamo/public API use.
|
||||
- **P2:** reproducibility or maintenance drag.
|
||||
- **P3:** optional parity or future memory/perf improvement.
|
||||
|
||||
### Findings: zero core typed API drift
|
||||
|
||||
The public branch is aligned with the goal on the **core typed API
|
||||
surface** (construction, request, continuation state, async events).
|
||||
The table below lists items that look like drift only if compared by
|
||||
path name or legacy field name. They are intentional public refactors
|
||||
or already guarded by tests. **Note:** the realtime-runtime _contract_
|
||||
surface (FE-required health routes) is a separate matter — see "real
|
||||
drift items" §4 below.
|
||||
|
||||
| Investigated item | Drift? | Evidence | Conclusion |
|
||||
|---|---:|---|---|
|
||||
| Dreamverse surface 1: pipeline construction | No | Dreamverse migrated from flat kwargs to typed `GeneratorConfig` at `d80c2a8`; mapping documented in [cross-repo surfaces](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/cross-repo-surfaces.md#L22-L47). | Stable public surface exists. |
|
||||
| Dreamverse surface 2: realtime runtime | No on architecture; some route work remains | Runtime migration target is public `streaming/`, not internal `realtime/`: [streaming upstream](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/streaming-server.md#L14-L31). | Rename/refactor is intentional. |
|
||||
| Dreamverse surface 3: continuation state | No | Public typed `ContinuationState` plus LTX-2 state mapping are documented in [cross-repo surfaces](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/cross-repo-surfaces.md#L108-L146). | Public state is a superset of Dreamverse's data carrier. |
|
||||
| Internal `fastvideo/entrypoints/realtime/` | No | Public design chooses parallel `fastvideo/entrypoints/streaming/`: [layout decision](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/streaming-server.md#L55-L60), [current build_app](file:///home/william5lin/FastVideo/fastvideo/entrypoints/streaming/server.py#L88-L160). | Intentional rename plus typed-config rewrite. |
|
||||
| Internal `configs/sample/` presets | No | Public PR 4 intentionally deleted `configs/sample/` and moved defaults to per-family presets plus `fastvideo/api/sampling_param.py`: [design](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/design.md#L194-L205), [PR roadmap](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/pr-roadmap.md#L21-L29). | Intentional consolidation. |
|
||||
| LTX-2 pipeline presets | No | Public target is model-owned named presets and per-family colocation: [design](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/design.md#L175-L205). | Public layout matches design. |
|
||||
| Internal `use_fp4_linear` flag | No | Public typed quant carrier is `engine.quantization.transformer_quant`; schema field exists in [schema](file:///home/william5lin/FastVideo/fastvideo/api/schema.py#L68-L85), compat resolves it in [compat.py](file:///home/william5lin/FastVideo/fastvideo/api/compat.py#L267-L279). | Replaced by typed NVFP4 surface. |
|
||||
| Public-only `transformer_quant` field | No | Public `FastVideoArgs` pins typed quant to `dit_config.quant_config`: [fastvideo_args](file:///home/william5lin/FastVideo/fastvideo/fastvideo_args.py#L220-L228), [apply logic](file:///home/william5lin/FastVideo/fastvideo/fastvideo_args.py#L260-L279). | Public superset, not drift. |
|
||||
| Internal `config_model_path` | No | Public typed home is `generator.pipeline.components.config_root`: [design mapping](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/design.md#L261-L269), [compat mapping](file:///home/william5lin/FastVideo/fastvideo/api/compat.py#L295-L299). | Alias is covered. |
|
||||
| Internal flat video request fields | No | Public `GenerationRequest` nests `inputs`, `sampling`, `runtime`, `output`, `state`, `extensions`: [schema](file:///home/william5lin/FastVideo/fastvideo/api/schema.py#L193-L204). Internal legacy fields live in internal protocol at [protocol.py](file:///home/william5lin/FastVideo-internal/fastvideo/entrypoints/openai/protocol.py#L64-L82). | Intentional request refactor. |
|
||||
| Dreamverse typed init kwargs | No | Contract test asserts current Dreamverse load kwargs all land on typed fields, not `experimental`: [test_dreamverse_shape](file:///home/william5lin/FastVideo/fastvideo/tests/contract/test_dreamverse_shape.py#L44-L135). | Guard in place. |
|
||||
| Dreamverse request path | No | Contract test asserts request fields round-trip through typed `GenerationRequest`: [test_dreamverse_shape](file:///home/william5lin/FastVideo/fastvideo/tests/contract/test_dreamverse_shape.py#L153-L197). | Guard in place. |
|
||||
| Dynamo native backend shape | No | FastVideo's only obligation is stable typed Python API: [cross-repo contract](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/cross-repo-surfaces.md#L188-L210). | Dynamo should stay out of FastVideo. |
|
||||
| `generate_async` event API | No | API exists in [video_generator](file:///home/william5lin/FastVideo/fastvideo/entrypoints/video_generator.py#L264-L332), event types exist in [results.py](file:///home/william5lin/FastVideo/fastvideo/api/results.py#L109-L164). | #1288 covers the async contract. |
|
||||
| Dynamo request mapping | No | Authoritative source is the contract test [test_dynamo_shape](file:///home/william5lin/FastVideo/fastvideo/tests/contract/test_dynamo_shape.py#L90-L175) which asserts `req.prompt`, `req.sampling.{height,width,num_frames,fps,num_inference_steps,guidance_scale,seed,negative_prompt}`, and `req.inputs.{image_path,video_path}` against the actual nested [`GenerationRequest` schema](file:///home/william5lin/FastVideo/fastvideo/api/schema.py#L193-L204). The `streaming-server.md` Dynamo mapping table mis-cites a `prompt -> sampling.prompt` path that no longer exists; the test is correct, the doc is stale and tracked for refresh. | Guard in place; companion doc needs minor refresh. |
|
||||
| Public API exports | No | `VideoEvent`, `VideoResult`, and typed schema classes are exported from [fastvideo.api](file:///home/william5lin/FastVideo/fastvideo/api/__init__.py#L49-L109). | Integration imports resolve. |
|
||||
| FastVideo-internal FP4/NVFP4 paths | No | Public NVFP4 files and roles are documented in [quantization](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/quantization.md#L24-L35); actual `NVFP4Config` documents lazy FlashInfer and public naming in [nvfp4_config.py](file:///home/william5lin/FastVideo/fastvideo/layers/quantization/nvfp4_config.py#L1-L19). | Public is typed superset. |
|
||||
| AbsMaxFP8 refactor | No for Dreamverse | Public quant registry includes `AbsMaxFP8` and `NVFP4`: [quantization init](file:///home/william5lin/FastVideo/fastvideo/layers/quantization/__init__.py#L1-L8). AbsMaxFP8 failure is tracked as separate tech debt: [open threads](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/open-threads.md#L100-L117). | Not Dreamverse blocker. |
|
||||
| Internal realtime API regression test | No | Public contract tests replace it: [Dreamverse contract](file:///home/william5lin/FastVideo/fastvideo/tests/contract/test_dreamverse_shape.py#L1-L26), [Dynamo contract](file:///home/william5lin/FastVideo/fastvideo/tests/contract/test_dynamo_shape.py#L1-L19), [generate_async tests](file:///home/william5lin/FastVideo/fastvideo/tests/contract/test_generate_async.py#L91-L230). | Better scoped guards exist. |
|
||||
| Dreamverse dependency declaration | No | `server` extra declares `fastvideo>=0.1.7`: [pyproject](file:///home/william5lin/Dreamverse/pyproject.toml#L17-L22). Dev lock resolves editable public `../FastVideo`: [uv.lock](file:///home/william5lin/Dreamverse/uv.lock#L716-L722), [package source](file:///home/william5lin/Dreamverse/uv.lock#L777-L780). | Dependency is already switched in metadata/lock. |
|
||||
|
||||
#### Core conclusion for the zero-typed-drift section
|
||||
|
||||
The public typed API no longer needs to mirror `FastVideo-internal` file
|
||||
paths. The correct test is whether Dreamverse and Dynamo can express their
|
||||
needs through public typed objects and public entrypoints. On that test,
|
||||
the **typed core** is covered (construction, request, continuation,
|
||||
async events). The **runtime contract** still has health-route gaps —
|
||||
see real drift §4. On the **typed core**:
|
||||
|
||||
- `GeneratorConfig` and `GenerationRequest` cover construction and calls:
|
||||
[schema surface](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/design.md#L45-L72).
|
||||
- `ServeConfig.streaming` covers the server envelope:
|
||||
[schema](file:///home/william5lin/FastVideo/fastvideo/api/schema.py#L244-L279).
|
||||
- `generate_async` covers streaming, OpenAI, and Dynamo on one substrate:
|
||||
[video_generator](file:///home/william5lin/FastVideo/fastvideo/entrypoints/video_generator.py#L264-L332).
|
||||
- Contract tests now encode the cross-repo shapes:
|
||||
[Dreamverse](file:///home/william5lin/FastVideo/fastvideo/tests/contract/test_dreamverse_shape.py#L70-L214),
|
||||
[Dynamo](file:///home/william5lin/FastVideo/fastvideo/tests/contract/test_dynamo_shape.py#L170-L331),
|
||||
[async events](file:///home/william5lin/FastVideo/fastvideo/tests/contract/test_generate_async.py#L91-L273).
|
||||
|
||||
### Findings: real drift items requiring action
|
||||
|
||||
#### 1. Dreamverse README and bootstrap script still point at FastVideo-internal
|
||||
|
||||
- **Priority:** P0 for a clean public-dependency story.
|
||||
- **Effort:** Small.
|
||||
- **Owner:** Dreamverse repo.
|
||||
- **Evidence:** Dreamverse metadata already points at public FastVideo:
|
||||
[pyproject](file:///home/william5lin/Dreamverse/pyproject.toml#L17-L22),
|
||||
[uv source](file:///home/william5lin/Dreamverse/pyproject.toml#L54-L61),
|
||||
[uv.lock](file:///home/william5lin/Dreamverse/uv.lock#L716-L722).
|
||||
- **Drift:** README still tells users that `uv` resolves from
|
||||
`../FastVideo-internal` and that bootstrap expects `../FastVideo-internal`:
|
||||
[README](file:///home/william5lin/Dreamverse/README.md#L76-L109).
|
||||
- **Drift:** bootstrap script still defaults to cloning the private repo and
|
||||
verifying imports from that clone:
|
||||
[script defaults](file:///home/william5lin/Dreamverse/.agents/skills/bootstrap-fastvideo-private-fork/scripts/bootstrap_fastvideo_private.sh#L7-L11),
|
||||
[script clone flow](file:///home/william5lin/Dreamverse/.agents/skills/bootstrap-fastvideo-private-fork/scripts/bootstrap_fastvideo_private.sh#L33-L63),
|
||||
[script import assertion](file:///home/william5lin/Dreamverse/.agents/skills/bootstrap-fastvideo-private-fork/scripts/bootstrap_fastvideo_private.sh#L66-L88).
|
||||
- **Action:** Replace private-fork bootstrap with public FastVideo bootstrap
|
||||
or delete the bootstrap once PyPI publication is the default path.
|
||||
- **Do not overreach:** no FastVideo code change required.
|
||||
|
||||
#### 2. Dreamverse carries a 1933-line prompt-enhancer fork
|
||||
|
||||
- **Priority:** P1.
|
||||
- **Effort:** Medium.
|
||||
- **Owner:** Dreamverse repo, after public prompt enhancer is available.
|
||||
- **Evidence:** Dreamverse local fork starts at
|
||||
[server/prompt_enhancer.py](file:///home/william5lin/Dreamverse/server/prompt_enhancer.py#L1-L80).
|
||||
- **Public replacement:** FastVideo now has provider-agnostic
|
||||
`PromptEnhancer` with `enhance`, `auto_extend`, `rewrite`, and
|
||||
`register_provider`:
|
||||
[public enhancer](file:///home/william5lin/FastVideo/fastvideo/entrypoints/streaming/prompt/enhancer.py#L66-L142).
|
||||
- **Provider extension point:** custom providers implement `LLMProvider`:
|
||||
[provider protocol](file:///home/william5lin/FastVideo/fastvideo/entrypoints/streaming/prompt/providers/base.py#L63-L75).
|
||||
- **Tracking:** DR-1 in open threads already defines the compat-shim shape:
|
||||
[DR-1](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/open-threads.md#L174-L206).
|
||||
- **Action:** Replace the fork with a small Dreamverse shim that adapts
|
||||
public `LLMResponse` to Dreamverse's product response objects and keeps
|
||||
only product-only extras.
|
||||
- **Do not overreach:** do not merge Dreamverse's full prompt product layer
|
||||
into FastVideo unless a second consumer needs the same semantics.
|
||||
|
||||
#### 3. `cerebras_ifm` provider is unresolved
|
||||
|
||||
- **Priority:** P1 if Dreamverse needs IFM in production; P2 otherwise.
|
||||
- **Effort:** Small decision plus small/medium implementation.
|
||||
- **Owner:** Team decision; implementation either Dreamverse-side or public.
|
||||
- **Public state:** `PromptEnhancerConfig.provider` is currently
|
||||
`Literal["cerebras", "groq"]`:
|
||||
[schema](file:///home/william5lin/FastVideo/fastvideo/api/schema.py#L229-L235).
|
||||
- **Design note:** public Literal excludes `cerebras_ifm` today:
|
||||
[streaming-server D-3](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/streaming-server.md#L61-L100).
|
||||
- **Tracking:** DR-2 already frames the public-vs-Dreamverse decision:
|
||||
[DR-2](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/open-threads.md#L211-L229).
|
||||
- **Recommended default:** implement IFM as a Dreamverse-side custom provider
|
||||
registered through `enhancer.register_provider(...)` unless there is a
|
||||
non-Dreamverse public user.
|
||||
|
||||
#### 4. `/healthz`, `/readyz`, and `/status` are not in public `build_app`
|
||||
|
||||
- **Priority:** P1 for `BE_FLAVOR=fastvideo` frontend compatibility.
|
||||
- **Effort:** Medium/Large because route shapes need tests.
|
||||
- **Owner:** FastVideo public.
|
||||
- **Public current state:** `build_app` exposes `GET /health` and
|
||||
`WS /v1/stream`:
|
||||
[server.py](file:///home/william5lin/FastVideo/fastvideo/entrypoints/streaming/server.py#L126-L160).
|
||||
- **Dreamverse expected state:** Dreamverse exposes `GET /healthz`,
|
||||
`GET /readyz`, and `GET /status`:
|
||||
[routes/health.py](file:///home/william5lin/Dreamverse/server/routes/health.py#L34-L79).
|
||||
- **Tracking:** open item #1 documents route ownership and files likely to
|
||||
touch:
|
||||
[open threads](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/open-threads.md#L69-L99).
|
||||
- **Design note:** `/curated-presets`, `/prompt-system-config`, and devtools
|
||||
stay Dreamverse-side, with feature detection:
|
||||
[streaming route contract](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/streaming-server.md#L232-L259).
|
||||
|
||||
#### 5. `fastvideo/models/layerwise_offload.py` exists only internally
|
||||
|
||||
- **Priority:** P3 unless memory-tight Dreamverse deployments require it.
|
||||
- **Effort:** Medium if adopted; low if documented as deferred.
|
||||
- **Owner:** FastVideo public only if a concrete deployment needs it.
|
||||
- **Internal evidence:** internal file defines async layerwise CPU offload
|
||||
manager with pinned CPU memory and prefetch stream:
|
||||
[layerwise_offload.py](file:///home/william5lin/FastVideo-internal/fastvideo/models/layerwise_offload.py#L1-L20),
|
||||
[prefetch path](file:///home/william5lin/FastVideo-internal/fastvideo/models/layerwise_offload.py#L127-L180).
|
||||
- **Public state:** no equivalent public file was identified in this audit.
|
||||
- **Action:** defer unless Dreamverse or another public deployment hits a
|
||||
memory ceiling that cannot be handled by existing offload knobs.
|
||||
- **Decision rule:** if adopted, port as a generic offload utility with
|
||||
tests; do not make it Dreamverse-specific.
|
||||
|
||||
#### 6. Standalone LTX-2 upsampler CLI exists only internally
|
||||
|
||||
- **Priority:** P2 for reproducibility; P3 for product runtime.
|
||||
- **Effort:** Small/Medium after scope decision.
|
||||
- **Owner:** FastVideo public if standalone upsampling is a supported user
|
||||
workflow.
|
||||
- **Internal utility:** `upscale_video_file(...)` reads an existing video,
|
||||
prepares frame count/resolution, loads VAE + upsampler, and writes an mp4:
|
||||
[upsample.py](file:///home/william5lin/FastVideo-internal/fastvideo/entrypoints/upsample.py#L120-L180),
|
||||
[write tail](file:///home/william5lin/FastVideo-internal/fastvideo/entrypoints/upsample.py#L181-L202).
|
||||
- **Internal CLI:** `fastvideo upsample` wrapper exists internally:
|
||||
[cli/upsample.py](file:///home/william5lin/FastVideo-internal/fastvideo/entrypoints/cli/upsample.py#L15-L35),
|
||||
[CLI args](file:///home/william5lin/FastVideo-internal/fastvideo/entrypoints/cli/upsample.py#L48-L130).
|
||||
- **Public related functionality:** LTX-2 SR refine stage covers the
|
||||
in-pipeline latent upsample/refine path:
|
||||
[ltx2_refine.py](file:///home/william5lin/FastVideo/fastvideo/pipelines/basic/ltx2/stages/ltx2_refine.py#L1-L22),
|
||||
[upsample stage](file:///home/william5lin/FastVideo/fastvideo/pipelines/basic/ltx2/stages/ltx2_refine.py#L116-L180).
|
||||
- **Action:** decide whether standalone file-to-file upsampling is a public
|
||||
CLI promise or whether the SR refine stage is sufficient.
|
||||
|
||||
#### 7. Reproducible streaming demo config lives only in Dreamverse
|
||||
|
||||
- **Priority:** P2.
|
||||
- **Effort:** Small.
|
||||
- **Owner:** FastVideo public.
|
||||
- **Evidence:** canonical demo config currently lives at
|
||||
[Dreamverse/serve_configs/streaming_demo.yaml](file:///home/william5lin/Dreamverse/serve_configs/streaming_demo.yaml#L1-L12).
|
||||
- **Config content:** it documents LTX-2 distilled model, one GPU,
|
||||
no offload, compile settings, NVFP4, refine overrides, default request,
|
||||
and streaming settings:
|
||||
[generator block](file:///home/william5lin/Dreamverse/serve_configs/streaming_demo.yaml#L31-L87),
|
||||
[streaming block](file:///home/william5lin/Dreamverse/serve_configs/streaming_demo.yaml#L108-L149).
|
||||
- **Memory pointer:** design.md already treats this as the canonical
|
||||
example:
|
||||
[design YAML example](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/design.md#L235-L239).
|
||||
- **Action:** copy/adapt it into
|
||||
`examples/serving/streaming_demo.yaml` with public-safe comments.
|
||||
|
||||
#### 8. LTX-2 stage equivalence is a verification gap, not proven drift
|
||||
|
||||
- **Priority:** P2.
|
||||
- **Effort:** Medium if parity checks are added; small if only manual audit.
|
||||
- **Owner:** FastVideo public.
|
||||
- **Public state:** model-specific LTX-2 stages are colocated under
|
||||
`fastvideo/pipelines/basic/ltx2/stages/`, consistent with the target
|
||||
layout in [design.md](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/design.md#L175-L205).
|
||||
- **Example public stage:** `ltx2_refine.py` explicitly says it is a
|
||||
public-side port of the internal stage and describes the three-stage SR
|
||||
flow:
|
||||
[ltx2_refine.py](file:///home/william5lin/FastVideo/fastvideo/pipelines/basic/ltx2/stages/ltx2_refine.py#L1-L22).
|
||||
- **Action:** verify behavior for the six internal `ltx2_*` stage files
|
||||
against public colocated stages. If a mismatch is found, file it as a
|
||||
real drift item with a failing parity test.
|
||||
|
||||
### Findings: deferred / accepted residual
|
||||
|
||||
These items should not block the public-dependency transition.
|
||||
|
||||
1. **StepVideo residual.**
|
||||
- Dreamverse's model registry is LTX-2/LTX-2.3 only:
|
||||
[Dreamverse config](file:///home/william5lin/Dreamverse/server/config.py#L28-L45).
|
||||
- Internal local tests even stub StepVideo modules to keep LTX registry
|
||||
tests focused:
|
||||
[test_ltx2_registry.py](file:///home/william5lin/FastVideo-internal/tests/local_tests/test_ltx2_registry.py#L38-L61).
|
||||
- Conclusion: accepted low-priority deferral unless Dreamverse adds a
|
||||
StepVideo model.
|
||||
|
||||
2. **Internal debug-only `FastVideoArgs` fields.**
|
||||
- Internal debug fields exist around `FastVideoArgs` and stage/model sums:
|
||||
[internal grep source](file:///home/william5lin/FastVideo-internal/fastvideo/fastvideo_args.py#L200-L203).
|
||||
- They are debug-only and not a public user-facing integration surface.
|
||||
- Conclusion: low-priority; do not add to public schema unless a debug
|
||||
workflow requires them.
|
||||
|
||||
3. **Private request aliases.**
|
||||
- Public request schema is nested and strict:
|
||||
[GenerationRequest](file:///home/william5lin/FastVideo/fastvideo/api/schema.py#L193-L204).
|
||||
- Legacy OpenAI flat fields are compatibility input, not the canonical
|
||||
public API:
|
||||
[internal protocol](file:///home/william5lin/FastVideo-internal/fastvideo/entrypoints/openai/protocol.py#L64-L82).
|
||||
- Conclusion: no action beyond current compat tests.
|
||||
|
||||
4. **`experimental["pipeline_config"]` escape hatch.**
|
||||
- Dreamverse currently uses an explicit in-memory quant config because
|
||||
typed `transformer_quant: "NVFP4"` does not expose `layer_profile`:
|
||||
[quantization](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/quantization.md#L86-L97).
|
||||
- Open thread #4 tracks `layer_profile`:
|
||||
[open threads](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/open-threads.md#L244-L260).
|
||||
- Conclusion: defer broader typed carrier design; add `layer_profile`
|
||||
first if Dreamverse needs base/refine profile selection.
|
||||
|
||||
5. **Router sticky routing and active-active semantics.**
|
||||
- Public router intentionally ships active-passive first and defers
|
||||
sticky/weighted routing:
|
||||
[D-15](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/decisions-log.md#L118-L155).
|
||||
- Follow-ups are tracked:
|
||||
[D-15 action items](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/decisions-log.md#L175-L201).
|
||||
- Conclusion: not drift; defer until load-balancing needs are real.
|
||||
|
||||
6. **AbsMaxFP8 failure.**
|
||||
- Pre-existing and not introduced by NVFP4:
|
||||
[state](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/state.md#L154-L159),
|
||||
[quantization](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/quantization.md#L202-L214).
|
||||
- Conclusion: fix separately; not a Dreamverse public-dependency blocker.
|
||||
|
||||
### Drift summary table
|
||||
|
||||
| # | Item | Priority | Effort | Status | Tracked where | Next action |
|
||||
|---:|---|---|---|---|---|---|
|
||||
| 1 | Dreamverse README still names `../FastVideo-internal` | P0 | S | Real drift | [README lines](file:///home/william5lin/Dreamverse/README.md#L76-L109) | Update docs to public FastVideo / PyPI path. |
|
||||
| 2 | Dreamverse private bootstrap clones internal repo | P0 | S | Real drift | [bootstrap script](file:///home/william5lin/Dreamverse/.agents/skills/bootstrap-fastvideo-private-fork/scripts/bootstrap_fastvideo_private.sh#L7-L11) | Replace or delete private bootstrap. |
|
||||
| 3 | Dreamverse `prompt_enhancer.py` fork | P1 | M | Real drift | [DR-1](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/open-threads.md#L174-L206) | Build compat shim over public enhancer. |
|
||||
| 4 | `cerebras_ifm` provider path | P1/P2 | S-M | Real drift / decision | [DR-2](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/open-threads.md#L211-L229) | Choose public provider vs Dreamverse custom provider. |
|
||||
| 5 | Health route mismatch | P1 | M-L | Real drift | [open item #1](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/open-threads.md#L69-L99) | Add `/healthz`, `/readyz`, `/status` to public build_app. |
|
||||
| 6 | Missing public streaming demo config | P2 | S | Real drift | [Dreamverse config](file:///home/william5lin/Dreamverse/serve_configs/streaming_demo.yaml#L1-L12) | Add `examples/serving/streaming_demo.yaml`. |
|
||||
| 7 | Standalone upsampler CLI | P2/P3 | S-M | Real drift if standalone CLI is desired | [internal CLI](file:///home/william5lin/FastVideo-internal/fastvideo/entrypoints/cli/upsample.py#L15-L35) | Decide CLI promise; port or defer. |
|
||||
| 8 | Layerwise offload utility | P3 | M | Optional internal-only residual (no Dreamverse deployment requires it today) | [internal manager](file:///home/william5lin/FastVideo-internal/fastvideo/models/layerwise_offload.py#L15-L20) | Defer until memory-tight deployment needs it. |
|
||||
| 9 | LTX-2 stage equivalence | P2 | S-M | Verification gap | [public refine stage](file:///home/william5lin/FastVideo/fastvideo/pipelines/basic/ltx2/stages/ltx2_refine.py#L1-L22) | Add targeted parity audit/test if needed. |
|
||||
| 10 | StepVideo | P3 | M | Accepted residual | [Dreamverse model registry](file:///home/william5lin/Dreamverse/server/config.py#L28-L45) | No action unless Dreamverse adds StepVideo. |
|
||||
| 11 | Debug-only fields | P3 | S | Accepted residual | [internal args](file:///home/william5lin/FastVideo-internal/fastvideo/fastvideo_args.py#L200-L203) | Do not publicize unless needed. |
|
||||
| 12 | `layer_profile` typed quant knob | P2 | M | Tracked gap | [open item #4](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/open-threads.md#L244-L260) | Add typed layer profile if Dreamverse drops escape hatch. |
|
||||
| 13 | `ltx2_image_crf` per-segment field flow (D-8) | P1 | S | Open verification gap — Dreamverse still passes `ltx2_image_crf=0.0` per [Dreamverse video_generation.py](file:///home/william5lin/Dreamverse/server/video_generation.py#L420-L435); needs trace-through to confirm it lands on `request.stage_overrides.refine.image_crf` rather than being silently dropped | [D-8 in open-threads](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/open-threads.md#L52-L68) | 10-min trace + add a Dreamverse-shape contract test pinning the field. |
|
||||
| 14 | `video_position_offset_sec` semantics (VPO) | P1 | S | Open decision — persistent-vs-per-segment ambiguity unresolved | [VPO in open-threads](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/open-threads.md#L120-L144) | Confirm semantics with audio team; document + add test. Decision deadline was "before PR 7.6 emits state" — that PR (7.6 / #1257) is now MERGED, so the decision is overdue. |
|
||||
| 15 | `GpuPool` ABC docstring missing experimental caveat (D-12-A) | P3 | trivial | Tracked gap — `GpuPool` ABC at [gpu_pool.py:74-83](file:///home/william5lin/FastVideo/fastvideo/entrypoints/streaming/gpu_pool.py#L74-L83) lacks the "API may change post-PR-7.10; experimental / server-internal" caveat | [D-12-A](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/open-threads.md#L301-L313) | Edit docstring; trivial. |
|
||||
| 16 | `GpuPool.run_async()` migration (D-12-B) | P2 | M | Tracked gap — `GpuPool.run() -> Any` should become `run_async() -> AsyncIterator[VideoEvent]` per D-12 / D-12-B | [D-12-B](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/open-threads.md#L317-L327) | Land alongside #1288 merge or in immediate follow-up. |
|
||||
| 17 | `SessionStore` / `BlobStore` lifecycle policy (SBS) | P2 | M | Tracked gap — in-memory defaults have no eviction/TTL/blob-cleanup policy | [SBS](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/open-threads.md#L278-L294) | Streaming server design pass needed before high-traffic deployment. |
|
||||
| 13 | Router sticky / active-active | P3 | M | Deferred | [D-15](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/decisions-log.md#L175-L201) | Defer until reconnect/load evidence. |
|
||||
| 14 | AbsMaxFP8 test failure | P2 | S | Separate tech debt | [state](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/state.md#L154-L159) | Fix outside Dreamverse migration. |
|
||||
| 15 | Dynamo backend package | P1 | External | Not FastVideo drift | [Dynamo contract](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/cross-repo-surfaces.md#L188-L210) | Reopen Dynamo-side PR after public API lands. |
|
||||
|
||||
---
|
||||
|
||||
## Part 2 — Integration path tradeoffs
|
||||
|
||||
### The four options
|
||||
|
||||
#### Option A — Status quo: Dreamverse stays separate and depends on `fastvideo`
|
||||
|
||||
**Shape**
|
||||
|
||||
- FastVideo remains the Python library and reusable backend runtime.
|
||||
- Dreamverse remains the product repo with FastAPI product glue and Next.js
|
||||
frontend.
|
||||
- Dreamverse `server` extra depends on `fastvideo>=0.1.7`:
|
||||
[pyproject](file:///home/william5lin/Dreamverse/pyproject.toml#L17-L22).
|
||||
- Local development can keep using editable `../FastVideo` until PyPI
|
||||
publication catches up:
|
||||
[uv.lock](file:///home/william5lin/Dreamverse/uv.lock#L716-L722).
|
||||
|
||||
**What it solves**
|
||||
|
||||
- Directly satisfies "Dreamverse depends on public FastVideo".
|
||||
- Keeps frontend release cadence independent.
|
||||
- Keeps product-specific prompts, routes, and UI in the product repo.
|
||||
- Minimizes FastVideo packaging and CI growth.
|
||||
|
||||
**What it does not solve by itself**
|
||||
|
||||
- Does not remove Dreamverse prompt-enhancer fork unless DR-1 is executed.
|
||||
- Does not give Dreamverse FE compatibility with public `build_app` until
|
||||
health routes migrate.
|
||||
- Does not make Dreamverse server itself reusable as a public entrypoint.
|
||||
|
||||
**Best fit**
|
||||
|
||||
- Default for the next release if the goal is to stop using
|
||||
FastVideo-internal quickly and safely.
|
||||
|
||||
#### Option B — Dreamverse as a subfolder under FastVideo
|
||||
|
||||
**Shape**
|
||||
|
||||
- One repository: FastVideo contains `dreamverse/server/` and
|
||||
`dreamverse/apps/web/`.
|
||||
- Dreamverse can remain a separate package in the same repo, or FastVideo's
|
||||
build can ignore Dreamverse by default.
|
||||
- CI must understand Python library tests plus Next.js install/build/test.
|
||||
|
||||
**What it solves**
|
||||
|
||||
- Eliminates sibling-checkout drift.
|
||||
- Makes cross-repo integration changes atomic.
|
||||
- Easier for a single reviewer to see library and product changes together.
|
||||
|
||||
**Costs**
|
||||
|
||||
- Adds frontend dependency management to a Python ML library repo.
|
||||
- Couples clone size, CI setup, issue tracking, and review load.
|
||||
- Forces maintainers to decide whether product assets are included in source
|
||||
distributions, wheels, docs, and release notes.
|
||||
|
||||
**Best fit**
|
||||
|
||||
- Only if Dreamverse becomes the primary FastVideo product surface and the
|
||||
team accepts a product monorepo.
|
||||
|
||||
#### Option C — Full merge into `fastvideo.entrypoints.dreamverse.*`
|
||||
|
||||
**Shape**
|
||||
|
||||
- Dreamverse backend becomes FastVideo code.
|
||||
- Public import becomes something like
|
||||
`from fastvideo.entrypoints.dreamverse import build_app`.
|
||||
- CLI becomes `fastvideo dreamverse-serve --config dreamverse.yaml`.
|
||||
- Frontend either ships as static assets in the package or as a frontend
|
||||
extra.
|
||||
|
||||
**What it solves**
|
||||
|
||||
- One namespace and one release train for library plus product backend.
|
||||
- No dependency boundary between Dreamverse server and FastVideo internals.
|
||||
- Product route contract can be tested entirely inside FastVideo CI.
|
||||
|
||||
**Costs**
|
||||
|
||||
- Maximally expands FastVideo's public/security surface.
|
||||
- Locks product experiments to FastVideo release cadence.
|
||||
- Makes private prompt/provider/product assumptions look like framework API.
|
||||
- Has weak precedent for a Python ML library plus Next.js product being merged
|
||||
into the library namespace.
|
||||
|
||||
**Best fit**
|
||||
|
||||
- Only if Dreamverse is no longer a separate product and becomes the
|
||||
canonical FastVideo UI/serving mode.
|
||||
|
||||
#### Option D — Hybrid: backend merges, frontend stays separate
|
||||
|
||||
**Shape**
|
||||
|
||||
- Reusable backend components merge into public FastVideo.
|
||||
- Frontend stays in a separate Dreamverse UI repo or Dreamverse product repo.
|
||||
- The backend should be generic where possible: `fastvideo.entrypoints.streaming`,
|
||||
not product-only names, unless product-only routes are intentionally
|
||||
accepted as public API.
|
||||
- This matches the current trajectory: streaming server, GPU pool, prompt
|
||||
enhancer, safety/rewrite/session logging, router, NVFP4, and
|
||||
`generate_async` are public-side work already tracked in the PR roadmap:
|
||||
[pr-roadmap](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/pr-roadmap.md#L19-L42).
|
||||
|
||||
**What it solves**
|
||||
|
||||
- Removes FastVideo-internal dependency for reusable backend pieces.
|
||||
- Keeps product frontend cadence independent.
|
||||
- Gives non-Dreamverse users a streaming backend and typed API without
|
||||
carrying the Dreamverse app.
|
||||
- Gives Dynamo a stable library API while leaving Dynamo package code in
|
||||
Dynamo:
|
||||
[Dynamo contract](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/cross-repo-surfaces.md#L188-L210).
|
||||
|
||||
**Costs**
|
||||
|
||||
- Requires careful boundary discipline: generic streaming/server code in
|
||||
FastVideo; product routes/prompts/presets in Dreamverse.
|
||||
- Requires contract tests to prevent drift.
|
||||
- Some Dreamverse compatibility routes may become public and need support.
|
||||
|
||||
**Best fit**
|
||||
|
||||
- Best long-term target if the team wants FastVideo to own serving/runtime
|
||||
infrastructure while keeping Dreamverse as a separately evolving product.
|
||||
|
||||
### Comparison matrix
|
||||
|
||||
| Criterion | A. Separate dep | B. Subfolder monorepo | C. Full namespace merge | D. Hybrid backend merge |
|
||||
|---|---|---|---|---|
|
||||
| Alignment with stated goal | High: Dreamverse depends on public package | Medium: no external dep, but product becomes repo-local | Medium: dependency disappears by absorption | High: reusable backend in public, product separate |
|
||||
| Time to remove `FastVideo-internal` | Fastest | Medium | Slowest | Medium-fast |
|
||||
| Build complexity | Low | High: Python + Next.js in one repo | High: Python package plus static/frontend extras | Medium: Python backend only in FastVideo |
|
||||
| Release cadence | Independent | Coupled clone; releases can still be separate but more friction | Fully coupled | Backend coupled to FastVideo, frontend independent |
|
||||
| Security surface in FastVideo | Low | Medium/High | Highest | Medium |
|
||||
| Contributor friction | Low for both repos | Higher for library contributors | Highest; product assumptions in library | Medium; clear backend boundary needed |
|
||||
| Dependency management | Normal package pin | Workspace/monorepo tooling needed | FastVideo extras/static asset decisions needed | FastVideo extras for backend; FE out-of-tree |
|
||||
| CI cost | Low/medium | High | High | Medium |
|
||||
| Contract-test value | High; cross-repo contract tests are essential | Medium; same repo but still useful | Medium; less boundary pressure | High; generic backend vs product boundary |
|
||||
| Precedent strength | Strong: library/server plus external UI patterns exist | Mixed | Weak for Python ML library + Next.js inside namespace | Strongest match: in-tree server/backend, external UI |
|
||||
| Packaging risk | Low | Medium/high | High | Medium |
|
||||
| Future Dynamo fit | Strong | Strong if API remains clean | Risky if product API bleeds in | Strong |
|
||||
| Frontend iteration speed | Highest | Lower | Lowest | Highest |
|
||||
| Risk of product-specific API leakage | Low | Medium | High | Medium; controllable with naming discipline |
|
||||
| Reversibility | High | Medium | Low | Medium/high |
|
||||
|
||||
### OSS precedents (with citations)
|
||||
|
||||
| Pattern | Project | What it supports | Citation |
|
||||
|---|---|---|---|
|
||||
| Library plus in-tree server | vLLM | A Python ML library can ship an in-tree OpenAI-compatible server while clients remain external. | https://github.com/vllm-project/vllm/blob/bcf5cac9fb956788f649d1f5297b74c886a9d6d3/README.md#L64-L74 |
|
||||
| Service packaging | BentoML | Packaging model + service + dependencies is supported, but CWD packaging creates discipline needs. | https://github.com/bentoml/BentoML/blob/32230a5276a8da8b23c4a06a9ec6272c1993451a/docs/source/build-with-bentoml/asgi.rst#L5-L18 |
|
||||
| YAML-driven production serving | Ray Serve | Production updates should avoid in-place mutation; use new deployment/traffic switch. | https://docs.ray.io/en/latest/serve/advanced-guides/inplace-updates.html |
|
||||
| Library/server plus external UI | TGI + ChatUI | Server can live with backend project while UI is separate. | https://github.com/huggingface/text-generation-inference/blob/b4adbf2f6e2e721280bd0ea5f91d70f7d033f5ed/docs/source/basic_tutorials/consuming_tgi.md#L182-L186 |
|
||||
| Lean library plus examples elsewhere | Transformers.js | Library stays lean; demos/examples can live outside core. | https://github.com/huggingface/transformers.js/blob/f7487c737aa8cafbc106c9adf69dc9578c8f3fe0/README.md#L26-L34 |
|
||||
| Product monorepo that later split frontend | ComfyUI | Product UI/server monorepo can hit release-cadence mismatch and split FE later. | https://github.com/comfyanonymous/ComfyUI/blob/fed8d5efa6b70d5b24c4c33cb643bfccc39d45b5/README.md#L131-L149 and https://github.com/Comfy-Org/ComfyUI_frontend/blob/60f789d58070a9d1d789b260f83c36d7293a39f0/README.md#L31-L60 |
|
||||
| Tightly coupled UI/server product | AUTOMATIC1111 SD WebUI | Product repos can couple UI/server tightly, but security surface becomes product-sized. | https://github.com/AUTOMATIC1111/stable-diffusion-webui/blob/82a973c04367123ae98bd9abdf80d9eda9b910e2/webui.py#L48-L104 |
|
||||
|
||||
#### Precedent synthesis
|
||||
|
||||
- Strong precedents exist for a Python ML library shipping a server entrypoint.
|
||||
- Strong precedents exist for keeping frontend/product UI out of the backend
|
||||
library repo.
|
||||
- The cited set does not contain a clean precedent for merging a Next.js
|
||||
product into a Python ML library namespace.
|
||||
- The most applicable pattern is **backend/server in the ML project,
|
||||
product UI outside**.
|
||||
|
||||
### Recommendation
|
||||
|
||||
#### Recommend Option D, constrained: backend merges as generic FastVideo streaming; frontend stays separate
|
||||
|
||||
Recommendation: follow **Option D** as the long-term architecture, but keep
|
||||
the backend merge generic. In practice, this means continuing the current
|
||||
public FastVideo path:
|
||||
|
||||
- `fastvideo.entrypoints.streaming.*` owns reusable streaming runtime.
|
||||
- `fastvideo.entrypoints.streaming.gpu_pool` owns generic GPU worker pools.
|
||||
- `fastvideo.entrypoints.streaming.prompt.*` owns provider-agnostic prompt
|
||||
operations.
|
||||
- `fastvideo.entrypoints.streaming.router.*` owns FastVideo-aware routing.
|
||||
- `fastvideo.api` owns typed construction, requests, results, events, and
|
||||
continuation state.
|
||||
- Dreamverse keeps product-only FE, curated presets, prompt UX, product
|
||||
routes, and launch scripts.
|
||||
|
||||
This is effectively the path already underway in PRs #1257, #1258, #1284,
|
||||
#1286, and #1288:
|
||||
[PR roadmap](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/pr-roadmap.md#L21-L42).
|
||||
|
||||
#### Why not Option A as the final answer?
|
||||
|
||||
Option A is the fastest near-term release posture and should be used as the
|
||||
immediate migration posture. However, plain status quo is not enough for
|
||||
the ultimate goal because reusable backend pieces still need to live in
|
||||
public FastVideo so Dreamverse can stop reaching into internal code. That
|
||||
work is already partly complete:
|
||||
|
||||
- GPU pool: [D-12](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/decisions-log.md#L47-L116).
|
||||
- Prompt enhancer: [PR roadmap 7.7](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/pr-roadmap.md#L32-L35).
|
||||
- Streaming auxiliaries: [PR roadmap 7.8](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/pr-roadmap.md#L35-L36).
|
||||
- Router: [D-15](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/decisions-log.md#L118-L155).
|
||||
- `generate_async`: [streaming-server unlock PR](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/streaming-server.md#L314-L345).
|
||||
|
||||
So the practical answer is:
|
||||
|
||||
- **Near term:** Option A operationally, after docs/scripts are fixed.
|
||||
- **Architecture target:** Option D, with generic backend ownership in
|
||||
FastVideo and product ownership in Dreamverse.
|
||||
|
||||
#### Why not Option B?
|
||||
|
||||
Option B makes cross-repo coordination easier but imports frontend build,
|
||||
package, and CI complexity into FastVideo. That is unnecessary while a
|
||||
normal package dependency plus contract tests can guard the integration.
|
||||
FastVideo's current repo structure is a Python package with examples and
|
||||
docs, not a product monorepo:
|
||||
[codebase map](file:///home/william5lin/FastVideo/.agents/memory/codebase-map/README.md#L5-L75).
|
||||
|
||||
#### Why not Option C?
|
||||
|
||||
Option C makes the product backend a public FastVideo namespace. That is
|
||||
only appropriate if the team wants to support Dreamverse as a first-class
|
||||
FastVideo product surface. Today the known public obligations are generic:
|
||||
typed requests, streaming server, GPU pool, prompt provider protocol,
|
||||
router, NVFP4, and Dynamo event APIs. Product-only Dreamverse behavior does
|
||||
not need to become framework API.
|
||||
|
||||
#### Conditions that would change the recommendation
|
||||
|
||||
Move from constrained D toward **C** only if all of these become true:
|
||||
|
||||
1. Dreamverse is declared the canonical FastVideo serving product.
|
||||
2. Product routes such as curated presets and prompt-system config are
|
||||
accepted as public FastVideo API.
|
||||
3. FastVideo maintainers accept the security and support surface.
|
||||
4. Release cadence for product UX and FastVideo core is intentionally
|
||||
coupled.
|
||||
5. Frontend packaging/static asset strategy is explicitly owned by
|
||||
FastVideo.
|
||||
|
||||
Move from constrained D back toward **A** if any of these become true:
|
||||
|
||||
1. Prompt enhancement, router, or GPU pool turn out to be Dreamverse-only.
|
||||
2. No second user appears for the streaming backend outside Dreamverse.
|
||||
3. FastVideo maintainers want to minimize serving surface and publish only
|
||||
Python library APIs.
|
||||
4. Dreamverse needs product changes faster than FastVideo can release.
|
||||
5. Security review rejects in-tree serving/router responsibilities.
|
||||
|
||||
### Migration sketch for the recommended path
|
||||
|
||||
#### Phase 0 — Land the public backend stack
|
||||
|
||||
- **Effort:** Large, already in flight.
|
||||
- **Owner:** FastVideo public.
|
||||
- **Files:** #1288 scope, especially `fastvideo/api/`,
|
||||
`fastvideo/entrypoints/video_generator.py`,
|
||||
`fastvideo/entrypoints/streaming/`, LTX-2 pipeline stages, NVFP4 files,
|
||||
and contract tests.
|
||||
- **Exit criteria:** #1288 merges; public `fastvideo.api.VideoEvent` and
|
||||
`VideoGenerator.generate_async` are available:
|
||||
[results.py](file:///home/william5lin/FastVideo/fastvideo/api/results.py#L109-L164),
|
||||
[video_generator.py](file:///home/william5lin/FastVideo/fastvideo/entrypoints/video_generator.py#L264-L332).
|
||||
|
||||
#### Phase 1 — Fix Dreamverse dependency docs and bootstrap
|
||||
|
||||
- **Effort:** Small.
|
||||
- **Owner:** Dreamverse.
|
||||
- **Files:**
|
||||
- [README.md](file:///home/william5lin/Dreamverse/README.md#L76-L109)
|
||||
- [bootstrap script](file:///home/william5lin/Dreamverse/.agents/skills/bootstrap-fastvideo-private-fork/scripts/bootstrap_fastvideo_private.sh#L7-L11)
|
||||
- [pyproject.toml](file:///home/william5lin/Dreamverse/pyproject.toml#L17-L22)
|
||||
- [uv.lock](file:///home/william5lin/Dreamverse/uv.lock#L716-L722)
|
||||
- **Exit criteria:** no user-facing docs or scripts mention
|
||||
`FastVideo-internal` as the expected dependency path.
|
||||
|
||||
#### Phase 2 — Add public health/readiness/status route compatibility
|
||||
|
||||
- **Effort:** Medium/Large.
|
||||
- **Owner:** FastVideo public.
|
||||
- **Files likely to touch:**
|
||||
- `fastvideo/entrypoints/streaming/server.py::build_app`
|
||||
- new `fastvideo/entrypoints/streaming/health.py`
|
||||
- tests under `fastvideo/tests/entrypoints/streaming/`
|
||||
- **Source route shapes:**
|
||||
[Dreamverse health routes](file:///home/william5lin/Dreamverse/server/routes/health.py#L34-L79).
|
||||
- **Tracking:** [open item #1](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/open-threads.md#L69-L99).
|
||||
- **Exit criteria:** Dreamverse FE can target public `build_app` for
|
||||
`/healthz`, `/readyz`, `/status`, and `/v1/stream`; product-only routes
|
||||
remain feature-detected.
|
||||
|
||||
#### Phase 3 — Replace Dreamverse prompt enhancer fork
|
||||
|
||||
- **Effort:** Medium.
|
||||
- **Owner:** Dreamverse.
|
||||
- **Files likely to touch:**
|
||||
- new `Dreamverse/server/prompting/_internal_compat.py`
|
||||
- `Dreamverse/server/runtime.py`
|
||||
- `Dreamverse/server/main.py`
|
||||
- `Dreamverse/server/prompt_enhancer.py`
|
||||
- **Public API:**
|
||||
[PromptEnhancer](file:///home/william5lin/FastVideo/fastvideo/entrypoints/streaming/prompt/enhancer.py#L66-L142),
|
||||
[LLMProvider](file:///home/william5lin/FastVideo/fastvideo/entrypoints/streaming/prompt/providers/base.py#L63-L75).
|
||||
- **Tracking:** [DR-1](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/open-threads.md#L174-L206).
|
||||
- **Exit criteria:** Dreamverse no longer carries a full local fork for the
|
||||
generic prompt operations public FastVideo already owns.
|
||||
|
||||
#### Phase 4 — Decide and implement `cerebras_ifm`
|
||||
|
||||
- **Effort:** Small decision plus small/medium implementation.
|
||||
- **Owner:** Team decision, then Dreamverse or FastVideo.
|
||||
- **Default recommendation:** Dreamverse-side custom provider.
|
||||
- **Public schema source:**
|
||||
[PromptEnhancerConfig](file:///home/william5lin/FastVideo/fastvideo/api/schema.py#L229-L235).
|
||||
- **Tracking:** [DR-2](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/open-threads.md#L211-L229).
|
||||
- **Exit criteria:** Dreamverse IFM provider works after prompt fork removal.
|
||||
|
||||
#### Phase 5 — Move streaming demo config into FastVideo examples
|
||||
|
||||
- **Effort:** Small.
|
||||
- **Owner:** FastVideo public.
|
||||
- **Source:**
|
||||
[Dreamverse streaming_demo.yaml](file:///home/william5lin/Dreamverse/serve_configs/streaming_demo.yaml#L1-L149).
|
||||
- **Target:** `examples/serving/streaming_demo.yaml`.
|
||||
- **Exit criteria:** users can reproduce the typed streaming path from the
|
||||
FastVideo repo without checking out Dreamverse.
|
||||
|
||||
#### Phase 6 — Remove `experimental["pipeline_config"]` where practical
|
||||
|
||||
- **Effort:** Medium for `layer_profile`; Large for a full typed
|
||||
`dit_config.quant_config` carrier.
|
||||
- **Owner:** FastVideo public, then Dreamverse cleanup.
|
||||
- **Tracking:**
|
||||
[open item #4](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/open-threads.md#L244-L260),
|
||||
[quantization follow-up](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/quantization.md#L175-L200).
|
||||
- **Exit criteria:** Dreamverse can express its quant layer profile through
|
||||
typed config instead of in-memory mutation.
|
||||
|
||||
#### Phase 7 — Decide standalone upsampler CLI
|
||||
|
||||
- **Effort:** Small/Medium.
|
||||
- **Owner:** FastVideo public.
|
||||
- **Input:** internal standalone utility
|
||||
[upsample.py](file:///home/william5lin/FastVideo-internal/fastvideo/entrypoints/upsample.py#L120-L180)
|
||||
and internal CLI
|
||||
[cli/upsample.py](file:///home/william5lin/FastVideo-internal/fastvideo/entrypoints/cli/upsample.py#L48-L130).
|
||||
- **Public alternative:** SR refine stage already covers in-pipeline latent
|
||||
upsampling:
|
||||
[ltx2_refine.py](file:///home/william5lin/FastVideo/fastvideo/pipelines/basic/ltx2/stages/ltx2_refine.py#L116-L180).
|
||||
- **Exit criteria:** explicit decision: port CLI, document refine-stage-only
|
||||
support, or defer.
|
||||
|
||||
#### Phase 8 — Validate LTX-2 stage parity and offload residuals
|
||||
|
||||
- **Effort:** Small/Medium for stage parity; Medium for layerwise offload.
|
||||
- **Owner:** FastVideo public.
|
||||
- **Stage source:**
|
||||
[public LTX-2 stages](file:///home/william5lin/FastVideo/fastvideo/pipelines/basic/ltx2/stages/).
|
||||
- **Offload source:**
|
||||
[internal layerwise offload](file:///home/william5lin/FastVideo-internal/fastvideo/models/layerwise_offload.py#L15-L20).
|
||||
- **Exit criteria:** no known behavior gap between internal and public LTX-2
|
||||
stages; offload is either deliberately deferred or ported with tests.
|
||||
|
||||
### Open questions
|
||||
|
||||
1. **Which provider path for `cerebras_ifm`?**
|
||||
- Public provider or Dreamverse-side custom provider?
|
||||
- Default recommendation: Dreamverse-side unless there is another user.
|
||||
- Source: [DR-2](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/open-threads.md#L211-L229).
|
||||
|
||||
2. **Should public FastVideo support standalone LTX-2 file upsampling?**
|
||||
- If yes, port internal CLI.
|
||||
- If no, document that SR support is pipeline-refine only.
|
||||
- Sources: [internal CLI](file:///home/william5lin/FastVideo-internal/fastvideo/entrypoints/cli/upsample.py#L15-L35),
|
||||
[public refine stage](file:///home/william5lin/FastVideo/fastvideo/pipelines/basic/ltx2/stages/ltx2_refine.py#L1-L22).
|
||||
|
||||
3. **Does Dreamverse need layerwise CPU offload?**
|
||||
- If memory-tight deployments require it, port as generic FastVideo.
|
||||
- Otherwise defer.
|
||||
- Source: [internal offload manager](file:///home/william5lin/FastVideo-internal/fastvideo/models/layerwise_offload.py#L15-L20).
|
||||
|
||||
4. **How much Dreamverse route surface should FastVideo own?**
|
||||
- Health/readiness/status should migrate because they are part of
|
||||
streaming-server compatibility.
|
||||
- Curated presets and prompt-system config should stay Dreamverse-side
|
||||
unless product policy changes.
|
||||
- Source: [route contract](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/streaming-server.md#L232-L259).
|
||||
|
||||
5. **Should `layer_profile` be the only near-term quant typed addition?**
|
||||
- Adding `layer_profile` is bounded.
|
||||
- A typed carrier for arbitrary mutated `PipelineConfig` is larger design
|
||||
work.
|
||||
- Source: [quantization follow-ups](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/quantization.md#L175-L200).
|
||||
|
||||
6. **When does Option D become Option C?**
|
||||
- Only if Dreamverse backend routes become public FastVideo product API.
|
||||
- Until then, keep generic streaming code in FastVideo and product code in
|
||||
Dreamverse.
|
||||
|
||||
---
|
||||
|
||||
## Part 3 — Action items
|
||||
|
||||
1. **P0 / S — Update Dreamverse README dependency notes.**
|
||||
- Replace `../FastVideo-internal` with public FastVideo instructions.
|
||||
- Preserve local editable `../FastVideo` dev flow where useful.
|
||||
- Source: [README stale lines](file:///home/william5lin/Dreamverse/README.md#L76-L109).
|
||||
|
||||
2. **P0 / S — Replace or remove private FastVideo bootstrap script.**
|
||||
- Current script clones `FastVideo-internal` and verifies imports from it.
|
||||
- Source: [script](file:///home/william5lin/Dreamverse/.agents/skills/bootstrap-fastvideo-private-fork/scripts/bootstrap_fastvideo_private.sh#L7-L11).
|
||||
|
||||
3. **P1 / M-L — Add `/healthz`, `/readyz`, and `/status` to public `build_app`.**
|
||||
- Keep `/curated-presets` and prompt-system config in Dreamverse.
|
||||
- Source: [open item #1](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/open-threads.md#L69-L99).
|
||||
|
||||
4. **P1 / M — Replace Dreamverse prompt-enhancer fork with compat shim.**
|
||||
- Wrap public `PromptEnhancer`.
|
||||
- Keep only Dreamverse-specific metadata and product fallback behavior.
|
||||
- Source: [DR-1](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/open-threads.md#L174-L206).
|
||||
|
||||
5. **P1 / S-M — Decide `cerebras_ifm` provider path.**
|
||||
- Default: Dreamverse custom provider via `register_provider`.
|
||||
- Source: [DR-2](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/open-threads.md#L211-L229).
|
||||
|
||||
6. **P2 / S — Add `examples/serving/streaming_demo.yaml` to FastVideo.**
|
||||
- Start from Dreamverse config and remove Dreamverse-private comments.
|
||||
- Source: [streaming_demo.yaml](file:///home/william5lin/Dreamverse/serve_configs/streaming_demo.yaml#L1-L149).
|
||||
|
||||
7. **P2 / M — Verify each public LTX-2 colocated stage against internal behavior.**
|
||||
- Start with refine, denoising, latent prep, image conditioning, text
|
||||
encoding, and audio decoding.
|
||||
- Source: [public refine stage](file:///home/william5lin/FastVideo/fastvideo/pipelines/basic/ltx2/stages/ltx2_refine.py#L1-L22).
|
||||
|
||||
8. **P2 / M — Add typed `transformer_quant_layer_profile` if Dreamverse needs it.**
|
||||
- Thread schema → compat → `FastVideoArgs._apply_transformer_quant`.
|
||||
- Source: [open item #4](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/open-threads.md#L244-L260).
|
||||
|
||||
9. **P2 / S-M — Decide standalone LTX-2 upsampler CLI support.**
|
||||
- Port internal CLI only if file-to-file upsampling is a public workflow.
|
||||
- Source: [internal upsample CLI](file:///home/william5lin/FastVideo-internal/fastvideo/entrypoints/cli/upsample.py#L15-L35).
|
||||
|
||||
10. **P2 / S — Fix pre-existing AbsMaxFP8 test failure separately.**
|
||||
- Do not block Dreamverse migration on it.
|
||||
- Source: [open item #2](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/open-threads.md#L100-L117).
|
||||
|
||||
11. **P2 / S-M — Document public streaming install extras and dependencies.**
|
||||
- Include router `websockets`, prompt enhancer provider SDKs, and optional
|
||||
safety classifier extras.
|
||||
- Source: [D-16 dependency note](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/decisions-log.md#L224-L242).
|
||||
|
||||
12. **P2 / S — Keep contract tests in the FastVideo CI path.**
|
||||
- Guard Dreamverse shape, Dynamo shape, and async events.
|
||||
- Sources: [Dreamverse test](file:///home/william5lin/FastVideo/fastvideo/tests/contract/test_dreamverse_shape.py#L1-L26),
|
||||
[Dynamo test](file:///home/william5lin/FastVideo/fastvideo/tests/contract/test_dynamo_shape.py#L1-L19),
|
||||
[async test](file:///home/william5lin/FastVideo/fastvideo/tests/contract/test_generate_async.py#L1-L7).
|
||||
|
||||
13. **P3 / M — Defer layerwise offload until a deployment needs it.**
|
||||
- Port only as generic FastVideo utility with tests.
|
||||
- Source: [internal offload](file:///home/william5lin/FastVideo-internal/fastvideo/models/layerwise_offload.py#L15-L20).
|
||||
|
||||
14. **P3 / M — Defer StepVideo public parity for this integration.**
|
||||
- Dreamverse model registry is LTX-2/LTX-2.3 only.
|
||||
- Source: [Dreamverse config](file:///home/william5lin/Dreamverse/server/config.py#L28-L45).
|
||||
|
||||
15. **P3 / S — Do not add debug-only fields to public schema by default.**
|
||||
- Keep them private unless there is a user-facing debugging workflow.
|
||||
- Source: [internal debug args](file:///home/william5lin/FastVideo-internal/fastvideo/fastvideo_args.py#L200-L203).
|
||||
|
||||
16. **P3 / M — Keep router active-active and sticky routing deferred.**
|
||||
- Add only when session-routing evidence justifies it.
|
||||
- Source: [D-15 action items](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/decisions-log.md#L175-L201).
|
||||
|
||||
17. **P3 / S — Preserve Dynamo as an external backend package.**
|
||||
- FastVideo should expose typed API; Dynamo code lives in Dynamo.
|
||||
- Source: [Dynamo contract](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/cross-repo-surfaces.md#L188-L210).
|
||||
|
||||
18. **P3 / S — After #1288 merges, update memory-dir state.**
|
||||
- Mark item D resolved and update branch tips.
|
||||
- Source: [runbook post-merge steps](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/runbook.md#L51-L70).
|
||||
|
||||
19. **P3 / S — Remove stale split-PR mental model from follow-up docs.**
|
||||
- #1288 is the current vehicle; split bookmarks are historical.
|
||||
- Source: [D-17 implications](file:///home/william5lin/FastVideo/.agents/memory/dreamverse-integration/decisions-log.md#L34-L45).
|
||||
|
||||
20. **P3 / S — Keep product-only Dreamverse frontend out of FastVideo unless explicitly re-scoped.**
|
||||
- This preserves release cadence and avoids packaging bloat.
|
||||
- Source: [Dreamverse baseline](file:///home/william5lin/Dreamverse/README.md#L5-L16).
|
||||
@@ -0,0 +1,553 @@
|
||||
# Open Threads — Active Follow-Ups
|
||||
|
||||
Live work items with priority, effort estimate, dependencies, and
|
||||
recommended next action.
|
||||
|
||||
For why each item is open see [decisions-log.md](decisions-log.md). For
|
||||
PR-level context see [pr-roadmap.md](pr-roadmap.md).
|
||||
|
||||
**Last updated:** 2026-05-12 (added DR-4 follow-up for PR #1330's skipped
|
||||
`App websocket integration` suite. Earlier: 2026-05-05 D-20 broken-pipe root cause + fix landed on
|
||||
`will/dreamverse-monorepo` @ `5eaf0a13`; added new thread D-20-CP for
|
||||
cherry-picking the public-API audio routing fix to `will/ltx2_sr_port`
|
||||
so it lands in PR #1288. Earlier: strategy reversal — PR #1287 CLOSED,
|
||||
replaced by mega-PR #1288 on `will/ltx2_sr_port` @ `b36bdbc9` covering
|
||||
the full 6-layer stack at once. See [decisions-log.md D-17](decisions-log.md#d-17).
|
||||
Item D resolution gate is now #1288 merge instead of #1287; same content,
|
||||
different vehicle.).
|
||||
|
||||
## Priority overview
|
||||
|
||||
| # | Pri | Item | Effort | Unblocks |
|
||||
|---|---|---|---|---|
|
||||
| **D-20-CP** | High | Cherry-pick `[fix] api: route LTX-2 audio kwargs through batch.extra; strict update` (`265ce1a6`) onto `will/ltx2_sr_port` so it lands in PR #1288 | 15 min | Surfaces the public-API audio-conditioning fix in the mega-PR rather than waiting for `will/dreamverse-monorepo` to fold in. Tests already pass (185/185 api). |
|
||||
| **~~D-22~~** | ~~Med~~ | ~~Per-chunk timing instrumentation in `stream_fmp4` and the controller's AV relay loop~~ | ~~M~~ | ✅ **Resolved 2026-05-06** in `bade2c0a`. `av_streaming.stream_fmp4` now records `av_wav_write_ms`, `av_ffmpeg_spawn_ms`, `av_first_chunk_ms`, `av_chunk_interval_ms_{min,median,p95,max}`, `av_chunk_publish_ms_{median,p95}`, `av_chunk_read_ms_{median,p95}` into the timings dict; `gpu_pool.handle_command` prints a one-line summary per segment so they show up directly in the deploy log. **Controller-side WS-send instrumentation is the remaining unaddressed slice** — see new D-22-CTL. |
|
||||
| **D-22-CTL** | Low | Controller-side AV relay timing in `apps/dreamverse/server/session/controller.py` (between media event arrival and ws_send_bytes). | S | The worker side is now fully attributed by D-22. The controller-side per-chunk WS-send overhead is the still-unmeasured slice of the 700ms gap between `worker_e2e` and `main_user_step`. Add `t_ws_send_ms` per chunk in the AV relay loop, surface as `controller_ws_send_ms_{median,p95}` in the segment summary log line. |
|
||||
| **D-23** | Med | On NVENC-capable hosts (RTX 50-series / T4 / A10 / H100 PCIe), benchmark `h264_nvenc` vs `libx264` and decide whether to default `--nvenc=true` for that SKU | S benchmark + S decision | Stutter mitigation depends on hardware; B200 needs different fix path (D-24). |
|
||||
| **D-24** | Low | For B200-class deploys without NVENC, prototype gen-N+1 // encode-N pipelining or FE buffer pre-fill | L | Architectural change required; benchmark suggests ~700ms can be hidden if we pipeline. Ship-blocker only when realtime stutter becomes user-visible on B200 deploys. |
|
||||
| **D-8** | High | Verify `ltx2_image_crf` post-`d80c2a8` | 10 min | Confirms typed stage-override path actually flows; closes a latent silent-drop bug |
|
||||
| **1** | High | Migrate `/healthz`+`/readyz`+`/status` into FastVideo `build_app` | M-L | Closes BE_FLAVOR=fastvideo FE-compatibility; closes streaming-upstream contract debt |
|
||||
| **2** | High | Fix pre-existing AbsMaxFP8 test failure | S | Self-contained quantization tech debt |
|
||||
| **VPO** | High | Decide `video_position_offset_sec` semantics (a vs b) | 30 min | Unblocks PR 7.6 state emission |
|
||||
| **D** | 🟢 in flight | Implement `generate_async` — content shipped in mega-PR **#1288** on `will/ltx2_sr_port` @ `b36bdbc9` (was #1287, CLOSED + re-routed per [D-17](decisions-log.md#d-17)) | L | Closes Q-5/Q-9/PR-7.5 TODOs simultaneously; enables Dynamo backend; unblocks audio re-encode; enables `GpuPool.run_async()` migration (D-12-B). Resolution gate: #1288 merge. |
|
||||
| **DR-1** | High | Dreamverse: create `prompting/_internal_compat.py` shim + replace local `prompt_enhancer.py` (1933 LOC) — **PR #1258 has merged (`f673423b`); now actionable** | M (~150-200 LOC shim, replace upstream wiring) | Lets Dreamverse stop carrying a 1933-LOC fork |
|
||||
| **DR-2** | Med | Decide `cerebras_ifm` provider path: (a) public Literal + `CerebrasIFMProvider` shipped, OR (b) Dreamverse-side custom provider via `enhancer.register_provider(...)` | S (decision) + S-M (impl) | Resolves the cerebras_ifm gap left by PR #1258. Same item as legacy #3 below; DR-2 is the Dreamverse-side framing. |
|
||||
| **DR-3** | Low | Replace Dreamverse `PromptEnhancer._run_blocking_request` manual thread/queue polling with `asyncio.to_thread` after the PR #1327 prompt-enhancer compatibility surface is retired or isolated | S | Review comment #1327 (`prompt_enhancer.py`) is valid, but deferred to avoid patching the local fork in this PR. |
|
||||
| **DR-4** | Low | Investigate unskipping PR #1330's skipped public `App websocket integration` suite | S-M | Public PR #1330 has `describe.skip(...)` around 27 websocket tests while the internal equivalent suite is active with 25 tests. The 2 public-only tests cover backend unreachable / GPU workers not ready. Not blocking while skipped, but stale assertions may need safe refresh before unskip. |
|
||||
| **3** | Med | Add `cerebras_ifm` to `PromptEnhancerConfig.provider` Literal + provider | S-M | Public-side resolution if DR-2 picks (a) |
|
||||
| **4** | Med | Expose `layer_profile` on typed `engine.quantization` | M | Removes Dreamverse's `experimental["pipeline_config"]` dodge for stage profiles |
|
||||
| **5** | Med | Design typed `dit_config.quant_config` carrier | L design + L impl | Removes broader `experimental["pipeline_config"]` escape hatch |
|
||||
| **SBS** | Med | `SessionStore` / `BlobStore` lifecycle policy | M design | Needed in PR 7.5 design pass |
|
||||
| **D-12-A** | Med | Update `GpuPool` ABC docstring: mark "API may change post-PR-7.10; experimental / server-internal" | trivial | Prevents accidental promotion of streaming-internal API to framework-level |
|
||||
| **D-12-B** | Med | Replace `GpuPool.run() -> Any` with `run_async() -> AsyncIterator[VideoEvent]` in PR 7.10 cycle | M | Closes the streaming-server cancellation TODO; converges with `generate_async` |
|
||||
| **D-13-A** | Med | Document `fastvideo.entrypoints.streaming.prompt.*` in user-facing docs as "streaming-server scoped"; avoid framework-level framing | trivial (docs only) | Keeps future move to `fastvideo.prompt.*` cheap |
|
||||
| **D-13-B** | Low | Add optional `client_factory` parameter to `LLMProvider` for `httpx.AsyncClient` pooling | S | Only if metrics show connect/TLS overhead is meaningful |
|
||||
| **D-12-C** | Low | Avoid locking `PoolAssignment.gpu_id: int` as public; rename to `worker_id` (already exists) or add `device_ids: list[int]` for topology-aware pooling | S | Future multi-GPU-per-worker refactor stays cheap |
|
||||
| **6** | Low | Audio attention quantization profile + test update | S | Future audio quant exploration |
|
||||
| **7** | Low | Schema parity inventory cleanup (env-driven prompt fields) | S-M | Long-term consistency |
|
||||
| **8** | Low | Stale `apps/web/test-results/` dir cleanup | trivial | Cosmetic |
|
||||
| **11** | Low | Promote LTX-2 prompt orchestration (locked segments, segment_prompts JSON shape, rollout id/label) to `fastvideo.entrypoints.streaming.prompt.ltx2_orchestration` | M | Resolves Q-2 from decisions-log when a second LTX-2-style consumer appears |
|
||||
| **12** | Low | When streaming server starts using `PromptSafetyFilter`, ensure operator-visible logging on `SafetyDecision.UNAVAILABLE` results | trivial | Surfaces degraded-safety state to operators (per D-14 Watch-Out item) |
|
||||
| **13** | Low | When sticky session routing is needed, add `ReplicaRegistry.select(routing_key: str | None = None)` and document where `session_id` lives (WS URL/header preferred over first JSON frame) | M | Forward-compat from D-15 — keeps the door open without buffering/peeking |
|
||||
| **14** | Low | At higher load, add `_bridge_session()` max-size + timeout limits OR recommend Envoy/HAProxy in front | S-M | The libraries' basic backpressure suffices for MVP; document the limit per D-15 |
|
||||
| **15** | Low | If active-active multi-primary becomes a requirement, define behavior (round-robin within healthy primaries, weighted, sticky-by-key) | M | Currently `RouterConfig.__post_init__` rejects multi-primary; D-15 deferred until evidence |
|
||||
| **~~Source-doc disposition~~** | ~~Med~~ | ~~Disposition of 7 untracked source docs~~ | ~~trivial~~ | ✅ **Resolved 2026-05-03** — moved into [source-archive/](source-archive/) |
|
||||
| **~~9~~** | ~~Low~~ | ~~Commit-message cleanup: PR 8's 3 commits still have `[8/n] Improve API:` prefix~~ | ~~S~~ | ✅ **Resolved 2026-05-04** — bundled into the will/api_7.8 prep rebase. PR 8's 3 commits now read `[type] streaming: ...` |
|
||||
| **~~10~~** | ~~Low~~ | ~~Commit-message cleanup: PR 7.8/7.9 commits have `streaming: streaming X` duplication~~ | ~~S~~ | ✅ **Resolved 2026-05-04** — bundled into the will/api_7.8 prep rebase. 3 commits dedup'd. |
|
||||
|
||||
---
|
||||
|
||||
## High priority
|
||||
|
||||
### D-20-CP: Cherry-pick public-API audio routing fix to `will/ltx2_sr_port`
|
||||
|
||||
**Why:** Commit `265ce1a6` (`[fix] api: route LTX-2 audio kwargs through batch.extra; strict update`) lives only on `will/dreamverse-monorepo` today. It touches public FastVideo surface (`fastvideo/entrypoints/video_generator.py`, `fastvideo/api/sampling_param.py`, plus a new regression test). For PR #1288 to ship a coherent public API — including the strict `SamplingParam.update()` — this commit needs to also land on `will/ltx2_sr_port`.
|
||||
|
||||
**Action:**
|
||||
|
||||
1. `git checkout will/ltx2_sr_port`
|
||||
2. `git cherry-pick 265ce1a6` (clean; only touches `fastvideo/` paths that exist on both branches)
|
||||
3. `pre-commit run --files fastvideo/entrypoints/video_generator.py fastvideo/api/sampling_param.py fastvideo/tests/api/test_extra_overrides_routing.py`
|
||||
4. `pytest fastvideo/tests/api/ -q` (expect 185 passed)
|
||||
5. `git push origin will/ltx2_sr_port` (fast-forward, no force)
|
||||
6. `git checkout will/dreamverse-monorepo` (return to default working branch per runbook)
|
||||
|
||||
**Outcome:** PR #1288 picks up the fix automatically (its head IS `will/ltx2_sr_port`). The cherry-pick lives on both branches as separate SHAs; they'll dedupe naturally on any future rebase.
|
||||
|
||||
**Effort:** 15 minutes (cherry-pick + lint + test + push).
|
||||
|
||||
**Dependencies:** None. Tests already pass; no rebase conflicts expected.
|
||||
|
||||
**Files touched (same on both branches):**
|
||||
|
||||
- `fastvideo/entrypoints/video_generator.py`
|
||||
- `fastvideo/api/sampling_param.py`
|
||||
- `fastvideo/tests/api/test_extra_overrides_routing.py` (new file)
|
||||
|
||||
### D-8: Verify `ltx2_image_crf` typed flow post-`d80c2a8`
|
||||
|
||||
**Why:** Apr 26 dreamverse_review documented this field getting silently
|
||||
dropped by the public `SamplingParam`. May 2 `d80c2a8` (Dreamverse)
|
||||
refactored to typed `GeneratorConfig` + `preset_overrides`. Whether
|
||||
`image_crf` now flows through `request.stage_overrides.refine.image_crf`
|
||||
(per [design.md](design.md) mapping) or is still dropped is unverified.
|
||||
|
||||
**Action:**
|
||||
1. Read [`Dreamverse/server/video_generation.py`](file:///home/william5lin/Dreamverse/server/video_generation.py)
|
||||
post-`d80c2a8` for `image_crf` handling
|
||||
2. Trace through to FastVideo's `request.stage_overrides.refine.image_crf`
|
||||
3. Confirm runtime consumption in [`fastvideo/pipelines/basic/ltx2/`](file:///home/william5lin/FastVideo/fastvideo/pipelines/basic/ltx2/)
|
||||
|
||||
**Effort:** 10 min, no code changes.
|
||||
|
||||
**Outcome:** Either confirms working OR identifies bug → opens fix item.
|
||||
|
||||
### Item #1: Migrate `/healthz`+`/readyz`+`/status` into `build_app`
|
||||
|
||||
**Why:** Today
|
||||
[`fastvideo.entrypoints.streaming.server.build_app`](file:///home/william5lin/FastVideo/fastvideo/entrypoints/streaming/server.py)
|
||||
exposes only `/health` + `/v1/stream`. Dreamverse FE expects all of
|
||||
`/healthz`, `/readyz`, `/status`, `/curated-presets`,
|
||||
`/prompt-system-config`, devtools.
|
||||
|
||||
The streaming-server-upstream plan (line 84) explicitly lists
|
||||
`/healthz`+`/readyz`+`/status` as part of the contract that the upstream
|
||||
of `realtime/` → `streaming/` must preserve. They were deferred from
|
||||
PR 7.5's MVP. `/curated-presets` and `/prompt-system-config` are
|
||||
operator-side and stay in Dreamverse (FE feature-detects).
|
||||
|
||||
**Action:**
|
||||
1. Read PR 7.5 (#1251) `build_app` to scope what's there
|
||||
2. Read [`Dreamverse/server/routes/health.py`](file:///home/william5lin/Dreamverse/server/routes/health.py)
|
||||
for the route shapes Dreamverse already consumes
|
||||
3. Propose route migration as commit on top of `will/api_7.5` or as
|
||||
part of PR 7.10 cycle
|
||||
4. Land
|
||||
|
||||
**Effort:** Medium-Large (route shapes need preservation; tests).
|
||||
|
||||
**Dependencies:** None blocking; can land anytime.
|
||||
|
||||
**Files likely to touch:**
|
||||
- `fastvideo/entrypoints/streaming/server.py::build_app`
|
||||
- New `fastvideo/entrypoints/streaming/health.py`
|
||||
- Tests in `fastvideo/tests/entrypoints/streaming/`
|
||||
|
||||
### Item #2: AbsMaxFP8 pre-existing test failure
|
||||
|
||||
**Why:** [`fastvideo/tests/ops/quantization/test_absmax_fp8.py::test_create_weights_rejects_invalid_dtype`](file:///home/william5lin/FastVideo/fastvideo/tests/ops/quantization/test_absmax_fp8.py)
|
||||
fails with `AssertionError not raised`. Pre-existing on `main`; verified
|
||||
NOT introduced by NVFP4 work via `git stash`.
|
||||
|
||||
**Action:**
|
||||
1. `git log --oneline fastvideo/tests/ops/quantization/test_absmax_fp8.py`
|
||||
to find when it last passed
|
||||
2. Either:
|
||||
- Restore the assert in `AbsMaxFP8LinearMethod.create_weights` if
|
||||
intentional behavior was lost
|
||||
- Drop the test if assert is no longer correct
|
||||
3. Verify
|
||||
|
||||
**Effort:** Small.
|
||||
|
||||
**Dependencies:** None.
|
||||
|
||||
### Item VPO: `video_position_offset_sec` semantics
|
||||
|
||||
**Why:** Per
|
||||
[`fastvideo/pipelines/basic/ltx2/continuation.py`](file:///home/william5lin/FastVideo/fastvideo/pipelines/basic/ltx2/continuation.py),
|
||||
`LTX2ContinuationState.video_position_offset_sec` exists as a state
|
||||
field. Two valid interpretations:
|
||||
|
||||
- **(a) Persistent across segments** — accumulating time offset for long
|
||||
sessions; useful for time-coherent audio chaining.
|
||||
- **(b) Per-segment hint that rides on the carrier** — runtime
|
||||
overwrites every time; field is harmless redundancy.
|
||||
|
||||
Dreamverse computes `prefix_sec = float(audio_extra) / 24.0` per segment
|
||||
in `apply_audio` and currently does NOT persist it on
|
||||
`ContinuationState`. Field's docstring leans toward (b).
|
||||
|
||||
**Decision deadline:** before PR 7.6 starts emitting/consuming the
|
||||
field (PR 7.6 branch is ready, not yet PR'd).
|
||||
|
||||
**Action:**
|
||||
1. Confirm field's intended semantics with audio team
|
||||
2. If (a): document the accumulation rule explicitly + add tests
|
||||
3. If (b): leave docstring as-is + add test confirming overwrite
|
||||
|
||||
**Effort:** 30 min discussion + small implementation.
|
||||
|
||||
### Item D: Implement `generate_async` (PR 7.10)
|
||||
|
||||
**Why:** Highest leverage. Closes:
|
||||
|
||||
- D-5 / Q-5: audio re-encode for cross-segment continuity
|
||||
- Q-9: Dynamo progress passthrough (deferred)
|
||||
- PR 7.5's mid-segment cancellation TODO
|
||||
- Unblocks Dynamo native backend integration
|
||||
- **D-12-B**: enables `GpuPool.run() -> run_async() -> AsyncIterator[VideoEvent]` migration
|
||||
|
||||
**Action:** See [streaming-server.md](streaming-server.md) "PR 7.10 — the
|
||||
unlock PR" section for scoping.
|
||||
|
||||
**Effort:** Large.
|
||||
|
||||
**Dependencies:** Best after PR 7.6 lands (gpu_pool upstream).
|
||||
|
||||
**Files:**
|
||||
- `fastvideo/entrypoints/video_generator.py` — add `generate_async`,
|
||||
refactor `generate_video` as wrapper
|
||||
- `fastvideo/api/results.py` — add `VideoEvent`/`VideoProgressEvent`/
|
||||
`VideoPartialEvent`/`VideoFinalEvent`
|
||||
- `fastvideo/entrypoints/streaming/server.py` — consume `generate_async`,
|
||||
remove TODO markers
|
||||
- `fastvideo/entrypoints/streaming/gpu_pool.py` — add `run_async()`
|
||||
forwarding events from worker to caller
|
||||
- New `fastvideo/tests/entrypoints/test_generate_async.py`
|
||||
- New `fastvideo/tests/contract/test_dynamo_shape.py` (already in PR 8)
|
||||
|
||||
### Item DR-1: Dreamverse — replace local `prompt_enhancer.py` with public + compat shim
|
||||
|
||||
**Why:** Today Dreamverse carries `Dreamverse/server/prompt_enhancer.py`
|
||||
(1933 LOC) — a local copy/derivative of the FastVideo-internal version.
|
||||
After PR #1258 merges, Dreamverse should switch to the public
|
||||
`fastvideo.entrypoints.streaming.prompt.PromptEnhancer` and delete most
|
||||
of the local module.
|
||||
|
||||
**Migration shape:**
|
||||
|
||||
1. **Create** `Dreamverse/server/prompting/_internal_compat.py` (~150-200 LOC):
|
||||
- Wraps public `PromptEnhancer.enhance()` → returns `EnhanceResult` shape Dreamverse expects
|
||||
- Wraps public `PromptEnhancer.auto_extend()` — JSON-parses `LLMResponse.content` into `{"next_prompt": "..."}`
|
||||
- Wraps public `PromptEnhancer.rewrite()` — JSON-parses into `{"segment_prompts": [...]}` with lenient fallback for malformed JSON
|
||||
- Layers locked-segment + rollout_id + rollout_label metadata back on top
|
||||
2. **Update** `Dreamverse/server/runtime.py + main.py` — replace `from prompt_enhancer import PromptEnhancer` with `from prompting._internal_compat import PromptEnhancer`
|
||||
3. **Delete most of** `Dreamverse/server/prompt_enhancer.py` (1933 LOC). Keep only the bits that don't have a public equivalent:
|
||||
- Race-based parallel fallback (`_run_provider_race`) — Dreamverse-specific tail-latency optimization
|
||||
- `cerebras_ifm` provider — pending DR-2 decision
|
||||
- Multi-classifier prompt safety (NSFW + hate-speech chained) — public ships single classifier
|
||||
4. **Tests** — verify Dreamverse session controllers still see the expected response shapes through the shim
|
||||
|
||||
**Effort:** Medium (~150-200 LOC shim + replace upstream wiring + delete 1700+ LOC local module + test fixture updates).
|
||||
|
||||
**Dependencies:**
|
||||
- PR #1258 must merge first (publishes `fastvideo.entrypoints.streaming.prompt.*`)
|
||||
- DR-2 informs the cerebras_ifm path
|
||||
|
||||
**Files:**
|
||||
- New: `Dreamverse/server/prompting/_internal_compat.py`
|
||||
- Modified: `Dreamverse/server/runtime.py`, `Dreamverse/server/main.py`
|
||||
- Mostly deleted: `Dreamverse/server/prompt_enhancer.py`
|
||||
|
||||
---
|
||||
|
||||
## Medium priority
|
||||
|
||||
### Item DR-2: Decide `cerebras_ifm` provider path
|
||||
|
||||
**Why:** Public PR #1258's `PromptEnhancerConfig.provider` is
|
||||
`Literal["cerebras", "groq"]`. Internal supports `"cerebras_ifm"` (the
|
||||
Cerebras IFM API endpoint with different auth). Dreamverse needs
|
||||
`cerebras_ifm` working post-migration.
|
||||
|
||||
Two options:
|
||||
|
||||
| Option | Approach | Pros | Cons |
|
||||
|---|---|---|---|
|
||||
| **(a) Public** | Add `"cerebras_ifm"` to public Literal + ship `CerebrasIFMProvider` in `fastvideo/entrypoints/streaming/prompt/providers/cerebras_ifm.py` | Discoverable; users with IFM access can use typed config | Adds ~50 LOC + Literal extension to public surface |
|
||||
| **(b) Dreamverse-side** | Implement `CerebrasIFMProvider` Dreamverse-side as a custom `LLMProvider`, register via `enhancer.register_provider(CerebrasIFMProvider())` | Zero public surface change; private endpoint stays private | Slightly more boilerplate Dreamverse-side; not surfaced to non-Dreamverse users |
|
||||
|
||||
**Recommendation:** Option (b) is more contained. Option (a) is more
|
||||
discoverable. Default to (b) unless there's a third-party user who needs
|
||||
IFM access. The Dreamverse-side PR carrying DR-1 is the natural place to
|
||||
make this decision.
|
||||
|
||||
**Effort:** Small (decision) + Small-Medium (implementation).
|
||||
|
||||
**Dependencies:** DR-1 (compat shim creation).
|
||||
|
||||
### Item #3: `cerebras_ifm` provider in public Literal
|
||||
|
||||
**Why:** Same item as DR-2 from the public-side framing. If DR-2 picks
|
||||
option (a), this is the implementation. If DR-2 picks option (b), this
|
||||
item is closed without implementation.
|
||||
|
||||
**Action:** See DR-2.
|
||||
|
||||
**Effort:** S-M.
|
||||
|
||||
### Item #4: Expose `layer_profile` on typed `engine.quantization`
|
||||
|
||||
**Why:** Today `transformer_quant: "NVFP4"` always constructs
|
||||
`NVFP4Config()` with default `layer_profile="refine"`. Dreamverse
|
||||
dodges via `experimental["pipeline_config"]`.
|
||||
|
||||
**Action:**
|
||||
1. Add `transformer_quant_layer_profile: str | None = None` to
|
||||
`QuantizationConfig` in [`schema.py`](file:///home/william5lin/FastVideo/fastvideo/api/schema.py)
|
||||
2. Thread through [`compat.py`](file:///home/william5lin/FastVideo/fastvideo/api/compat.py)
|
||||
3. Update `_apply_transformer_quant` in
|
||||
[`fastvideo_args.py`](file:///home/william5lin/FastVideo/fastvideo/fastvideo_args.py)
|
||||
to pass profile
|
||||
4. Update Dreamverse to drop the `experimental["pipeline_config"]`
|
||||
dodge in favor of typed knob
|
||||
5. Tests in [`test_typed_quant_flow.py`](file:///home/william5lin/FastVideo/fastvideo/tests/api/test_typed_quant_flow.py)
|
||||
|
||||
**Effort:** Medium.
|
||||
|
||||
**Files:** schema.py, compat.py, fastvideo_args.py, test_typed_quant_flow.py,
|
||||
+ Dreamverse/server/video_generation.py.
|
||||
|
||||
### Item #5: Typed `dit_config.quant_config` carrier
|
||||
|
||||
**Why:** The `experimental["pipeline_config"]` escape hatch in
|
||||
Dreamverse should eventually become a typed field. Design TBD.
|
||||
|
||||
**Action:** Heaviest design work. Should consult Oracle.
|
||||
|
||||
**Effort:** Large design + Large implementation.
|
||||
|
||||
**Dependencies:** #4 should land first; this is the "final form" of #4.
|
||||
|
||||
### Item SBS: `SessionStore` / `BlobStore` lifecycle policy
|
||||
|
||||
**Why:** PR 7's in-memory implementations have no eviction, no TTL, no
|
||||
automatic blob cleanup on state replacement. Documented as per-deployment
|
||||
policy decision.
|
||||
|
||||
When PR 7.5/7.6 land the live consumer, who owns:
|
||||
|
||||
- bounded session capacity (LRU? TTL? hard max?)
|
||||
- blob `drop()` chained when state is replaced
|
||||
- session expiry on websocket disconnect
|
||||
|
||||
**Recommendation:** streaming server's session manager. Worth stating
|
||||
explicitly in PR 7.5's design.
|
||||
|
||||
**Effort:** Medium design + small implementation.
|
||||
|
||||
### Item D-12-A: Update `GpuPool` ABC docstring — mark experimental
|
||||
|
||||
**Why:** Per D-12 in [decisions-log.md](decisions-log.md), `GpuPool`
|
||||
should be documented as "API may change post-PR-7.10; experimental /
|
||||
server-internal" to prevent accidental promotion of streaming-internal
|
||||
API to framework-level. PR #1257 merged without this caveat.
|
||||
|
||||
**Action:** Edit
|
||||
[`fastvideo/entrypoints/streaming/gpu_pool.py`](file:///home/william5lin/FastVideo/fastvideo/entrypoints/streaming/gpu_pool.py)
|
||||
class docstring on `GpuPool` ABC. Add a note: "API may change post-PR-7.10
|
||||
when run_async() lands; treat as server-internal for now."
|
||||
|
||||
**Effort:** Trivial.
|
||||
|
||||
**Dependencies:** None.
|
||||
|
||||
### Item D-12-B: Replace `GpuPool.run() -> Any` with `run_async() -> AsyncIterator[VideoEvent]`
|
||||
|
||||
**Why:** Per D-12, this is the canonical evolution post-PR-7.10. Closes
|
||||
the streaming server's cancellation TODO and converges the streaming +
|
||||
OpenAI + Dynamo consumers on a single async API.
|
||||
|
||||
**Action:** As part of PR 7.10 cycle:
|
||||
1. Add `GpuPool.run_async(session_id, request) -> AsyncIterator[VideoEvent]`
|
||||
2. Worker forwards events through `result_queue` with type discriminator
|
||||
3. Streaming server replaces `await pool.run(...)` with `async for event in pool.run_async(...)`
|
||||
4. Sync `run()` becomes a thin compat wrapper that collects events and returns the final
|
||||
5. Cancellation propagates: client disconnect → `asyncio.CancelledError` → worker stops mid-step
|
||||
|
||||
**Effort:** Medium. Adds ~50-100 LOC + tests.
|
||||
|
||||
**Dependencies:** Item D (PR 7.10 — `generate_async` on `VideoGenerator`).
|
||||
|
||||
### Item D-13-A: Document `streaming/prompt/*` as streaming-scoped
|
||||
|
||||
**Why:** Per D-13 in [decisions-log.md](decisions-log.md), the prompt
|
||||
enhancer is currently scoped to streaming-server use even though the
|
||||
abstraction is general. Phrase user-facing docs as "streaming-server
|
||||
prompt enhancement" to keep future move to `fastvideo.prompt.*` cheap.
|
||||
|
||||
**Action:** When PR 12 (docs migration) is written, the prompt enhancer
|
||||
section should:
|
||||
- Be titled "Streaming Server Prompt Enhancement", not "Prompt API"
|
||||
- Note the 3 fixed operations (`enhance` / `auto_extend` / `rewrite`) are
|
||||
shaped by LTX-2 streaming session needs
|
||||
- Note that consumers wanting custom prompt operations can use
|
||||
`provider.complete()` directly with their own LLMRequest
|
||||
- Avoid `from fastvideo import LLMProvider` exports until a second
|
||||
consumer exists
|
||||
|
||||
**Effort:** Trivial (docs only).
|
||||
|
||||
**Dependencies:** PR 12 (docs migration).
|
||||
|
||||
---
|
||||
|
||||
## Low priority
|
||||
|
||||
### Item D-13-B: Optional `client_factory` parameter for `httpx.AsyncClient` pooling
|
||||
|
||||
**Why:** Today `_openai_compat.py` instantiates `httpx.AsyncClient` per
|
||||
call (no connection pooling). Reviewer flagged inefficient. Team chose
|
||||
simplicity for the expected scale (~6-10 enhancer calls per LTX-2
|
||||
session). If real-world metrics show connect/TLS overhead is meaningful,
|
||||
add an optional `client_factory: Callable[[], httpx.AsyncClient] | None`
|
||||
parameter to providers so they can share a pool.
|
||||
|
||||
**Action:** Only when metrics justify. Add `client_factory=None` parameter
|
||||
to `CerebrasProvider` / `GroqProvider` constructors and pass through to
|
||||
`complete_openai_compatible()`. Default to current per-call behavior.
|
||||
|
||||
**Effort:** Small.
|
||||
|
||||
**Dependencies:** None blocking; only act on real perf data.
|
||||
|
||||
### Item DR-4: Unskip PR #1330 `App websocket integration` suite safely
|
||||
|
||||
**Why:** Public PR #1330 currently wraps `App websocket integration` in
|
||||
`describe.skip(...)`, so the 27-test websocket integration suite does not
|
||||
run. The internal repo has the equivalent suite active via `describe(...)`
|
||||
with 25 tests. The two public-only cases cover backend unreachable and GPU
|
||||
workers not ready readiness/reachability behavior.
|
||||
|
||||
**Action:** Later, investigate whether the public suite can be unskipped and
|
||||
refresh any stale assertions without changing the intended websocket contract.
|
||||
Because the suite is skipped today, this is not blocking the current stack or
|
||||
current PR #1330 review.
|
||||
|
||||
**Effort:** Small-Medium.
|
||||
|
||||
**Dependencies:** Best handled when someone can run the frontend integration
|
||||
stack end-to-end and compare the public assertions against the internal active
|
||||
suite.
|
||||
|
||||
### Item D-12-C: Avoid locking `PoolAssignment.gpu_id: int` as public
|
||||
|
||||
**Why:** Today `PoolAssignment` exposes `gpu_id: int`, assuming
|
||||
one-GPU-per-worker. Future topology-aware pooling may need
|
||||
`device_ids: list[int]` (one worker = group of GPUs running internal
|
||||
`MultiprocExecutor`). Don't freeze the int field as public API.
|
||||
|
||||
**Action:**
|
||||
- Treat `gpu_id` as a current-impl detail; prefer `worker_id` (already
|
||||
exists, is stable identifier)
|
||||
- When a worker actually spans multiple GPUs, add
|
||||
`PoolAssignment.device_ids: list[int]` and let `gpu_id` be `device_ids[0]`
|
||||
for backward compat
|
||||
- Or rename to `gpu_id` → `device_id` with deprecation alias
|
||||
|
||||
**Effort:** Small (1 field rename + alias).
|
||||
|
||||
**Dependencies:** Driven by an actual future "one worker = many GPUs" use case. Don't act preemptively.
|
||||
|
||||
### Item #6: Audio attention quantization profile
|
||||
|
||||
**Why:** Today audio attn and FFN are bf16. If an audio-quant profile
|
||||
is added to `NVFP4Config.fp4_layers`, update
|
||||
[`test_basic_av_block_propagates_quant_config_to_all_children`](file:///home/william5lin/FastVideo/fastvideo/tests/ops/quantization/test_nvfp4_ltx2_wiring.py).
|
||||
|
||||
**Effort:** Small (one test + one config field).
|
||||
|
||||
### Item #7: Schema parity inventory cleanup
|
||||
|
||||
**Why:** A few internal-only fields are not exposed publicly:
|
||||
|
||||
- `PROMPT_HTTP_TIMEOUT_MS`
|
||||
- `PROMPT_INITIAL_STAGE_TIMEOUT_MS`
|
||||
- `PROMPT_TEMPERATURE`
|
||||
- `PROMPT_MAX_COMPLETION_TOKENS`
|
||||
- `PROMPT_AUTO_SLEEP_MS`
|
||||
- `PROMPT_AUTO_TIMEOUT_MS`
|
||||
- curated-presets file paths
|
||||
|
||||
These flow via env vars on `dreamverse-server` today. If
|
||||
`fastvideo serve --config` becomes the canonical entrypoint, they need
|
||||
typed homes.
|
||||
|
||||
**Effort:** Small-Medium.
|
||||
|
||||
### Item #8: Stale `apps/web/test-results/` directory
|
||||
|
||||
**Why:** Cosmetic. `.gitignore` entry hides it from `git status`, but
|
||||
the dir has stale `.last-run.json` (45 bytes) from a prior Playwright
|
||||
run.
|
||||
|
||||
**Action:** `rm -rf apps/web/test-results` whenever convenient.
|
||||
|
||||
**Effort:** Trivial.
|
||||
|
||||
### ~~Item #9~~ + ~~#10~~: Commit-message cleanups — ✅ Resolved 2026-05-04
|
||||
|
||||
Both items resolved during the `will/api_7.8` prep rebase. A targeted
|
||||
conditional script (`/tmp/opencode/cleanup_subjects_v2.sh` — only amends
|
||||
when text actually changes) ran across 33 commits, modified 6:
|
||||
|
||||
- **#9 fix**: extended the regex from `\[\d+\.\d+/n\]` to
|
||||
`\[\d+(\.\d+)?/n\]` so single-digit prefixes match. PR 8's 3 commits
|
||||
now read `[type] streaming: ...` instead of `[type] [8/n] Improve API: ...`.
|
||||
- **#10 fix**: added second substitution `streaming: streaming X` →
|
||||
`streaming: X`. PR 7.8 / 7.9 commits no longer have the duplication.
|
||||
|
||||
The conditional check skipped pre-commit-hook flakiness on no-op amends
|
||||
(unlike the earlier first attempt). All affected commits verified clean
|
||||
post-rebase.
|
||||
|
||||
### Item #11: Promote LTX-2 prompt orchestration to public (when 2nd consumer exists)
|
||||
|
||||
**Why:** Per Q-2 in [decisions-log.md](decisions-log.md) and D-13's
|
||||
"missing alternative", the LTX-2-specific orchestration (locked
|
||||
segments, segment_prompts JSON shape, rollout id/label, lenient JSON
|
||||
parsing) currently stays Dreamverse-side per DR-1. If a second
|
||||
LTX-2-style consumer appears (e.g. another video model with multi-segment
|
||||
continuation needing the same prompt orchestration), promote this layer
|
||||
to `fastvideo.entrypoints.streaming.prompt.ltx2_orchestration`.
|
||||
|
||||
**Action:** Wait for a second consumer to materialize. Until then, the
|
||||
orchestration stays in Dreamverse's `_internal_compat.py` shim (DR-1).
|
||||
|
||||
**Effort:** Medium when triggered.
|
||||
|
||||
**Dependencies:** A second consumer.
|
||||
|
||||
---
|
||||
|
||||
## Recommended pull order
|
||||
|
||||
If you have unbounded time and want to maximize forward progress:
|
||||
|
||||
1. **D-8 verify** (10 min) — eliminates uncertainty
|
||||
2. **D-12-A docstring** (trivial) — caveat the GpuPool API publicly
|
||||
3. **Item #2 AbsMaxFP8** (S) — clears tech debt
|
||||
4. **Item VPO video_position_offset_sec** (30 min) — unblocks PR 7.10 (since 7.6 has merged, this is now scoped to whatever consumer first reads the field)
|
||||
5. **DR-1 + DR-2 Dreamverse migration** (M) — **now unblocked since PR #1258 merged**; replaces 1700+ LOC of local fork
|
||||
6. **Item #4 layer_profile** (M) — closes Dreamverse quant escape hatch
|
||||
7. **Item #1 build_app routes** (M-L) — closes FE-compat
|
||||
8. **Item D generate_async** (L) — unlock PR; brings along D-12-B (run_async) + closes Q-5/Q-9/PR-7.5 TODOs
|
||||
9. **Item #5 typed quant_config carrier** (L+L) — final form
|
||||
10. **Items #6/#7/#8 + D-12-C/D-13-A/D-13-B + #11** — cleanup polish (#9, #10 resolved 2026-05-04)
|
||||
|
||||
If you have a specific user goal (e.g. "ship `BE_FLAVOR=fastvideo`
|
||||
flavor end-to-end"), that goal dictates the order — read this list as a
|
||||
menu, not a prescription.
|
||||
|
||||
---
|
||||
|
||||
## Verification gates per item
|
||||
|
||||
When implementing any item above, evidence required:
|
||||
|
||||
| Phase | Check |
|
||||
|---|---|
|
||||
| Build | `lsp_diagnostics` clean on changed files |
|
||||
| Test | new + relevant existing tests pass; output captured |
|
||||
| Manual QA | actually run the affected feature end-to-end (per AGENTS.md MANUAL_QA_MANDATE) |
|
||||
| Regression | full `fastvideo/tests/api/` + `contract/` + relevant SSIM (if NVFP4 touch) |
|
||||
|
||||
For NVFP4 touches: re-run `test_nvfp4_ltx2_wiring.py` +
|
||||
`test_typed_quant_flow.py` (CPU) + ideally a flashinfer-enabled path
|
||||
test (manual, not in CI).
|
||||
|
||||
For Dreamverse-side items (DR-1, DR-2): re-run
|
||||
`Dreamverse/apps/web/npx playwright test e2e/preset-prompt-generation.spec.ts`
|
||||
end-to-end against the live BE+FE — this is the contract test that
|
||||
exercises the prompt enhancer through a real session.
|
||||
@@ -0,0 +1,141 @@
|
||||
# PR Roadmap
|
||||
|
||||
Status of all 17 PRs in the FastVideo public API refactor + streaming
|
||||
server upstream + Dynamo backend contract + post-deprecation cleanup.
|
||||
|
||||
For design rationale see [design.md](design.md). For streaming-specific
|
||||
PRs (7.5-7.10) see [streaming-server.md](streaming-server.md). For NVFP4
|
||||
work that runs parallel to this sequence see [quantization.md](quantization.md).
|
||||
|
||||
**Last updated:** 2026-05-05 (strategy reversal — single mega-PR #1288 replaces planned splits 7.10/8/LTX-2/NVFP4/post-fixes/agents_cleanup; see [decisions-log.md D-17](decisions-log.md#d-17)).
|
||||
|
||||
## Status legend
|
||||
|
||||
- ✅ **Landed on `origin/main`**
|
||||
- 🟢 **Open / in flight** — branch exists, may have open PR
|
||||
- 🟡 **Planned** — designed, not started
|
||||
- 🔵 **Future** — deferred to post-PR-13 cleanup
|
||||
|
||||
## Landed PRs (0 → 7.7)
|
||||
|
||||
| # | PR | Status | Merge commit | Scope |
|
||||
|---|---|---|---|---|
|
||||
| 0 | #1218 [1/n] | ✅ | merged | Parity inventory + typed inference schema |
|
||||
| 1 | #1218 [1/n] | ✅ | merged | Strict parser/validation/overrides + API tests |
|
||||
| 2 | #1220 [2/n] | ✅ | merged | Typed `VideoGenerator` constructors + request path + compat |
|
||||
| 3 | #1226 [3/n] | ✅ | merged | CLI/YAML-first typed config loading for `generate` and `serve` |
|
||||
| 4 | #1234 [4/n] | ✅ | merged | Preset registry + presets for all 13 model families; `SamplingParam` moved to `fastvideo/api/`; `configs/sample/` deleted entirely |
|
||||
| 5 | #1237 [5/n] | ✅ | merged | `ServeConfig.default_request` wired into stateless OpenAI server |
|
||||
| 5.5 | (`5d1d71fc`) | ✅ | merged | Streaming server package skeleton, typed `StreamingConfig`/`GpuPoolConfig`/`PromptEnhancerConfig`/`PromptSafetyConfig`/`WarmupConfig`, `streaming-serve` CLI stub |
|
||||
| 6 | #1239 [6/n] | ✅ | merged | LTX2 public preset + asset wiring + `gpu_pool.py` typed-kwarg translation |
|
||||
| 7 | #1250 [7/n] | ✅ | merged | Typed LTX2 continuation state + streaming session store + blob store |
|
||||
| **7.5** | **#1251** | ✅ | `95fd29e0` (merged 2026-04-26) | Streaming server skeleton (WebSocket + fMP4 + single generator). 8 commits. Deferred TODOs (per-step progress, mid-segment cancellation) carried forward to PR 7.10. |
|
||||
| **7.6** | **#1257** | ✅ | `eb0a4152` (merged 2026-05-04) | GPU pool upstream + worker subprocess + two-segment warmup. 7 commits squashed. APPROVED by Eigensystem. See [decisions-log.md D-12](decisions-log.md#d-12) for the architectural review. |
|
||||
| **7.7** | **#1258** | ✅ | `f673423b` (merged 2026-05-04) | Prompt enhancer with `LLMProvider` abstraction. Built-in providers: cerebras, groq. 3 commits squashed. **Public Literal does NOT include `cerebras_ifm`** — open-threads.md item DR-2 covers the gap. See [decisions-log.md D-13](decisions-log.md#d-13) for the architectural review. |
|
||||
| **7.8** | **#1284** | ✅ | `eb3a3942` (merged 2026-05-04) | Streaming auxiliaries — `prompt/safety.py` (optional fasttext, lazy import), `prompt/rewrite.py`, `session_logger.py` (thread-safe JSONL), `mock_server.py` (build_mock_app + MockGenerator for FE dev). 730 LOC, 2 commits. See [decisions-log.md D-14](decisions-log.md#d-14). |
|
||||
| **7.9** | **#1286** | ✅ | `2aaeee2a` (merged 2026-05-05) | Streaming router (multi-replica load balancer + WS proxy + `fastvideo router-serve` CLI). Squashed `cd76cf51 + 1ac1e732 + b0b7f59c + a152cb77` (router-polish second-pass; cherry-pick of `40e265b8` from `will/ltx2_sr_port`). See [decisions-log.md D-15](decisions-log.md#d-15) (structural review) + [D-16](decisions-log.md#d-16) (second-pass polish). |
|
||||
|
||||
## In flight (mega-PR #1288)
|
||||
|
||||
| # | PR | Status | Branch | Scope |
|
||||
|---|---|---|---|---|
|
||||
| **mega** | **#1288** | 🟢 OPEN, MERGEABLE | `will/ltx2_sr_port` (head `b36bdbc9`) | **Single consolidated landing of the full `will/ltx2_sr_port` chain.** Was originally planned as 6 stacked PRs (slices 1-3 / 4-6 / 7-15 / 16-21 / 22-23 / 24-34). Now landing as one PR — see [decisions-log.md D-17](decisions-log.md#d-17) for the strategy decision. **Contents** (commit-ordered): (1) streaming `generate_async` + `VideoEvent` + Dynamo backend contract (3 commits, was PR 7.10/#1287 closed); (2) server contract docs + Dreamverse/Dynamo shape tests (3 commits, was PR 8); (3) LTX-2 SR runtime port + i2v conditioning + alignment harness (9 commits); (4) NVFP4 wire-up + per-component compile + typed `transformer_quant` flow (6 commits); (5) LTX-2 post-handoff parity fixes — Gemma `to()`, list-of-generators (2 commits); (6) `.agents/memory/dreamverse-integration/` knowledge base + agents Phase 1 cleanup (11 commits). 34 commits total, 71 files, +13,074/-583 LOC. |
|
||||
|
||||
## Closed PRs in this scope
|
||||
|
||||
| # | PR | Status | Why closed |
|
||||
|---|---|---|---|
|
||||
| **7.10** | **#1287** | ❌ CLOSED 2026-05-05 | Superseded by mega-PR #1288 — strategy reversal to land everything in one go. Same 3 commits now form the head of #1288. |
|
||||
|
||||
## Deprecated split bookmarks (D-17)
|
||||
|
||||
`will/api_7.10` / `will/api_8` / `will/ltx2_sr_runtime` / `will/ltx2_nvfp4` / `will/ltx2_post_fixes` / `will/agents_cleanup` were the split-PR bookmarks under the abandoned 6-PR plan. They remain locally as historical references but are no longer maintained. STACK.md (top-level) is similarly deprecated.
|
||||
|
||||
## Planned (post-#1288 merge)
|
||||
|
||||
| # | Status | Branch | Scope |
|
||||
|---|---|---|---|
|
||||
| 9 | 🟡 | — | LongCat preset migration + colocation (9 model-specific stage files) |
|
||||
| 10 | 🟡 | — | Hunyuan15 SR preset migration + colocation + SR field migration POC |
|
||||
| 11 | 🟡 | — | SSIM/performance test migration off legacy `generate_video(..., **kwargs)` |
|
||||
| 12 | 🟡 | — | Docs + examples migration (includes streaming server + Dynamo) |
|
||||
| 13 | 🟡 | — | Deprecation cleanup (includes flat LTX2 kwargs the internal `gpu_pool.py` used to consume) |
|
||||
|
||||
## Future (compat.py death sequence)
|
||||
|
||||
After PR 13 lands deprecation warnings, `fastvideo/api/compat.py` (~370
|
||||
lines) is the last translation shim between typed public API and legacy
|
||||
internals (`FastVideoArgs`, `SamplingParam`).
|
||||
|
||||
| # | Status | Scope | Lines removed |
|
||||
|---|---|---|---|
|
||||
| 14 | 🔵 reachable | Strip forward translation: `legacy_from_pretrained_to_config`, `legacy_generate_call_to_request`, `_sampling_param_to_request_raw`, `_LEGACY_REQUEST_ALIASES`, `_LTX2_REFINE_FLAT_KEYS`. Depends on PRs 11/12/7.6 callers being migrated. | ~100 |
|
||||
| 15 | 🔵 | `FastVideoArgs` becomes a `@dataclass` view over `GeneratorConfig` with `@property` accessors backing legacy field names. ~600-line god-object refactor. Depends on PR 14. | reverse-translation half (~150) trivial |
|
||||
| 16 | 🔵 | `ForwardBatch` reads `GenerationRequest` by reference; kills `request_to_sampling_param` and the `ForwardBatch(**shallow_asdict(sampling_param), …)` spread. `SamplingParam` demoted or deleted. Depends on PR 15. | rest |
|
||||
| 17 | 🔵 | Move `normalize_generator_config`, `normalize_generation_request`, `load_generator_config_from_file` to `parser.py`. Delete `compat.py`. | file gone |
|
||||
|
||||
PRs 15-17 touch training, distributed, and worker code in addition to
|
||||
inference path; realistically 1-2 quarters beyond the current plan.
|
||||
|
||||
## Dependency chain
|
||||
|
||||
```
|
||||
PR 13 (deprecation)
|
||||
↓
|
||||
PRs 11, 12, 7.6 (migrate callers)
|
||||
↓
|
||||
PR 14 (forward translation gone) ─── ~100 lines out of compat.py
|
||||
↓
|
||||
PR 15 (FastVideoArgs as view) ─── reverse-translation trivial
|
||||
↓
|
||||
PR 16 (ForwardBatch reads request) ─── SamplingParam demoted
|
||||
↓
|
||||
PR 17 (move normalizers, delete file)
|
||||
```
|
||||
|
||||
## NVFP4 work (out-of-band, parallel to PR 7.5+)
|
||||
|
||||
NOT in the canonical PR sequence. Lives on `will/ltx2_sr_port`
|
||||
(currently @ `156103b9`) — a separate stack alongside the public-API
|
||||
upstreaming. See [quantization.md](quantization.md) for what each commit
|
||||
locks in.
|
||||
|
||||
| Commit range | Topic |
|
||||
|---|---|
|
||||
| `cfccd292..b6ac7630` | LTX-2 i2v + SR runtime port + alignment harness |
|
||||
| `a4760bae..c6c14c55` | NVFP4 LTX-2 wire-up + per-component compile + parity fixes (May 2 handoff) |
|
||||
| `a5fcd19c..156103b9` | Post-handoff parity/perf fixes |
|
||||
|
||||
## Key landed artifacts (reference points)
|
||||
|
||||
- Parity inventory: [`docs/design/inference_schema_parity_inventory.yaml`](file:///home/william5lin/FastVideo/docs/design/inference_schema_parity_inventory.yaml) + guard [`fastvideo/tests/api/test_schema_parity_inventory.py`](file:///home/william5lin/FastVideo/fastvideo/tests/api/test_schema_parity_inventory.py)
|
||||
- Typed schema: [`fastvideo/api/schema.py`](file:///home/william5lin/FastVideo/fastvideo/api/schema.py)
|
||||
- Compat layer: [`fastvideo/api/compat.py`](file:///home/william5lin/FastVideo/fastvideo/api/compat.py)
|
||||
- Preset system: [`fastvideo/api/presets.py`](file:///home/william5lin/FastVideo/fastvideo/api/presets.py) + per-family `pipelines/basic/<family>/presets.py`
|
||||
- Streaming package skeleton (PR 5.5): [`fastvideo/entrypoints/streaming/`](file:///home/william5lin/FastVideo/fastvideo/entrypoints/streaming/)
|
||||
- LTX2 typed continuation state (PR 7): [`fastvideo/pipelines/basic/ltx2/continuation.py`](file:///home/william5lin/FastVideo/fastvideo/pipelines/basic/ltx2/continuation.py)
|
||||
|
||||
## Known notable decisions carried forward
|
||||
|
||||
- **Public inference boundary stays plain dataclasses + plain dict/YAML/JSON**
|
||||
— not OmegaConf, not runtime config wrappers.
|
||||
- **Every public entrypoint normalizes into typed config objects** before
|
||||
touching legacy `FastVideoArgs` or `SamplingParam`.
|
||||
- **Legacy `generate_video(..., **kwargs)` stays on direct legacy execution
|
||||
path until PR 11**'s SSIM/performance migration. Prevents golden
|
||||
baselines from drifting during compat period.
|
||||
- **Typed requests use schema defaults**; legacy `generate_video(...)`
|
||||
continues to inherit model-specific `SamplingParam` defaults during
|
||||
compat period.
|
||||
- **Preset registry uses explicit `_register_presets()` pattern** matching
|
||||
`_register_configs()`; lookup keyed by `model_family`.
|
||||
- **Stateless OpenAI server clones `ServeConfig.default_request`** and
|
||||
merges user overrides; preset validation runs before legacy generation.
|
||||
- **Streaming server added as sibling `fastvideo/entrypoints/streaming/`**
|
||||
rather than extending `fastvideo/entrypoints/openai/` (PR 5.5).
|
||||
|
||||
## Per-PR commit-level detail
|
||||
|
||||
For per-PR commit lists, test plans, and merge criteria, the archived
|
||||
source [`source-archive/PR-plan.md`](source-archive/PR-plan.md) (1145 lines)
|
||||
remains the deepest reference. This file is the navigable summary.
|
||||
@@ -0,0 +1,229 @@
|
||||
# Quantization — NVFP4, LinearBase Fallback, Layer Profiles
|
||||
|
||||
What landed in the May 2 NVFP4 stack, why it's load-bearing, and what's
|
||||
still owed (`layer_profile`, typed quant carrier, AbsMaxFP8 cleanup).
|
||||
|
||||
For overall API design see [design.md](design.md). For the open
|
||||
follow-ups see [open-threads.md](open-threads.md).
|
||||
|
||||
**Last updated:** 2026-05-03.
|
||||
|
||||
## NVFP4 — what it is
|
||||
|
||||
NVIDIA's specific block-scaled FP4 format:
|
||||
|
||||
- e2m1 mantissa
|
||||
- fp32 alpha
|
||||
- `layout_128x4` scale layout
|
||||
- group size 16
|
||||
|
||||
Distinct from MX-FP4 / OCP-FP4 / generic e3m0. The May 2 rename
|
||||
(`94c983a2`) disambiguated the naming throughout FastVideo's public
|
||||
surface.
|
||||
|
||||
## Files (current)
|
||||
|
||||
| File | Role |
|
||||
|---|---|
|
||||
| [`fastvideo/layers/quantization/nvfp4_config.py`](file:///home/william5lin/FastVideo/fastvideo/layers/quantization/nvfp4_config.py) | `NVFP4Config`, `NVFP4QuantizeMethod`, `convert_model_to_nvfp4` |
|
||||
| [`fastvideo/layers/quantization/__init__.py`](file:///home/william5lin/FastVideo/fastvideo/layers/quantization/__init__.py) | `QuantizationMethods` literal includes `"NVFP4"`; `get_quantization_config` resolves it |
|
||||
| [`fastvideo/layers/linear.py`](file:///home/william5lin/FastVideo/fastvideo/layers/linear.py) | `LinearBase.__init__` falls back to `UnquantizedLinearMethod` when `quant_config.get_quant_method` returns None — **load-bearing** |
|
||||
| [`fastvideo/models/loader/fsdp_load.py`](file:///home/william5lin/FastVideo/fastvideo/models/loader/fsdp_load.py) | `_maybe_convert_model_to_nvfp4` helper detects via `isinstance(quant_method, NVFP4QuantizeMethod)`; calls `convert_model_to_nvfp4` to materialize buffers |
|
||||
| [`fastvideo/models/dits/ltx2.py`](file:///home/william5lin/FastVideo/fastvideo/models/dits/ltx2.py) | `nn.Linear` → `ReplicatedLinear` for FP4-eligible subset; `_supports_prequantized_input` + `_linear_project_with_optional_prequant` helpers; quant_config + prefix= plumbing |
|
||||
| [`fastvideo/api/compat.py`](file:///home/william5lin/FastVideo/fastvideo/api/compat.py) | Typed `engine.quantization.transformer_quant: "NVFP4"` resolves to `NVFP4Config()` instance |
|
||||
| [`fastvideo/fastvideo_args.py`](file:///home/william5lin/FastVideo/fastvideo/fastvideo_args.py) | `__post_init__._apply_transformer_quant` pins `pipeline_config.dit_config.quant_config = NVFP4Config()` |
|
||||
|
||||
## Buffer naming (post-rename)
|
||||
|
||||
| Old | New |
|
||||
|---|---|
|
||||
| `_fp4_weight` / `_fp4_alpha` | `_nvfp4_weight` / `_nvfp4_alpha` |
|
||||
| `_weight_global_sf` | unchanged |
|
||||
| `convert_model_to_fp4` | `convert_model_to_nvfp4` |
|
||||
| `FP4QuantizeMethod` | `NVFP4QuantizeMethod` |
|
||||
| `QuantizationMethods` literal `"FP4"` | `"NVFP4"` |
|
||||
|
||||
Internal-scope torch op namespace `fastvideo_fp4::*` and
|
||||
`_get_ltx2_fp4_stage_profile` deliberately left as-is — purely internal
|
||||
naming that mirrors FastVideo-internal.
|
||||
|
||||
## Layer set asymmetry — by design
|
||||
|
||||
`NVFP4Config.fp4_layers` (default `layer_profile="refine"`) covers:
|
||||
|
||||
- `attn1.{to_q,to_k,to_v,to_out}` — full self-attention
|
||||
- `attn2.{to_q,to_out}` — cross-attn Q + out only (text context not quantized)
|
||||
- `audio_to_video_attn.{to_q,to_out}` — AV cross Q + out
|
||||
- `video_to_audio_attn.{to_k,to_v}` — VA cross K + V
|
||||
- `ffn.{fc_in,fc_out}` — video FFN
|
||||
- `adaln_single.linear` — but this is `nn.Linear` (not `LinearBase`),
|
||||
so it never actually gets FP4'd. List entry has no effect; matches
|
||||
internal.
|
||||
|
||||
**NOT in the set:**
|
||||
|
||||
- audio self-attention (`audio_attn1.*`)
|
||||
- audio cross-attention (`audio_attn2.*`)
|
||||
- audio FFN (`audio.ffn.*`)
|
||||
|
||||
Audio path is cheap enough that quant overhead isn't worth it. Test
|
||||
[`test_basic_av_block_propagates_quant_config_to_all_children`](file:///home/william5lin/FastVideo/fastvideo/tests/ops/quantization/test_nvfp4_ltx2_wiring.py)
|
||||
locks this in — if you add audio quantization later, update the test.
|
||||
|
||||
## `LinearBase` fallback — DO NOT REMOVE
|
||||
|
||||
[`fastvideo/layers/linear.py:191-202`](file:///home/william5lin/FastVideo/fastvideo/layers/linear.py#L191-L202): when `quant_config.get_quant_method` returns
|
||||
`None` (layer not in the quant config's set), we fall back to
|
||||
`UnquantizedLinearMethod`.
|
||||
|
||||
**Removing this fallback would break every non-tagged
|
||||
`ReplicatedLinear` constructed with an `NVFP4Config`** — the previous
|
||||
`assert quant_method is not None` would crash on unmatched layers (e.g.
|
||||
text-encoder K/V projections, audio attention, etc.).
|
||||
|
||||
This is one of the load-bearing changes from `42b30bf9`.
|
||||
|
||||
## `transformer_quant` precedence rules
|
||||
|
||||
`FastVideoArgs._apply_transformer_quant` only writes
|
||||
`dit_config.quant_config` when it's currently `None`. **If a caller has
|
||||
explicitly set** `pipeline_config.dit_config.quant_config = NVFP4Config(...)`,
|
||||
the explicit setter wins.
|
||||
|
||||
Dreamverse's `video_generation.py` relies on this precedence — it sets
|
||||
`NVFP4Config()` directly via `experimental["pipeline_config"]` because
|
||||
typed `transformer_quant: "NVFP4"` doesn't yet expose `layer_profile`.
|
||||
See "Open follow-ups" below.
|
||||
|
||||
## Attention forward optimization
|
||||
|
||||
[`models/dits/ltx2.py`](file:///home/william5lin/FastVideo/fastvideo/models/dits/ltx2.py)
|
||||
ports `_supports_prequantized_input` and
|
||||
`_linear_project_with_optional_prequant`. Attention forward
|
||||
pre-quantizes input once (`quantize_input`), reuses the
|
||||
`(x_fp4, x_scale, x_global_sf)` tuple for k/v projections when
|
||||
`context is x` — bit-matches internal's fused path.
|
||||
|
||||
## `prepare_for_compile` protocol
|
||||
|
||||
[`composed_pipeline_base._maybe_compile_pipeline_module`](file:///home/william5lin/FastVideo/fastvideo/pipelines/composed_pipeline_base.py)
|
||||
calls `getattr(module, "prepare_for_compile", None)` before invoking
|
||||
`torch.compile`. Defined as a duck-type protocol — no base class method.
|
||||
|
||||
Currently only **Gemma3** implements it (to materialize HF weights
|
||||
outside Dynamo's tracer). Add to other models that have lazy external
|
||||
state if you observe compile-time graph breaks.
|
||||
|
||||
## Per-component compile flags
|
||||
|
||||
`CompileConfig` (in
|
||||
[`fastvideo/api/schema.py`](file:///home/william5lin/FastVideo/fastvideo/api/schema.py))
|
||||
gained per-component knobs in `221cb20a`:
|
||||
|
||||
```python
|
||||
@dataclass
|
||||
class CompileConfig:
|
||||
enabled: bool = False # master DiT switch
|
||||
backend: str = "inductor"
|
||||
fullgraph: bool = False
|
||||
mode: str | None = None
|
||||
dynamic: bool | None = None
|
||||
extras: dict = field(default_factory=dict)
|
||||
|
||||
# Per-component overlays, None = inherit master `enabled`
|
||||
text_encoder_enabled: bool | None = None
|
||||
vae_enabled: bool | None = None
|
||||
audio_vae_enabled: bool | None = None
|
||||
|
||||
# Per-component kwargs override master when non-empty
|
||||
dit_kwargs: dict = field(default_factory=dict)
|
||||
text_encoder_kwargs: dict = field(default_factory=dict)
|
||||
vae_kwargs: dict = field(default_factory=dict)
|
||||
audio_vae_kwargs: dict = field(default_factory=dict)
|
||||
```
|
||||
|
||||
**`transformer_refine` is auto-compiled with the master DiT flag.** No
|
||||
separate `enable_torch_compile_refine` flag — by design, refine inherits
|
||||
DiT compile state to keep typed surface small. Decoupling would add a
|
||||
new flag, not repurpose existing ones.
|
||||
|
||||
## Quantization commit chain (`will/ltx2_sr_port`)
|
||||
|
||||
| Commit | Locks in |
|
||||
|---|---|
|
||||
| `365a66c7 feat(quantization): upstream LTX-2 FP4Config with lazy flashinfer` | Public colocation of FP4Config (resolves dreamverse_review Q-6 option 1); flashinfer lazy-imported in loader helper, no public hard-dep |
|
||||
| `a4760bae fix(api): propagate generic refine_*` | `_resolve_refine_args()` copies generic `refine_*` knobs onto `ltx2_refine_*` runtime carriers; `_randn_ltx2_video_latents` reverts to `torch.randn` to bit-match internal under single-generator inference |
|
||||
| `221cb20a feat(api): typed per-component CompileConfig` | `CompileConfig` per-component knobs; matching `FastVideoArgs` carriers; compat layer round-trip |
|
||||
| `6da342ba feat(compile): per-component compile + transformer_refine + prepare hook` | `composed_pipeline_base.post_init` compiles `transformer_refine` alongside `transformer`/`transformer_2`; per-component compile loops; `prepare_for_compile` hook on Gemma3 |
|
||||
| `42b30bf9 feat(ltx2): wire FP4 inference` (largest) | `nn.Linear` → `ReplicatedLinear` for FP4-eligible LTX2 subset; `quant_config` + `prefix=` plumbing; `_maybe_convert_model_to_nvfp4` helper; `LinearBase` fallback to `UnquantizedLinearMethod`; typed `transformer_quant` resolution |
|
||||
| `94c983a2 refactor(quant): rename FP4 → NVFP4` | Mechanical rename across config, methods, buffers, tests |
|
||||
| `c6c14c55 test(nvfp4): lock LTX-2 wiring + typed transformer_quant flow` | 6+4 tests in `test_nvfp4_ltx2_wiring.py` + `test_typed_quant_flow.py` |
|
||||
| `a5fcd19c [fix]: lazy-import flash_attn 2 fallback in attention backend` | post-handoff: lazy import to avoid hard flash_attn 2 dep |
|
||||
| `d4ee5be2 [fix]: avoid model.to() round-trip in Gemma encoder forward` | post-handoff: parity / perf fix |
|
||||
| `156103b9 [fix]: unwrap list-of-generator before torch.randn in LTX-2 latent prep` | post-handoff: parity fix for list-of-generators (was bit-matching only single-generator path) |
|
||||
|
||||
## Tests
|
||||
|
||||
| Test | Asserts |
|
||||
|---|---|
|
||||
| [`fastvideo/tests/ops/quantization/test_nvfp4_ltx2_wiring.py`](file:///home/william5lin/FastVideo/fastvideo/tests/ops/quantization/test_nvfp4_ltx2_wiring.py) (6 tests) | `LTXSelfAttention.to_q/to_k/to_v/to_out` are `ReplicatedLinear`; `NVFP4Config()` attaches `NVFP4QuantizeMethod` on the quantized subset with correct `layer_prefix`; non-tagged projections (cross-attn K/V, audio attn, audio FFN) fall back to `UnquantizedLinearMethod`; `BasicAVTransformerBlock` propagates `quant_config`+`prefix` correctly to all 4 attention modules + FFN |
|
||||
| [`fastvideo/tests/api/test_typed_quant_flow.py`](file:///home/william5lin/FastVideo/fastvideo/tests/api/test_typed_quant_flow.py) (4 tests) | typed `engine.quantization.transformer_quant: "NVFP4"` → `NVFP4Config()` instance flow; default leaves `transformer_quant` None; explicit `dit_config.quant_config = ...` wins over typed carrier |
|
||||
|
||||
CPU-only by design; do NOT exercise actual FP4 kernels (no flashinfer in
|
||||
CI). Real kernel coverage requires a CI run with flashinfer installed.
|
||||
|
||||
## Open follow-ups (quantization-specific)
|
||||
|
||||
### #4: Expose `layer_profile` on typed `engine.quantization`
|
||||
|
||||
Today `transformer_quant: "NVFP4"` always constructs `NVFP4Config()`
|
||||
with default `layer_profile="refine"`. To support stage-1 profiles (no
|
||||
`attn2.to_out`, no cross-modal AV) via typed config, add
|
||||
`transformer_quant_layer_profile: str | None = None` and thread it
|
||||
through:
|
||||
|
||||
- `fastvideo/api/schema.py` — `QuantizationConfig` field
|
||||
- `fastvideo/api/compat.py` — typed → flat translation
|
||||
- `fastvideo/fastvideo_args.py` — `_apply_transformer_quant` consumes it
|
||||
|
||||
Dreamverse currently dodges this by setting `NVFP4Config()` directly via
|
||||
`experimental["pipeline_config"]`. Exposing `layer_profile` removes the
|
||||
dodge. See [open-threads.md](open-threads.md) #4.
|
||||
|
||||
### #5: Typed `dit_config.quant_config` carrier (replace `experimental["pipeline_config"]`)
|
||||
|
||||
Long-term: design a typed home for an in-memory `PipelineConfig`
|
||||
instance with mutated `dit_config`. Today `compat.py` recognizes the
|
||||
`pipeline_config` key in `experimental` and threads it through to
|
||||
`FastVideoArgs.from_kwargs`. This is fine for short-term but not pretty.
|
||||
|
||||
Heaviest design work in the open queue. May need Oracle consult.
|
||||
|
||||
### #2: AbsMaxFP8 pre-existing test failure
|
||||
|
||||
`fastvideo/tests/ops/quantization/test_absmax_fp8.py::test_create_weights_rejects_invalid_dtype`
|
||||
fails on `main` and on `will/ltx2_sr_port` with the same error
|
||||
(`AssertionError not raised`). Verified via `git stash` that the
|
||||
failure pre-dates NVFP4 work.
|
||||
|
||||
Either:
|
||||
- Fix the test (`AbsMaxFP8LinearMethod.create_weights` no longer
|
||||
asserts on invalid dtype — restore the assert if intentional, or drop
|
||||
the test).
|
||||
|
||||
Self-contained tech debt; small fix.
|
||||
|
||||
## Don't / Cautions
|
||||
|
||||
- **Don't change `NVFP4Config` buffer names back to `_fp4_*`.** Rename
|
||||
is intentional to disambiguate from MX-FP4 / OCP-FP4.
|
||||
- **Don't remove the `LinearBase` `UnquantizedLinearMethod` fallback.**
|
||||
Load-bearing for non-tagged layers when a `quant_config` is set.
|
||||
- **Don't repurpose `enable_torch_compile` to mean DiT-only.** It also
|
||||
drives `transformer_refine` and `transformer_2` compile.
|
||||
- **Don't bypass the typed surface for new options.** New compile /
|
||||
quant / refine knobs should land on the dataclass + compat.py +
|
||||
parity inventory together. The existing test suite locks this in.
|
||||
- **Don't merge to main without a CI run that covers FP4.** Current CI
|
||||
doesn't run flashinfer-dependent paths; the wiring tests are CPU-only
|
||||
by design.
|
||||
@@ -0,0 +1,385 @@
|
||||
# Runbook — How to Do Work in This Scope
|
||||
|
||||
Operational how-to for the dreamverse-integration scope. Read after
|
||||
[state.md](state.md) and [open-threads.md](open-threads.md).
|
||||
|
||||
For design rationale see [design.md](design.md). For who to credit see
|
||||
[authors.md](authors.md). For PR status see [pr-roadmap.md](pr-roadmap.md).
|
||||
|
||||
**Last updated:** 2026-05-05 (strategy reversed to single mega-PR #1288 on `will/ltx2_sr_port`; #1287 closed; STACK.md split model deprecated per [decisions-log.md D-17](decisions-log.md#d-17)).
|
||||
|
||||
## Worktree contract
|
||||
|
||||
```
|
||||
Repo: /home/william5lin/FastVideo
|
||||
Branch: will/ltx2_sr_port
|
||||
```
|
||||
|
||||
Other agents and the user share this worktree concurrently. If `git status`
|
||||
shows changes you don't recognize, they belong to **someone else's work** —
|
||||
don't revert, don't `git stash drop`, don't `git checkout -- <file>`.
|
||||
Switch to `will/ltx2_sr_port` cleanly with `git checkout will/ltx2_sr_port`
|
||||
(safe if your own working tree is clean) and proceed.
|
||||
|
||||
If your task requires a different branch (e.g. cherry-pick to
|
||||
`will/api_7.9` for PR #1286 propagation), return to `will/ltx2_sr_port`
|
||||
when done — that is the assumed default.
|
||||
|
||||
## Branch topology (single mega-PR model)
|
||||
|
||||
The dreamverse-integration work now ships as one PR (#1288) off
|
||||
`will/ltx2_sr_port`. The split-PR model documented in earlier revisions
|
||||
of this runbook (and in top-level `STACK.md`) is **abandoned** —
|
||||
see [decisions-log.md D-17](decisions-log.md#d-17).
|
||||
|
||||
```
|
||||
origin/main
|
||||
↓ [public-API refactor: PRs 0..7.9 merged on main, latest #1286 = 2aaeee2a]
|
||||
will/ltx2_sr_port (**PR #1288 head** — single mega-PR, 34 commits, 71 files, +13,074/-583)
|
||||
```
|
||||
|
||||
| Branch | Role | Status |
|
||||
|---|---|---|
|
||||
| `will/ltx2_sr_port` | **PR #1288 head**, default working branch | OPEN, MERGEABLE |
|
||||
| `will/api_7.10` / `will/api_8` / `will/ltx2_sr_runtime` / `will/ltx2_nvfp4` / `will/ltx2_post_fixes` / `will/agents_cleanup` | deprecated split-PR bookmarks | local-only historical references; safe to delete |
|
||||
| `will/ltx2_sr_port-pre-1286-rebase` | safety backup | local-only; preserves the 4 commits dropped during the post-#1286 rebase |
|
||||
|
||||
**Sanity check:** `git merge-base --is-ancestor origin/main will/ltx2_sr_port`
|
||||
should exit 0. If it doesn't, the branch is in an unexpected state — read
|
||||
[state.md](state.md) before continuing.
|
||||
|
||||
## After PR #1288 merges
|
||||
|
||||
When the mega-PR squash-merges into `main`:
|
||||
|
||||
1. `git fetch origin main` to pull the merge commit.
|
||||
2. The entire `will/ltx2_sr_port` content is now on main; the branch can
|
||||
be deleted (locally + on origin) once all consumers are notified.
|
||||
3. Delete deprecated split bookmarks: `git branch -D will/api_7.10
|
||||
will/api_8 will/ltx2_sr_runtime will/ltx2_nvfp4 will/ltx2_post_fixes
|
||||
will/agents_cleanup` (local-only, no remote).
|
||||
4. Optionally remove top-level `STACK.md` (now a historical artifact).
|
||||
Keep [co-authors.md](co-authors.md) — still the canonical roster reference.
|
||||
5. Decide whether to keep `will/ltx2_sr_port-pre-1286-rebase` (safety
|
||||
backup of the pre-rebase chain) — recommend deleting once #1288 is
|
||||
merged and verified on main.
|
||||
6. Update memory dir to reflect the post-merge state — bump
|
||||
`Last reconciled` headers, mark Item D resolved in
|
||||
[open-threads.md](open-threads.md), record the merge commit in
|
||||
[decisions-log.md](decisions-log.md).
|
||||
|
||||
## Historical: split-PR re-slice protocol (deprecated)
|
||||
|
||||
Prior revisions of this runbook documented a 10-step re-slice protocol
|
||||
for the abandoned 6-PR split model. That protocol is now obsolete.
|
||||
The post-#1286 rebase (2026-05-05) was the last execution of it; details
|
||||
are preserved in [state.md](state.md) "Post-#1286 rebase summary" and
|
||||
git history at commit `b34d9704`.
|
||||
|
||||
## Verification
|
||||
|
||||
### Lint (pre-commit)
|
||||
|
||||
```bash
|
||||
pre-commit run --files <changed-paths...>
|
||||
```
|
||||
|
||||
- Binary: `/home/william5lin/miniconda3/envs/fv-main/bin/pre-commit`.
|
||||
NOT `.venv/bin/pre-commit` — that doesn't exist in this worktree.
|
||||
- Auto-applies yapf reformatting; re-stage modified files after.
|
||||
- Hook chain: yapf → ruff → codespell → mypy → spaces-check.
|
||||
- Memory dir (`.agents/memory/`) is yapf/ruff/mypy excluded — only
|
||||
"spaces" runs. Memory edits don't need lint, but DO use UTF-8 and
|
||||
consistent line endings.
|
||||
|
||||
### Tests
|
||||
|
||||
Router tests (PR #1286 scope):
|
||||
```bash
|
||||
.venv/bin/python -m pytest fastvideo/tests/entrypoints/streaming/test_router.py -v --no-header
|
||||
```
|
||||
|
||||
Stack baseline (May 2 handoff suite — re-run when you change anything in
|
||||
api/, contract/, or LTX-2 paths):
|
||||
```bash
|
||||
.venv/bin/python -m pytest \
|
||||
fastvideo/tests/api/ \
|
||||
fastvideo/tests/contract/ \
|
||||
fastvideo/tests/ops/quantization/test_nvfp4_*.py \
|
||||
tests/local_tests/pipelines/test_ltx2_pipeline_smoke.py \
|
||||
-q --no-header
|
||||
```
|
||||
|
||||
Expected baselines:
|
||||
- May 2 handoff (`156103b9`): 222 passed, 1 skipped.
|
||||
- Post-D-16 (`a152cb77` / `09647a30`): +7 router tests pass on top.
|
||||
|
||||
### LSP
|
||||
|
||||
Use `lsp_diagnostics` on changed files BEFORE running build. Pre-existing
|
||||
warnings to ignore (predate this work):
|
||||
|
||||
- `fastvideo/entrypoints/streaming/router/main.py:37` — `Task` generic.
|
||||
- `fastvideo/entrypoints/cli/router_serve.py:55` — `_SubParsersAction` generic.
|
||||
|
||||
### gh CLI for PR status
|
||||
|
||||
```bash
|
||||
# PR #1286 quick status
|
||||
gh pr view 1286 --json headRefOid,mergeable,statusCheckRollup \
|
||||
--jq '{headRefOid, mergeable, checks: [.statusCheckRollup[] | {name, status, conclusion}]}'
|
||||
|
||||
# All commits in a PR + co-author check
|
||||
gh pr view 1286 --json commits \
|
||||
--jq '.commits[] | {oid: .oid[0:8], msg: .messageHeadline, author: .authors[0].login}'
|
||||
```
|
||||
|
||||
## Commit workflow
|
||||
|
||||
### Subject convention
|
||||
|
||||
`[type] <scope>: <imperative summary>` — keep ≤ 72 chars.
|
||||
|
||||
Types observed in this scope: `feat`, `fix`, `test`, `docs`, `chore`,
|
||||
`refactor`. Scopes observed: `streaming`, `dreamverse-integration`,
|
||||
`api`, `quant`, `ltx2`, `nvfp4`, etc.
|
||||
|
||||
Examples:
|
||||
- `[fix] streaming: router polish — bridge cancel + state machine + deps`
|
||||
- `[docs] dreamverse-integration: add authors.md + track D-16 router polish`
|
||||
|
||||
### Body convention
|
||||
|
||||
Bullet list, one bullet per file or concern. Why-before-what. Wrap at
|
||||
~80 chars (yapf doesn't reformat commit messages; readability is on you).
|
||||
|
||||
### Co-author trailers (REQUIRED on every commit)
|
||||
|
||||
The 4 trailers in [authors.md](authors.md) MUST appear on every commit
|
||||
in this scope. Use `--trailer` flags or write the body to a file with
|
||||
`-F` — DO NOT use multiple `-m` blocks for the trailers (each `-m` is
|
||||
its own paragraph and git's trailer parser only reads the LAST paragraph,
|
||||
yielding 1 trailer parsed instead of 4).
|
||||
|
||||
**Inline `--trailer` form (preferred for short commits):**
|
||||
|
||||
```bash
|
||||
git commit -m "subject" -m "body..." \
|
||||
--trailer "Co-authored-by: Junda (David) Su <90978028+Davids048@users.noreply.github.com>" \
|
||||
--trailer "Co-authored-by: Matthew Noto <99706358+RandNMR73@users.noreply.github.com>" \
|
||||
--trailer "Co-authored-by: XOR-op <17672363+XOR-op@users.noreply.github.com>" \
|
||||
--trailer "Co-authored-by: Zhang Peiyuan <42993249+jzhang38@users.noreply.github.com>"
|
||||
```
|
||||
|
||||
**File form (preferred for multi-paragraph bodies):**
|
||||
|
||||
```bash
|
||||
cat > /tmp/opencode/msg.txt <<'EOF'
|
||||
[type] scope: subject
|
||||
|
||||
* Bullet one with rationale.
|
||||
* Bullet two with rationale.
|
||||
|
||||
Co-authored-by: Junda (David) Su <90978028+Davids048@users.noreply.github.com>
|
||||
Co-authored-by: Matthew Noto <99706358+RandNMR73@users.noreply.github.com>
|
||||
Co-authored-by: XOR-op <17672363+XOR-op@users.noreply.github.com>
|
||||
Co-authored-by: Zhang Peiyuan <42993249+jzhang38@users.noreply.github.com>
|
||||
EOF
|
||||
git commit -F /tmp/opencode/msg.txt
|
||||
```
|
||||
|
||||
The trailers MUST be a single block at the end of the message with no
|
||||
blank lines between them.
|
||||
|
||||
**Verify trailers parsed:**
|
||||
|
||||
```bash
|
||||
git log -1 --format='%(trailers:key=Co-authored-by,valueonly)'
|
||||
```
|
||||
|
||||
Should print 4 lines (one per author). If only 1 line, you have the
|
||||
multi-`-m` bug — amend with `-F` to fix (allowed if commit is unpushed
|
||||
and you authored it in this session per AGENTS.md amend rules).
|
||||
|
||||
### NEVER add to commits
|
||||
|
||||
Per [`AGENTS.md`](../../../AGENTS.md):
|
||||
|
||||
- AI co-authors (Claude, GPT, Codex, Cursor, etc.) — explicitly forbidden
|
||||
- "Generated with Claude Code" footer — explicitly forbidden
|
||||
- `--no-verify` to skip pre-commit — explicitly forbidden
|
||||
|
||||
## Push + PR propagation
|
||||
|
||||
### Pushing `will/ltx2_sr_port` (top of stack)
|
||||
|
||||
```bash
|
||||
git push origin will/ltx2_sr_port # fast-forward, no force needed
|
||||
```
|
||||
|
||||
If git wants to force-push, you've rewritten history. STOP and verify:
|
||||
|
||||
```bash
|
||||
git log origin/will/ltx2_sr_port..will/ltx2_sr_port # local-only commits
|
||||
git log will/ltx2_sr_port..origin/will/ltx2_sr_port # remote-only commits
|
||||
```
|
||||
|
||||
Force-push requires explicit user confirmation per `AGENTS.md`.
|
||||
|
||||
### Propagating fixes to PR #1286 (`will/api_7.9`)
|
||||
|
||||
When a fix is in router code (`fastvideo/entrypoints/streaming/router/`,
|
||||
`cli/router_serve.py`, `tests/entrypoints/streaming/test_router.py`,
|
||||
or `pyproject.toml` router-related), it must land on BOTH branches.
|
||||
Cherry-pick avoids any force-push:
|
||||
|
||||
```bash
|
||||
# 1. Commit on will/ltx2_sr_port first (working branch)
|
||||
git add <files...>
|
||||
git commit -F /tmp/opencode/msg.txt # with trailers per above
|
||||
|
||||
# 2. Cherry-pick onto will/api_7.9 (creates a separate SHA, identical diff)
|
||||
git checkout will/api_7.9
|
||||
git cherry-pick <ltx2_sr_port-sha>
|
||||
git push origin will/api_7.9 # fast-forward, no force
|
||||
|
||||
# 3. Return to working branch
|
||||
git checkout will/ltx2_sr_port
|
||||
|
||||
# 4. Verify PR #1286 picked it up
|
||||
gh pr view 1286 --json headRefOid --jq '.headRefOid'
|
||||
```
|
||||
|
||||
Two SHAs for the same diff — they'll dedupe naturally on the next
|
||||
bulk-rebase via the trailer-injection rebase command in
|
||||
[authors.md](authors.md).
|
||||
|
||||
### When a fix is memory-dir-only
|
||||
|
||||
`.agents/memory/dreamverse-integration/` lives in the `agents_cleanup`
|
||||
layer of the stack — it does NOT belong on `will/api_7.9`. Memory updates
|
||||
stay on `will/ltx2_sr_port` only.
|
||||
|
||||
### When a fix is non-router code in the integration scope
|
||||
|
||||
Land on `will/ltx2_sr_port`. If that fix needs to ship as a separate PR
|
||||
(e.g. extending PR 7.10 or starting PR 9), open a new branch off the
|
||||
right base per [pr-roadmap.md](pr-roadmap.md).
|
||||
|
||||
## Memory dir maintenance
|
||||
|
||||
When state changes, update the memory dir BEFORE moving on. Every file
|
||||
has a "Last updated" header — bump when you edit.
|
||||
|
||||
| Change | File to update |
|
||||
|---|---|
|
||||
| Branch tip moves | [state.md](state.md) "Branch tips" + "Last reconciled" |
|
||||
| PR opens / merges | [pr-roadmap.md](pr-roadmap.md) status table |
|
||||
| New decision made | [decisions-log.md](decisions-log.md) — add D-N entry, bump header |
|
||||
| Open thread resolved | [open-threads.md](open-threads.md) — strikethrough + "Resolved" note |
|
||||
| New open thread | [open-threads.md](open-threads.md) — priority overview + section |
|
||||
| New collaborator credited | [authors.md](authors.md) roster + trailer block + bulk-rebase |
|
||||
| Source doc archived | [source-archive/README.md](source-archive/README.md) + [README.md](README.md) sources table |
|
||||
| Process / runbook detail changes | [runbook.md](runbook.md) (this file) |
|
||||
|
||||
Cross-link siblings via relative paths. Never duplicate content — link.
|
||||
|
||||
## Common pitfalls
|
||||
|
||||
### `pre-commit` not in `.venv/bin`
|
||||
|
||||
`pre-commit` lives at `/home/william5lin/miniconda3/envs/fv-main/bin/pre-commit`.
|
||||
The `.venv` here is for the FastVideo package itself, not pre-commit.
|
||||
|
||||
### Trailers split across paragraphs
|
||||
|
||||
`git commit -m A -m B -m C` makes A, B, C separate paragraphs. Git's
|
||||
trailer parser only reads the LAST paragraph — multiple `-m
|
||||
"Co-authored-by: ..."` produces 1 trailer parsed, not 4. Use `--trailer`
|
||||
flags or `-F` with the trailers in a single block at the end.
|
||||
|
||||
### Stash 0 on FastVideo IS NOT yours
|
||||
|
||||
`stash@{0}: WIP on main: 71bfc13d HunyuanVideo plugin` predates this work.
|
||||
**DO NOT POP.** See [state.md](state.md) "Stashes — DO NOT POP".
|
||||
|
||||
### `AbsMaxFP8` test "failure" is pre-existing
|
||||
|
||||
`fastvideo/tests/ops/quantization/test_absmax_fp8.py::test_create_weights_rejects_invalid_dtype`
|
||||
fails on `main` and on every branch in this scope. NOT introduced by
|
||||
integration work. See [open-threads.md](open-threads.md) item #2.
|
||||
|
||||
### Untracked nested clones at repo root
|
||||
|
||||
`dynamo/`, `ray/`, `vllm-omni/` are untracked nested git clones at the
|
||||
FastVideo repo root. Reference repos for cross-repo work. **Do not
|
||||
`rm -rf`** — they're someone else's working state.
|
||||
|
||||
### Live services on 8009 / 5274
|
||||
|
||||
`dreamverse-server` runs on 8009 (warmed GPU worker), Next.js dev server
|
||||
on 5274. Don't start new instances on those ports without checking
|
||||
[state.md](state.md) "Live services" first.
|
||||
|
||||
### Branch may have been switched by another agent
|
||||
|
||||
Other agents share this worktree. If `git branch --show-current` returns
|
||||
something other than `will/ltx2_sr_port`, switch back cleanly with
|
||||
`git checkout will/ltx2_sr_port` — don't disturb their work, don't
|
||||
discard their uncommitted changes.
|
||||
|
||||
### Force-push policy
|
||||
|
||||
Per `AGENTS.md`: never force-push without explicit user confirmation.
|
||||
For trailer fixes on already-pushed commits, prefer the bulk-rebase
|
||||
command in [authors.md](authors.md) — safe to re-run.
|
||||
|
||||
### Two trailerless commits in PR #1286
|
||||
|
||||
`a152cb77` (on `will/api_7.9`) and `40e265b8` (now-superseded ancestor
|
||||
on `will/ltx2_sr_port`) lack the 4 co-author trailers. **Accepted gap**
|
||||
per user decision — see [authors.md](authors.md) "Known gaps".
|
||||
|
||||
## Self-test (verify your context is loaded)
|
||||
|
||||
After reading the memory dir, you should be able to answer:
|
||||
|
||||
1. What branch should I be on? → `will/ltx2_sr_port`
|
||||
2. What's the active open PR in this scope? → #1286 on `will/api_7.9`
|
||||
3. Where does PR #1286 land in the stack? → Bottom; ancestor of `will/ltx2_sr_port`
|
||||
4. Who do I credit on every commit? → 4 authors per [authors.md](authors.md)
|
||||
5. Where do memory updates land? → `will/ltx2_sr_port` only (NOT api_7.9)
|
||||
6. What's the next-priority open thread? → See [open-threads.md](open-threads.md) "Recommended pull order" — D-8 verify is current top
|
||||
7. What pre-existing failure can I ignore? → AbsMaxFP8 test (item #2)
|
||||
8. What's the bulk-rebase command for adding trailers across the stack? → See [authors.md](authors.md) "How the trailers were applied"
|
||||
|
||||
If you can't answer one of these from the memory dir alone, the dir has
|
||||
a gap — file it as a new entry in [open-threads.md](open-threads.md)
|
||||
before continuing.
|
||||
|
||||
## First 60 seconds — copy-paste orientation
|
||||
|
||||
```bash
|
||||
# 1. Confirm branch
|
||||
cd /home/william5lin/FastVideo
|
||||
git branch --show-current # should print: will/ltx2_sr_port
|
||||
# If not, recover: git checkout will/ltx2_sr_port
|
||||
|
||||
# 2. Confirm worktree clean (untracked nested clones expected)
|
||||
git status --short
|
||||
|
||||
# 3. Confirm PR #1286 head matches expected api_7.9 tip
|
||||
gh pr view 1286 --json headRefOid --jq '.headRefOid'
|
||||
git rev-parse will/api_7.9 # should match PR head
|
||||
|
||||
# 4. Confirm your context vs the memory dir
|
||||
git log -1 --oneline
|
||||
cat .agents/memory/dreamverse-integration/state.md | head -30
|
||||
|
||||
# 5. Confirm live services still running
|
||||
curl -s http://localhost:8009/readyz | head -c 200
|
||||
curl -s http://localhost:5274/ -o /dev/null -w "%{http_code}\n"
|
||||
```
|
||||
|
||||
If any of those produce unexpected output, read [state.md](state.md)
|
||||
before changing anything.
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,44 @@
|
||||
# Source Archive
|
||||
|
||||
These are the original unsynthesized design and integration docs that
|
||||
predate the consolidation in
|
||||
[`../`](../). They are **NOT** the source of truth — the synthesized
|
||||
sibling files in the parent directory are.
|
||||
|
||||
Archived 2026-05-03. All previously untracked.
|
||||
|
||||
## Contents
|
||||
|
||||
| File | Original location | Date | Synthesized into |
|
||||
|---|---|---|---|
|
||||
| `apirefactor.md` | `FastVideo/` (repo root) | 2026-04-21 | [`../design.md`](../design.md) |
|
||||
| `PR-plan.md` (was `PR plan.md` at repo root) | `FastVideo/` (repo root) | 2026-04-25 | [`../pr-roadmap.md`](../pr-roadmap.md) |
|
||||
| `dreamverse_review.md` | `FastVideo/` (repo root) | 2026-04-26 | [`../decisions-log.md`](../decisions-log.md) + [`../state.md`](../state.md) |
|
||||
| `handoff-nvfp4-launch-demo.md` | `.agents/exploration/` | 2026-05-02 | [`../state.md`](../state.md) + [`../quantization.md`](../quantization.md) + [`../open-threads.md`](../open-threads.md) |
|
||||
| `streaming-server-upstream-plan.md` | `.agents/exploration/` | 2026-04-17 | [`../streaming-server.md`](../streaming-server.md) + [`../decisions-log.md`](../decisions-log.md) |
|
||||
| `dreamverse_integration.md` | `.agents/exploration/` | 2026-04-23 | [`../cross-repo-surfaces.md`](../cross-repo-surfaces.md) |
|
||||
| `video-generator-config-api-design.md` | `.agents/exploration/` | 2026-04-02 | [`../design.md`](../design.md) (early-draft material) |
|
||||
|
||||
## Why archived (not deleted)
|
||||
|
||||
- Future agents may want the **full unsynthesized rationale** for a
|
||||
decision the synthesis abbreviated.
|
||||
- The originals remain useful as a **time machine** for understanding
|
||||
how the design evolved.
|
||||
- These docs were never committed to git, so leaving them on disk costs
|
||||
nothing.
|
||||
|
||||
## When to read the archive vs. the synthesis
|
||||
|
||||
- **Read the synthesis (`../*.md`)** for: current state, decision
|
||||
status, action items, design rationale at the conceptual level.
|
||||
- **Read the archive (here)** for: deep historical context, exact wording
|
||||
of design decisions, full PR plan with all sub-PR commit details,
|
||||
the original Q-1..Q-9 / D-1..D-11 prose.
|
||||
|
||||
## Maintenance rule
|
||||
|
||||
Do NOT edit files in this archive. They are point-in-time snapshots.
|
||||
If new design material appears that supersedes an entry here, update the
|
||||
synthesis (the parent dir) and append a note to that synthesis file —
|
||||
do not mutate this archive.
|
||||
@@ -0,0 +1,838 @@
|
||||
# FastVideo API Refactor Design
|
||||
|
||||
## Related Documents
|
||||
- [PR plan.md](PR%20plan.md) — PR-by-PR implementation plan for this design
|
||||
- [.agents/exploration/streaming-server-upstream-plan.md](.agents/exploration/streaming-server-upstream-plan.md) — streaming-server upstream + Dynamo backend contract (shapes PRs 5.5-7.10)
|
||||
- `../FastVideo-internal/.agents/exploration/rebase-upstream-fastvideo.md` — rebasing FastVideo-internal onto upstream (enables PRs 6-8)
|
||||
- `../FastVideo-internal/ui/ltx2-streaming/` — source for the streaming server being upstreamed (PRs 7.5-7.9)
|
||||
- `../dynamo/` — local clone of ai-dynamo/dynamo; `components/src/dynamo/sglang/` is the template for FastVideo's native backend landed in PR 7.10
|
||||
- https://github.com/ai-dynamo/dynamo/pull/7544 — closed draft PR that establishes the Dynamo backend shape this design must satisfy
|
||||
|
||||
## Status
|
||||
|
||||
Design spec for the public inference API refactor. PRs 0-5.5 are landed; see [PR plan.md](PR%20plan.md) for rollout status and the PR 6+ roadmap. The typed schema, strict parser, preset system, typed VideoGenerator, typed CLI, and stateless OpenAI server default-request merge are all implemented. Streaming package skeleton + typed streaming config types are in place; live streaming server + Dynamo contract are the next milestones.
|
||||
|
||||
## Executive Summary
|
||||
|
||||
FastVideo should move to a single typed nested inference schema that is shared across:
|
||||
|
||||
- Python API
|
||||
- CLI
|
||||
- YAML/JSON config files
|
||||
- OpenAI/server request translation
|
||||
|
||||
The core split is:
|
||||
|
||||
- `GeneratorConfig`: generator-instance lifetime settings
|
||||
- `GenerationRequest`: per-call inputs, sampling, outputs, and continuation
|
||||
- `InferencePreset`: model-owned named multi-stage defaults
|
||||
|
||||
The canonical user experience should be:
|
||||
|
||||
1. Choose a model.
|
||||
2. Choose a pipeline preset.
|
||||
3. Override a few typed fields.
|
||||
4. Generate.
|
||||
|
||||
FastVideo should not make a raw free-form string dict the primary API. Dicts and YAML/JSON should be supported as serialization/interchange layers, but they must be parsed immediately into typed config objects with strict unknown-key validation.
|
||||
|
||||
The repo should also shift model-specific preset/default definitions closer to their pipeline implementations, while keeping the shared public schema and parsers centralized.
|
||||
|
||||
## Why This Refactor Is Needed
|
||||
|
||||
Today the public inference boundary is too flat and too forgiving.
|
||||
|
||||
- `VideoGenerator.from_pretrained(..., **kwargs)` mixes:
|
||||
- engine/runtime settings
|
||||
- pipeline init settings
|
||||
- component overrides
|
||||
- `VideoGenerator.generate_video(..., **kwargs)` mixes:
|
||||
- prompt and inputs
|
||||
- sampling parameters
|
||||
- output settings
|
||||
- model-specific workflow knobs
|
||||
- unknown or drifting keys can be silently filtered or merely logged instead of failing fast
|
||||
- model-specific multi-stage behavior is exposed through ad hoc top-level flags instead of a stable preset/stage abstraction
|
||||
|
||||
This is already painful in LTX2/Dreamverse, and it will get worse as more multi-stage pipelines are upstreamed.
|
||||
|
||||
## Design Goals
|
||||
|
||||
- Keep the Python API typed and editor-friendly.
|
||||
- Make YAML/JSON a first-class serialization of the same schema.
|
||||
- Support CLI overrides cleanly without flattening the schema into hundreds of canonical flags.
|
||||
- Separate init-time config from request-time config.
|
||||
- Provide a stable public abstraction for multi-stage pipelines.
|
||||
- Support LTX2 two-stage and continuation behavior cleanly.
|
||||
- Keep the simple case simple.
|
||||
- Co-locate model-owned defaults and stage topology with the relevant pipeline.
|
||||
- Protect current public/server behavior with an explicit schema parity audit before freezing the new surface.
|
||||
- Preserve backward compatibility long enough to migrate examples, internal users, and servers safely.
|
||||
|
||||
## Non-Goals
|
||||
|
||||
- Do not make Ray a structural dependency or copy its package layout.
|
||||
- Do not make a raw free-form dict the primary Python API.
|
||||
- Do not force all models into one universal `RefineConfig`.
|
||||
- Do not expose stage indices as the primary user interface.
|
||||
- Do not move every shared config class into per-model directories.
|
||||
|
||||
## External Inspiration
|
||||
|
||||
### Ray
|
||||
|
||||
Borrow only the ergonomic idea that user-facing config can be expressed as a string-keyed dict or YAML/JSON config. Do not copy Ray's structure into FastVideo.
|
||||
|
||||
### SGL Multimodal Gen
|
||||
|
||||
Useful ideas: split instance config from request config; allow dict input at the boundary; parse dicts immediately into typed request objects; merge request overrides onto model defaults; validate request params against pipeline/task type. Do not copy: request objects depending on server/engine config; broad weakly typed request bags as the canonical API.
|
||||
|
||||
### vLLM-Omni
|
||||
|
||||
Useful ideas: model-owned pipeline presets; explicit stage topology; per-stage default sampling params; clean separation between stage topology, engine defaults, and runtime overrides. Do not copy: positional `sampling_params_list` as the primary public API; serving-engine-oriented stage index semantics in the main Python interface.
|
||||
|
||||
## Core Decision
|
||||
|
||||
FastVideo should have:
|
||||
|
||||
1. A shared typed public schema.
|
||||
2. Model-owned named pipeline presets.
|
||||
3. Semantic stage overrides by stage name.
|
||||
4. Optional advanced explicit plans for power users.
|
||||
5. YAML-first config loading with dotted CLI overrides.
|
||||
|
||||
The public API should be stable at the schema level, while model-specific behavior should be contained in preset definitions and model-specific typed override classes.
|
||||
|
||||
## Schema Parity Requirement
|
||||
|
||||
Before the new schema is declared canonical, FastVideo should build a parity inventory across all current public inference surfaces (Python `VideoGenerator` kwargs, CLI flags, YAML/JSON config inputs, OpenAI/server request models, model-specific sampling/runtime fields). Each field must be marked: kept as-is, renamed, moved to a nested path, preset-owned, private-only adapter field, or intentionally dropped. No field should disappear implicitly.
|
||||
|
||||
For any public field that remains supported, there should be either a normalized-config equivalence test, or an explicit parser/translation test. Fields that exist only in private Dreamverse integration code should be handled by a private adapter layer, not quietly converted into public FastVideo compatibility guarantees.
|
||||
|
||||
Landed artifact: [inference_schema_parity_inventory.yaml](docs/design/inference_schema_parity_inventory.yaml) + guard [test_schema_parity_inventory.py](fastvideo/tests/api/test_schema_parity_inventory.py).
|
||||
|
||||
## Canonical Public Schema
|
||||
|
||||
The typed schema is implemented in [fastvideo/api/schema.py](fastvideo/api/schema.py). Envelope types:
|
||||
|
||||
- `RunConfig` — offline: `generator` (GeneratorConfig) + `request` (GenerationRequest)
|
||||
- `ServeConfig` — serving: `generator` + `server` (ServerConfig) + `default_request` (GenerationRequest) + optional `streaming` (StreamingConfig)
|
||||
|
||||
Key nested types (summary; full fields in `schema.py`):
|
||||
|
||||
- `GeneratorConfig` → `model_path`, `revision`, `trust_remote_code`, `engine` (EngineConfig: parallelism/offload/compile/quantization/flags), `pipeline` (PipelineSelection: workload_type, preset, preset_version, components, preset_overrides, experimental)
|
||||
- `GenerationRequest` → `prompt`, `negative_prompt`, `inputs` (InputConfig), `sampling` (SamplingConfig), `runtime` (RequestRuntimeConfig), `output` (OutputConfig), `stage_overrides`, `state` (ContinuationState), `plan` (GenerationPlan), `extensions`
|
||||
- `ContinuationState` → opaque `{kind: str, payload: dict[str, Any]}`
|
||||
- `GenerationPlan` → `{stages: list[PlannedStage], final_stage: str | None}`; advanced/escape-hatch only
|
||||
|
||||
### Important Semantics
|
||||
|
||||
- Dataclasses are canonical for Python users.
|
||||
- Dict and YAML/JSON are parsed into these dataclasses immediately.
|
||||
- Unknown keys must raise validation errors.
|
||||
- Typed `GenerationRequest` defaults come from the public schema, not from model-specific `SamplingParam.from_pretrained(...)` defaults.
|
||||
- Legacy `generate_video(...)` continues to inherit model-specific sampling defaults until the SSIM/performance migration lands (PR 11).
|
||||
- The only open-ended escape hatches are:
|
||||
- `generator.pipeline.experimental`
|
||||
- `request.extensions`
|
||||
|
||||
That keeps the public contract strict without blocking experimental work.
|
||||
|
||||
### Request Mutation Tracking
|
||||
|
||||
When a `GenerationRequest` is parsed from a raw dict (YAML, JSON, or Python mapping), FastVideo records which fields the user explicitly provided versus which received schema defaults. This matters because `request_to_sampling_param()` must distinguish user-provided values (which should override model defaults) from schema defaults (which should NOT override model defaults).
|
||||
|
||||
The tracking contract:
|
||||
|
||||
- At parse time, the original raw dict and a baseline snapshot of the parsed object are stored on the request.
|
||||
- Dataclass field mutations after parsing (e.g., `request.sampling.seed = 7`) are captured via lightweight `__setattr__` dirty-path recording.
|
||||
- Dict-typed field mutations (e.g., `del request.stage_overrides["refine"]`) are detected at access time by diffing the current dict against the baseline snapshot.
|
||||
- Setting a field to the schema default value IS captured as explicit, so it will override model defaults.
|
||||
- The raw dict is reconciled lazily when `normalize_generation_request()` is called, not on every individual mutation.
|
||||
|
||||
### Schema Purity and Model-Specific Fields
|
||||
|
||||
The shared schema currently contains fields that are specific to one or two model families. These remain for backward compatibility during the initial migration (PRs 0-3) but should migrate to preset-owned typed override classes as the preset system lands (PRs 4-10).
|
||||
|
||||
**SamplingConfig fields to migrate:**
|
||||
|
||||
- `height_sr`, `width_sr`, `num_inference_steps_sr`: Hunyuan15 SR only. Target: `HunyuanSRStageOverride` in PR 10.
|
||||
- `guidance_scale_2`, `boundary_ratio`: Wan2.2 and LingBotWorld only. Target: preset-owned overrides in the relevant model migration PR.
|
||||
|
||||
**InputConfig fields to migrate:**
|
||||
|
||||
- `mouse_cond`, `keyboard_cond`, `grid_sizes`: MatrixGame action control only. Target: `request.extensions` or a typed MatrixGame input config.
|
||||
- `c2ws_plucker_emb`: LingBotWorld camera control only. Target: `request.extensions` or a typed LingBotWorld input config.
|
||||
- `refine_from`, `stage1_video`: LongCat refinement only. Target: `LongCatRefineStageOverride` inputs or keep in `InputConfig` if they remain a public contract.
|
||||
|
||||
**Universal fields that stay in the shared schema:**
|
||||
|
||||
- `guidance_rescale`: used by multiple denoising stages across models, default 0.0. Universally applicable.
|
||||
- `true_cfg_scale`: OpenAI adapter surface. Keep for protocol compatibility.
|
||||
|
||||
### Escape Hatch Sunset
|
||||
|
||||
`generator.pipeline.experimental` and `request.extensions` are intentional escape hatches for experimental and private work. They bypass strict validation by design.
|
||||
|
||||
Rules for escape hatch usage:
|
||||
|
||||
- New fields should not be added to `experimental` or `extensions` without a plan to either promote them to typed fields or remove them within two PR cycles.
|
||||
- Each model migration PR (PRs 6-10) should review and shrink escape hatch usage for that model family.
|
||||
- The compatibility layer currently routes unrecognized legacy kwargs into `experimental`. This pass-through should shrink as presets absorb model-specific fields.
|
||||
|
||||
## Public Python API
|
||||
|
||||
### New Canonical API
|
||||
|
||||
```python
|
||||
from fastvideo import VideoGenerator
|
||||
from fastvideo.api import (
|
||||
GeneratorConfig, GenerationRequest,
|
||||
EngineConfig, OutputConfig,
|
||||
PipelineSelection, SamplingConfig,
|
||||
)
|
||||
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
config=GeneratorConfig(
|
||||
model_path="/models/ltx2",
|
||||
engine=EngineConfig(num_gpus=1),
|
||||
pipeline=PipelineSelection(
|
||||
workload_type="t2v",
|
||||
preset="ltx2_two_stage",
|
||||
),
|
||||
)
|
||||
)
|
||||
|
||||
result = generator.generate(
|
||||
GenerationRequest(
|
||||
prompt="a fox running through snow",
|
||||
sampling=SamplingConfig(
|
||||
num_frames=121, height=1024, width=1536,
|
||||
num_inference_steps=8, seed=42,
|
||||
),
|
||||
output=OutputConfig(save_video=True, return_state=True),
|
||||
)
|
||||
)
|
||||
```
|
||||
|
||||
### Accepted Construction Forms
|
||||
|
||||
Canonical:
|
||||
|
||||
```python
|
||||
VideoGenerator.from_pretrained(config=GeneratorConfig(...))
|
||||
VideoGenerator.from_config(GeneratorConfig(...))
|
||||
VideoGenerator.from_file("run.yaml")
|
||||
```
|
||||
|
||||
Stable convenience constructor:
|
||||
|
||||
```python
|
||||
VideoGenerator.from_pretrained("model-id")
|
||||
VideoGenerator.from_pretrained("model-id", num_gpus=2, use_fsdp_inference=False, ...)
|
||||
```
|
||||
|
||||
Legacy compatibility:
|
||||
|
||||
```python
|
||||
VideoGenerator.from_pretrained(model_path, **legacy_kwargs)
|
||||
```
|
||||
|
||||
All constructor forms normalize through the same typed path. Stable convenience kwargs remain supported with no deprecation warning. Advanced model/pipeline-specific kwargs are accepted during migration but only as compatibility inputs that normalize into `GeneratorConfig`. The thing being deprecated over time is the unbounded legacy kwarg surface, not the `from_pretrained(...)` entrypoint itself.
|
||||
|
||||
### Generation Entry Point
|
||||
|
||||
Canonical: `generator.generate(request: GenerationRequest) -> GenerationResult`.
|
||||
|
||||
Compatibility alias: `generator.generate_video(prompt=..., **legacy_kwargs)` — converts legacy calls into a `GenerationRequest` and emits a deprecation warning.
|
||||
|
||||
During the compat period, `generate(request=...)` uses schema defaults while `generate_video(...)` preserves legacy model-default behavior. These paths intentionally differ until preset-owned defaults replace the remaining `SamplingParam` default logic (migrated in PR 11).
|
||||
|
||||
### Boundary Normalization Rule
|
||||
|
||||
Every public inference entrypoint normalizes into typed config objects before touching legacy internals. That includes Python constructors, generation calls, CLI `generate`, CLI `serve`, and OpenAI/server request translation. Legacy internals (`FastVideoArgs`, `SamplingParam`) may remain temporarily, but only behind a typed normalization boundary.
|
||||
|
||||
## Pipeline Presets
|
||||
|
||||
### Definition
|
||||
|
||||
An `InferencePreset` is a named model-owned preset that defines:
|
||||
|
||||
- workload selection
|
||||
- stage topology
|
||||
- per-stage defaults
|
||||
- stage names
|
||||
- allowed stage override types
|
||||
- init-time feature requirements
|
||||
|
||||
The preset is not user-authored by default. It is supplied by the model integration.
|
||||
|
||||
### Why Presets Are The Right Abstraction
|
||||
|
||||
Users usually do not want to assemble a stage graph by hand. They want to say:
|
||||
|
||||
- use LongCat distill + refine
|
||||
- use Hunyuan 1080p SR
|
||||
- use LTX2 two-stage continuation mode
|
||||
|
||||
Presets provide a stable public noun for that behavior.
|
||||
|
||||
### Preset Naming Rules
|
||||
|
||||
- Use semantic names, not stage indices.
|
||||
- Keep names stable across releases.
|
||||
- If semantics change incompatibly, change `preset_version` or create a new preset name.
|
||||
|
||||
Examples: `ltx2_base`, `ltx2_two_stage`, `longcat_distill_refine`, `hunyuan15_sr_720p`, `hunyuan15_sr_1080p`.
|
||||
|
||||
### Preset-Owned Stage Names
|
||||
|
||||
Stage names are public and stable within a preset.
|
||||
|
||||
- LTX2: `base`, `refine`
|
||||
- LongCat: `distill`, `refine`
|
||||
- Hunyuan15: `base`, `sr_720p`, `sr_1080p`
|
||||
|
||||
Public overrides should reference these stage names, never stage indices.
|
||||
|
||||
## Stage Overrides
|
||||
|
||||
The main user override surface for multi-stage pipelines is:
|
||||
|
||||
```python
|
||||
request.stage_overrides["refine"] = ...
|
||||
```
|
||||
|
||||
Each model family should expose typed override classes for its stage names. Examples for the model families that land in PRs 6/9/10:
|
||||
|
||||
```python
|
||||
@dataclass
|
||||
class LTX2RefineStageOverride:
|
||||
enabled: bool | None = None
|
||||
num_inference_steps: int | None = None
|
||||
guidance_scale: float | None = None
|
||||
add_noise: bool | None = None
|
||||
image_crf: int | None = None
|
||||
video_position_offset_sec: float | None = None
|
||||
|
||||
@dataclass
|
||||
class LongCatRefineStageOverride:
|
||||
t_thresh: float | None = None
|
||||
spatial_refine_only: bool | None = None
|
||||
num_cond_frames: int | None = None
|
||||
|
||||
@dataclass
|
||||
class HunyuanSRStageOverride:
|
||||
num_inference_steps: int | None = None
|
||||
guidance_scale: float | None = None
|
||||
```
|
||||
|
||||
### Strictness Rules
|
||||
|
||||
- Stage names must exist in the selected preset.
|
||||
- Override fields must be valid for that stage type.
|
||||
- Unknown stage names and unknown fields must error.
|
||||
|
||||
## Advanced Explicit Plans
|
||||
|
||||
Presets should be the default API. `GenerationPlan` exists only for advanced composition or experimentation:
|
||||
|
||||
- building a custom workflow that is not yet standardized as a preset
|
||||
- debugging or benchmarking stage combinations
|
||||
- prototyping a future preset
|
||||
|
||||
Do not require `GenerationPlan` for normal users.
|
||||
|
||||
## Continuation State
|
||||
|
||||
Continuation must be a first-class part of the API.
|
||||
|
||||
### Public Contract
|
||||
|
||||
- `GenerationResult.state` may return a `ContinuationState`.
|
||||
- `GenerationRequest.state` may accept a previously returned state.
|
||||
- Most users should treat `state` as opaque and round-trip it back into the next request.
|
||||
|
||||
### Why This Matters
|
||||
|
||||
Dreamverse/LTX2 currently leaks continuation internals into app-level request fields like video conditions, audio clean latent, audio denoise mask, and segment offsets. Those should not remain top-level app-owned public API.
|
||||
|
||||
### State Design
|
||||
|
||||
Public surface:
|
||||
|
||||
```python
|
||||
@dataclass
|
||||
class ContinuationState:
|
||||
kind: str
|
||||
payload: dict[str, Any]
|
||||
```
|
||||
|
||||
Internally, FastVideo should also define typed model-specific state subclasses, e.g. `LTX2ContinuationState` (PR 7) and `LongCatIntermediateState` if ever needed. Minimal stable surface: return state, pass state back in, validate that the state is compatible with the active preset.
|
||||
|
||||
Payload serialization: fields must be JSON-serializable or use an opaque blob-ID indirection for large tensors. This supports both the stateless OpenAI client round-trip AND future Dynamo prefill/decode disaggregation where prefill yields a state that decode hydrates across workers.
|
||||
|
||||
## YAML / JSON Design
|
||||
|
||||
YAML and JSON should be exact serializations of the typed schema, not a second unrelated config system. YAML is the primary documented format. JSON is accepted with the same schema.
|
||||
|
||||
### Run Config Example
|
||||
|
||||
```yaml
|
||||
generator:
|
||||
model_path: /models/ltx2
|
||||
engine:
|
||||
num_gpus: 1
|
||||
parallelism: {tp_size: -1, sp_size: -1}
|
||||
offload: {dit: false, text_encoder: false, vae: false, pin_cpu_memory: true}
|
||||
pipeline:
|
||||
workload_type: t2v
|
||||
preset: ltx2_two_stage
|
||||
components:
|
||||
config_root: /models/ltx2-config
|
||||
upsampler_weights: /models/ltx2-refine
|
||||
lora_path: /models/ltx2-refine-lora
|
||||
preset_overrides:
|
||||
refine: {enabled: true, add_noise: true}
|
||||
|
||||
request:
|
||||
prompt: "a fox running through snow"
|
||||
sampling:
|
||||
num_frames: 121
|
||||
height: 1024
|
||||
width: 1536
|
||||
num_inference_steps: 8
|
||||
seed: 42
|
||||
output: {save_video: true, return_state: true}
|
||||
stage_overrides:
|
||||
refine: {num_inference_steps: 2, guidance_scale: 1.0}
|
||||
```
|
||||
|
||||
### Serve Config Example
|
||||
|
||||
```yaml
|
||||
generator:
|
||||
model_path: /models/ltx2
|
||||
engine: {num_gpus: 1}
|
||||
pipeline: {workload_type: t2v, preset: ltx2_two_stage}
|
||||
|
||||
server: {host: 0.0.0.0, port: 8000, output_dir: outputs/}
|
||||
|
||||
default_request:
|
||||
sampling: {num_frames: 121, height: 1024, width: 1536, num_inference_steps: 8}
|
||||
output: {save_video: false, return_frames: false}
|
||||
```
|
||||
|
||||
### Validation Rules
|
||||
|
||||
- top-level schema must match `RunConfig` or `ServeConfig`
|
||||
- unknown keys must fail
|
||||
- dotted CLI overrides are applied to the nested config before typed parsing
|
||||
- parse errors must include the exact nested path that failed
|
||||
|
||||
## CLI Design
|
||||
|
||||
Inference CLI reuses the best parts of the current training authoring flow (YAML-first authoring, dotted nested overrides, typed parsing after merge) but stays stricter than training at the public boundary because it is a user-facing API surface for Python, CLI, YAML/JSON, and serving.
|
||||
|
||||
### Canonical CLI Forms
|
||||
|
||||
```bash
|
||||
fastvideo generate --config run.yaml
|
||||
fastvideo generate --config run.yaml --request.sampling.seed 42
|
||||
fastvideo generate --config run.yaml --generator.engine.num_gpus 2
|
||||
|
||||
fastvideo serve --config serve.yaml
|
||||
fastvideo serve --config serve.yaml --server.port 8090
|
||||
```
|
||||
|
||||
The CLI is config-only. Beyond `--config`, CLI input uses dotted override paths into the nested schema rather than maintaining a second flat flag surface.
|
||||
|
||||
Implementation: YAML/JSON is loaded into a nested dict, dotted CLI overrides are applied to the nested dict, then the result is parsed into typed config objects. Flat CLI flags are rejected so the nested schema stays canonical.
|
||||
|
||||
## OpenAI / Server Mapping
|
||||
|
||||
`fastvideo serve` loads `ServeConfig`. Incoming HTTP requests are translated into `GenerationRequest` by:
|
||||
|
||||
1. cloning `default_request`
|
||||
2. applying API request fields onto that request
|
||||
3. validating against the selected preset
|
||||
|
||||
This is similar in spirit to the SGL pattern of merging user overrides onto model defaults.
|
||||
|
||||
Rules:
|
||||
|
||||
- HTTP request translation must not bypass typed validation.
|
||||
- multi-stage defaults should come from the preset and `default_request`, not from ad hoc server-local logic.
|
||||
- stateful continuation requests should accept and return typed `ContinuationState` payloads.
|
||||
|
||||
Landed in PR 5 for the stateless OpenAI server at `fastvideo/entrypoints/openai/`. The streaming/session server (PRs 7.5-7.9) uses the same preset/default_request merge through `ServeConfig.streaming`.
|
||||
|
||||
## Streaming Server + Dynamo Backend
|
||||
|
||||
The typed public API is consumed by three server-class integrations. They must share one execution substrate so we don't grow three near-duplicate progress loops.
|
||||
|
||||
### The three consumers
|
||||
|
||||
| Consumer | Transport | Request shape | State |
|
||||
|---|---|---|---|
|
||||
| Stateless OpenAI (`fastvideo/entrypoints/openai/`) | HTTP POST | `GenerationRequest` merged onto `ServeConfig.default_request` | Stateless; continuation via opaque payload if needed |
|
||||
| Streaming WebSocket (`fastvideo/entrypoints/streaming/`) | WebSocket JSON + binary fMP4 | `GenerationRequest` per segment, session-scoped | Server-held session (per-GPU continuation cache); snapshot on demand |
|
||||
| Dynamo native backend (`ai-dynamo/dynamo/components/src/dynamo/fastvideo/`) | Dynamo RPC endpoint | `NvCreateVideoRequest` ↔ adapter ↔ `GenerationRequest` | Aggregated today; disaggregated prefill/decode later via `ContinuationState` |
|
||||
|
||||
### Shared execution substrate: `VideoGenerator.generate_async`
|
||||
|
||||
The OpenAI server, streaming server, and Dynamo backend all want the same thing: a typed async API that yields progress events and a typed final result. FastVideo exposes exactly one canonical entry point:
|
||||
|
||||
```python
|
||||
async def generate_async(
|
||||
self,
|
||||
request: GenerationRequest,
|
||||
) -> AsyncGenerator[VideoEvent, None]: ...
|
||||
```
|
||||
|
||||
Events:
|
||||
|
||||
```python
|
||||
@dataclass
|
||||
class VideoProgressEvent:
|
||||
step: int
|
||||
total_steps: int
|
||||
stage: str # "denoise" | "refine" | "decode" | ...
|
||||
|
||||
@dataclass
|
||||
class VideoPartialEvent:
|
||||
frames: np.ndarray # shape: (num_frames, H, W, 3)
|
||||
index: int # monotonic chunk index
|
||||
|
||||
@dataclass
|
||||
class VideoFinalEvent:
|
||||
video_bytes: bytes | None # mp4-encoded if requested
|
||||
tensor: torch.Tensor | None # raw if requested
|
||||
metadata: dict[str, Any]
|
||||
continuation_state: ContinuationState | None
|
||||
|
||||
VideoEvent = VideoProgressEvent | VideoPartialEvent | VideoFinalEvent
|
||||
```
|
||||
|
||||
The sync `generate_video(request=...) -> VideoResult` becomes a thin `asyncio.run` wrapper over `generate_async` that collects events and returns the final.
|
||||
|
||||
### Streaming server mapping
|
||||
|
||||
`fastvideo/entrypoints/streaming/` owns per-session state:
|
||||
|
||||
- `SessionStore.hydrate(state: ContinuationState) -> session_id`
|
||||
- `SessionStore.snapshot(session_id) -> ContinuationState`
|
||||
- Per-GPU implicit continuation cache (today's internal behavior) is wrapped as a `SessionStore` implementation.
|
||||
|
||||
Per-segment, the session writes a `GenerationRequest`, pipes the event stream to the WebSocket (progress → JSON messages, partial → fMP4 frames), and persists the final's `ContinuationState` into the session.
|
||||
|
||||
### Dynamo backend mapping
|
||||
|
||||
Dynamo's backend pattern (from `components/src/dynamo/sglang/`) is a pure Python import. FastVideo does not host a `fastvideo/entrypoints/dynamo/` subpackage; the integration lives in the Dynamo repo. FastVideo exposes a stable contract:
|
||||
|
||||
| Surface | Exposed as |
|
||||
|---|---|
|
||||
| Construction | `VideoGenerator.from_pretrained(model_path, **typed_kwargs)` |
|
||||
| Execution (async) | `VideoGenerator.generate_async(request) -> AsyncGenerator[VideoEvent, None]` |
|
||||
| Execution (sync) | `VideoGenerator.generate_video(request=...) -> VideoResult` |
|
||||
| Typed request | `fastvideo.api.GenerationRequest`, `SamplingConfig`, `InputConfig` |
|
||||
| Typed result | `fastvideo.api.VideoResult`, `VideoEvent`, `ContinuationState` |
|
||||
| Health-check input | `VideoGenerator.default_health_check_request() -> GenerationRequest` |
|
||||
| Config dump | `config_to_dict(cfg)` (already exists) |
|
||||
|
||||
Request/response mapping the Dynamo adapter must perform:
|
||||
|
||||
```
|
||||
NvCreateVideoRequest -> fastvideo.api.GenerationRequest
|
||||
prompt -> sampling.prompt
|
||||
size="WxH" -> sampling.width, sampling.height
|
||||
seconds -> seconds * nvext.fps -> sampling.num_frames
|
||||
input_reference -> input.image_path | input.video_path
|
||||
nvext.fps -> sampling.fps
|
||||
nvext.num_frames -> sampling.num_frames (overrides seconds*fps)
|
||||
nvext.num_inference_steps -> sampling.num_inference_steps
|
||||
nvext.guidance_scale -> sampling.guidance_scale
|
||||
nvext.seed -> sampling.seed
|
||||
nvext.negative_prompt -> sampling.negative_prompt
|
||||
response_format -> (handled at the adapter's output stage)
|
||||
|
||||
VideoFinalEvent -> NvVideosResponse
|
||||
video_bytes -> data[0].b64_json (if response_format=b64_json)
|
||||
uploaded URL -> data[0].url (if response_format=url)
|
||||
metadata.inference_time_s -> inference_time_s
|
||||
continuation_state -> (reserved for future disaggregation)
|
||||
```
|
||||
|
||||
All fields already exist (or will exist after PR 6's typed-kwarg expansion) on FastVideo's typed schema. **The adapter lives entirely in the Dynamo repo** at `components/src/dynamo/fastvideo/` — FastVideo does not host any Dynamo subpackage, dep, or CLI. The only FastVideo obligation is the stable public Python API listed above.
|
||||
|
||||
### Constraints this places on other sections
|
||||
|
||||
- **Continuation State** (see earlier section): `ContinuationState.payload` must be JSON-serializable or use an opaque blob-ID indirection for large tensors. This supports both the stateless OpenAI client round-trip *and* future Dynamo prefill/decode disaggregation, where prefill yields a state that decode hydrates across workers.
|
||||
- **Typed GeneratorConfig** (see Public Python API): every flat legacy LTX2 kwarg currently used by the internal `gpu_pool.py` must have a typed home reachable from `GeneratorConfig`. Dynamo's `FastVideoArgGroup` builds the config from its CLI and must not have to know any legacy LTX2 name.
|
||||
- **Public exports**: `from fastvideo import VideoGenerator`; `from fastvideo.api import GenerationRequest, SamplingConfig, ContinuationState, VideoResult, VideoEvent, VideoProgressEvent, VideoPartialEvent, VideoFinalEvent`.
|
||||
|
||||
## Repo Layout
|
||||
|
||||
### Shared Public API
|
||||
|
||||
`fastvideo/api/` contains the shared public API package. Current files:
|
||||
|
||||
- `schema.py` — `RunConfig`, `ServeConfig`, `ServerConfig`, `GeneratorConfig`, and all nested typed config dataclasses
|
||||
- `sampling_param.py` — `SamplingParam` + `CacheParams` (canonical home since PR 4; former `configs/sample/base.py` location removed)
|
||||
- `presets.py` — `InferencePreset`, `PresetStageSpec`, registry APIs
|
||||
- `results.py` — `GenerationResult` / `VideoResult`
|
||||
- `parser.py` — `from_dict`, `to_dict`, `load_yaml`, `load_json`, validation
|
||||
- `overrides.py` — dotted override application
|
||||
- `compat.py` — legacy Python kwargs translation
|
||||
- `errors.py` — path-aware validation errors
|
||||
|
||||
May split further by concern in a future cleanup.
|
||||
|
||||
### Pipeline-Local Model-Owned Config
|
||||
|
||||
Model-owned presets and override types live next to the model pipeline:
|
||||
|
||||
```text
|
||||
fastvideo/pipelines/basic/ltx2/
|
||||
ltx2_pipeline.py, presets.py, stage_overrides.py, continuation.py
|
||||
|
||||
fastvideo/pipelines/basic/longcat/
|
||||
longcat_pipeline.py, presets.py, stage_overrides.py
|
||||
|
||||
fastvideo/pipelines/basic/hunyuan15/
|
||||
hunyuan15_pipeline.py, hunyuan15_sr_pipeline.py, hunyuan15_2sr_pipeline.py,
|
||||
presets.py, stage_overrides.py
|
||||
```
|
||||
|
||||
PR 4 landed `presets.py` for all 13 model families. Remaining colocation targets are `pipeline_configs.py` (moving `configs/pipelines/<family>.py`) and model-specific stages (moving `pipelines/stages/<family>_*.py`); see [PR plan.md](PR%20plan.md) "Pipeline Package Structure".
|
||||
|
||||
### Registry
|
||||
|
||||
Central registry (`fastvideo/registry.py`) registers preset providers rather than owning all model-specific defaults directly. It answers:
|
||||
|
||||
- which pipeline class corresponds to a model path
|
||||
- which presets are available for that model family
|
||||
- which override/state classes are valid for a selected preset
|
||||
|
||||
## Relationship To Current Internal Classes
|
||||
|
||||
This refactor does not require deleting current internals immediately.
|
||||
|
||||
- `FastVideoArgs` is an internal compatibility/input adapter, no longer the primary public inference type.
|
||||
- `SamplingParam` now lives in `fastvideo/api/sampling_param.py` and gets model-specific defaults from presets via `_from_preset()`. All 12 `SamplingParam` subclasses have been removed and the former `fastvideo/configs/sample/` directory has been deleted entirely (PR 4). It remains an internal adapter between the preset system and the runtime.
|
||||
- current `PipelineConfig` classes can remain temporarily as internal component config carriers
|
||||
- the new public schema is the stable boundary above them
|
||||
|
||||
`VideoGenerator` accepts the new schema and translates down into current execution internals. Legacy `generate_video(..., **kwargs)` stays on the direct execution path during the compat period until SSIM/performance tests migrate in PR 11.
|
||||
|
||||
## Model-Specific Design
|
||||
|
||||
### LTX2 / Dreamverse
|
||||
|
||||
LTX2 needs both:
|
||||
|
||||
- init-time two-stage feature wiring
|
||||
- request-time continuation/refine behavior
|
||||
|
||||
Expressed as:
|
||||
|
||||
- preset: `ltx2_two_stage`
|
||||
- init-time fields: refine assets, optional config root, stage enablement
|
||||
- request-time fields: stage override for refine behavior, optional returned continuation state
|
||||
|
||||
#### LTX2 Preset Example
|
||||
|
||||
```yaml
|
||||
generator:
|
||||
pipeline:
|
||||
preset: ltx2_two_stage
|
||||
components:
|
||||
config_root: /models/ltx2-config
|
||||
upsampler_weights: /models/ltx2-refine
|
||||
lora_path: /models/ltx2-refine-lora
|
||||
preset_overrides:
|
||||
refine: {enabled: true, add_noise: true}
|
||||
```
|
||||
|
||||
#### LTX2 Request Example
|
||||
|
||||
```yaml
|
||||
request:
|
||||
prompt: "continue the previous sequence"
|
||||
state: ${previous_result.state}
|
||||
stage_overrides:
|
||||
refine:
|
||||
num_inference_steps: 2
|
||||
guidance_scale: 1.0
|
||||
image_crf: 18
|
||||
output:
|
||||
return_state: true
|
||||
```
|
||||
|
||||
#### LTX2 Explicit Decisions
|
||||
|
||||
- `config_model_path` becomes `generator.pipeline.components.config_root`
|
||||
- `ltx2_refine_*` stops being a pile of top-level kwargs
|
||||
- continuation internals move into `ContinuationState`
|
||||
- app-level code should pass `state`, not raw latent/audio condition payloads
|
||||
|
||||
### LongCat
|
||||
|
||||
LongCat should expose a named preset like `longcat_distill_refine` with stage topology `distill` and `refine`.
|
||||
|
||||
User-facing override knobs remain model-specific (`t_thresh`, `spatial_refine_only`, `num_cond_frames`) but live under:
|
||||
|
||||
```yaml
|
||||
request:
|
||||
stage_overrides:
|
||||
refine:
|
||||
t_thresh: 0.5
|
||||
spatial_refine_only: false
|
||||
num_cond_frames: 8
|
||||
```
|
||||
|
||||
### Hunyuan 1.5 SR
|
||||
|
||||
Hunyuan already behaves like an integrated multi-stage pipeline. Expose it via presets: `hunyuan15_sr_720p`, `hunyuan15_sr_1080p`. Users should not need to know the exact internal pipeline class split between base and SR stages. Per-stage override surface should stay small and mostly sampling-focused.
|
||||
|
||||
Hunyuan15 presets (`hunyuan15_t2v_480p`, `hunyuan15_i2v_480p_distilled`, `hunyuan15_t2v_720p`, `hunyuan15_i2v_720p_distilled`, `hunyuan15_sr_1080p`) are implemented (PR 4). The `Hunyuan15_*_SamplingParam` subclasses have been removed; defaults (including precomputed sigmas) come from preset `defaults` dicts. Remaining work: adding typed `HunyuanSRStageOverride` classes and colocating PipelineConfig (PR 10).
|
||||
|
||||
## Exact Compatibility Mapping
|
||||
|
||||
Intended translation layer for common current fields.
|
||||
|
||||
| Legacy Field | New Path |
|
||||
| --- | --- |
|
||||
| `model_path` | `generator.model_path` |
|
||||
| `revision` | `generator.revision` |
|
||||
| `trust_remote_code` | `generator.trust_remote_code` |
|
||||
| `workload_type` | `generator.pipeline.workload_type` |
|
||||
| `num_gpus` | `generator.engine.num_gpus` |
|
||||
| `tp_size` | `generator.engine.parallelism.tp_size` |
|
||||
| `sp_size` | `generator.engine.parallelism.sp_size` |
|
||||
| `dit_cpu_offload` | `generator.engine.offload.dit` |
|
||||
| `dit_layerwise_offload` | `generator.engine.offload.dit_layerwise` |
|
||||
| `text_encoder_cpu_offload` | `generator.engine.offload.text_encoder` |
|
||||
| `image_encoder_cpu_offload` | `generator.engine.offload.image_encoder` |
|
||||
| `vae_cpu_offload` | `generator.engine.offload.vae` |
|
||||
| `pin_cpu_memory` | `generator.engine.offload.pin_cpu_memory` |
|
||||
| `enable_torch_compile` | `generator.engine.compile.enabled` |
|
||||
| `torch_compile_kwargs` | split across `generator.engine.compile.backend`, `.fullgraph`, `.mode`, `.dynamic`; uncommon keys land in `.extras` |
|
||||
| `enable_torch_compile_text_encoder` | `generator.engine.compile.text_encoder_enabled` |
|
||||
| `enable_stage_verification` | `generator.engine.enable_stage_verification` |
|
||||
| `prompt_txt` | `request.inputs.prompt_path` |
|
||||
| `prompt` | `request.prompt` |
|
||||
| `negative_prompt` | `request.negative_prompt` |
|
||||
| `image_path` | `request.inputs.image_path` |
|
||||
| `video_path` | `request.inputs.video_path` |
|
||||
| `output_path` | `request.output.output_path` |
|
||||
| `output_video_name` | `request.output.output_video_name` |
|
||||
| `save_video` | `request.output.save_video` |
|
||||
| `return_frames` | `request.output.return_frames` |
|
||||
| `num_videos_per_prompt` | `request.sampling.num_videos_per_prompt` |
|
||||
| `seed` | `request.sampling.seed` |
|
||||
| `num_frames` | `request.sampling.num_frames` |
|
||||
| `height` | `request.sampling.height` |
|
||||
| `width` | `request.sampling.width` |
|
||||
| `fps` | `request.sampling.fps` |
|
||||
| `num_inference_steps` | `request.sampling.num_inference_steps` |
|
||||
| `guidance_scale` | `request.sampling.guidance_scale` |
|
||||
| `guidance_scale_2` | `request.sampling.guidance_scale_2` |
|
||||
| `guidance_rescale` | `request.sampling.guidance_rescale` |
|
||||
| `true_cfg_scale` | `request.sampling.true_cfg_scale` |
|
||||
| `boundary_ratio` | `request.sampling.boundary_ratio` |
|
||||
| `sigmas` | `request.sampling.sigmas` |
|
||||
| `enable_teacache` | `request.runtime.enable_teacache` |
|
||||
| `return_trajectory_latents` | `request.runtime.return_trajectory_latents` |
|
||||
| `return_trajectory_decoded` | `request.runtime.return_trajectory_decoded` |
|
||||
|
||||
### Private Dreamverse Adapter Mapping
|
||||
|
||||
The mappings below are useful for private Dreamverse migration, but they should not be treated as a public FastVideo backward-compatibility promise unless and until those fields actually exist in the public repo surfaces.
|
||||
|
||||
| Private Adapter Field | New Path |
|
||||
| --- | --- |
|
||||
| `config_model_path` | `generator.pipeline.components.config_root` |
|
||||
| `ltx2_refine_enabled` | `generator.pipeline.preset_overrides.refine.enabled` |
|
||||
| `ltx2_refine_upsampler_path` | `generator.pipeline.components.upsampler_weights` |
|
||||
| `ltx2_refine_lora_path` | `generator.pipeline.components.lora_path` |
|
||||
| `ltx2_refine_num_inference_steps` | `request.stage_overrides.refine.num_inference_steps` |
|
||||
| `ltx2_refine_guidance_scale` | `request.stage_overrides.refine.guidance_scale` |
|
||||
| `ltx2_refine_add_noise` | `generator.pipeline.preset_overrides.refine.add_noise` |
|
||||
| `ltx2_image_crf` | `request.stage_overrides.refine.image_crf` |
|
||||
| `return_continuation_state` | `request.output.return_state` |
|
||||
|
||||
### LongCat Legacy Mapping
|
||||
|
||||
| Legacy Field | New Path |
|
||||
| --- | --- |
|
||||
| `refine_from` | `request.inputs.refine_from` |
|
||||
| `stage1_video` | `request.inputs.stage1_video` |
|
||||
| `t_thresh` | `request.stage_overrides.refine.t_thresh` |
|
||||
| `spatial_refine_only` | `request.stage_overrides.refine.spatial_refine_only` |
|
||||
| `num_cond_frames` | `request.stage_overrides.refine.num_cond_frames` |
|
||||
|
||||
## Validation and Error Handling
|
||||
|
||||
### Strict by Default
|
||||
|
||||
All structured inputs should be strict by default: unknown keys error, wrong types error, invalid stage names error, incompatible state/preset combinations error.
|
||||
|
||||
### Exceptions
|
||||
|
||||
The only intentionally open-ended fields are `generator.pipeline.experimental` and `request.extensions`. These must be clearly documented as unstable and unsupported for long-term API compatibility.
|
||||
|
||||
### Error Quality
|
||||
|
||||
Validation errors should include the full nested path, expected type or valid choices, and preset/stage context when relevant:
|
||||
|
||||
```text
|
||||
Invalid field: request.stage_overrides.refine.num_inference_steps
|
||||
Expected int, got "two"
|
||||
Preset: ltx2_two_stage
|
||||
Stage: refine
|
||||
```
|
||||
|
||||
## Implementation Plan
|
||||
|
||||
### Phases 0-5: Landed
|
||||
|
||||
- Phase 0 — Schema Parity Inventory: inventory complete; field classifications live in `docs/design/inference_schema_parity_inventory.yaml`; parity test guard in `fastvideo/tests/api/test_schema_parity_inventory.py`.
|
||||
- Phase 1 — Shared Schema: `fastvideo/api/` with typed dataclasses, parser, validation, dotted overrides, `RunConfig`/`ServeConfig`.
|
||||
- Phase 2 — VideoGenerator Compat: `from_config`, `from_file`, `generate(request=...)`, legacy `from_pretrained(..., **kwargs)` and `generate_video(..., **kwargs)` as compat shims routed through typed normalization.
|
||||
- Phase 3 — CLI Refactor: `fastvideo generate` and `fastvideo serve` parse nested YAML/JSON with training-style dotted overrides; flat flag expansion removed as the canonical path.
|
||||
- Phase 4 — Preset System: shared registry + pipeline-local `presets.py` for all 13 families; all 12 `SamplingParam` subclasses removed; `SamplingParam` moved to `fastvideo/api/sampling_param.py`.
|
||||
- Phase 5 — Server Request Translation: `fastvideo serve` loads `ServeConfig`; stateless OpenAI endpoint clones `default_request` and merges validated user overrides.
|
||||
|
||||
### Remaining Phases
|
||||
|
||||
- **Phase 6 — LTX2 Public Upstream Path** (PR 6): upstream `ltx2_two_stage` preset; upstream continuation-state contract; upstream only repo-visible/public LTX2 surfaces into FastVideo.
|
||||
- **Phase 7 — Dreamverse Adapter Migration** (PR 7 + private repo work): translate private Dreamverse-only request/config fields in a private adapter; replace raw app-owned continuation kwargs with `state` in the private server; do not expand the public FastVideo compatibility promise just to match private adapter fields.
|
||||
- **Phase 7.5-7.10 — Streaming Server and Dynamo Contract** (PRs 7.5-7.10): upstream the streaming server (skeleton, GPU pool, prompt enhancer, auxiliaries, router) consuming `generate_async`; land the Dynamo backend contract (`VideoGenerator.generate_async`, health-check helper) with the Dynamo backend package itself living in the Dynamo repo.
|
||||
- **Phase 8 — Model Migration and Docs** (PRs 9-10, 12): colocate `configs/pipelines/<family>.py` with pipeline implementations; add typed stage override classes for multi-stage models; update basic examples to the new API; document YAML-first inference config and migration guidance.
|
||||
- **Phase 8.5 — Golden-Test Migration** (PR 11): keep SSIM/performance regression tests on legacy Python generation while preset defaults are still settling; one dedicated migration pass after the preset system and model-default behavior are stable; complete this migration before removing legacy Python inference entrypoints or kwargs.
|
||||
- **Phase 9 — Deprecation and Cleanup** (PR 13): deprecate direct public use of `FastVideoArgs`; deprecate direct public use of `SamplingParam`; gradually reduce public documentation for flat flags; eventually remove legacy kwargs after downstream migration is complete.
|
||||
|
||||
## Final Recommendation
|
||||
|
||||
The public FastVideo inference API is being rebuilt around:
|
||||
|
||||
- typed nested configs
|
||||
- model-owned named presets
|
||||
- semantic stage overrides
|
||||
- first-class continuation state
|
||||
- YAML-first CLI with dotted overrides
|
||||
|
||||
The primary abstraction is `InferencePreset`, not raw kwargs and not a fully manual stage graph.
|
||||
|
||||
The repo is moving model-specific defaults closer to each pipeline, while keeping the public schema and parsing logic centralized.
|
||||
|
||||
Regression and quality tests follow the rollout. Unit/entrypoint tests migrated to the typed API early, but SSIM/performance suites only move once the typed path can express all current knobs without compatibility exceptions and produces stable defaults through presets (PR 11).
|
||||
|
||||
End state:
|
||||
|
||||
- stable Python typing
|
||||
- clean YAML/JSON support
|
||||
- a much better CLI story
|
||||
- a sane path for Dreamverse/LTX2
|
||||
- a unified abstraction for LongCat, Hunyuan, and future multi-stage models
|
||||
@@ -0,0 +1,285 @@
|
||||
# Dreamverse ↔ FastVideo Integration
|
||||
|
||||
## Status
|
||||
|
||||
Working integration record. Captures how Dreamverse consumes the
|
||||
FastVideo public API today, what's already shared, what's still ad
|
||||
hoc, and what migrations land alongside each PR in the API refactor
|
||||
sequence.
|
||||
|
||||
Pinned versions (last reconciled this session):
|
||||
|
||||
| Repo | Branch | Commit | Note |
|
||||
|---|---|---|---|
|
||||
| FastVideo (public) | `origin/main` | `70ee5d23` | PR 6 merged |
|
||||
| FastVideo (public) | `will/api_7` | `3de5f833` | PR 7 in flight (typed continuation state) |
|
||||
| FastVideo-internal | `will/rebase-nbv` | `1adc513e` | pre-PR-1 on the API refactor; has live realtime runtime |
|
||||
| Dreamverse | `master` | `dc500330` | uses local + remote FastVideo runtimes via `server/runtime/` |
|
||||
|
||||
## Related Documents
|
||||
|
||||
- [PR plan.md](../../PR%20plan.md) — PR-by-PR sequence for the API refactor
|
||||
- [apirefactor.md](../../apirefactor.md) — design spec
|
||||
- [streaming-server-upstream-plan.md](streaming-server-upstream-plan.md) — upstream plan for `ui/ltx2-streaming/server/`
|
||||
- `../../../Dreamverse/server/video_generation.py` — Dreamverse's worker + local `ContinuationState`
|
||||
- `../../../Dreamverse/server/runtime/{factory,backend,gpu_pool,interfaces}.py` — runtime abstraction
|
||||
- `../../../FastVideo-internal/fastvideo/entrypoints/realtime/{api_server,local_runtime}.py` — internal's realtime runtime (PR 7.5/7.6 upstream source)
|
||||
|
||||
## Surface Area
|
||||
|
||||
Dreamverse depends on FastVideo across three surfaces. Listed in order
|
||||
of how stable each is.
|
||||
|
||||
### 1. Pipeline construction (stable)
|
||||
|
||||
`Dreamverse/server/video_generation.py:VideoGenerationWorker` calls
|
||||
`VideoGenerator.from_pretrained(...)` with flat LTX-2 kwargs today.
|
||||
After PR 6 the typed `GeneratorConfig` path exists; Dreamverse can
|
||||
migrate at its own pace.
|
||||
|
||||
| Dreamverse usage | FastVideo public surface (post-PR 6) |
|
||||
|---|---|
|
||||
| `VideoGenerator.from_pretrained(model_path, ltx2_refine_enabled=…, …)` | `VideoGenerator.from_pretrained(config=GeneratorConfig(...))` |
|
||||
| Flat `torch_compile_kwargs={…}` dict | `engine.compile.{backend,fullgraph,mode,dynamic,extras}` |
|
||||
| `ltx2_vae_tiling=True` | `pipeline.vae_tiling=True` |
|
||||
| `ltx2_refine_*` family | `pipeline.preset_overrides.refine.*` + `pipeline.components.upsampler_weights` |
|
||||
| `enable_torch_compile_text_encoder` | `engine.compile.text_encoder_enabled` |
|
||||
|
||||
The legacy flat-kwarg path stays supported via `compat.py`; migration
|
||||
is opt-in. PR 13's deprecation warnings are the eventual nudge.
|
||||
|
||||
### 2. Realtime runtime (in flight: PRs 7.5–7.6)
|
||||
|
||||
`Dreamverse/server/runtime/factory.py` selects a runtime backend at
|
||||
process start:
|
||||
|
||||
```python
|
||||
def create_runtime_pool() -> RuntimePool:
|
||||
if os.getenv("FASTVIDEO_REALTIME_BASE_URL"):
|
||||
return FastVideoRealtimePool(base_url=..., ws_url=..., default_model_id=...)
|
||||
return GPUPool(get_available_gpus()) # in-process, wraps fastvideo.entrypoints.realtime.local_runtime
|
||||
```
|
||||
|
||||
Both backends speak the same `RuntimePool` / `RuntimeSlot` Protocol
|
||||
(`server/runtime/interfaces.py`):
|
||||
|
||||
- `acquire(client_id, websocket=None) -> (gpu_id, RuntimeSlot)`
|
||||
- `release(client_id)`
|
||||
- `RuntimeSlot.{join_user, user_step, leave_user, register_stream_queue, …}`
|
||||
|
||||
Today both impls reach into FastVideo-internal's
|
||||
`fastvideo.entrypoints.realtime.local_runtime` (which exposes
|
||||
`RealtimeRuntimeConfig`, `GPUPool`, `GPUSlot`). The remote backend
|
||||
talks HTTP+WS to a separately-deployed runtime of the same shape.
|
||||
|
||||
**Contract that PR 7.5/7.6 must preserve:**
|
||||
|
||||
- `RealtimeRuntimeConfig` accepts `model_registry`, `default_model_id`,
|
||||
`default_height/width/num_frames/fps/num_inference_steps/guidance_scale/seed/negative_prompt`,
|
||||
`default_ltx2_image_crf`, `startup_warmup_{enabled,prompt,timeout_seconds}`.
|
||||
- `GPUPool(gpu_ids: list[int], config: RealtimeRuntimeConfig)` constructor.
|
||||
- `pool.initialize() / shutdown() / acquire() / release() / get_status()`.
|
||||
- HTTP endpoints on the remote variant: `GET /healthz`, `GET /readyz`,
|
||||
`GET /status`, `WS /ws`. (These already match what
|
||||
`Dreamverse/server/routes/health.py` consumes.)
|
||||
|
||||
When PR 7.6 lands the upstream of `fastvideo/entrypoints/realtime/`,
|
||||
Dreamverse should not need any code change unless we rename the import
|
||||
path. **Open: do we rename `realtime/` → `streaming/` to match the
|
||||
public package introduced in PR 5.5?** A deprecation alias module
|
||||
keeps both working during transition.
|
||||
|
||||
### 3. Continuation state (PR 7)
|
||||
|
||||
`Dreamverse/server/video_generation.py:89 ContinuationState` is
|
||||
Dreamverse's hand-rolled per-session state holder. PR 7 introduces
|
||||
the typed equivalent at `fastvideo/pipelines/basic/ltx2/continuation.py`.
|
||||
|
||||
#### Field mapping
|
||||
|
||||
| Dreamverse | PR 7 `LTX2ContinuationState` | Notes |
|
||||
|---|---|---|
|
||||
| `video_images: list[PIL.Image]` | `video_frames: list[np.ndarray]` (uint8 H×W×3) | numpy is leaner; Dreamverse already round-trips PIL→numpy→PIL just to add noise |
|
||||
| `audio_latents: torch.Tensor` `[B, C, T, mel]` | `audio_latents: torch.Tensor` (safetensors-serialized; bf16-safe) | unchanged shape; safetensors preserves dtype incl. `bfloat16` |
|
||||
| `LTX2_VIDEO_CONDITIONING_FRAME_IDX` (env) | `video_conditioning_frame_idx: int` | env constant → per-state field |
|
||||
| `LTX2_VIDEO_CONDITIONING_STRENGTH` (env) | `video_conditioning_strength: float` | env constant → per-state field |
|
||||
| `AUDIO_CONDITIONING_NUM_FRAMES` (env) | `audio_conditioning_num_frames: int` | env constant → per-state field |
|
||||
| `AUDIO_CONDITIONING_STRENGTH` (env) | `audio_conditioning_strength: float` | env constant → per-state field |
|
||||
| `audio_lps` (passed into `apply_audio`) | `audio_sample_rate: int \| None` | analogous; rename worth confirming with audio team |
|
||||
| Computed `prefix_sec` per segment | `video_position_offset_sec: float` | **see open question below** |
|
||||
| `segment_idx` (param to apply_*) | `segment_index: int` | per-state field |
|
||||
| `VIDEO_CONTEXT_NOISE`, `AUDIO_CONTEXT_NOISE`, `ENABLE_AUDIO_COND` | not on state | runtime policy / regularization knobs, not portable session data |
|
||||
| `apply_video / apply_audio / save_video / save_audio_latents / clear` | not on PR-7 state class | state is a pure data carrier; runtime owns lifecycle policy |
|
||||
|
||||
PR-7 is a strict superset of Dreamverse's data model **plus** lifts
|
||||
several env globals into per-session typed fields.
|
||||
|
||||
#### Lifecycle mapping
|
||||
|
||||
| Dreamverse pattern | `SessionStore` API |
|
||||
|---|---|
|
||||
| `self.continuation = ContinuationState()` per session | `state = session_store.snapshot(sid) or LTX2ContinuationState()` |
|
||||
| `apply_video(req_kwargs, segment_idx)` + `apply_audio(req_kwargs, segment_idx, audio_lps)` | `state = session_store.snapshot(sid)`; runtime builds request from `state.video_frames` / `state.audio_latents` etc. |
|
||||
| `save_video(frames)` + `save_audio_latents(latents)` | runtime constructs new `LTX2ContinuationState`, then `session_store.store(sid, new_state.to_continuation_state())` |
|
||||
| `clear()` at end of session | `session_store.drop(sid)` |
|
||||
|
||||
`SessionStore` and `BlobStore` ABCs ship with thread-safe in-memory
|
||||
defaults (`InMemorySessionStore`, `InMemoryBlobStore`). Dreamverse can
|
||||
adopt them as-is for the local runtime; remote runtimes can plug in
|
||||
redis-backed implementations later.
|
||||
|
||||
#### Wire format (HTTP/WS round-trip)
|
||||
|
||||
Dreamverse's `FastVideoRealtimePool` already speaks the realtime
|
||||
runtime's HTTP+WS protocol. When PR 7.5/7.6 land state emission on
|
||||
the server side, the on-the-wire payload is the public envelope:
|
||||
|
||||
```json
|
||||
{
|
||||
"kind": "ltx2.v1",
|
||||
"payload": {
|
||||
"schema_version": 1,
|
||||
"segment_index": 3,
|
||||
"video_conditioning_frame_idx": 9,
|
||||
"video_conditioning_strength": 0.75,
|
||||
"audio_sample_rate": 24000,
|
||||
"audio_conditioning_num_frames": 5,
|
||||
"audio_conditioning_strength": 0.5,
|
||||
"video_position_offset_sec": 0.2,
|
||||
"video": {"frames_b64": ["..."]},
|
||||
"audio": {"safetensors_b64": "..."},
|
||||
"metadata": {}
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
JSON-serializable end-to-end; safetensors blob preserves audio dtype
|
||||
(incl. bf16). For payloads above the inline threshold a `BlobStore`
|
||||
indirection replaces the b64-encoded body with `{"blob_id": "..."}`;
|
||||
the blob itself stays inside the runtime that produced it.
|
||||
|
||||
## Migration Plan
|
||||
|
||||
Per PR landed, Dreamverse adoption is opt-in.
|
||||
|
||||
### After PR 7 merges
|
||||
|
||||
Single-file change in Dreamverse, ~50-line PR:
|
||||
|
||||
1. Replace `server/video_generation.py:89 ContinuationState` import
|
||||
with `from fastvideo.pipelines.basic.ltx2.continuation import LTX2ContinuationState`.
|
||||
2. Move `apply_video`, `apply_audio`, `save_video`, `save_audio_latents`,
|
||||
`clear` off the state class onto `VideoGenerationWorker` (these are
|
||||
runtime policy that uses the state, not part of the state itself).
|
||||
3. Update `apply_audio` to read knobs from `state.audio_conditioning_num_frames`
|
||||
and `state.audio_conditioning_strength` instead of the env globals
|
||||
`AUDIO_CONDITIONING_NUM_FRAMES` / `AUDIO_CONDITIONING_STRENGTH`. The
|
||||
env globals can stay as defaults that populate the state when a new
|
||||
session starts.
|
||||
4. Same treatment for video knobs: `state.video_conditioning_frame_idx`,
|
||||
`state.video_conditioning_strength`.
|
||||
5. Frame storage swaps `list[PIL.Image]` for `list[np.ndarray]` —
|
||||
simpler `save_video` (no PIL conversion) and simpler `clear` (no
|
||||
`.close()` loop).
|
||||
|
||||
### After PR 7.5 lands streaming server skeleton
|
||||
|
||||
Dreamverse's runtime/factory.py either:
|
||||
|
||||
- Continues to construct `GPUPool` from `RealtimeRuntimeConfig` (the
|
||||
current path), now backed by the upstreamed `fastvideo/entrypoints/realtime/`.
|
||||
- Or migrates to the upstream's `ServeConfig.streaming` shape and
|
||||
invokes `fastvideo serve --config realtime.yaml` as the launch path.
|
||||
|
||||
Either way, `Dreamverse/server/runtime/interfaces.py` `RuntimePool` /
|
||||
`RuntimeSlot` Protocol can stay in place — it was modeled after the
|
||||
realtime runtime's surface. No interface change needed.
|
||||
|
||||
### After PR 7.6 lands the GPU pool upstream
|
||||
|
||||
- The `local_runtime.py` import in
|
||||
`Dreamverse/server/runtime/gpu_pool.py:24` becomes a public import
|
||||
with the same symbols (`RealtimeRuntimeConfig`, `GPUPool`,
|
||||
`get_available_gpus`).
|
||||
- Per-GPU continuation state inside the worker (`ltx2_continuation_images`,
|
||||
`ltx2_continuation_audio_latents`) gets replaced by a `SessionStore`
|
||||
reference. Dreamverse doesn't see this change — it's runtime-internal.
|
||||
- `request.state` / `result.state` round-trip starts working end-to-end
|
||||
on the local runtime. Dreamverse's worker can begin reading
|
||||
`result.state` and feeding `request.state` between segments.
|
||||
|
||||
### After PR 7.10 lands the Dynamo backend contract
|
||||
|
||||
- `VideoGenerator.generate_async(...) -> AsyncGenerator[VideoEvent, None]`
|
||||
is the canonical API.
|
||||
- Dreamverse's per-segment `user_step` flow can migrate from the legacy
|
||||
sync `generate_video(..., **kwargs)` path to consuming the typed
|
||||
event stream. Optional; the sync wrapper stays.
|
||||
|
||||
## Open Questions
|
||||
|
||||
### `video_position_offset_sec` semantics
|
||||
|
||||
Dreamverse computes `prefix_sec = float(audio_extra) / 24.0` per
|
||||
segment in `apply_audio`. Not persisted on `ContinuationState`.
|
||||
|
||||
PR-7 has `video_position_offset_sec` as a **state field**. Two valid
|
||||
interpretations:
|
||||
|
||||
(a) **Persistent across segments** — accumulating time offset for
|
||||
long sessions; useful for time-coherent audio chaining.
|
||||
(b) **Per-segment hint that rides on the carrier** — runtime
|
||||
overwrites every time; field is harmless redundancy.
|
||||
|
||||
Field's docstring leans toward (b). Decide before PR 7.6 starts
|
||||
emitting/consuming it. If we land on (a), document the accumulation
|
||||
rule explicitly.
|
||||
|
||||
### `BlobStore` / `SessionStore` lifecycle ownership
|
||||
|
||||
PR 7's in-memory implementations have no eviction, no TTL, no
|
||||
automatic blob cleanup on state replacement. Documented as a
|
||||
per-deployment policy decision.
|
||||
|
||||
When PR 7.5/7.6 land the live consumer, who owns:
|
||||
|
||||
- bounded session capacity (LRU? TTL? hard max?)
|
||||
- blob `drop()` chained when a state is replaced
|
||||
- session expiry on websocket disconnect
|
||||
|
||||
Probably the streaming server's session manager, but worth stating
|
||||
explicitly in PR 7.5's design.
|
||||
|
||||
### `realtime/` vs `streaming/` package naming
|
||||
|
||||
Currently:
|
||||
|
||||
- Public PR 5.5 introduced `fastvideo/entrypoints/streaming/` (skeleton + typed config).
|
||||
- Internal has `fastvideo/entrypoints/realtime/` (live runtime).
|
||||
- Dreamverse imports from `fastvideo.entrypoints.realtime` (per the internal name).
|
||||
|
||||
PR 7.5 either picks one or ships a deprecation alias module.
|
||||
Recommendation in `streaming-server-upstream-plan.md`: keep
|
||||
`streaming/` (it's the post-PR-5.5 public name), provide
|
||||
`realtime/__init__.py` as a re-export with a `DeprecationWarning` for
|
||||
one release cycle so internal/Dreamverse can land import updates.
|
||||
|
||||
## Test Coverage on the FastVideo Side
|
||||
|
||||
PR 7 ships:
|
||||
|
||||
- `fastvideo/tests/api/test_ltx2_continuation.py` — typed
|
||||
state round-trip (inline + blob), bf16 preservation, JSON
|
||||
serializability, kind/version validation, schema_version guard.
|
||||
- `fastvideo/tests/entrypoints/streaming/test_session_store.py` —
|
||||
store/snapshot/hydrate/drop behavior on `InMemorySessionStore`;
|
||||
put/get/drop on `InMemoryBlobStore`; thread-safety of both.
|
||||
|
||||
PR 7.5+ should add a contract test that exercises the round-trip via
|
||||
the same wire format Dreamverse's `FastVideoRealtimePool` consumes.
|
||||
|
||||
## Changelog
|
||||
|
||||
| Date | Change |
|
||||
|------|--------|
|
||||
| 2026-04-23 | Initial draft. Captures PR 6 / PR 7 mapping; open questions on `video_position_offset_sec`, lifecycle ownership, and `realtime/` vs `streaming/` naming. |
|
||||
@@ -0,0 +1,390 @@
|
||||
# Dreamverse Integration Review Log
|
||||
|
||||
This document tracks design decisions, open questions, and integration-time
|
||||
choices made while landing the public-side stacked PRs (7.7 → 8) and switching
|
||||
Dreamverse from `FastVideo-internal` to public `FastVideo`. The user will
|
||||
review this carefully — entries are deliberately verbose about *why*.
|
||||
|
||||
## Goal
|
||||
|
||||
Replace Dreamverse's dependency on `FastVideo-internal` with the public
|
||||
`FastVideo` package, using the upstreamed streaming server stack
|
||||
(`fastvideo.entrypoints.streaming.*`) where Dreamverse currently has local
|
||||
copies or imports private modules.
|
||||
|
||||
## Surfaces Dreamverse currently uses from FastVideo-internal
|
||||
|
||||
(from `/home/william5lin/Dreamverse/server/`, scanned 2026-04-26):
|
||||
|
||||
| Dreamverse import | Internal path | Public replacement |
|
||||
|---|---|---|
|
||||
| `fastvideo.entrypoints.realtime.local_runtime.RealtimeRuntimeConfig` | `FastVideo-internal/fastvideo/entrypoints/realtime/local_runtime.py` | (none) — Dreamverse rewires through `streaming.gpu_pool.SubprocessGpuPool` |
|
||||
| `fastvideo.entrypoints.realtime.local_runtime.GPUPool` | same as above | `fastvideo.entrypoints.streaming.gpu_pool.SubprocessGpuPool` (PR 7.6) |
|
||||
| `fastvideo.configs.pipelines.base.PipelineConfig` | already in public | unchanged |
|
||||
| `fastvideo.entrypoints.video_generator.VideoGenerator` | already in public | unchanged |
|
||||
| `fastvideo.layers.quantization.fp4_config.FP4Config` | already in public | unchanged |
|
||||
| `fastvideo.utils.maybe_download_model` | already in public | unchanged |
|
||||
| `fastvideo.models.audio.ltx2_audio_processing.AudioProcessor` | already in public | unchanged |
|
||||
| `fastvideo.models.loader.component_loader.ComponentLoader` | already in public | unchanged |
|
||||
| `fastvideo.models.dits.ltx2.*` | already in public | unchanged |
|
||||
| local copy: `Dreamverse/server/prompt_enhancer.py` (1933 lines) | mirrors `FastVideo-internal/.../prompt_enhancer.py` | `fastvideo.entrypoints.streaming.prompt.*` (PR 7.7) |
|
||||
| local copy: `Dreamverse/server/prompt_safety.py` | mirrors `FastVideo-internal/.../prompt_safety.py` | `fastvideo.entrypoints.streaming.prompt.safety` (PR 7.8) |
|
||||
| local copy: `Dreamverse/server/session_logger.py` | mirrors `FastVideo-internal/.../session_logger.py` | `fastvideo.entrypoints.streaming.session_logger` (PR 7.8) |
|
||||
| local copy: `Dreamverse/server/rewrite_prompt_payload.py` | mirrors `FastVideo-internal/.../rewrite_prompt_payload.py` | `fastvideo.entrypoints.streaming.prompt.rewrite` (PR 7.8) |
|
||||
| local copy: `Dreamverse/server/mock_server.py` (1200 lines) | mirrors `FastVideo-internal/.../mock_server.py` | `fastvideo.entrypoints.streaming.mock_server` (PR 7.8) |
|
||||
| local copy: `Dreamverse/server/session_init_image.py` | mirrors `FastVideo-internal/.../session_init_image.py` | `fastvideo.entrypoints.streaming.session_init_image` (PR 7.5 — already public) |
|
||||
|
||||
## Design decisions made (auto-resolved)
|
||||
|
||||
### D-1: Realtime runtime → streaming GpuPool migration shape
|
||||
|
||||
**Context.** Dreamverse's `server/runtime/gpu_pool.py` thin-wraps
|
||||
`fastvideo.entrypoints.realtime.local_runtime.GPUPool`, which takes a
|
||||
`RealtimeRuntimeConfig(model_registry=…, default_model_id=…, default_height=…,
|
||||
default_width=…, default_num_frames=…, default_num_inference_steps=…,
|
||||
startup_warmup_*…)`. The public `streaming.gpu_pool.SubprocessGpuPool` takes a
|
||||
typed `GeneratorConfig` + `GpuPoolConfig` + `WarmupConfig`.
|
||||
|
||||
The shapes differ in two important ways:
|
||||
|
||||
1. The internal version had a multi-model registry (`model_id → model_config`
|
||||
dict). The public version is single-model (one `GeneratorConfig`).
|
||||
2. The internal version flattened a few sampling defaults (height/width/frames/
|
||||
steps) into the runtime config. The public version expects them as part of
|
||||
the per-request `SamplingConfig`.
|
||||
|
||||
**Decision.** Dreamverse will:
|
||||
1. Drop the multi-model registry on the integration branch (it is not used in
|
||||
production today — Dreamverse boots one model per replica).
|
||||
2. Construct a `GeneratorConfig` for the chosen model from `MODEL_REGISTRY[id]`
|
||||
and pass it to `SubprocessGpuPool`.
|
||||
3. Move the `default_height` / `default_width` / `default_num_frames` /
|
||||
`default_num_inference_steps` defaults into a server-side
|
||||
`default_request: GenerationRequest` template the session controller fills
|
||||
from per-request input.
|
||||
|
||||
**Why.** Multi-model is feasible to add back later (one pool per model id,
|
||||
acquire by `(session_id, model_id)`), but not on the migration branch — that
|
||||
would couple the upstream switch to a feature redesign. Punting keeps the
|
||||
upstream switch a pure mechanical refactor.
|
||||
|
||||
**Risk.** If a Dreamverse code path silently relied on the registry to swap
|
||||
models per-session, the migration branch will surface that as a missing-model
|
||||
error. The integration tests must exercise at least one segment per supported
|
||||
model id before merging the Dreamverse branch.
|
||||
|
||||
### D-2: PR 7.7 prompt enhancer API surface narrower than the internal one
|
||||
|
||||
**Context.** The upstreamed `PromptEnhancer.enhance/auto_extend/rewrite` returns
|
||||
`LLMResponse(content, provider, model, latency_ms, fallback_used)`. The internal
|
||||
`enhance_prompt` / `generate_auto_prompt` / `rewrite_prompt_sequence` returns
|
||||
`EnhanceResult(prompt, fallback_used, error, provider, model, latency_ms)` /
|
||||
`RewriteResult(prompts, …, rollout_id, rollout_label, raw_response_text)`.
|
||||
|
||||
**Decision.** The Dreamverse integration branch will adapt at the call site:
|
||||
- `enhancer.enhance_prompt(...)` → `enhancer.enhance(prompt)` + a thin shim
|
||||
that maps the structured response into the existing `EnhanceResult` shape
|
||||
for the session-controller code path. Move the shim to
|
||||
`Dreamverse/server/prompting/_internal_compat.py`.
|
||||
- The locked-segment / next-segment-index plumbing the internal version
|
||||
built into the user payload becomes Dreamverse-side template logic in
|
||||
the shim.
|
||||
- The JSON-shaped responses the internal prompts assume (`{"next_prompt":
|
||||
"..."}` / `{"segment_prompts": [...]}`) become Dreamverse-side
|
||||
parsing in the shim, since the public `LLMResponse` is intentionally raw.
|
||||
|
||||
**Why.** The public surface stays minimal and provider-agnostic; the
|
||||
LTX-2-specific orchestration (locked segments, rollout id/label, JSON
|
||||
schemas) is an internal-UI concern, not something every public consumer
|
||||
should wear. Dreamverse keeps its existing call shape; the public stays
|
||||
clean.
|
||||
|
||||
**Open question for review:** Should we promote some of this into
|
||||
`fastvideo.entrypoints.streaming.prompt.ltx2_orchestration` (or similar)
|
||||
once a second consumer appears? Logging here so we have the option.
|
||||
|
||||
### D-3: Multi-stage provider race (Dreamverse) vs sequential fallback (public)
|
||||
|
||||
**Context.** The internal enhancer runs all providers in a stage in parallel
|
||||
and returns the first to succeed (`_run_provider_race`). The public
|
||||
enhancer runs providers strictly sequentially with retryable-error fallback.
|
||||
|
||||
**Decision.** Public stays sequential for PR 7.7. The race-based fallback is
|
||||
a Dreamverse-specific tail-latency optimization that depends on parallel API
|
||||
budgets; promoting it would force every public consumer to have multiple
|
||||
provider keys configured. Dreamverse can keep `_run_provider_race` as an
|
||||
internal optimization on its side.
|
||||
|
||||
**Risk.** First-segment latency on Dreamverse may regress slightly when
|
||||
Cerebras is having a bad minute (sequential fallback waits the full
|
||||
20s timeout before trying Groq). If this is a real production concern,
|
||||
add a public knob like `concurrency: int = 1` on `PromptEnhancer` that
|
||||
gates a race path — but only after measuring.
|
||||
|
||||
### D-4: Skipping PR 7.9 router for the integration branch
|
||||
|
||||
**Context.** The internal stack ships a `router/main.py` that load-balances
|
||||
across replicas with health checks. Dreamverse's deployment uses a single
|
||||
replica per region (per `gpu_pool.py:_parse_requested_gpu_limit`).
|
||||
|
||||
**Decision.** Land PR 7.9 on the public side (so the surface is upstreamed)
|
||||
but skip wiring it into the Dreamverse integration branch. Dreamverse's
|
||||
`server/main.py` does not import from `router/`.
|
||||
|
||||
### D-5: Audio re-encode (PR 7.10) needed for streaming, deferred
|
||||
|
||||
**Context.** The internal streaming server's per-step path runs an audio
|
||||
re-encode (`_re_encode_audio` inside `_stream_av_fmp4_events` /
|
||||
`do_step_ltx2`) so each fMP4 segment ships with continuation-conditioning
|
||||
audio. The whole-segment `pool.run()` path the public streaming server
|
||||
currently uses doesn't need this. The PR plan defers re-encode integration
|
||||
to PR 7.10 (`generate_async` / per-step streaming).
|
||||
|
||||
**Decision.** Land PR 7.10's `generate_async` on the public side. The
|
||||
Dreamverse integration branch initially keeps using `pool.run()` (whole
|
||||
segment, no re-encode); a follow-up branch swaps it to
|
||||
`generate_async` + audio re-encode once that path is exercised end-to-end.
|
||||
|
||||
### D-6: `realtime/local_runtime.py` is *not* upstreamed
|
||||
|
||||
**Context.** It is the FastVideo-internal precursor to `streaming.gpu_pool`.
|
||||
Upstreaming both would create two GPU pool implementations in the public
|
||||
repo.
|
||||
|
||||
**Decision.** Don't upstream `realtime/local_runtime.py`. Dreamverse switches
|
||||
to `streaming.gpu_pool.SubprocessGpuPool` on the integration branch. The
|
||||
internal module can be deleted from FastVideo-internal at a follow-up.
|
||||
|
||||
## Open questions for user review
|
||||
|
||||
Each section below is a place the auto-decision could plausibly be wrong.
|
||||
Please flip / annotate these in review.
|
||||
|
||||
### Q-1 Multi-model GPU pool (D-1)
|
||||
|
||||
Does any current Dreamverse production flow load multiple model ids
|
||||
concurrently? If yes, we need to either (a) keep `realtime/local_runtime`
|
||||
alive on the internal side until the public side gains a multi-model pool,
|
||||
or (b) build the multi-model abstraction upstream as part of PR 7.6 follow-up
|
||||
work.
|
||||
|
||||
### Q-2 Promoting LTX-2 prompt orchestration (D-2)
|
||||
|
||||
The locked-segments / next-segment-index / JSON-response orchestration is
|
||||
LTX-2-specific. If Cosmos / Wan / Hunyuan ever grow a similar continuation
|
||||
flow, we'll regret keeping the orchestration on the consumer side. Worth
|
||||
promoting now?
|
||||
|
||||
### Q-3 Race-based provider fallback (D-3)
|
||||
|
||||
The sequential fallback in the public enhancer adds up to `timeout_ms` of
|
||||
extra latency per failing provider before the next is tried. For Dreamverse
|
||||
that's 20s. Should we land the race path now behind a `concurrency: int = 1`
|
||||
knob, or wait until we have data?
|
||||
|
||||
### Q-4 Router upstream skip on Dreamverse branch (D-4)
|
||||
|
||||
We're upstreaming PR 7.9 (router) but not consuming it in the Dreamverse
|
||||
integration branch. Is that right? Dreamverse currently has no router
|
||||
component, so the answer is probably yes — but flagging.
|
||||
|
||||
### Q-5 generate_async cutover for the streaming path (D-5)
|
||||
|
||||
The plan leaves Dreamverse using `pool.run` (whole segment) initially.
|
||||
Audio re-encode for cross-segment continuity is deferred to a follow-up.
|
||||
Is that acceptable for the first switch, or does Dreamverse audio quality
|
||||
regress relative to the internal path until 7.10 is wired in?
|
||||
|
||||
## PR-by-PR execution log
|
||||
|
||||
### PR 7.6 — already opened (#1257)
|
||||
|
||||
`will/api_7.6` rebased onto `origin/main`, with subprocess-pool robustness
|
||||
review fixes pushed (boot_ok event, dead-worker detection, parallel shutdown,
|
||||
reader-exit pending-job cleanup). 17/17 gpu_pool tests + 89/89 streaming
|
||||
tests green at head.
|
||||
|
||||
### PR 7.7 — already opened (#1258)
|
||||
|
||||
`will/api_7.7` rebased onto the new 7.6 + LLM provider review fixes applied
|
||||
locally (per-instance `retryable`, 4xx-non-retryable, json-decode wrap,
|
||||
shared `_openai_compat.complete_openai_compatible`, `dataclasses.replace`
|
||||
for the fallback marker). 29/29 prompt tests + 120/120 streaming tests green.
|
||||
**Pending push** — the user opted to push this branch themselves.
|
||||
|
||||
### PR 7.8 — rebased onto new 7.7
|
||||
|
||||
`will/api_7.8` two commits replayed cleanly on the new 7.7. Adds
|
||||
`fastvideo/entrypoints/streaming/{prompt/safety,prompt/rewrite,session_logger,
|
||||
mock_server}.py` plus `test_auxiliaries.py`. 141/141 streaming tests green.
|
||||
|
||||
Notable gap vs internal version: the public `PromptSafetyFilter` ships one
|
||||
classifier slot (`unsafe` label, single threshold) whereas the internal
|
||||
version chained an NSFW filter and a hate-speech filter with marker-based
|
||||
label matching. Multi-classifier composition is left to Dreamverse —
|
||||
operators chain two filters explicitly. See **D-7** below.
|
||||
|
||||
### PR 7.9 — rebased onto new 7.8
|
||||
|
||||
`will/api_7.9` three commits replayed cleanly. Adds streaming router
|
||||
(`router/{config,registry,main}.py`), `fastvideo router-serve` CLI
|
||||
subcommand, and `test_router.py`. 151/151 streaming tests green.
|
||||
|
||||
Caveat: router/main.py uses the deprecated FastAPI `app.on_event("shutdown")`
|
||||
hook — emits a DeprecationWarning. Migration to lifespan handlers is a
|
||||
pre-merge cleanup item but not a blocker.
|
||||
|
||||
### PR 7.10 — rebased onto new 7.9
|
||||
|
||||
`will/api_7.10` three commits replayed with two trivial conflicts (line
|
||||
wrap in `server.py`, redundant test in `test_cli_translation.py`). Adds
|
||||
`VideoEvent` hierarchy, `VideoGenerator.generate_async`,
|
||||
`default_health_check_request`, plus `test_generate_async.py` (273-line
|
||||
contract test). 184/184 streaming + contract tests green.
|
||||
|
||||
### PR 8 — rebased onto new 7.10
|
||||
|
||||
`will/api_8` four commits → three (the 4th was a duplicate
|
||||
`streaming.md` doc that 7.5 already shipped, dropped during rebase).
|
||||
Adds `docs/design/server_contracts/{dynamo,index,openai}.md`,
|
||||
`mkdocs.yml` entries, and `fastvideo/tests/contract/test_{dreamverse,
|
||||
dynamo}_shape.py`. 206/206 streaming + contract tests green.
|
||||
|
||||
### Dreamverse `will/integrate-public-fastvideo`
|
||||
|
||||
Branch created from Dreamverse `master`. Single change: `pyproject.toml`
|
||||
swaps `fastvideo = { path = "../FastVideo-internal", editable = true }`
|
||||
to point at `../FastVideo`. Comment added linking back to this review
|
||||
doc.
|
||||
|
||||
**Verified:** every TRACKED `from fastvideo.*` import in Dreamverse
|
||||
(`server/video_generation.py` only) resolves against the public
|
||||
package — except `fastvideo.layers.quantization.fp4_config.FP4Config`
|
||||
(see **D-7** / Q-6 below).
|
||||
|
||||
**Untracked WIP** in `Dreamverse/server/{config,prompting,runtime,session}/`
|
||||
imports `fastvideo.entrypoints.realtime.local_runtime` (D-6); this
|
||||
branch does not migrate that WIP. The user's existing untracked work
|
||||
stays untouched and will need a separate follow-up to consume
|
||||
`streaming.gpu_pool.SubprocessGpuPool`.
|
||||
|
||||
## Test ladder (built-up to e2e per user request)
|
||||
|
||||
Each rung verifies the integration switch at one layer. Run from the
|
||||
narrowest to the broadest before running the full e2e against real
|
||||
GPU + model weights.
|
||||
|
||||
| # | Layer | Command | Status against the switched stack |
|
||||
|---|---|---|---|
|
||||
| 1 | Public FastVideo unit + contract tests | `pytest fastvideo/tests/api/ fastvideo/tests/entrypoints/streaming/ fastvideo/tests/contract/` | 358/358 passing on `will/api_8` |
|
||||
| 2 | Public FastVideo FP4 lazy-import | `pytest fastvideo/tests/ops/quantization/test_fp4_config.py` | 3/3 passing |
|
||||
| 3 | Dreamverse Python tests | `cd Dreamverse && uv run pytest server/tests/ -k "not stress and not benchmark and not health_endpoint"` | 73/73 passing against public FastVideo |
|
||||
| 4 | Dreamverse FE unit/integration (vitest) | `cd Dreamverse/apps/web && npm test` | 54/86 passing — 32 failures are pre-existing copy-mismatches in `reducer.test.ts` etc., not caused by the switch |
|
||||
| 5 | Backend HTTP smoke (Playwright) | `cd Dreamverse/apps/web && PLAYWRIGHT_SKIP_WEBSERVER=1 PLAYWRIGHT_BASE_URL=http://127.0.0.1:8009 npx playwright test e2e/backend-health.spec.ts` | 4/4 passing (5th correctly skipped because devtools-only route is off) |
|
||||
| 6 | Frontend shell smoke (Playwright) | `npx playwright test e2e/frontend-shell.spec.ts` | Pending — requires Next.js dev server to be reachable; was stuck during this run, needs a clean restart |
|
||||
| 7 | Full e2e preset generation | `npx playwright test e2e/preset-prompt-generation.spec.ts` | **8/8 passing** end-to-end after restart with `CUDA_VISIBLE_DEVICES=4 ENABLE_TORCH_COMPILE=0 FASTVIDEO_GPU_COUNT=1 FASTVIDEO_ENABLE_DEVTOOLS=1`. BE warmup + GPU 4 idle slot let `/readyz` flip green; the spec verifies preset → WS → backend handshake → "Generating video…" state. |
|
||||
|
||||
### How to reproduce e2e tier 7 from cold
|
||||
|
||||
```
|
||||
# 1. BE — picks an idle GPU and skips torch.compile (avoids the
|
||||
# aarch64 cross-compiler bug in the conda env's triton stack).
|
||||
cd ~/Dreamverse
|
||||
set -a; source ~/.env; set +a
|
||||
CUDA_VISIBLE_DEVICES=4 ENABLE_TORCH_COMPILE=0 \
|
||||
FASTVIDEO_ENABLE_DEVTOOLS=1 FASTVIDEO_GPU_COUNT=1 \
|
||||
uv run dreamverse-server &
|
||||
|
||||
# 2. Wait for /readyz (~2 min for warmup x2 segments)
|
||||
until curl -fsS http://127.0.0.1:8009/readyz >/dev/null; do sleep 5; done
|
||||
|
||||
# 3. FE
|
||||
cd ~/Dreamverse/apps/web
|
||||
BACKEND_URL=http://127.0.0.1:8009 NEXT_PUBLIC_INCLUDE_DEVTOOLS=1 \
|
||||
npm run dev:devtools &
|
||||
|
||||
# 4. Playwright
|
||||
cd ~/Dreamverse/apps/web
|
||||
PLAYWRIGHT_SKIP_WEBSERVER=1 \
|
||||
PLAYWRIGHT_BASE_URL=http://127.0.0.1:5274 \
|
||||
BACKEND_URL=http://127.0.0.1:8009 \
|
||||
npx playwright test --project=chromium --reporter=list
|
||||
```
|
||||
|
||||
### Surfaced during the e2e debug pass (logged here for follow-up)
|
||||
|
||||
* **`SamplingParam has no field ltx2_image_crf`** — Dreamverse's
|
||||
`server/video_generation.py:406` passes `ltx2_image_crf=0.0` to a
|
||||
`SamplingParam(...)` constructor. The internal SamplingParam (in
|
||||
`fastvideo/configs/sample/base.py`) declared this field; the public
|
||||
`fastvideo.api.sampling_param.SamplingParam` does not. Currently
|
||||
the BE logs an `ERROR` and silently drops the kwarg; warmup still
|
||||
succeeds because the field is non-load-bearing for FP4-disabled
|
||||
inference. Either re-add the field to the public schema or update
|
||||
Dreamverse to stop passing it. **D-8.**
|
||||
|
||||
* **`aarch64-conda-linux-gnu-cc` triton compile failure** — the conda
|
||||
env we boot from injects an ARM cross-compiler ahead of `gcc` on
|
||||
`$PATH`, so `torch._inductor`'s triton launcher fails compilation.
|
||||
Setting `ENABLE_TORCH_COMPILE=0` bypasses it. Long-term fix: clean
|
||||
the conda env's compiler shadowing or add a `CC=gcc` override in
|
||||
Dreamverse's worker bootstrap. **D-9.**
|
||||
|
||||
* **GPU pool starts but warmup OOMs on a shared GPU** — when
|
||||
`CUDA_VISIBLE_DEVICES` lands on a GPU another tenant is using
|
||||
(107 GiB-pegged training run on GPU 0 in this case), LTX-2 warmup
|
||||
fails with OOM. Picking an idle GPU (4-7 here) is a manual step.
|
||||
A pre-warm probe that checks free memory before booting the pool
|
||||
would prevent this. **D-10.**
|
||||
|
||||
* **ffmpeg fragment write `Broken pipe`** — when the WS client closes
|
||||
before the backend finishes streaming the first segment, ffmpeg
|
||||
hits `[Errno 32] Broken pipe`. Currently Dreamverse's
|
||||
`gpu_pool.handle_command` re-raises this as a session error,
|
||||
which then propagates to "User step failed". Cosmetic for now —
|
||||
swallowing pipe-broken on intentional disconnect would clean up
|
||||
the logs. **D-11.**
|
||||
|
||||
## Additional integration gaps surfaced during the switch
|
||||
|
||||
### D-7: `FP4Config` is private-only
|
||||
|
||||
**Context.** `Dreamverse/server/video_generation.py:271` imports
|
||||
`fastvideo.layers.quantization.fp4_config.FP4Config` and assigns it to
|
||||
`pipeline_config.dit_config.quant_config`. The 411-line module lives only
|
||||
in `FastVideo-internal/fastvideo/layers/quantization/fp4_config.py` and
|
||||
hard-imports `flashinfer` at module top — it never made the public
|
||||
upstream pass. Public has `base_config.py` and `absmax_fp8.py` only.
|
||||
|
||||
**Decision (provisional).** Don't upstream `fp4_config.py` in this
|
||||
session. Reasons:
|
||||
1. It introduces a new external dependency (`flashinfer`) the public
|
||||
package has avoided so far.
|
||||
2. The class hard-codes LTX-2 layer paths
|
||||
(`ltx2.blocks.{i}.attn1.to_q` etc.) — this is "LTX-2-specific FP4",
|
||||
not generic FP4. Belongs colocated with `pipelines/basic/ltx2/` if
|
||||
it goes anywhere.
|
||||
3. The FP4 pre-quantize/forward op surface is the kind of thing where
|
||||
a careful review pass matters more than a bulk copy.
|
||||
|
||||
**What this means for the integration branch.** Dreamverse will boot
|
||||
fine; only the FP4-quantized path inside `video_generation.py:283`
|
||||
will fail (lazy import). For workflows that don't enable FP4
|
||||
quantization, the integration is complete.
|
||||
|
||||
### Q-6 (review): how to land FP4Config publicly?
|
||||
|
||||
Two reasonable next steps:
|
||||
1. **Colocate.** Move FP4 code to `fastvideo/pipelines/basic/ltx2/quantization.py`
|
||||
with `flashinfer` as an optional extra: `pip install fastvideo[fp4]`.
|
||||
Refactor `FP4QuantizeMethod` to take its layer-prefix list from a
|
||||
pipeline-config field instead of hardcoding ltx2 paths so the
|
||||
approach generalizes.
|
||||
2. **Keep private.** Treat FP4 as a Dreamverse-side concern — Dreamverse
|
||||
imports `fp4_config` from the internal repo via a thin shim. Public
|
||||
FastVideo stays focused on generic surfaces. This means the
|
||||
"FastVideo-internal removable" goal is partially undone.
|
||||
|
||||
Recommendation: option 1 once the API refactor settles — wait until
|
||||
the LTX-2 colocation step (PR 9 / 10 territory) and land FP4 there.
|
||||
|
||||
@@ -0,0 +1,518 @@
|
||||
# Handoff: LTX-2 NVFP4 wire-up + Dreamverse launch-demo skill
|
||||
|
||||
This document hands off in-flight work to the next coding agent. It covers
|
||||
two related streams that landed across two repos:
|
||||
|
||||
1. **FastVideo** (`will/ltx2_sr_port`): wire NVFP4 (NVIDIA's block-scaled
|
||||
FP4) inference + per-component torch.compile + supporting parity fixes
|
||||
so the public package matches `FastVideo-internal` for the LTX-2
|
||||
distilled streaming path used by Dreamverse.
|
||||
2. **Dreamverse** (`will/integrate-public-fastvideo`): switch the GPU
|
||||
worker to the typed `GeneratorConfig` API, rename `FP4Config` →
|
||||
`NVFP4Config`, add a `launch-demo` skill + canonical
|
||||
`serve_configs/streaming_demo.yaml` for `fastvideo serve --config`.
|
||||
|
||||
Stack remains green: 222/222 FastVideo unit/contract/api tests pass; 8/8
|
||||
Playwright e2e tests pass against the live `dreamverse-server` + Next.js
|
||||
stack.
|
||||
|
||||
---
|
||||
|
||||
## Repo + branch state
|
||||
|
||||
| Repo | Path | Branch | Tip |
|
||||
| --- | --- | --- | --- |
|
||||
| FastVideo | `/home/william5lin/FastVideo` | `will/ltx2_sr_port` | `c6c14c55` |
|
||||
| Dreamverse | `/home/william5lin/Dreamverse` | `will/integrate-public-fastvideo` | `3d7fd89` |
|
||||
| Reference (read-only) | `/home/william5lin/FastVideo-internal` | (their) `main` | source of truth for parity |
|
||||
|
||||
> **Working branch on FastVideo is `will/ltx2_sr_port`, not the default checkout.**
|
||||
> The shell may report `will/uv-pip-install-everywhere` because that was
|
||||
> the earlier checkout. Run `git checkout will/ltx2_sr_port` before
|
||||
> picking up FastVideo work.
|
||||
|
||||
### Live processes (do not duplicate)
|
||||
|
||||
```
|
||||
:8009 dreamverse-server pid 2453227 (warmed, /readyz returns 200)
|
||||
:5274 next-server (dev) pid 2399103 (devtools build)
|
||||
```
|
||||
|
||||
### Stashes
|
||||
|
||||
* FastVideo: `stash@{0}: WIP on main: …HunyuanVideo plugin…` — pre-existing,
|
||||
unrelated to this work, do not pop.
|
||||
* Dreamverse: `stash@{0}: wip: server modular refactor (split
|
||||
config/prompting/runtime/session)` — 3867 lines of orphan modular split
|
||||
off this branch. Do not pop on this branch; recover on a separate
|
||||
feature branch if anyone wants to resurrect it.
|
||||
|
||||
---
|
||||
|
||||
## What landed (FastVideo: `cfccd292..c6c14c55`)
|
||||
|
||||
Six commits on top of the i2v / continuation latent port:
|
||||
|
||||
```
|
||||
c6c14c55 test(nvfp4): lock LTX-2 wiring + typed transformer_quant flow
|
||||
94c983a2 refactor(quant): rename FP4 → NVFP4 to disambiguate from other FP4 variants
|
||||
42b30bf9 feat(ltx2): wire FP4 inference through fastvideo.layers.quantization
|
||||
6da342ba feat(compile): per-component compile + transformer_refine + prepare hook
|
||||
221cb20a feat(api): typed per-component CompileConfig + FastVideoArgs carriers
|
||||
a4760bae fix(api): propagate generic refine_* args + match internal randn
|
||||
```
|
||||
|
||||
Each commit message has the rationale. Highlights below.
|
||||
|
||||
### `a4760bae` — three small parity fixes
|
||||
|
||||
* `FastVideoArgs.__post_init__` now calls `_resolve_refine_args()` which
|
||||
copies the public-facing generic `refine_*` knobs onto their
|
||||
`ltx2_refine_*` runtime carriers. Was missing → callers that set
|
||||
`refine_lora_path=...` saw "applied to 0 layers" warnings as the value
|
||||
was silently dropped.
|
||||
* `_randn_ltx2_video_latents` patch path reverted from `randn_tensor` →
|
||||
`torch.randn` to bit-match internal under single-generator inference.
|
||||
Identical for a single `torch.Generator` but diverges for
|
||||
`list[Generator]` (per-sample seeds).
|
||||
* Classified 19 `refine_*` / `ltx2_refine_*` / i2v / `ltx2_audio_*` /
|
||||
`ltx2_conditioning_latent_*` / `ltx2_video_conditions` fields in the
|
||||
schema-parity inventory yaml.
|
||||
|
||||
### `221cb20a` — typed CompileConfig + FastVideoArgs carriers
|
||||
|
||||
`CompileConfig` (in `fastvideo/api/schema.py`) gained per-component knobs:
|
||||
|
||||
```python
|
||||
@dataclass
|
||||
class CompileConfig:
|
||||
enabled: bool = False # master DiT switch
|
||||
backend / fullgraph / mode / dynamic / extras # master kwargs
|
||||
|
||||
# Per-component overlays, None = inherit master `enabled`
|
||||
text_encoder_enabled: bool | None = None
|
||||
vae_enabled: bool | None = None
|
||||
audio_vae_enabled: bool | None = None
|
||||
|
||||
# Per-component kwargs override master when non-empty
|
||||
dit_kwargs: dict = ...
|
||||
text_encoder_kwargs: dict = ...
|
||||
vae_kwargs: dict = ...
|
||||
audio_vae_kwargs: dict = ...
|
||||
```
|
||||
|
||||
Matching carrier fields on `FastVideoArgs`:
|
||||
`enable_torch_compile_text_encoder/vae/audio_vae` and
|
||||
`torch_compile_kwargs_dit/text_encoder/vae/audio_vae`. Compat layer
|
||||
round-trips them through `legacy_from_pretrained_to_config` and
|
||||
`generator_config_to_fastvideo_args`. **No behavior change yet** — these
|
||||
are surface ports only; consumed in the next commit.
|
||||
|
||||
### `6da342ba` — refine + per-component compile + prepare_for_compile
|
||||
|
||||
`composed_pipeline_base.post_init` now:
|
||||
|
||||
* compiles `transformer_refine` alongside `transformer` and
|
||||
`transformer_2` whenever the DiT compile flag is on (closes the LTX-2
|
||||
stage-2 silent-eager bug);
|
||||
* dispatches per-component compile loops (text encoder, VAE, audio VAE)
|
||||
with per-component kwargs falling back to master when empty;
|
||||
* calls `module.prepare_for_compile()` on each compiled submodule
|
||||
before invoking `torch.compile` (hook protocol — model-specific).
|
||||
Implemented on `Gemma3` to materialize HF weights outside Dynamo's
|
||||
tracer.
|
||||
|
||||
### `42b30bf9` — NVFP4 LTX-2 inference wire-up *(largest)*
|
||||
|
||||
End-to-end:
|
||||
|
||||
1. `models/dits/ltx2.py` — swap `nn.Linear` → `ReplicatedLinear` for the
|
||||
FP4-eligible subset (`LTXSelfAttention`, `LTXDistributedSelfAttention`,
|
||||
`FeedForward`/`GELUApprox`); plumb `quant_config` and `prefix=` from
|
||||
`BasicAVTransformerBlock` → `_init_transformer_blocks` → `LTXModel`
|
||||
→ `LTX2Transformer3DModel`. Other linears
|
||||
(`TimestepEmbedding`, `PixArtAlphaTextProjection`, `patchify_proj`,
|
||||
`proj_out`, `AdaLayerNormSingle.linear`) stay `nn.Linear` —
|
||||
matches internal exactly.
|
||||
2. Port `_supports_prequantized_input` and
|
||||
`_linear_project_with_optional_prequant` helpers. Attention forward
|
||||
pre-quantizes input once (`quantize_input`), reuses the
|
||||
`(x_fp4, x_scale, x_global_sf)` tuple for k/v projections when
|
||||
`context is x` — bit-matches internal's fused path.
|
||||
3. `models/loader/fsdp_load.py` — new `_maybe_convert_model_to_nvfp4`
|
||||
helper detects via `isinstance(quant_method, NVFP4QuantizeMethod)`
|
||||
(no flag); calls `convert_model_to_nvfp4` to materialize
|
||||
`_nvfp4_weight*` / `_nvfp4_alpha` / `_weight_global_sf` buffers.
|
||||
`flashinfer` import is lazy (inside the helper), so the loader is a
|
||||
no-op on hosts without flashinfer.
|
||||
4. `layers/quantization/__init__.py` — registered `"NVFP4"` in
|
||||
`QuantizationMethods` literal + `get_quantization_config`.
|
||||
5. `api/compat.py` + `fastvideo_args.py` — typed
|
||||
`engine.quantization.transformer_quant: "NVFP4"` resolves to a
|
||||
concrete `NVFP4Config()` instance, carried on `FastVideoArgs.transformer_quant`,
|
||||
pinned onto `pipeline_config.dit_config.quant_config` in
|
||||
`__post_init__._apply_transformer_quant`. **The explicit setter
|
||||
(legacy mutation pattern) wins** if `dit_config.quant_config` is
|
||||
already non-None.
|
||||
6. `layers/linear.py` — `LinearBase.__init__` now falls back to
|
||||
`UnquantizedLinearMethod` when `quant_config.get_quant_method` returns
|
||||
`None`. `NVFP4Config` only tags a curated subset of LTX-2 layers, and
|
||||
the previous `assert quant_method is not None` would crash any
|
||||
non-tagged layer that received a quant_config.
|
||||
|
||||
### `94c983a2` — FP4 → NVFP4 rename
|
||||
|
||||
NVIDIA's specific block-scaled fp4 format (e2m1 mantissa, fp32 alpha,
|
||||
`layout_128x4` scale layout, group size 16) — distinct from MX-FP4 /
|
||||
OCP-FP4 / generic e3m0. Mechanical rename, no behavior change:
|
||||
|
||||
* `fp4_config.py` → `nvfp4_config.py`
|
||||
* `FP4Config` → `NVFP4Config`; `get_name()` returns `"nvfp4"`
|
||||
* `FP4QuantizeMethod` → `NVFP4QuantizeMethod`
|
||||
* `convert_model_to_fp4` → `convert_model_to_nvfp4`
|
||||
* `QuantizationMethods` literal: `"FP4"` → `"NVFP4"`
|
||||
* registered buffer names: `_fp4_weight`/`_fp4_alpha` →
|
||||
`_nvfp4_weight`/`_nvfp4_alpha`
|
||||
* loader helper renamed
|
||||
* test file rename + symbol updates
|
||||
|
||||
Internal-scope torch op namespace `fastvideo_fp4::*` and
|
||||
`_get_ltx2_fp4_stage_profile` deliberately left as-is — purely
|
||||
internal naming that mirrors FastVideo-internal.
|
||||
|
||||
### `c6c14c55` — contract + numerical lock-in tests
|
||||
|
||||
* `fastvideo/tests/ops/quantization/test_nvfp4_ltx2_wiring.py` (6 tests):
|
||||
asserts that `LTXSelfAttention.to_q/to_k/to_v/to_out` are
|
||||
`ReplicatedLinear`; `NVFP4Config()` attaches `NVFP4QuantizeMethod`
|
||||
on the quantized subset with the correct `layer_prefix`; non-tagged
|
||||
projections (cross-attn K/V, audio attn, audio FFN) fall back to
|
||||
`UnquantizedLinearMethod`; `BasicAVTransformerBlock` propagates
|
||||
`quant_config` and `prefix` correctly to all 4 attention modules +
|
||||
FFN at once.
|
||||
* `fastvideo/tests/api/test_typed_quant_flow.py` (4 tests): asserts
|
||||
typed `engine.quantization.transformer_quant: "NVFP4"` →
|
||||
`NVFP4Config()` instance flow; default leaves `transformer_quant`
|
||||
None; explicit `dit_config.quant_config = …` wins over typed carrier.
|
||||
|
||||
---
|
||||
|
||||
## What landed (Dreamverse: `248060b..3d7fd89`)
|
||||
|
||||
Three commits on top of the e2e tier:
|
||||
|
||||
```
|
||||
3d7fd89 feat(skill): launch-demo orchestrator + fastvideo serve YAML
|
||||
d80c2a8 refactor(server): drive FP4 + per-component compile via typed GeneratorConfig
|
||||
4cc6b30 chore: gitignore Playwright + Next.js build artifacts under apps/web
|
||||
```
|
||||
|
||||
### `d80c2a8` — server/video_generation.py refactor
|
||||
|
||||
Three coordinated changes in the GPU worker:
|
||||
|
||||
* Replace legacy `load_kwargs` dict + `VideoGenerator.from_pretrained(model_root, **kwargs)`
|
||||
call with the typed `GeneratorConfig` (`EngineConfig` /
|
||||
`OffloadConfig` / `CompileConfig` / `PipelineSelection` /
|
||||
`ComponentConfig`). Refine knobs move from `ltx2_refine_*` flat
|
||||
kwargs into `preset_overrides["refine"]`. **The in-memory
|
||||
`pipeline_config` pin** (`dit_config.quant_config = NVFP4Config()`)
|
||||
keeps using the legacy `experimental["pipeline_config"]` carrier
|
||||
because typed `transformer_quant: "NVFP4"` doesn't yet support
|
||||
setting `layer_profile`.
|
||||
* Rename FP4 → NVFP4.
|
||||
* Re-enable `"mode": "max-autotune-no-cudagraphs"` (was commented out).
|
||||
Closes the last known divergence vs FastVideo-internal in the
|
||||
worker-level path trace.
|
||||
|
||||
### `4cc6b30` — gitignore Playwright/Next.js artifacts
|
||||
|
||||
Added `apps/web/{node_modules,.next,test-results,playwright-report}` to
|
||||
`.gitignore`. Mirror of the existing `prod-ui/` ignore set.
|
||||
|
||||
### `3d7fd89` — launch-demo skill
|
||||
|
||||
```
|
||||
.agents/skills/launch-demo/
|
||||
├── SKILL.md
|
||||
└── scripts/
|
||||
├── launch_demo.sh # orchestrator: BE + FE + health probes + Ctrl-C trap
|
||||
├── launch_backend_dreamverse.sh # uv run dreamverse-server (default)
|
||||
├── launch_backend_fastvideo.sh # uv run fastvideo serve --config (typed path)
|
||||
└── launch_frontend.sh # next dev (devtools/dev/single5s)
|
||||
serve_configs/
|
||||
└── streaming_demo.yaml # canonical ServeConfig matching internal/ui
|
||||
```
|
||||
|
||||
YAML has every field annotated with the internal source line it mirrors:
|
||||
LTX-2 distilled, NVFP4, 121 frames @ 1088×1920 24fps, 5 inference steps,
|
||||
2-step refine gs=1.0 add_noise=true, max-autotune-no-cudagraphs compile,
|
||||
121-frame default request, 300s session timeout, 6 segment cap, av_fmp4
|
||||
streaming, cinematic-drone warmup prompt, 2400s warmup timeout, 9
|
||||
conditioning frames + 0 end-offset, prompt enhancer on with cerebras /
|
||||
gpt-oss-120b / 20s timeout.
|
||||
|
||||
**Two BE flavors documented in SKILL.md:**
|
||||
|
||||
| `BE_FLAVOR=` | Boots | Routes served | FE compatible |
|
||||
| --- | --- | --- | --- |
|
||||
| `dreamverse` (default) | `dreamverse-server` | `/healthz`, `/readyz`, `/curated-presets`, `/v1/stream`, devtools, session monitor | ✓ full |
|
||||
| `fastvideo` | `fastvideo serve --config <yaml>` | `/health`, `/v1/stream` | ⚠ FE will surface fetch errors for `/curated-presets`, `/readyz` until those routes migrate into FastVideo's `build_app` |
|
||||
|
||||
The fastvideo flavor exists today as the verifiable typed-config path
|
||||
(YAML parses, streaming worker boots, dotted overrides work). It is not
|
||||
yet a drop-in for the FE — see "Open follow-ups" below.
|
||||
|
||||
---
|
||||
|
||||
## Verified
|
||||
|
||||
* `222 passed, 1 skipped` across `fastvideo/tests/api/`,
|
||||
`fastvideo/tests/contract/`,
|
||||
`fastvideo/tests/ops/quantization/test_nvfp4_*`,
|
||||
`tests/local_tests/pipelines/test_ltx2_pipeline_smoke.py`.
|
||||
* `8 passed` Playwright e2e (backend-health 5, frontend-shell 2,
|
||||
preset-prompt-generation 1) against the live `dreamverse-server`
|
||||
+ Next.js stack.
|
||||
* `streaming_demo.yaml` parses cleanly against `ServeConfig`; the
|
||||
validation path of `fastvideo serve --config <yaml>` runs without
|
||||
error and accepts dotted overrides like `--server.port 8010`.
|
||||
* FastVideo `bash -n` clean across all four launch scripts.
|
||||
|
||||
---
|
||||
|
||||
## Critical context (gotchas a successor should know)
|
||||
|
||||
### NVFP4 layer set is asymmetric — by design
|
||||
|
||||
`NVFP4Config.fp4_layers` covers:
|
||||
|
||||
* `attn1.{to_q,to_k,to_v,to_out}` — full self-attention
|
||||
* `attn2.{to_q,to_out}` — cross-attn Q + out only (text context not quantized)
|
||||
* `audio_to_video_attn.{to_q,to_out}` — AV cross Q + out
|
||||
* `video_to_audio_attn.{to_k,to_v}` — VA cross K + V
|
||||
* `ffn.{fc_in,fc_out}` — video FFN
|
||||
* `adaln_single.linear` — but this is `nn.Linear` (not `LinearBase`),
|
||||
so it never actually gets FP4'd. List entry has no effect; matches
|
||||
internal.
|
||||
|
||||
**NOT in the set:** audio self-attention (`audio_attn1.*`), audio
|
||||
cross-attention (`audio_attn2.*`), audio FFN (`audio.ffn.*`). Audio
|
||||
path is cheap enough that quant overhead isn't worth it. Test
|
||||
`test_basic_av_block_propagates_quant_config_to_all_children` locks
|
||||
this in — if you add audio quantization later, update the test.
|
||||
|
||||
### `LinearBase` fallback is load-bearing
|
||||
|
||||
`fastvideo/layers/linear.py:191-202`: when `quant_config.get_quant_method`
|
||||
returns `None` (layer not in the quant config's set), we fall back to
|
||||
`UnquantizedLinearMethod`. **Do not remove this fallback** — it would
|
||||
break every non-tagged `ReplicatedLinear` constructed with a
|
||||
`NVFP4Config`, and `assert quant_method is not None` in
|
||||
`ReplicatedLinear.__init__` would fire on unmatched layers.
|
||||
|
||||
### Typed `transformer_quant` precedence
|
||||
|
||||
`FastVideoArgs._apply_transformer_quant` only writes
|
||||
`dit_config.quant_config` when it's currently `None`. If a caller has
|
||||
explicitly set `pipeline_config.dit_config.quant_config = NVFP4Config(...)`,
|
||||
the explicit setter wins. Dreamverse's `video_generation.py` relies on
|
||||
this — it sets `NVFP4Config()` directly because the typed
|
||||
`transformer_quant: "NVFP4"` doesn't expose `layer_profile`.
|
||||
|
||||
### Pre-existing AbsMaxFP8 test failure is NOT mine
|
||||
|
||||
`fastvideo/tests/ops/quantization/test_absmax_fp8.py::test_create_weights_rejects_invalid_dtype`
|
||||
fails on `main` and on this branch with the same error
|
||||
("AssertionError not raised"). I confirmed via `git stash` that the
|
||||
failure pre-dates my changes. Not blocking; tracked as separate tech
|
||||
debt.
|
||||
|
||||
### `transformer_refine` is auto-compiled with the master DiT flag
|
||||
|
||||
Set `enable_torch_compile=True` and `transformer_refine` compiles
|
||||
along with `transformer` and `transformer_2`. There is **no separate
|
||||
`enable_torch_compile_refine` flag** — by design, refine inherits the
|
||||
DiT compile state to keep the typed surface small. If you need them
|
||||
decoupled, add a new field; don't repurpose existing ones.
|
||||
|
||||
### `prepare_for_compile` is a duck-type protocol, not a base class method
|
||||
|
||||
Defined nowhere; called via `getattr(module, "prepare_for_compile", None)`
|
||||
in `composed_pipeline_base._maybe_compile_pipeline_module`. Currently
|
||||
only Gemma implements it (to materialize HF weights outside Dynamo).
|
||||
Add to other models that have lazy external state if you observe
|
||||
compile-time graph breaks.
|
||||
|
||||
### Public typed `PromptEnhancerConfig.provider` is `Literal["cerebras", "groq"]`
|
||||
|
||||
Internal supports `"cerebras_ifm"` (config.py:143). The public typed
|
||||
schema does not. The `streaming_demo.yaml` defaults to `"cerebras"`.
|
||||
For agents that need `cerebras_ifm`, the `dreamverse-server` flavor
|
||||
respects the `FASTVIDEO_PROMPT_PROVIDER` env var (legacy path);
|
||||
`fastvideo serve --config` does not currently expose it.
|
||||
|
||||
### Dreamverse `pipeline_config` is still a Python object passed via `experimental`
|
||||
|
||||
The typed `GeneratorConfig` doesn't have a clean home for an
|
||||
in-memory `PipelineConfig` instance with mutated `dit_config`. We
|
||||
pass it via `pipeline.experimental["pipeline_config"]` — the
|
||||
`compat.py` legacy adapter recognizes that key and threads it through
|
||||
to `FastVideoArgs.from_kwargs`. This is fine but not pretty; if
|
||||
someone designs a typed `dit_config` carrier later, this becomes
|
||||
obsolete.
|
||||
|
||||
### `fastvideo serve --config` is not yet a drop-in for the FE
|
||||
|
||||
`fastvideo.entrypoints.streaming.server.build_app` exposes only
|
||||
`/health` and `/v1/stream`. The Dreamverse Next.js shell expects
|
||||
`/healthz`, `/readyz`, `/status`, `/curated-presets`,
|
||||
`/curated-presets/append`, `/prompt-system-config`, and the devtools
|
||||
routes. These all live in `Dreamverse/server/main.py` +
|
||||
`Dreamverse/server/routes/`. Until they migrate into FastVideo's
|
||||
`build_app` (or are exposed via a Dreamverse-side proxy), the
|
||||
`BE_FLAVOR=fastvideo` flavor is for verifying the typed serve config
|
||||
path only — not for full FE compatibility.
|
||||
|
||||
---
|
||||
|
||||
## Open follow-ups (prioritized)
|
||||
|
||||
### High
|
||||
|
||||
1. **Migrate FE-required routes into FastVideo's `build_app`.**
|
||||
`/healthz`, `/readyz`, `/status` look obviously fastvideo-side
|
||||
(they're streaming-server health). `/curated-presets` and
|
||||
`/prompt-system-config` are operator-side surfaces and should
|
||||
probably stay in Dreamverse (or migrate as opt-in routes that the
|
||||
FE feature-detects). Without this, `BE_FLAVOR=fastvideo` is
|
||||
permanently a "diagnostic" flavor. Closes the
|
||||
`launch-demo` skill TODO.
|
||||
|
||||
2. **AbsMaxFP8 test failure cleanup.** Pre-existing. Either fix the
|
||||
test (`AbsMaxFP8LinearMethod.create_weights` no longer asserts on
|
||||
invalid dtype — restore the assert if intentional, otherwise drop
|
||||
the test).
|
||||
|
||||
### Medium
|
||||
|
||||
3. **Add `cerebras_ifm` to public `PromptEnhancerConfig.provider`
|
||||
Literal.** Trivial schema change; needs paired enhancer-side
|
||||
provider implementation in
|
||||
`fastvideo/entrypoints/streaming/prompt/providers/`.
|
||||
|
||||
4. **Expose `layer_profile` on typed `engine.quantization`.** Today
|
||||
`transformer_quant: "NVFP4"` always constructs `NVFP4Config()`
|
||||
with the default `layer_profile="refine"`. To support stage-1
|
||||
profiles (no `attn2.to_out`, no cross-modal AV) via typed config,
|
||||
add `transformer_quant_layer_profile: str | None = None` and
|
||||
thread it through `compat.py`. Dreamverse currently dodges this
|
||||
by setting `NVFP4Config()` directly via `experimental`.
|
||||
|
||||
5. **Typed `dit_config.quant_config` carrier.** The
|
||||
`experimental["pipeline_config"]` escape hatch in Dreamverse
|
||||
should eventually become a typed field. Design TBD.
|
||||
|
||||
### Low
|
||||
|
||||
6. **Audio attention quantization profile.** If an audio-quant
|
||||
profile is added to `NVFP4Config.fp4_layers` (currently audio attn
|
||||
and FFN are bf16), update
|
||||
`test_basic_av_block_propagates_quant_config_to_all_children`.
|
||||
|
||||
7. **Schema parity inventory.** A few internal-only fields are not
|
||||
exposed publicly (`PROMPT_HTTP_TIMEOUT_MS`,
|
||||
`PROMPT_INITIAL_STAGE_TIMEOUT_MS`, `PROMPT_TEMPERATURE`,
|
||||
`PROMPT_MAX_COMPLETION_TOKENS`, `PROMPT_AUTO_SLEEP_MS`,
|
||||
`PROMPT_AUTO_TIMEOUT_MS`, the curated-presets file paths).
|
||||
These all flow via env vars on `dreamverse-server` today; if
|
||||
`fastvideo serve --config` becomes the canonical entrypoint,
|
||||
they'll need typed homes.
|
||||
|
||||
8. **Empty `apps/web/test-results/` directory locally.** The
|
||||
`.gitignore` entry I added makes it invisible to `git status`,
|
||||
but the dir itself still has a stale `.last-run.json` (45 bytes)
|
||||
from a prior Playwright run. Harness blocked auto-cleanup
|
||||
("pre-existing files"); the user can `rm -rf
|
||||
apps/web/test-results` whenever convenient.
|
||||
|
||||
---
|
||||
|
||||
## How to pick up work
|
||||
|
||||
### Quick orientation (run these first)
|
||||
|
||||
```bash
|
||||
# FastVideo state
|
||||
cd /home/william5lin/FastVideo
|
||||
git checkout will/ltx2_sr_port
|
||||
git log --oneline cfccd292..HEAD # six commits added this round
|
||||
.venv/bin/python -m pytest fastvideo/tests/api/ \
|
||||
fastvideo/tests/contract/ \
|
||||
fastvideo/tests/ops/quantization/test_nvfp4_*.py \
|
||||
tests/local_tests/pipelines/test_ltx2_pipeline_smoke.py \
|
||||
-q --no-header # expect 222 passed, 1 skipped
|
||||
|
||||
# Dreamverse state
|
||||
cd /home/william5lin/Dreamverse
|
||||
git log --oneline 248060b..HEAD # three commits added this round
|
||||
cat serve_configs/streaming_demo.yaml | head -40
|
||||
ls .agents/skills/launch-demo/
|
||||
|
||||
# Live stack health (already running on this host)
|
||||
curl -s http://localhost:8009/readyz | head -c 200
|
||||
curl -s http://localhost:5274/ | head -c 100
|
||||
( cd apps/web && npx playwright test --reporter=line ) # expect 8 passed
|
||||
```
|
||||
|
||||
### Reference docs
|
||||
|
||||
* **FastVideo internal/ui parity source:** `../FastVideo-internal/ui/ltx2-streaming/server/config.py`
|
||||
* **NVFP4 source on internal:** `../FastVideo-internal/fastvideo/layers/quantization/fp4_config.py`
|
||||
* **Worker-trace audit:** `../FastVideo/dreamverse_review.md` (D-1
|
||||
multi-model, D-5 audio re-encode, prior gap inventory)
|
||||
* **Schema parity inventory:** `docs/design/inference_schema_parity_inventory.yaml`
|
||||
* **PR-plan for the broader migration:** `../FastVideo/PR plan.md`
|
||||
|
||||
### Files most likely to need touches in follow-ups
|
||||
|
||||
* `fastvideo/api/schema.py` — `CompileConfig`, `QuantizationConfig`,
|
||||
`PromptEnhancerConfig` Literal extension.
|
||||
* `fastvideo/api/compat.py` — typed → flat translation.
|
||||
* `fastvideo/fastvideo_args.py` — carrier fields and
|
||||
`_apply_transformer_quant`.
|
||||
* `fastvideo/entrypoints/streaming/server.py::build_app` — add
|
||||
`/healthz`, `/readyz`, `/status` routes for FE compatibility (high
|
||||
priority follow-up #1).
|
||||
* `Dreamverse/server/video_generation.py` — typed `GeneratorConfig`
|
||||
builder (current).
|
||||
* `Dreamverse/serve_configs/streaming_demo.yaml` — every parity
|
||||
knob; edit here, not in shell scripts.
|
||||
|
||||
---
|
||||
|
||||
## Don't / Cautions
|
||||
|
||||
* **Don't pop the Dreamverse stash on this branch.** It's 3867 lines
|
||||
of orphan modular refactor (server/{config,prompting,runtime,session}/)
|
||||
with broken absolute imports. If anyone wants to resurrect it, do so
|
||||
on a separate feature branch.
|
||||
* **Don't remove the `LinearBase` `UnquantizedLinearMethod` fallback.**
|
||||
See "Critical context" above.
|
||||
* **Don't repurpose `enable_torch_compile` to mean DiT-only.** It also
|
||||
drives `transformer_refine` and `transformer_2` compile. Add a new
|
||||
flag if decoupling is needed.
|
||||
* **Don't change `NVFP4Config` buffer names back to `_fp4_*`.** The
|
||||
rename is intentional to disambiguate from MX-FP4 / OCP-FP4.
|
||||
* **Don't bypass the typed surface for new options.** New compile /
|
||||
quant / refine knobs should land on the dataclass + compat.py +
|
||||
parity inventory together. The existing test suite locks this in.
|
||||
* **Don't merge to main without a CI run that covers FP4.** Current
|
||||
CI doesn't run flashinfer-dependent paths; the wiring tests in
|
||||
`test_nvfp4_ltx2_wiring.py` are CPU-only by design and don't
|
||||
exercise the actual FP4 kernels.
|
||||
|
||||
---
|
||||
|
||||
*Last updated: end of session that landed `c6c14c55` on FastVideo and
|
||||
`3d7fd89` on Dreamverse. Stack remains green; no dirty state.*
|
||||
+539
@@ -0,0 +1,539 @@
|
||||
# FastVideo Streaming Server Upstream — Design & Plan
|
||||
|
||||
## Status
|
||||
Exploration / design draft. Captures the re-evaluation triggered by the
|
||||
decision to upstream `FastVideo-internal/ui/ltx2-streaming/server/` into
|
||||
the public repo. Not yet approved for execution.
|
||||
|
||||
## Related Documents
|
||||
- [PR plan.md](../../PR%20plan.md) — PR-by-PR implementation plan for the API refactor
|
||||
- [apirefactor.md](../../apirefactor.md) — design spec this plan implements
|
||||
- `../../../FastVideo-internal/ui/ltx2-streaming/` — upstream source (server side)
|
||||
- `../../../dynamo/` — local clone of ai-dynamo/dynamo; backend patterns at
|
||||
`components/src/dynamo/{vllm,sglang,trtllm}/` and `CLAUDE.md` files
|
||||
- https://github.com/ai-dynamo/dynamo/pull/7544 — draft PR that promotes
|
||||
FastVideo to a native Dynamo backend (CLOSED, superseded — but establishes
|
||||
the integration shape)
|
||||
|
||||
## Context
|
||||
|
||||
The internal `FastVideo-internal/ui/ltx2-streaming/` directory contains a
|
||||
complete LTX2 streaming service. The user has decided:
|
||||
|
||||
- **Frontend clients** (`client/`, `prod-ui/`) stay in the internal repo
|
||||
- **Everything server-side** — FastAPI/WebSocket server, GPU pool, prompt
|
||||
enhancer, router, auxiliaries — will be upstreamed to FastVideo
|
||||
|
||||
In parallel, FastVideo is becoming a **first-class Dynamo backend** (same
|
||||
tier as vllm, sglang, trtllm). The refactor must produce an API that
|
||||
Dynamo's `components/src/dynamo/fastvideo/` package can consume as a
|
||||
pure Python import, without re-introducing the legacy flat-kwarg
|
||||
surface. Draft PR ai-dynamo/dynamo#7544 defines the concrete integration
|
||||
shape we need to support.
|
||||
|
||||
This materially changes the tail of the API refactor plan. The current
|
||||
PR 5 ("wire `ServeConfig.default_request` into the OpenAI-compatible
|
||||
HTTP server") addresses only the stateless endpoint; the real upstream
|
||||
target is a much larger, session-based stack **plus** a clean Dynamo
|
||||
backend contract.
|
||||
|
||||
This document captures:
|
||||
- what's being upstreamed and where it lands
|
||||
- four design decisions that shape the upstream (continuation model,
|
||||
streaming server layout, LLM provider abstraction, Dynamo backend
|
||||
integration)
|
||||
- a revised PR sequence for the tail of the refactor
|
||||
|
||||
## What's being upstreamed
|
||||
|
||||
| Internal path | Size | Role | Upstream target |
|
||||
|---|---|---|---|
|
||||
| `server/main.py` | 94KB | FastAPI + WebSocket, session lifecycle, segment orchestration | `fastvideo/entrypoints/streaming/server.py` + handlers |
|
||||
| `server/gpu_pool.py` | 66KB | GPU orchestration, subprocess workers | `fastvideo/entrypoints/streaming/gpu_pool.py` |
|
||||
| `server/prompt_enhancer.py` | 69KB | LLM orchestration (cerebras_ifm, cerebras, groq) | `fastvideo/entrypoints/streaming/prompt/` package |
|
||||
| `server/mock_server.py` | 45KB | Mock backend for dev/tests | `fastvideo/entrypoints/streaming/mock_server.py` |
|
||||
| `server/prompt_safety.py` | 7KB | Optional fasttext-gated prompt safety | `fastvideo/entrypoints/streaming/prompt/safety.py` |
|
||||
| `server/session_init_image.py` | 3KB | i2v init image handling | `fastvideo/entrypoints/streaming/session_init_image.py` |
|
||||
| `server/rewrite_prompt_payload.py` | 3KB | Rewrite flow payload builder | `fastvideo/entrypoints/streaming/prompt/rewrite.py` |
|
||||
| `server/session_logger.py` | 1KB | Session JSONL logs | `fastvideo/entrypoints/streaming/session_logger.py` |
|
||||
| `server/config.py` | 9KB | Env-driven server config | Typed `ServeConfig` extensions |
|
||||
| `router/main.py` | 27KB | Multi-replica load balancer + WS proxy | `fastvideo/entrypoints/streaming/router/` (or separate package) |
|
||||
| `slurm/` | — | Deployment scripts | Likely stays internal |
|
||||
|
||||
## FastVideo contact surface today
|
||||
|
||||
Direct calls from the internal stack into FastVideo, all in `gpu_pool.py`:
|
||||
|
||||
| Location | Call | Notes |
|
||||
|---|---|---|
|
||||
| `gpu_pool.py:164` | `from fastvideo.entrypoints.video_generator import VideoGenerator` | Subprocess-level import, post-`CUDA_VISIBLE_DEVICES` setup |
|
||||
| `gpu_pool.py:230` | `PipelineConfig.from_pretrained(config_model_path)` | Direct access to legacy `PipelineConfig` |
|
||||
| `gpu_pool.py:231` | `pipeline_config.dit_config.quant_config = FP4Config()` | Direct internals mutation |
|
||||
| `gpu_pool.py:264-267` | `VideoGenerator.from_pretrained(model_root, **load_kwargs)` | Flat legacy kwargs |
|
||||
| `gpu_pool.py:837` | `generator.generate_video(**request_kwargs)` | Per-segment flat kwargs |
|
||||
| `gpu_pool.py:282-288` | `LTX2AudioEncoder`, `AudioProcessor`, `get_diffusers_config` | Audio re-encode path |
|
||||
|
||||
`load_kwargs` at `gpu_pool.py:233-260` contains:
|
||||
`ltx2_refine_enabled`, `ltx2_refine_upsampler_path`, `ltx2_refine_lora_path`,
|
||||
`ltx2_refine_num_inference_steps`, `ltx2_refine_guidance_scale`,
|
||||
`ltx2_refine_add_noise`, `pipeline_config`, `torch_compile_kwargs`,
|
||||
`dit_cpu_offload`, `dit_layerwise_offload`, `vae_cpu_offload`,
|
||||
`text_encoder_cpu_offload`, `pin_cpu_memory`, `ltx2_vae_tiling`,
|
||||
`use_fsdp_inference`, `enable_torch_compile`.
|
||||
|
||||
`request_kwargs` at `gpu_pool.py:837` includes:
|
||||
`ltx2_audio_clean_latent`, `ltx2_audio_denoise_mask`,
|
||||
`ltx2_video_conditions`, `video_position_offset_sec`, standard sampling
|
||||
fields.
|
||||
|
||||
**Implication**: upstreaming `gpu_pool.py` as-is perpetuates the flat
|
||||
kwarg surface inside the public server. We need a typed translation
|
||||
(PR 6 expansion) at the worker boundary before, or as part of, the
|
||||
gpu_pool upstream.
|
||||
|
||||
## Session / continuation semantics today
|
||||
|
||||
Per-session state (in `server/main.py`):
|
||||
- `locked_segment_prompts`, `curated_prompts`, `segment_idx`,
|
||||
`generated_segment_count`, `loop_iteration`
|
||||
|
||||
Per-**GPU** (not per-session) continuation cache (in `gpu_pool.py`):
|
||||
- `ltx2_continuation_images` — last 9 decoded frames for clip conditioning
|
||||
- `ltx2_continuation_audio_latents` — denoised audio latents for audio conditioning
|
||||
|
||||
Segment N+1 automatically conditions on segment N's trailing frames and
|
||||
audio. On session reset or handoff (`USER_JOIN`), the per-GPU cache is
|
||||
cleared. There is currently **no way for a client to serialize and
|
||||
resume continuation state elsewhere** — it lives on the GPU only.
|
||||
|
||||
## Design Decision 1: Continuation model
|
||||
|
||||
### Options
|
||||
|
||||
**A. Opaque client-round-trip payload** (current plan PR 7 design)
|
||||
- Server returns `ContinuationState(kind, payload)`; client sends it back.
|
||||
- Pro: stateless server, trivially load-balanceable, survives disconnects.
|
||||
- Con: large payloads (frames + audio latents) over every request hop;
|
||||
bandwidth heavy on multi-segment WebSocket sessions.
|
||||
|
||||
**B. Server-held session state** (internal reality)
|
||||
- Continuation lives per-GPU; implicit between adjacent segments.
|
||||
- Pro: zero client bandwidth; fast; matches today.
|
||||
- Con: needs GPU affinity, no resume after disconnect, harder to scale horizontally.
|
||||
|
||||
**C. Hybrid** (recommended)
|
||||
- Server-held is the default for streaming WebSocket sessions.
|
||||
- Server exposes a `snapshot_state` message that returns the opaque
|
||||
payload form for migration/retry.
|
||||
- Stateless HTTP endpoints always use round-trip opaque payloads.
|
||||
- One serialization format underlies both surfaces.
|
||||
|
||||
### Decision: **C (Hybrid)**
|
||||
|
||||
Rationale: matches both internal streaming use (server-held, fast) and
|
||||
stateless API use (client-round-trip, resumable). Cost is one serialization
|
||||
layer that serves both.
|
||||
|
||||
### Implications
|
||||
- `ContinuationState.kind` identifies the payload schema
|
||||
(e.g. `"ltx2.v1"`).
|
||||
- `ContinuationState.payload` must cover:
|
||||
- trailing conditioning frames (or a tensor reference)
|
||||
- audio latents (or a tensor reference)
|
||||
- segment index / rollout position
|
||||
- any model-specific conditioning metadata (e.g. audio sample rate,
|
||||
`video_position_offset_sec`)
|
||||
- For large tensors, payload may reference a server-side blob by ID
|
||||
rather than inline everything.
|
||||
- Streaming server has a `SessionStore` keyed by session ID that holds
|
||||
a typed `LTX2ContinuationState` object.
|
||||
- `SessionStore.snapshot(session_id) -> ContinuationState` serializes
|
||||
the current state for export.
|
||||
- `SessionStore.hydrate(state: ContinuationState) -> session_id` loads
|
||||
a state into a new session.
|
||||
- Plan PR 7 expands to cover both surfaces and define the payload schema.
|
||||
|
||||
## Design Decision 2: Streaming server layout
|
||||
|
||||
### Options
|
||||
|
||||
- **A. `fastvideo/entrypoints/streaming/`** — parallel to
|
||||
`fastvideo/entrypoints/openai/`
|
||||
- **B. `fastvideo/entrypoints/server/{stateless,streaming}/`** — reorg both
|
||||
- **C. `fastvideo/streaming/`** — top-level package, not under entrypoints
|
||||
|
||||
### Decision: **A (parallel subpackage)**
|
||||
|
||||
Rationale: lowest-friction, no existing code moves, both servers share
|
||||
the same `fastvideo/entrypoints/*` namespace and import style. Shared
|
||||
utilities can be factored into `fastvideo/entrypoints/server_common/`
|
||||
later if needed. Option B creates churn across every openai/ import for
|
||||
marginal organizational win.
|
||||
|
||||
### Target layout
|
||||
|
||||
```text
|
||||
fastvideo/entrypoints/
|
||||
├── openai/ # existing: stateless HTTP POST
|
||||
│ ├── api_server.py
|
||||
│ ├── video_api.py
|
||||
│ ├── image_api.py
|
||||
│ ├── common_api.py
|
||||
│ ├── protocol.py
|
||||
│ ├── state.py
|
||||
│ ├── stores.py
|
||||
│ └── utils.py
|
||||
├── streaming/ # NEW: session WebSocket
|
||||
│ ├── server.py # FastAPI + WebSocket entry
|
||||
│ ├── session.py # session lifecycle, state machine
|
||||
│ ├── session_store.py # typed session state + snapshot/hydrate
|
||||
│ ├── protocol.py # JSON WebSocket message schemas
|
||||
│ ├── stream.py # fMP4 encoding (av_fmp4 mode)
|
||||
│ ├── gpu_pool.py # subprocess workers
|
||||
│ ├── worker.py # per-GPU worker loop
|
||||
│ ├── continuation.py # typed LTX2 state payload
|
||||
│ ├── session_init_image.py
|
||||
│ ├── session_logger.py
|
||||
│ ├── mock_server.py
|
||||
│ ├── prompt/
|
||||
│ │ ├── enhancer.py # provider-agnostic prompt ops
|
||||
│ │ ├── rewrite.py
|
||||
│ │ ├── safety.py # optional fasttext
|
||||
│ │ ├── payload.py # rewrite payload builder
|
||||
│ │ └── providers/
|
||||
│ │ ├── base.py # LLMProvider protocol
|
||||
│ │ ├── cerebras.py
|
||||
│ │ ├── cerebras_ifm.py
|
||||
│ │ └── groq.py
|
||||
│ └── router/ # or separate top-level package
|
||||
│ ├── main.py
|
||||
│ └── registry.py
|
||||
├── cli/ # existing
|
||||
└── video_generator.py # existing
|
||||
```
|
||||
|
||||
### Config integration
|
||||
|
||||
`ServeConfig` gets an optional `streaming: StreamingConfig | None` field:
|
||||
|
||||
```python
|
||||
@dataclass
|
||||
class StreamingConfig:
|
||||
session_timeout_seconds: int = 300
|
||||
generation_segment_cap: int = 6
|
||||
stream_mode: Literal["av_fmp4", "legacy_jpeg"] = "av_fmp4"
|
||||
warmup: WarmupConfig = field(default_factory=WarmupConfig)
|
||||
pool: GpuPoolConfig = field(default_factory=GpuPoolConfig)
|
||||
prompt: PromptEnhancerConfig | None = None
|
||||
safety: PromptSafetyConfig | None = None
|
||||
|
||||
@dataclass
|
||||
class GpuPoolConfig:
|
||||
num_workers: int | None = None # default: CUDA_VISIBLE_DEVICES count
|
||||
enable_audio_reencode: bool = True
|
||||
conditioning_num_frames: int = 9
|
||||
conditioning_end_offset: int = 0
|
||||
|
||||
@dataclass
|
||||
class PromptEnhancerConfig:
|
||||
provider: Literal["cerebras_ifm", "cerebras", "groq"] = "cerebras_ifm"
|
||||
model: str = "gpt-oss-120b"
|
||||
timeout_ms: int = 20000
|
||||
system_prompt_dir: str | None = None # hot-reloadable system prompts
|
||||
|
||||
@dataclass
|
||||
class PromptSafetyConfig:
|
||||
enabled: bool = False
|
||||
classifier_path: str | None = None
|
||||
```
|
||||
|
||||
## Design Decision 3: LLM provider abstraction
|
||||
|
||||
### Problem
|
||||
|
||||
`prompt_enhancer.py` (69KB) hard-codes three providers (cerebras_ifm,
|
||||
cerebras, groq) with provider-specific request/response handling
|
||||
scattered throughout. Upstreaming as-is locks FastVideo to those three
|
||||
providers and couples the prompt operations to their response shapes.
|
||||
|
||||
### Shape
|
||||
|
||||
Introduce an `LLMProvider` protocol:
|
||||
|
||||
```python
|
||||
from typing import Protocol, AsyncIterator, Literal
|
||||
from dataclasses import dataclass
|
||||
|
||||
@dataclass
|
||||
class LLMMessage:
|
||||
role: Literal["system", "user", "assistant"]
|
||||
content: str
|
||||
|
||||
@dataclass
|
||||
class LLMRequest:
|
||||
messages: list[LLMMessage]
|
||||
model: str
|
||||
max_tokens: int | None = None
|
||||
temperature: float | None = None
|
||||
timeout_ms: int | None = None
|
||||
|
||||
@dataclass
|
||||
class LLMResponse:
|
||||
content: str
|
||||
provider: str
|
||||
model: str
|
||||
latency_ms: float
|
||||
fallback_used: bool = False
|
||||
|
||||
class LLMProvider(Protocol):
|
||||
name: str
|
||||
async def complete(self, request: LLMRequest) -> LLMResponse: ...
|
||||
```
|
||||
|
||||
### Decision: **Protocol + built-in implementations for cerebras, cerebras_ifm, groq**
|
||||
|
||||
Rationale: keeps the prompt enhancer free of provider-specific branching;
|
||||
users (and future OpenAI/Anthropic/local additions) can register their
|
||||
own provider without modifying FastVideo. Each built-in provider is
|
||||
100-200 LOC; the enhancer becomes provider-agnostic prompt orchestration.
|
||||
|
||||
### Implications
|
||||
- `prompt_enhancer.py` splits into `enhancer.py` (prompt operations) +
|
||||
`providers/` (IO).
|
||||
- Config moves from scattered env vars to typed `PromptEnhancerConfig`
|
||||
under `ServeConfig.streaming.prompt`.
|
||||
- Hot-reloadable system prompts stay — exposed as a management endpoint
|
||||
on the streaming server.
|
||||
- Fallback behavior (retry across providers in priority order) moves
|
||||
into the enhancer layer, orthogonal to provider implementations.
|
||||
|
||||
## Design Decision 4 preamble: what Dynamo expects from FastVideo
|
||||
|
||||
Dynamo's backend pattern (observed in
|
||||
`dynamo/components/src/dynamo/sglang/` and confirmed by PR #7544) is a
|
||||
**pure Python import** pattern. Dynamo owns the backend subpackage in its
|
||||
own repo; FastVideo only needs to expose a stable, typed, aggregated
|
||||
and (later) streaming generation surface.
|
||||
|
||||
### Contract surface Dynamo consumes
|
||||
|
||||
| Surface | Shape | Notes |
|
||||
|---|---|---|
|
||||
| Constructor | `VideoGenerator.from_pretrained(model_path, **typed_kwargs)` | Already exists; `typed_kwargs` must be a stable subset from `GeneratorConfig` — no flat LTX2 legacy kwargs. |
|
||||
| Sync execution | `generator.generate_video(request: GenerationRequest) -> VideoResult` | Aggregated mode; Dynamo wraps in `asyncio.to_thread` under an `asyncio.Lock`. |
|
||||
| Async execution | `generator.generate_async(request: GenerationRequest) -> AsyncGenerator[VideoEvent, None]` | Needed for: (a) streaming server fMP4 chunks; (b) future Dynamo disaggregation. Events: `Progress`, `Partial?`, `Final`. |
|
||||
| Typed request | `fastvideo.api.GenerationRequest`, `SamplingConfig`, `InputConfig` | Stable import path; Dynamo's adapter builds this from `NvCreateVideoRequest` + `VideoNvExt`. |
|
||||
| Typed result | `VideoResult` with `video_bytes` or tensor frames, plus `ContinuationState?` | Must be picklable / JSON-serializable enough for Dynamo RPC. |
|
||||
| Continuation | `ContinuationState(kind, payload)` with schema-versioned payloads | Used by FastVideo's session store today; tomorrow by Dynamo disaggregated workers. |
|
||||
| Health check input | `VideoGenerator.default_health_check_request() -> GenerationRequest` | Minimal 256x256 / 8 frames / 1 step; lets Dynamo's `FastVideoHealthCheckPayload.to_dict()` produce the Dynamo `health_check_payload` kwarg without knowledge of FastVideo internals. |
|
||||
| Config dump | `GeneratorConfig.to_dict()` / `ServeConfig.to_dict()` | Dynamo calls `dynamo.common.config_dump.dump_config(path, config)` at worker start; we already have `config_to_dict()`. |
|
||||
|
||||
### Request/response mapping (Dynamo ↔ FastVideo)
|
||||
|
||||
Dynamo's video protocol (`NvCreateVideoRequest` / `NvVideosResponse`):
|
||||
|
||||
```
|
||||
NvCreateVideoRequest -> fastvideo.api.GenerationRequest
|
||||
prompt -> sampling.prompt
|
||||
size="WxH" -> sampling.width, sampling.height
|
||||
seconds -> (seconds * nvext.fps) -> sampling.num_frames
|
||||
input_reference -> input.image_path / input.video_path
|
||||
nvext.fps -> sampling.fps
|
||||
nvext.num_frames -> sampling.num_frames (overrides seconds*fps)
|
||||
nvext.num_inference_steps -> sampling.num_inference_steps
|
||||
nvext.guidance_scale -> sampling.guidance_scale
|
||||
nvext.seed -> sampling.seed
|
||||
nvext.negative_prompt -> sampling.negative_prompt
|
||||
response_format -> (handled by adapter at output)
|
||||
|
||||
VideoFinalEvent -> NvVideosResponse
|
||||
video_bytes -> data[0].b64_json (if response_format=b64_json)
|
||||
video_url (after upload) -> data[0].url (if response_format=url)
|
||||
metadata.inference_time_s -> inference_time_s
|
||||
```
|
||||
|
||||
All fields already exist (or will exist after PR 6 expansion) on
|
||||
FastVideo's typed schema. No FastVideo changes required beyond what the
|
||||
rest of this plan already covers **except**:
|
||||
|
||||
1. `generate_async` must exist (new in PR 7.10).
|
||||
2. `default_health_check_request()` helper (new in PR 7.10).
|
||||
3. The sync `generate_video(request=...)` path must be reachable without
|
||||
extra wrapping (exists since PR 2; confirm stability).
|
||||
|
||||
### Where the Dynamo subpackage lives
|
||||
|
||||
The Dynamo-side integration (`FastVideoHandler`, `register_fastvideo_model`,
|
||||
`FastVideoHealthCheckPayload`, args parsing, main.py, Dockerfile,
|
||||
request/response mapping) lives **entirely in the Dynamo repo** at
|
||||
`components/src/dynamo/fastvideo/`, matching the pattern used by vllm
|
||||
and sglang. FastVideo does **not** host any Dynamo-related subpackage,
|
||||
Dynamo dependency, or Dynamo-specific CLI. FastVideo's only obligation
|
||||
is to expose a clean, stable, typed Python API that Dynamo's backend
|
||||
package can import.
|
||||
|
||||
## Design Decision 4: Dynamo as first-class backend target
|
||||
|
||||
### Problem
|
||||
|
||||
PR #7544 (closed) shows two frictions with the pre-refactor API:
|
||||
|
||||
1. **Flat legacy kwargs** — the Dynamo handler had to know about
|
||||
LTX2-specific flat names.
|
||||
2. **Sync-only generation** — Dynamo's async handler wrapped
|
||||
`generator.generate(...)` in `asyncio.to_thread` under a lock; no
|
||||
progress streaming, no disaggregation path.
|
||||
|
||||
The refactor's stateless OpenAI server, WebSocket streaming server, and
|
||||
Dynamo backend all want the same thing: **a typed async API that yields
|
||||
progress events and a typed final result**. If we build it once in
|
||||
`VideoGenerator`, all three adapters become thin.
|
||||
|
||||
### Options
|
||||
|
||||
**A. Keep sync-only, each adapter wraps**
|
||||
- Simple; matches PR #7544.
|
||||
- Con: streaming server needs its own async runner; Dynamo loses progress
|
||||
streaming; no path to disaggregation.
|
||||
|
||||
**B. Add async event stream to `VideoGenerator`**
|
||||
- `generate_async(request) -> AsyncGenerator[VideoEvent, None]`.
|
||||
- Sync `generate_video` becomes a thin `asyncio.run` wrapper internally.
|
||||
- Pro: one canonical execution API; streaming server, OpenAI server,
|
||||
and Dynamo all consume events directly.
|
||||
- Con: larger delta in `VideoGenerator` — must thread async through the
|
||||
pipeline step loop.
|
||||
|
||||
**C. Queue-based `generate(request, event_cb)` callback**
|
||||
- Middle ground; callback receives events.
|
||||
- Pro: no async rewrite needed.
|
||||
- Con: callers have to invert control; awkward for Dynamo's async
|
||||
handler.
|
||||
|
||||
### Decision: **B (async event stream)**
|
||||
|
||||
Rationale: one substrate serves all three consumers. The cost is a
|
||||
`generate_async` implementation that runs the pipeline step loop in a
|
||||
thread and bridges events back via an asyncio queue — standard pattern,
|
||||
limited surface area.
|
||||
|
||||
### Implications
|
||||
|
||||
- New PR 7.10 adds `generate_async` on `VideoGenerator` with three event
|
||||
types: `VideoProgressEvent(step, total_steps, stage)`,
|
||||
`VideoPartialEvent(frames_ndarray, index)` (optional; emitted only in
|
||||
the streaming path), `VideoFinalEvent(video_bytes_or_tensor, metadata,
|
||||
continuation_state?)`.
|
||||
- Sync `generate_video(request=...)` becomes `asyncio.run(...)` over
|
||||
`generate_async`, collecting events and returning the final.
|
||||
- Streaming server's fMP4 encoder consumes `VideoPartialEvent` frames
|
||||
directly, never re-decoding through disk.
|
||||
- Dynamo adapter consumes `generate_async` and yields one
|
||||
`NvVideosResponse` per `VideoFinalEvent` (aggregated mode; ignores
|
||||
intermediate events today; can surface progress via Dynamo's
|
||||
status/progress fields in the future).
|
||||
- `ContinuationState` can be attached to `VideoFinalEvent.metadata`,
|
||||
giving Dynamo a first-class way to surface state for disaggregation
|
||||
later.
|
||||
- Stable public exports: `from fastvideo import VideoGenerator`;
|
||||
`from fastvideo.api import GenerationRequest, SamplingConfig,
|
||||
ContinuationState, VideoResult, VideoEvent`.
|
||||
- No Dynamo subpackage, dep, or CLI lives in FastVideo. The adapter
|
||||
(`NvCreateVideoRequest ↔ GenerationRequest` mapping, handler,
|
||||
registration) lives entirely in the Dynamo repo at
|
||||
`components/src/dynamo/fastvideo/`.
|
||||
|
||||
### Constraints this adds to earlier PRs
|
||||
|
||||
- **PR 6** (typed LTX2 kwargs): every flat kwarg must have a typed home
|
||||
**reachable from `GeneratorConfig`**, so Dynamo can construct the
|
||||
generator without importing internal compat paths.
|
||||
- **PR 7** (continuation state): `ContinuationState.payload` must be
|
||||
JSON/YAML serializable (no raw torch tensors inline; use blob
|
||||
indirection) so it survives Dynamo RPC transport.
|
||||
- **PR 7.5** (streaming skeleton): consume `generate_async` rather than
|
||||
re-implementing a progress loop around `generate_video`.
|
||||
- **PR 2/3/4 already landed**: the typed request shape is fixed and
|
||||
matches Dynamo's mapping needs — no backtracking required.
|
||||
|
||||
## Revised PR sequence (PR 5 onwards)
|
||||
|
||||
PRs 0-4 are unchanged and already landed. PR 5 is narrowed; PRs 5.5-7.9
|
||||
are new inserts; PRs 8-13 are reshaped or kept.
|
||||
|
||||
| # | Title | Change | Key deliverables |
|
||||
|---|---|---|---|
|
||||
| **5** | Stateless `ServeConfig.default_request` merge | **Narrowed.** Wire typed default-request into `fastvideo/entrypoints/openai/`. | `_merge_default_request` helper, validated-against-preset, tests for default+user-override precedence |
|
||||
| **5.5** | Server architecture split | **NEW.** Introduce `fastvideo/entrypoints/streaming/` subpackage skeleton. No behavior change. | Empty subpackage + stub server.py; CLI subcommand `fastvideo streaming-serve` (raises NotImplementedError); doc on layout |
|
||||
| **6** | LTX2 public preset + stage overrides + config colocation | **Expanded.** Also add typed replacements for every flat kwarg used by internal `gpu_pool.py`. | `ltx2_two_stage` preset, `LTX2RefineStageOverride`, `CompileConfig` field types, typed `FP4Config` integration, colocation |
|
||||
| **7** | Continuation state (public + session) | **Expanded.** Define both opaque payload AND server-held session store. | `ContinuationState.payload` schema, `LTX2ContinuationState` typed subclass, `SessionStore` interface, snapshot/hydrate APIs |
|
||||
| **7.5** | Streaming server skeleton | **NEW.** Minimum viable WebSocket server: session lifecycle, JSON messages, fMP4 output, single-generator. | `server.py`, `session.py`, `protocol.py`, `stream.py` (fMP4), typed `StreamingConfig` |
|
||||
| **7.6** | GPU pool upstream | **NEW.** Upstream `gpu_pool.py` with typed config boundary. | `gpu_pool.py`, `worker.py`, job queue, session-to-GPU binding, session timeout handling |
|
||||
| **7.7** | Prompt enhancer upstream | **NEW.** Upstream `prompt_enhancer.py` with `LLMProvider` abstraction. | `prompt/enhancer.py`, `prompt/providers/{base,cerebras,cerebras_ifm,groq}.py`, hot-reloadable system prompts |
|
||||
| **7.8** | Streaming auxiliaries | **NEW.** Small, isolated. | `prompt/safety.py`, `session_init_image.py`, `prompt/rewrite.py`, `session_logger.py`, `mock_server.py` |
|
||||
| **7.9** | Router upstream | **NEW.** Multi-replica load balancer + WS proxy. | `streaming/router/` (or separate top-level package), health checks, WS proxy |
|
||||
| **7.10** | Dynamo backend contract | **NEW.** Add `VideoGenerator.generate_async` event stream + `default_health_check_request()` helper. FastVideo exposes the async API only; the Dynamo backend package (handler, adapter, registration) lives entirely in the Dynamo repo at `components/src/dynamo/fastvideo/`. Streaming server (PR 7.5) and Dynamo backend both consume the same async API. | `generate_async` with `VideoProgressEvent`/`VideoPartialEvent`/`VideoFinalEvent`; sync `generate_video` becomes a thin wrapper; contract tests against a mock Dynamo-style handler that imports only public FastVideo APIs |
|
||||
| **8** | Internal-UI ↔ public-server contract docs & tests | **Reframed.** Was "Dreamverse Server Adaptation Layer." Also covers Dynamo integration reference. | WebSocket protocol reference, contract tests, migration examples, Dynamo adapter example that upstream PR can copy verbatim |
|
||||
| **9** | LongCat preset migration + colocation | **Keep.** | Stage overrides, colocation |
|
||||
| **10** | Hunyuan15 SR preset migration + colocation | **Keep.** | Stage overrides, SR field migration POC, colocation |
|
||||
| **11** | SSIM / perf test migration | **Keep.** Now blocked on PR 6 expansion. | Typed API migration of golden tests |
|
||||
| **12** | Docs + examples | **Keep, expand.** | Streaming server docs now part of scope |
|
||||
| **13** | Deprecation + cleanup | **Keep, expand.** | Also deprecate flat kwargs that internal gpu_pool uses today |
|
||||
|
||||
Total PR count: 13 → ~20 (13 original + 5 streaming-upstream inserts +
|
||||
1 architecture split + 1 Dynamo contract). Each new PR is small and
|
||||
self-contained because the streaming components are already cleanly
|
||||
separated in the internal repo, and the Dynamo contract rides on top of
|
||||
the async API that the streaming server already needs.
|
||||
|
||||
## Open questions
|
||||
|
||||
1. **Router: in-repo or separate package?** — It's orthogonal to inference;
|
||||
in-repo couples deploy cycles, separate leaves FastVideo cleaner.
|
||||
Recommendation: separate package `fastvideo-router/` or
|
||||
`fastvideo/contrib/router/`; defer final call to PR 7.9.
|
||||
2. **Session ID authority** — internal uses ad-hoc client IDs.
|
||||
Recommendation: server-generated UUID, accept externally provided
|
||||
session ID only for resume flows.
|
||||
3. **Torch compile kwargs typing** — `CompileConfig.kwargs: dict[str, Any]`
|
||||
today accepts `mode`, `backend`, `fullgraph`, `dynamic`. Options: keep
|
||||
as opaque dict; fully type; hybrid (type the common four + allow
|
||||
extras). Recommendation: hybrid, type common fields.
|
||||
4. **Prompt safety / fasttext dependency** — heavy for users who don't
|
||||
need it. Recommendation: ship as optional extra
|
||||
`pip install fastvideo[prompt-safety]`.
|
||||
5. **Audio-specific tensor payloads** — `ltx2_audio_clean_latent`,
|
||||
`ltx2_audio_denoise_mask`, `ltx2_audio_latents` are not in the current
|
||||
public schema. PR 7 should classify them (probably as opaque fields
|
||||
inside `LTX2ContinuationState.payload`, not top-level sampling fields).
|
||||
6. **Batching behavior** — internal `test_batching.py` suggests batching
|
||||
is exercised. Scope this into PR 7.5 or defer to a post-cleanup perf PR?
|
||||
7. ~~**Dynamo subpackage home**~~ — **Resolved.** No Dynamo code lives
|
||||
in FastVideo. The full backend package (handler, adapter,
|
||||
registration, health check) is owned by the Dynamo repo at
|
||||
`components/src/dynamo/fastvideo/`, same pattern as vllm/sglang.
|
||||
FastVideo only guarantees the public API contract listed above.
|
||||
8. **Disaggregation readiness** — PR #7544 is aggregated-only. Our
|
||||
`ContinuationState` hybrid already supports a future prefill/decode
|
||||
split (prefill yields state; decode hydrates it). Should PR 7.10
|
||||
explicitly validate that `ContinuationState` survives round-trip
|
||||
through a Dynamo-style RPC (pickle or JSON), even though Dynamo
|
||||
isn't using it today? Recommendation: yes; cheap contract test that
|
||||
prevents drift.
|
||||
9. **Dynamo progress/status passthrough** — `NvVideosResponse` has
|
||||
`status` and `progress` fields. Should PR 7.10's handler contract
|
||||
emit intermediate `NvVideosResponse` chunks keyed off
|
||||
`VideoProgressEvent`, or stay aggregated-final-only to match PR
|
||||
#7544? Recommendation: stay aggregated-final for PR 7.10; revisit
|
||||
after Dynamo clarifies their streaming/progress semantics.
|
||||
|
||||
## Immediate path forward
|
||||
|
||||
1. Land `will/api_5` cleanup commits — **done** (`e03ca7d9`, `41f93179`
|
||||
force-pushed without Claude co-author).
|
||||
2. Review this plan with a human — commit the doc to capture the state.
|
||||
3. Execute PR 5 (narrow stateless merge) and PR 5.5 (subpackage split)
|
||||
in parallel. Both small; both unblock the streaming upstream that
|
||||
follows.
|
||||
4. Start PR 6 expansion (typed replacements for flat LTX2 kwargs) as the
|
||||
critical path for PR 7.6 (gpu_pool upstream).
|
||||
+93
@@ -0,0 +1,93 @@
|
||||
# Exploration Log: Video Generator Config API Design
|
||||
|
||||
## Status: draft
|
||||
|
||||
## Context
|
||||
FastVideo's Python inference API currently mixes generator-instance settings,
|
||||
pipeline initialization settings, and per-request sampling/runtime settings
|
||||
through broad `**kwargs` surfaces on `VideoGenerator.from_pretrained(...)` and
|
||||
`VideoGenerator.generate_video(...)`.
|
||||
|
||||
This exploration compares the current FastVideo design with
|
||||
`sglang/multimodal_gen` and examines how to upstream multi-stage LTX2 /
|
||||
Dreamverse behavior without growing more ad hoc top-level flags.
|
||||
|
||||
## Progress
|
||||
- [x] Read FastVideo onboarding, codebase map, and relevant design docs.
|
||||
- [x] Inspect current FastVideo generator, args, sampling, registry, and
|
||||
workflow abstractions.
|
||||
- [x] Inspect internal LTX2 streaming server usage and current two-stage /
|
||||
continuation requirements.
|
||||
- [x] Inspect SGL diffusion generator, server args, sampling params, and
|
||||
request preparation boundary.
|
||||
- [x] Inspect vLLM-Omni stage config, stage metadata, request, and orchestration
|
||||
surfaces for multi-stage pipeline ideas.
|
||||
- [x] Inspect current FastVideo CLI/config-file loading and compare with the
|
||||
training YAML-only entrypoint.
|
||||
- [ ] Convert findings into a concrete implementation plan for FastVideo.
|
||||
|
||||
## Findings
|
||||
- FastVideo already has the right internal separation points:
|
||||
`FastVideoArgs`, `PipelineConfig`, `SamplingParam`, and `ForwardBatch`.
|
||||
- The public boundary is the unstable part:
|
||||
init-time and request-time knobs are mixed through `**kwargs`.
|
||||
- Unknown init keys can be silently filtered, while unknown request keys can be
|
||||
only logged rather than rejected. This makes API drift hard to detect.
|
||||
- SGL's split is cleaner:
|
||||
`ServerArgs` for engine/runtime, `PipelineConfig` for model-family wiring,
|
||||
and `SamplingParams` for per-request settings.
|
||||
- SGL also has better merge semantics for user request overrides:
|
||||
it preserves model defaults, tracks explicitly provided fields, and validates
|
||||
request params against pipeline task type.
|
||||
- SGL still has a design smell worth avoiding in FastVideo:
|
||||
`SamplingParams._adjust(...)` depends on `ServerArgs`, which leaks
|
||||
engine/pipeline concerns back into the request object.
|
||||
- vLLM-Omni contributes a useful extra abstraction beyond SGL:
|
||||
model-owned multi-stage topology via `ModelPipeline` and `StageConfig`,
|
||||
with per-stage defaults (`default_sampling_params`) and runtime override
|
||||
layering.
|
||||
- vLLM-Omni's best reusable idea for FastVideo is not the serving stack, but
|
||||
the separation between:
|
||||
1. model-defined stage topology and per-stage defaults,
|
||||
2. runtime engine overrides,
|
||||
3. request-time sampling/state handoff.
|
||||
- vLLM-Omni also shows the downside of exposing stage-indexed request lists too
|
||||
directly: `sampling_params_list` works for a serving engine, but is too
|
||||
positional and low-level for FastVideo's higher-level Python API.
|
||||
- FastVideo already supports YAML/JSON config files for inference CLI, but the
|
||||
current mechanism flattens nested documents back into argparse flags. This
|
||||
preserves backward compatibility but keeps the CLI surface as the canonical
|
||||
schema instead of a typed document model.
|
||||
- The training stack has a cleaner precedent: a YAML-first config loaded into a
|
||||
typed schema, with dotted CLI overrides applied onto the nested document
|
||||
before parsing. Inference can likely adopt a lighter variant of that pattern.
|
||||
- Multi-stage generation should be unified at the orchestration layer, not by
|
||||
forcing LongCat refine, Hunyuan SR, and LTX2 continuation into one leaf config.
|
||||
|
||||
## Mistakes / Dead Ends
|
||||
- A fully free-form string-dict API would lose too much type safety and would
|
||||
likely recreate the current drift problem under a different shape.
|
||||
- A single universal `RefineConfig` for all models would become a sparse bag of
|
||||
nullable fields and would not map cleanly to existing model families.
|
||||
|
||||
## Proposed Standardization
|
||||
- Introduce a typed public split:
|
||||
`GeneratorConfig` for instance-lifetime engine/init settings and
|
||||
`GenerationRequest` for per-call inputs/sampling/output.
|
||||
- Allow dict input only as an interchange layer that is parsed immediately into
|
||||
typed configs with strict unknown-key validation.
|
||||
- Add a typed `GenerationPlan` / multi-stage orchestration layer with
|
||||
discriminated stage configs:
|
||||
`SampleStageConfig`, `LongCatRefineStageConfig`,
|
||||
`HunyuanSRStageConfig`, `LTX2ContinuationStageConfig`.
|
||||
- Let model families own stage defaults and stage topology through named
|
||||
profiles or model-defined stage plans, similar in spirit to vLLM-Omni's
|
||||
pipeline YAMLs, but expose them through typed Python config objects rather
|
||||
than raw stage-indexed lists in the primary API.
|
||||
- Make YAML/JSON a first-class serialization of the same typed inference
|
||||
schema, not just a file format that expands into CLI flags.
|
||||
- Prefer a YAML-first CLI pattern for nested configs:
|
||||
`fastvideo generate --config run.yaml --request.sampling.seed 42`,
|
||||
while keeping a compatibility layer for existing flat flags during migration.
|
||||
- Upstream LTX2 two-stage / continuation behavior as a first-class stage or
|
||||
pipeline profile rather than more `ltx2_*` top-level kwargs.
|
||||
@@ -0,0 +1,196 @@
|
||||
# Current State — 2026-05-06 (D-26 rebase onto origin/main; PR base flipped)
|
||||
|
||||
Point-in-time snapshot of branches, commits, and live infrastructure.
|
||||
Update whenever commits land or services restart.
|
||||
|
||||
For HOW to commit / push / verify see [runbook.md](runbook.md). For
|
||||
roster of co-authors to credit on every commit see
|
||||
[authors.md](authors.md).
|
||||
|
||||
## Branch tips
|
||||
|
||||
| Repo | Branch | Tip | Distance |
|
||||
|---|---|---|---|
|
||||
| FastVideo | `will/ltx2_sr_port` (**PR #1288 head**) | `fbd823df` | merged-into-main pending; latest tip post-D-16 + integration-review + integration-plan + GPU4 smoke validation |
|
||||
| FastVideo | **`will/dreamverse-monorepo`** (**REBASED ONTO `origin/main`**, was forked from `will/ltx2_sr_port`) | `83829c5e` | 66 commits ahead of `origin/main` (`c17d33bf`). Contains the full LTX-2 SR port + NVFP4 + Dreamverse monorepo migration + audio kwarg fix + warmup + NVENC build + benchmarks + integration memory dir, all rebased to be openable as a single PR against main. PR #1288 on `will/ltx2_sr_port` is untouched. End-to-end verified on GPU4 pre-rebase with audio continuation across segments 1→2 (`Cached audio latents shape=(1, 8, 126, 16) for segment 2`, `Segment 2: relayed av chunks=22, bytes=3.8MB`, no BrokenPipeError). NVENC build supported but not usable on this dev host (B200 has no NVENC silicon — verified by direct ffmpeg probe). See [decisions-log.md D-19](decisions-log.md#d-19) + [D-20](decisions-log.md#d-20) + [D-21](decisions-log.md#d-21) + [D-26](decisions-log.md#d-26). |
|
||||
| FastVideo | `will/dreamverse-monorepo-pre-main-rebase-backup-20260506` | `2ee839a3` | **local-only safety backup** of the pre-rebase chain (the same 66 commits stacked on `2aaeee2a`). Keep until the new chain is fully verified by the next round of e2e on a non-stuck dev box or until the PR merges. |
|
||||
| FastVideo | `will/api_7.10` | `6ae7a99f` | **deprecated** — PR #1287 closed in favor of #1288. Branch can be deleted on origin and locally; kept for now as historical reference. |
|
||||
| FastVideo | `will/api_8`, `will/ltx2_sr_runtime`, `will/ltx2_nvfp4`, `will/ltx2_post_fixes`, `will/agents_cleanup` | (various) | **deprecated** split bookmarks. Strategy reversed to single mega-PR (D-17). Safe to delete locally; not pushed to origin. |
|
||||
| FastVideo | `will/ltx2_sr_port-pre-1286-rebase` | `1baa60bb` | **local-only safety backup** of pre-rebase chain (37 commits); keep until next slice merges |
|
||||
| Dreamverse | `will/integrate-public-fastvideo` | `ec8ef92` | 10 commits ahead of `737f3c1` (the dep switch) |
|
||||
| FastVideo-internal | their `main` | (read-only ref) | — |
|
||||
|
||||
FastVideo worktree default branch is `will/ltx2_sr_port`. Other agents
|
||||
share this worktree — if `git branch --show-current` shows something
|
||||
else, switch back cleanly with `git checkout will/ltx2_sr_port` (don't
|
||||
disturb their uncommitted work). I observed this happen repeatedly in
|
||||
the 2026-05-05 session — confirmed harmless; switching back was always
|
||||
safe with a clean working tree.
|
||||
|
||||
## Post-#1286 rebase summary
|
||||
|
||||
PR #1286 merged at `2aaeee2a` (squash). `will/ltx2_sr_port` was rebased
|
||||
onto new `origin/main`, dropping 4 commits whose content is now in main:
|
||||
|
||||
- `cd76cf51` `[feat] streaming: router (multi-replica load balancer)`
|
||||
- `1ac1e732` `[feat] streaming: fastvideo router-serve CLI`
|
||||
- `b0b7f59c` `[test] streaming: router registry + health loop ...`
|
||||
- `40e265b8` `[fix] streaming: router polish — bridge cancel + state
|
||||
machine + deps` (squashed into `2aaeee2a` via cherry-pick `a152cb77`)
|
||||
|
||||
Rebase was clean — no conflicts. All 33 surviving commits got new SHAs
|
||||
(rebase rewrites). The pre-rebase tip `1baa60bb` is preserved on the
|
||||
local backup branch `will/ltx2_sr_port-pre-1286-rebase`.
|
||||
|
||||
## New linearized chain (33 commits, slice indices for STACK.md)
|
||||
|
||||
| Slice | PR | Commits | Tip SHA | Subject |
|
||||
|---|---|---|---|---|
|
||||
| 1-3 | 7.10 (PR #1287) | 3 | `6ae7a99f` | `[test] streaming: generate_async coverage + refreshed streaming test` |
|
||||
| 4-6 | 8 | 3 | `f32e31ec` | `[test] streaming: contract tests for Dreamverse + Dynamo shapes` |
|
||||
| 7-15 | LTX-2 SR | 9 | `e7297519` | `feat(ltx2): full i2v conditioning + continuation latent port` |
|
||||
| 16-21 | NVFP4 | 6 | `6793166b` | `test(nvfp4): lock LTX-2 wiring + typed transformer_quant flow` |
|
||||
| 22-23 | LTX-2 post-fixes | 2 | `25897b67` | `[fix]: unwrap list-of-generator before torch.randn in LTX-2 latent prep` |
|
||||
| 24-33 | agents_cleanup | 10 | `b34d9704` | `[docs] dreamverse-integration: add runbook + fresh-context onboarding` |
|
||||
|
||||
5 PRs landed (7.5, 7.6, 7.7, 7.8, 7.9), 1 in flight (7.10), 5 remaining
|
||||
(8 / LTX-2 SR / NVFP4 / post-fixes / agents_cleanup).
|
||||
|
||||
## Historical commit chain analysis (pre-#1286 rebase)
|
||||
|
||||
The layered chain analysis below documented the pre-rebase SHAs (LTX-2
|
||||
SR layer, NVFP4 layer, post-handoff fixes layer). Those SHAs no longer
|
||||
exist on `will/ltx2_sr_port` — they live only on
|
||||
`will/ltx2_sr_port-pre-1286-rebase`. Content semantics are unchanged;
|
||||
SHAs were rewritten by the rebase. Kept here for narrative continuity.
|
||||
|
||||
## FastVideo: commit chain `cfccd292..156103b9`
|
||||
|
||||
Three layers since LTX-2 i2v port:
|
||||
|
||||
### Layer 1 — LTX-2 SR port + alignment harness (5 commits)
|
||||
|
||||
```
|
||||
365a66c7 feat(quantization): upstream LTX-2 FP4Config with lazy flashinfer
|
||||
433d26b2 feat(ltx2): port LTX-2 SR runtime — upsampler, refine stages, refine args
|
||||
751d05de feat(ltx2): wire SR pipeline graph + port denoising/latent-prep stages
|
||||
af6bbfea test(ltx2-sr): add numerical alignment harness — public vs internal
|
||||
974cd430 fix(ltx2-sr): close port gaps surfaced by alignment harness retries
|
||||
b6ac7630 test(ltx2-sr): pin ltx2 sampling knobs in harness for parity diff
|
||||
b043d550 fix(api): align public SamplingParam ltx2 defaults with distilled
|
||||
663dda80 fix(registry): order LTX-2 detectors so distilled wins for distilled paths
|
||||
cfccd292 feat(ltx2): full i2v conditioning + continuation latent port (BASE)
|
||||
```
|
||||
|
||||
(Predates the May 2 handoff.)
|
||||
|
||||
### Layer 2 — NVFP4 wire-up + per-component compile (6 commits, May 2 handoff)
|
||||
|
||||
```
|
||||
a4760bae fix(api): propagate generic refine_* args + match internal randn
|
||||
221cb20a feat(api): typed per-component CompileConfig + FastVideoArgs carriers
|
||||
6da342ba feat(compile): per-component compile + transformer_refine + prepare hook
|
||||
42b30bf9 feat(ltx2): wire FP4 inference through fastvideo.layers.quantization
|
||||
94c983a2 refactor(quant): rename FP4 → NVFP4 to disambiguate from other FP4 variants
|
||||
c6c14c55 test(nvfp4): lock LTX-2 wiring + typed transformer_quant flow
|
||||
```
|
||||
|
||||
See [quantization.md](quantization.md) for what each commit locks in.
|
||||
|
||||
### Layer 3 — Post-handoff parity/perf fixes (3 commits, since May 2)
|
||||
|
||||
```
|
||||
a5fcd19c [fix]: lazy-import flash_attn 2 fallback in attention backend
|
||||
d4ee5be2 [fix]: avoid model.to() round-trip in Gemma encoder forward
|
||||
156103b9 [fix]: unwrap list-of-generator before torch.randn in LTX-2 latent prep (HEAD)
|
||||
```
|
||||
|
||||
Three small fixes — no new features. Continued parity tightening with internal.
|
||||
|
||||
## Dreamverse: commit chain `737f3c1..ec8ef92`
|
||||
|
||||
```
|
||||
737f3c1 chore: switch fastvideo dep from FastVideo-internal to public FastVideo
|
||||
4cc6b30 chore: gitignore Playwright + Next.js build artifacts under apps/web
|
||||
33caa92 test(e2e): align Playwright specs with the actual production composer
|
||||
6fd137c test(e2e): tighten frontend-shell + preset specs to match actual UI
|
||||
248060b test(e2e): add Playwright tier with backend-health smoke + preset run
|
||||
d80c2a8 refactor(server): drive FP4 + per-component compile via typed GeneratorConfig
|
||||
3d7fd89 feat(skill): launch-demo orchestrator + fastvideo serve YAML
|
||||
72f69b9 Update ffmpeg installation instructions.
|
||||
1ba5635 fix(server): block startup on GPU warmup readiness, propagate failures
|
||||
ec8ef92 fix(server): detect worker death in _send_command via proc.sentinel (HEAD)
|
||||
```
|
||||
|
||||
The post-handoff trio (`72f69b9`, `1ba5635`, `ec8ef92`) hardens server
|
||||
startup robustness — ffmpeg install docs, GPU warmup readiness gate, and
|
||||
worker-death detection.
|
||||
|
||||
## Live services (do not duplicate)
|
||||
|
||||
| Port | Service | PID | Status |
|
||||
|---|---|---|---|
|
||||
| 8009 | `dreamverse-server` | 705513 | `/readyz` returns 200, 1 GPU worker on GPU 4, NVFP4 (50.9 GiB), `ENABLE_TORCH_COMPILE=0`, `FASTVIDEO_FFMPEG_BIN=$HOME/opt/ffmpeg-native/bin/ffmpeg`, `FASTVIDEO_VIDEO_CODEC=libx264` (post-D-20 deploy at 2026-05-05 14:??) |
|
||||
| 5274 | `next-server` (dev) | 707746 | 200 |
|
||||
| 8000 | unknown FastAPI | — | **Not in handoff.** Probably stray `fastvideo serve`. Verify with `lsof -i :8000` before launching a new BE on the default port. |
|
||||
|
||||
## Stashes — DO NOT POP
|
||||
|
||||
| Repo | Stash | Reason |
|
||||
|---|---|---|
|
||||
| FastVideo | `stash@{0}: WIP on main: 71bfc13d HunyuanVideo plugin` | Pre-existing, unrelated to integration work |
|
||||
| Dreamverse | `stash@{0}: wip: server modular refactor (split config/prompting/runtime/session)` | 3867-line orphan modular split, **not part of `will/integrate-public-fastvideo`**. Recover on a separate branch if needed. |
|
||||
|
||||
## Test status
|
||||
|
||||
| Suite | Status |
|
||||
|---|---|
|
||||
| FastVideo `fastvideo/tests/api/` (post-D-20) | 185 passed (was 222 before some tests moved; new `test_extra_overrides_routing.py` adds 7) |
|
||||
| FastVideo `contract/` + `nvfp4_*` + `ltx2_pipeline_smoke` (May 2 handoff) | 222 passed, 1 skipped |
|
||||
| Playwright e2e against live BE+FE (D-19) | 8 passed (5 backend-health + 2 frontend-shell + 1 preset-prompt-generation) |
|
||||
| Live segment-1→segment-2 audio continuation (D-20) | passes — `Cached audio latents shape=(1, 8, 126, 16) for segment 2` + `Segment 2: relayed av chunks=22, bytes=3.8MB`, no BrokenPipeError |
|
||||
| `dreamverse-deploy.sh` flag parser standalone test (D-20) | 13/13 permutations pass + bad-flag rejection (defaults / single flags / both flags / `--no-*` overrides / env-only / flag-overrides-env / both-env+both-no-flags / flags interleaved with positional args) |
|
||||
| `fastvideo serve --config streaming_demo.yaml` validation (May 2 handoff) | parses cleanly; dotted overrides work |
|
||||
| `bash -n` on `apps/dreamverse/scripts/install_native_ffmpeg.sh` + `dreamverse-deploy.sh` (D-20) | clean |
|
||||
|
||||
## Pre-existing failures (NOT caused by this work)
|
||||
|
||||
| Test | Failure | Notes |
|
||||
|---|---|---|
|
||||
| `fastvideo/tests/ops/quantization/test_absmax_fp8.py::test_create_weights_rejects_invalid_dtype` | `AssertionError not raised` | Pre-existing on `main`. Verified via `git stash` that NVFP4 work doesn't introduce it. See [open-threads.md](open-threads.md) item #2. |
|
||||
|
||||
## Source docs (archived 2026-05-03)
|
||||
|
||||
The 7 source docs that this memory dir consolidates have been moved into
|
||||
[`source-archive/`](source-archive/) — see the
|
||||
[archive README](source-archive/README.md) for the archive policy and
|
||||
synthesis mapping.
|
||||
|
||||
Other untracked items at the FastVideo repo root:
|
||||
- Nested clones: `dynamo/`, `ray/`, `vllm-omni/`
|
||||
- Lock files: `uv.lock`, `fastvideo/tests/ssim/.reference_videos_download.lock`
|
||||
- Skill dirs: `.agents/skills/diagnose-ssim-failure/`, `.agents/skills/review-pr-link/`
|
||||
- `.agents/exploration/pr-link-review.md` (kept; already promoted to a skill)
|
||||
|
||||
## Quick orientation commands
|
||||
|
||||
```bash
|
||||
# FastVideo state
|
||||
cd /home/william5lin/FastVideo
|
||||
git log --oneline cfccd292..HEAD # 14 commits this round
|
||||
|
||||
# Dreamverse state
|
||||
cd /home/william5lin/Dreamverse
|
||||
git log --oneline 737f3c1..HEAD # 10 commits this round
|
||||
|
||||
# Live stack health (already running)
|
||||
curl -s http://localhost:8009/readyz | head -c 300
|
||||
curl -s http://localhost:5274/ -o /dev/null -w "%{http_code}\n"
|
||||
|
||||
# Re-verify test suite
|
||||
.venv/bin/python -m pytest fastvideo/tests/api/ \
|
||||
fastvideo/tests/contract/ \
|
||||
fastvideo/tests/ops/quantization/test_nvfp4_*.py \
|
||||
tests/local_tests/pipelines/test_ltx2_pipeline_smoke.py \
|
||||
-q --no-header
|
||||
```
|
||||
@@ -0,0 +1,389 @@
|
||||
# Streaming Server Upstream — PRs 5.5 → 7.10
|
||||
|
||||
The `FastVideo-internal/ui/ltx2-streaming/server/` stack is being
|
||||
upstreamed into public FastVideo at `fastvideo/entrypoints/streaming/`.
|
||||
In parallel, FastVideo is becoming a first-class Dynamo backend (same
|
||||
tier as vllm, sglang, trtllm). This file covers both threads since they
|
||||
share `generate_async` as the substrate.
|
||||
|
||||
For PR sequence/status see [pr-roadmap.md](pr-roadmap.md). For the
|
||||
Dreamverse-side adoption see [cross-repo-surfaces.md](cross-repo-surfaces.md).
|
||||
|
||||
**Last updated:** 2026-05-03.
|
||||
|
||||
## What's being upstreamed
|
||||
|
||||
| Internal path | Size | Role | Public target |
|
||||
|---|---|---|---|
|
||||
| `server/main.py` | 94 KB | FastAPI + WebSocket, session lifecycle, segment orchestration | `fastvideo/entrypoints/streaming/server.py` + handlers |
|
||||
| `server/gpu_pool.py` | 66 KB | GPU orchestration, subprocess workers | `fastvideo/entrypoints/streaming/gpu_pool.py` |
|
||||
| `server/prompt_enhancer.py` | 69 KB | LLM orchestration (cerebras_ifm, cerebras, groq) | `fastvideo/entrypoints/streaming/prompt/` package |
|
||||
| `server/mock_server.py` | 45 KB | Mock backend for dev/tests | `fastvideo/entrypoints/streaming/mock_server.py` |
|
||||
| `server/prompt_safety.py` | 7 KB | Optional fasttext-gated prompt safety | `prompt/safety.py` |
|
||||
| `server/session_init_image.py` | 3 KB | i2v init image handling | `streaming/session_init_image.py` (PR 7.5, already public) |
|
||||
| `server/rewrite_prompt_payload.py` | 3 KB | Rewrite flow payload builder | `prompt/rewrite.py` |
|
||||
| `server/session_logger.py` | 1 KB | Session JSONL logs | `streaming/session_logger.py` |
|
||||
| `server/config.py` | 9 KB | Env-driven server config | typed `ServeConfig.streaming` extensions |
|
||||
| `router/main.py` | 27 KB | Multi-replica load balancer + WS proxy | `fastvideo/entrypoints/streaming/router/` |
|
||||
| `slurm/` | — | Deployment scripts | Stays internal |
|
||||
|
||||
Frontend clients (`client/`, `prod-ui/`) stay in the internal repo.
|
||||
|
||||
## Four design decisions that shape the upstream
|
||||
|
||||
### D-1: Continuation model — Hybrid (server-held + opaque client-round-trip)
|
||||
|
||||
Streaming WebSocket sessions hold continuation per-GPU (matches today's
|
||||
internal behavior, fast, zero client bandwidth). Stateless HTTP endpoints
|
||||
use opaque round-trip payloads. Server exposes a `snapshot_state` message
|
||||
that returns the opaque form for migration/retry.
|
||||
|
||||
One serialization layer underlies both surfaces.
|
||||
|
||||
Implementation: `SessionStore` (in-memory default, pluggable for
|
||||
redis/etc.) keyed by session ID, holds typed `LTX2ContinuationState`.
|
||||
- `snapshot(session_id) -> ContinuationState` exports for migration
|
||||
- `hydrate(state: ContinuationState) -> session_id` loads state into new session
|
||||
|
||||
Payload schema covers: trailing conditioning frames (or tensor-blob ID),
|
||||
audio latents (or blob ID), segment index, audio sample rate,
|
||||
`video_position_offset_sec`, model-specific metadata.
|
||||
|
||||
Landed in PR 7. See [cross-repo-surfaces.md](cross-repo-surfaces.md) for
|
||||
the full wire format.
|
||||
|
||||
### D-2: Streaming server layout — Parallel subpackage `fastvideo/entrypoints/streaming/`
|
||||
|
||||
Sits next to `fastvideo/entrypoints/openai/`. No existing code moves.
|
||||
Both servers share the `entrypoints/*` namespace. Shared utilities can be
|
||||
factored into `fastvideo/entrypoints/server_common/` later if needed.
|
||||
|
||||
### D-3: LLM provider abstraction — `LLMProvider` protocol + built-in providers
|
||||
|
||||
`prompt_enhancer.py` (69 KB) hard-coded three providers (cerebras_ifm,
|
||||
cerebras, groq) with provider-specific request/response handling
|
||||
scattered throughout. Upstreaming as-is would lock FastVideo to those
|
||||
providers.
|
||||
|
||||
Protocol shape:
|
||||
|
||||
```python
|
||||
@dataclass
|
||||
class LLMRequest:
|
||||
messages: list[LLMMessage]
|
||||
model: str
|
||||
max_tokens: int | None = None
|
||||
temperature: float | None = None
|
||||
timeout_ms: int | None = None
|
||||
|
||||
@dataclass
|
||||
class LLMResponse:
|
||||
content: str
|
||||
provider: str
|
||||
model: str
|
||||
latency_ms: float
|
||||
fallback_used: bool = False
|
||||
|
||||
class LLMProvider(Protocol):
|
||||
name: str
|
||||
async def complete(self, request: LLMRequest) -> LLMResponse: ...
|
||||
```
|
||||
|
||||
PR 7.7 ships built-in providers for cerebras, groq. **Public Literal
|
||||
currently restricts to `Literal["cerebras", "groq"]`** — `cerebras_ifm`
|
||||
is internal-only and remains environment-driven on `dreamverse-server`.
|
||||
See [open-threads.md](open-threads.md) follow-up #3.
|
||||
|
||||
Hot-reloadable system prompts via management endpoint. Sequential
|
||||
fallback across providers in priority order — race-based fallback (the
|
||||
internal optimization) deferred per [decisions-log.md](decisions-log.md)
|
||||
D-3.
|
||||
|
||||
### D-4: Dynamo as first-class backend target — async event stream
|
||||
|
||||
PR ai-dynamo/dynamo#7544 (closed draft) showed two frictions:
|
||||
|
||||
1. Flat legacy kwargs — Dynamo handler had to know LTX-2-specific names.
|
||||
2. Sync-only generation — Dynamo wrapped `generator.generate(...)` in
|
||||
`asyncio.to_thread` under a lock; no progress streaming, no
|
||||
disaggregation path.
|
||||
|
||||
Decision: **add `generate_async`** as the canonical execution API.
|
||||
|
||||
```python
|
||||
async def generate_async(
|
||||
self,
|
||||
request: GenerationRequest,
|
||||
) -> AsyncGenerator[VideoEvent, None]: ...
|
||||
```
|
||||
|
||||
Events:
|
||||
|
||||
```python
|
||||
@dataclass
|
||||
class VideoProgressEvent:
|
||||
step: int
|
||||
total_steps: int
|
||||
stage: str # "denoise" | "refine" | "decode" | ...
|
||||
|
||||
@dataclass
|
||||
class VideoPartialEvent:
|
||||
frames: np.ndarray # (num_frames, H, W, 3)
|
||||
index: int # monotonic chunk index
|
||||
|
||||
@dataclass
|
||||
class VideoFinalEvent:
|
||||
video_bytes: bytes | None
|
||||
tensor: torch.Tensor | None
|
||||
metadata: dict[str, Any]
|
||||
continuation_state: ContinuationState | None
|
||||
```
|
||||
|
||||
The sync `generate_video(request=...) -> VideoResult` becomes a thin
|
||||
`asyncio.run` wrapper over `generate_async` that collects events and
|
||||
returns the final.
|
||||
|
||||
**Three consumers, one substrate:**
|
||||
|
||||
| Consumer | Transport | Request shape | State |
|
||||
|---|---|---|---|
|
||||
| Stateless OpenAI (`fastvideo/entrypoints/openai/`) | HTTP POST | `GenerationRequest` merged onto `ServeConfig.default_request` | Stateless; opaque payload |
|
||||
| Streaming WebSocket (`fastvideo/entrypoints/streaming/`) | WebSocket JSON + binary fMP4 | `GenerationRequest` per segment, session-scoped | Server-held; per-GPU continuation cache |
|
||||
| Dynamo native backend (`ai-dynamo/dynamo/components/src/dynamo/fastvideo/`) | Dynamo RPC endpoint | `NvCreateVideoRequest` ↔ adapter ↔ `GenerationRequest` | Aggregated today; future disaggregated via `ContinuationState` |
|
||||
|
||||
**FastVideo does NOT host any Dynamo code.** The full backend package
|
||||
(`args.py`, `main.py`, `backend.py`, `register.py`, `health_check.py`)
|
||||
lives entirely in the Dynamo repo at `components/src/dynamo/fastvideo/`,
|
||||
matching the vllm/sglang pattern. FastVideo's only obligation is the
|
||||
stable public Python API.
|
||||
|
||||
PR 7.10 lands the FastVideo-side contract. Dynamo backend code lives in
|
||||
ai-dynamo/dynamo (next iteration of #7544 reopens against PR 8 reference
|
||||
docs).
|
||||
|
||||
## Target package layout
|
||||
|
||||
```
|
||||
fastvideo/entrypoints/
|
||||
├── openai/ # existing: stateless HTTP POST
|
||||
├── streaming/ # NEW: session WebSocket
|
||||
│ ├── server.py # FastAPI + WebSocket entry
|
||||
│ ├── session.py # session lifecycle, state machine
|
||||
│ ├── session_store.py # typed session state + snapshot/hydrate
|
||||
│ ├── protocol.py # JSON WebSocket message schemas
|
||||
│ ├── stream.py # fMP4 encoding (av_fmp4 mode)
|
||||
│ ├── gpu_pool.py # subprocess workers (PR 7.6)
|
||||
│ ├── worker.py # per-GPU worker loop
|
||||
│ ├── continuation.py # typed LTX2 state payload
|
||||
│ ├── session_init_image.py
|
||||
│ ├── session_logger.py
|
||||
│ ├── mock_server.py
|
||||
│ ├── prompt/
|
||||
│ │ ├── enhancer.py # provider-agnostic prompt ops
|
||||
│ │ ├── rewrite.py
|
||||
│ │ ├── safety.py # optional fasttext
|
||||
│ │ └── providers/
|
||||
│ │ ├── base.py # LLMProvider protocol
|
||||
│ │ ├── cerebras.py
|
||||
│ │ ├── cerebras_ifm.py
|
||||
│ │ └── groq.py
|
||||
│ └── router/
|
||||
│ ├── main.py
|
||||
│ └── registry.py
|
||||
├── cli/
|
||||
└── video_generator.py
|
||||
```
|
||||
|
||||
## Typed config integration
|
||||
|
||||
`ServeConfig` gets an optional `streaming: StreamingConfig | None`:
|
||||
|
||||
```python
|
||||
@dataclass
|
||||
class StreamingConfig:
|
||||
session_timeout_seconds: int = 300
|
||||
generation_segment_cap: int = 6
|
||||
stream_mode: Literal["av_fmp4", "legacy_jpeg"] = "av_fmp4"
|
||||
warmup: WarmupConfig = field(default_factory=WarmupConfig)
|
||||
pool: GpuPoolConfig = field(default_factory=GpuPoolConfig)
|
||||
prompt: PromptEnhancerConfig | None = None
|
||||
safety: PromptSafetyConfig | None = None
|
||||
|
||||
@dataclass
|
||||
class GpuPoolConfig:
|
||||
num_workers: int | None = None # default: CUDA_VISIBLE_DEVICES count
|
||||
enable_audio_reencode: bool = True
|
||||
conditioning_num_frames: int = 9
|
||||
conditioning_end_offset: int = 0
|
||||
|
||||
@dataclass
|
||||
class PromptEnhancerConfig:
|
||||
provider: Literal["cerebras", "groq"] = "cerebras" # cerebras_ifm pending
|
||||
model: str = "gpt-oss-120b"
|
||||
timeout_ms: int = 20000
|
||||
system_prompt_dir: str | None = None # hot-reloadable
|
||||
|
||||
@dataclass
|
||||
class PromptSafetyConfig:
|
||||
enabled: bool = False
|
||||
classifier_path: str | None = None
|
||||
```
|
||||
|
||||
## `build_app` route contract — open follow-up
|
||||
|
||||
Today `fastvideo.entrypoints.streaming.server.build_app` exposes only:
|
||||
|
||||
- `GET /health`
|
||||
- `WS /v1/stream`
|
||||
|
||||
The Dreamverse Next.js shell expects these additional routes that the
|
||||
upstream plan (and Dreamverse FE today) require:
|
||||
|
||||
| Route | Owner per upstream plan | Status |
|
||||
|---|---|---|
|
||||
| `GET /healthz` | Streaming-server-side health (FastVideo) | 🔴 NOT YET MIGRATED |
|
||||
| `GET /readyz` | Streaming-server-side health (FastVideo) | 🔴 NOT YET MIGRATED |
|
||||
| `GET /status` | Streaming-server-side health (FastVideo) | 🔴 NOT YET MIGRATED |
|
||||
| `GET /curated-presets` | Operator-side surface (Dreamverse) | 🟡 stays in Dreamverse, FE feature-detects |
|
||||
| `POST /curated-presets/append` | Operator-side surface (Dreamverse) | 🟡 stays in Dreamverse |
|
||||
| `GET /prompt-system-config` | Operator-side surface (Dreamverse) | 🟡 stays in Dreamverse |
|
||||
| Devtools routes | Dreamverse-only | 🟡 stays in Dreamverse |
|
||||
|
||||
Until the three health routes migrate into FastVideo's `build_app`, the
|
||||
`BE_FLAVOR=fastvideo` flavor of `launch_demo.sh` is a "diagnostic" flavor
|
||||
only (verifies typed serve-config path) — not FE-compatible. See
|
||||
[open-threads.md](open-threads.md) follow-up #1.
|
||||
|
||||
The streaming-upstream plan listed `/healthz`, `/readyz`, `/status`,
|
||||
`/ws` as the contract that the upstream of `realtime/` → `streaming/`
|
||||
must preserve. They were deferred from PR 7.5's MVP.
|
||||
|
||||
## PR 7.5 status — open as #1251
|
||||
|
||||
Single-generator WebSocket end-to-end shipped (8 commits):
|
||||
|
||||
1. `feat(streaming): protocol schemas + session state machine`
|
||||
2. `feat(streaming): fMP4 encoder + session init-image persistence`
|
||||
3. `feat(streaming): single-generator WebSocket server entry`
|
||||
4. `test(streaming): server lifecycle + protocol + fMP4 coverage`
|
||||
5. `docs(streaming): server contract spec`
|
||||
6. `fix(streaming): restore missing-streaming-block guard + retire stub-era test`
|
||||
7. `simplify(streaming): review follow-ups (idle timeout via asyncio.wait_for, _send_error helper, _cleanup_session, Protocol-typed generator, cleanup-on-disconnect)`
|
||||
8. `fix(streaming): enforce idle timeout on receive_json + flag generator-cancellation gap (TODO → PR 7.10)`
|
||||
|
||||
Deferred TODOs (intentionally) blocking on PR 7.10:
|
||||
|
||||
- **Per-step progress events** — only terminal `step_complete` today;
|
||||
needs `generate_async` for per-step `VideoProgressEvent` emission.
|
||||
- **Mid-segment cancellation on client disconnect** — TODO marker in
|
||||
`server.py` near `pool.run`. Needs `generate_async`'s cancellation
|
||||
propagation.
|
||||
|
||||
## PR 7.6 status — branch ready, not yet PR'd
|
||||
|
||||
`will/api_7.6` (5 commits, rebased on 7.5):
|
||||
|
||||
1. `feat [7.6/n]: GPU pool manager with typed worker boundary`
|
||||
2. `refactor [7.6/n]: route streaming server through GpuPool`
|
||||
3. `test [7.6/n]: GPU pool coverage (in-process + subprocess)`
|
||||
4. `fix(streaming): restore missing asyncio import in server` (rebase fixup)
|
||||
5. `feat(streaming): extract worker.py and add two-segment warmup`
|
||||
|
||||
Tests: 17/17 gpu_pool tests + 89/89 streaming tests green.
|
||||
|
||||
Ships:
|
||||
- `GpuPool` ABC + `InProcessGpuPool` + `SubprocessGpuPool` +
|
||||
`PoolAssignment` / `PoolHealth` / `PoolAcquireTimeout`
|
||||
- `worker.py` — per-GPU `worker_main` and two-segment warmup helper
|
||||
- Subprocess startup uses typed `GeneratorConfig`, NOT flat kwargs
|
||||
- Session-to-GPU binding with timeout + queue for contention
|
||||
- Two-segment startup warmup per worker (segment 1 fresh + segment 2
|
||||
with returned `ContinuationState` so both compile branches are primed)
|
||||
- `SessionStore` (from PR 7) wired for per-GPU continuation cache
|
||||
|
||||
Deferred to PR 7.10:
|
||||
|
||||
- **Audio re-encode (`LTX2AudioEncoder`, `AudioProcessor`)**: internal
|
||||
`_re_encode_audio` runs *inside* the per-step streaming loop
|
||||
(`_stream_av_fmp4_events` / `do_step_ltx2`). The whole-segment
|
||||
`pool.run()` path PR 7.6 ships doesn't need it. Re-encode is a
|
||||
per-step streaming concern that belongs with `generate_async`.
|
||||
- **Deprecate `VideoGenerator.from_pretrained(**flat_kwargs)`**: belongs
|
||||
with PR 13 cleanup.
|
||||
|
||||
## PR 7.10 — the unlock PR
|
||||
|
||||
PR 7.10 adds `generate_async` and closes three open threads
|
||||
simultaneously:
|
||||
|
||||
- Q-5 / D-5: audio re-encode for cross-segment continuity
|
||||
- Q-9: Dynamo progress passthrough
|
||||
- PR 7.5's mid-segment cancellation TODO (client disconnect →
|
||||
`asyncio.CancelledError` → GPU work stops)
|
||||
|
||||
Plus health-check helper:
|
||||
|
||||
```python
|
||||
def default_health_check_request(self) -> GenerationRequest: ...
|
||||
# Returns 256x256, 8 frames, 1 step. Lets Dynamo's
|
||||
# FastVideoHealthCheckPayload.to_dict() produce a Dynamo
|
||||
# health_check_payload kwarg without knowledge of FastVideo internals.
|
||||
```
|
||||
|
||||
Stable public exports:
|
||||
|
||||
```python
|
||||
from fastvideo import VideoGenerator
|
||||
from fastvideo.api import (
|
||||
GenerationRequest, SamplingConfig, ContinuationState,
|
||||
VideoResult, VideoEvent,
|
||||
VideoProgressEvent, VideoPartialEvent, VideoFinalEvent,
|
||||
)
|
||||
```
|
||||
|
||||
Streaming server (PR 7.5) gets rewired to consume `generate_async`
|
||||
directly — no wrapper duplication.
|
||||
|
||||
## Dynamo request/response mapping
|
||||
|
||||
```
|
||||
NvCreateVideoRequest -> fastvideo.api.GenerationRequest
|
||||
prompt -> sampling.prompt
|
||||
size="WxH" -> sampling.width, sampling.height
|
||||
seconds -> seconds * nvext.fps -> sampling.num_frames
|
||||
input_reference -> input.image_path | input.video_path
|
||||
nvext.fps -> sampling.fps
|
||||
nvext.num_frames -> sampling.num_frames (overrides seconds*fps)
|
||||
nvext.num_inference_steps -> sampling.num_inference_steps
|
||||
nvext.guidance_scale -> sampling.guidance_scale
|
||||
nvext.seed -> sampling.seed
|
||||
nvext.negative_prompt -> sampling.negative_prompt
|
||||
response_format -> (handled at adapter's output stage)
|
||||
|
||||
VideoFinalEvent -> NvVideosResponse
|
||||
video_bytes -> data[0].b64_json (response_format=b64_json)
|
||||
uploaded URL -> data[0].url (response_format=url)
|
||||
metadata.inference_time_s -> inference_time_s
|
||||
continuation_state -> (reserved for future disaggregation)
|
||||
```
|
||||
|
||||
## Open questions
|
||||
|
||||
1. **Router placement** — in-tree at `fastvideo/entrypoints/streaming/router/`
|
||||
(current implementation per PR 7.9) or separate package
|
||||
`fastvideo-router/` / `fastvideo/contrib/router/`. Effectively
|
||||
resolved in-tree by the PR 7.9 implementation.
|
||||
2. **Session ID authority** — server-generated UUID; accept externally
|
||||
provided session ID only for resume flows.
|
||||
3. **Disaggregation readiness contract test** — should PR 7.10 validate
|
||||
`ContinuationState` survives round-trip through Dynamo-style RPC
|
||||
(pickle or JSON), even though Dynamo isn't using it today?
|
||||
Recommended: yes; cheap regression guard.
|
||||
4. **Dynamo progress/status passthrough** — should PR 7.10's handler
|
||||
contract emit intermediate `NvVideosResponse` chunks keyed off
|
||||
`VideoProgressEvent`, or stay aggregated-final-only? Recommended:
|
||||
stay aggregated-final for PR 7.10; revisit after Dynamo clarifies.
|
||||
5. **`video_position_offset_sec` semantics** — see [decisions-log.md](decisions-log.md)
|
||||
open question; needs decision before PR 7.6 emits state.
|
||||
6. **`SessionStore` / `BlobStore` lifecycle** — eviction, TTL, blob-drop
|
||||
on state replacement; defer to PR 7.5 design pass.
|
||||
@@ -0,0 +1,327 @@
|
||||
# Evaluation Metrics Registry
|
||||
|
||||
Living catalog of all evaluation metrics for FastVideo-WorldModel video quality
|
||||
assessment. Each metric includes a detailed explanation, implementation status,
|
||||
usage instructions, and interpretation guide.
|
||||
|
||||
_Last updated: 2026-03-02_
|
||||
|
||||
---
|
||||
|
||||
## Metric Summary
|
||||
|
||||
| Metric | Category | Status | Location | Trust |
|
||||
|--------|----------|--------|----------|-------|
|
||||
| **FVD** | Distribution | ✅ Implemented | `benchmarks/fvd/` | High |
|
||||
| **SSIM** | Reference | ✅ Implemented | `fastvideo/tests/ssim/` | High |
|
||||
| **LPIPS** | Perceptual | ✅ Implemented | `scripts/lora_extraction/` | Medium |
|
||||
| **Loss trajectory** | Training signal | ✅ Implemented | W&B `train_loss` | Medium |
|
||||
| **Grad norm stability** | Training signal | ✅ Implemented | W&B `grad_norm` | Medium |
|
||||
| **GameWorld Score** | Multi-dim benchmark | 🟡 External | Matrix-Game repo | Low |
|
||||
| **Human preference** | Gold standard | 🔴 Manual | N/A | Highest |
|
||||
|
||||
---
|
||||
|
||||
## Implemented Metrics
|
||||
|
||||
### FVD — Fréchet Video Distance
|
||||
|
||||
**Category**: Distribution-level quality metric
|
||||
**Status**: ✅ Fully implemented in `benchmarks/fvd/`
|
||||
**Trust**: High — standard protocol, I3D feature extractor
|
||||
|
||||
#### What It Measures
|
||||
FVD measures the distance between the **distribution** of generated videos and
|
||||
a distribution of real/reference videos. It works by:
|
||||
1. Extracting spatiotemporal features from both real and generated video sets
|
||||
using a pretrained **I3D** (Inflated 3D ConvNet) model.
|
||||
2. Modeling each set of features as a multivariate Gaussian (mean + covariance).
|
||||
3. Computing the **Fréchet distance** between the two Gaussians.
|
||||
|
||||
Lower FVD = generated videos are more statistically similar to real videos.
|
||||
|
||||
#### Why It Matters
|
||||
- FVD is the **de facto standard** for benchmarking video generation models.
|
||||
- It captures both **visual quality** (are individual frames realistic?) and
|
||||
**temporal coherence** (do frames flow naturally?).
|
||||
- Matrix-Game 2.0, Open-Sora, and most video generation papers report FVD.
|
||||
|
||||
#### Limitations
|
||||
- Requires a **large sample set** (standard protocol uses 2048 videos) to
|
||||
produce stable statistics. Small sample sizes yield noisy results.
|
||||
- Measures **distributional similarity**, not per-video quality. A model could
|
||||
have low FVD by generating a diverse set of "roughly okay" videos.
|
||||
- The I3D model was trained on Kinetics-400 (human actions). It may be less
|
||||
sensitive to domain-specific artifacts in non-human-action videos (e.g.,
|
||||
driving, game environments).
|
||||
- Does not directly measure text-video alignment or action controllability.
|
||||
|
||||
#### How to Use
|
||||
|
||||
```python
|
||||
# Programmatic
|
||||
from benchmarks.fvd import compute_fvd_with_config, FVDConfig
|
||||
|
||||
config = FVDConfig.fvd2048_16f() # Standard: 2048 videos, 16 frames
|
||||
results = compute_fvd_with_config('data/real/', 'outputs/gen/', config)
|
||||
print(f"FVD: {results['fvd']:.2f}")
|
||||
```
|
||||
|
||||
```bash
|
||||
# CLI
|
||||
python -m benchmarks.fvd.cli \
|
||||
--real-path data/real/ \
|
||||
--gen-path outputs/gen/ \
|
||||
--protocol fvd2048_16f
|
||||
```
|
||||
|
||||
**Preset protocols**:
|
||||
| Protocol | Videos | Frames | Use Case |
|
||||
|----------|--------|--------|----------|
|
||||
| `fvd2048_16f` | 2048 | 16 | Standard benchmark (papers) |
|
||||
| `fvd2048_128f` | 2048 | 128 | Long video evaluation |
|
||||
| `quick_test` | 100 | 16 | Fast dev iteration |
|
||||
|
||||
**Feature extractors**: `i3d` (default, standard), `clip`, `videomae`
|
||||
|
||||
#### Interpretation
|
||||
| FVD Range | Interpretation |
|
||||
|-----------|---------------|
|
||||
| < 100 | Excellent — near-real quality |
|
||||
| 100–300 | Good — competitive with SOTA |
|
||||
| 300–600 | Fair — noticeable gap from real |
|
||||
| > 600 | Poor — significant quality issues |
|
||||
|
||||
> FVD values are dataset-dependent. Always compare against baselines evaluated
|
||||
> on the same real video distribution.
|
||||
|
||||
---
|
||||
|
||||
### SSIM — Structural Similarity Index
|
||||
|
||||
**Category**: Per-frame reference comparison
|
||||
**Status**: ✅ Implemented in `fastvideo/tests/ssim/`
|
||||
**Trust**: High — used in CI regression tests
|
||||
|
||||
#### What It Measures
|
||||
SSIM compares two images (or video frames) based on three components:
|
||||
1. **Luminance**: brightness similarity
|
||||
2. **Contrast**: dynamic range similarity
|
||||
3. **Structure**: spatial pattern similarity
|
||||
|
||||
The final score is a value in [0, 1] where 1.0 = identical.
|
||||
|
||||
#### Why It Matters
|
||||
- Used as a **regression guard** in CI: ensures model updates don't degrade
|
||||
visual output below a threshold.
|
||||
- More perceptually meaningful than raw pixel MSE.
|
||||
- Fast to compute — suitable for automated testing.
|
||||
|
||||
#### Limitations
|
||||
- Requires a **pixel-aligned reference** video. Cannot compare videos with
|
||||
different seeds, prompts, or angles.
|
||||
- Operates **per-frame** — does not capture temporal coherence.
|
||||
- Insensitive to some perceptual artifacts (color shifts, high-frequency noise).
|
||||
|
||||
#### How to Use
|
||||
|
||||
```bash
|
||||
pytest fastvideo/tests/ssim/ -vs
|
||||
```
|
||||
|
||||
#### Interpretation
|
||||
| SSIM Range | Quality |
|
||||
|------------|---------|
|
||||
| > 0.90 | Excellent — very close to reference |
|
||||
| 0.80–0.90 | Good — acceptable for most uses |
|
||||
| 0.70–0.80 | Fair — noticeable differences |
|
||||
| < 0.70 | Poor — significant divergence |
|
||||
|
||||
---
|
||||
|
||||
### LPIPS — Learned Perceptual Image Patch Similarity
|
||||
|
||||
**Category**: Per-frame perceptual distance
|
||||
**Status**: ✅ Implemented in `scripts/lora_extraction/lora_inference_comparison.py`
|
||||
**Trust**: Medium — available but only used for LoRA comparison currently
|
||||
|
||||
#### What It Measures
|
||||
LPIPS uses a pretrained neural network (AlexNet by default) to extract
|
||||
deep features from two images and computes the distance between them in
|
||||
feature space. Unlike SSIM, LPIPS correlates much more strongly with
|
||||
**human perceptual judgments**.
|
||||
|
||||
Lower LPIPS = more perceptually similar.
|
||||
|
||||
#### Why It Matters
|
||||
- Best available automated proxy for **human visual judgments** at the frame
|
||||
level.
|
||||
- Captures semantic and structural differences that SSIM misses (e.g., texture
|
||||
changes, minor recoloring).
|
||||
- Used for validating LoRA merge quality.
|
||||
|
||||
#### Limitations
|
||||
- Per-frame metric — no temporal awareness.
|
||||
- Requires reference video (paired comparison only).
|
||||
- Slightly slower than SSIM due to neural network forward pass.
|
||||
|
||||
#### How to Use
|
||||
|
||||
```bash
|
||||
python scripts/lora_extraction/lora_inference_comparison.py \
|
||||
--base merged_model \
|
||||
--ft path/to/finetuned \
|
||||
--adapter NONE \
|
||||
--output-dir results \
|
||||
--prompt "A cat" \
|
||||
--compute-lpips
|
||||
```
|
||||
|
||||
#### Interpretation
|
||||
| LPIPS Range | Quality |
|
||||
|-------------|---------|
|
||||
| < 0.10 | Excellent — nearly indistinguishable |
|
||||
| 0.10–0.20 | Good — minor perceptual differences |
|
||||
| 0.20–0.40 | Fair — noticeable differences |
|
||||
| > 0.40 | Poor — clearly different |
|
||||
|
||||
---
|
||||
|
||||
### Loss Trajectory
|
||||
|
||||
**Category**: Training signal proxy
|
||||
**Status**: ✅ Active (from W&B `train_loss`)
|
||||
**Trust**: Medium — proxy, not direct quality measure
|
||||
|
||||
#### What It Measures
|
||||
Tracks the training loss over time. A healthy training run shows:
|
||||
- **Decreasing loss** over the first hundreds of steps.
|
||||
- **Stable gradient norms** (no wild spikes).
|
||||
- **Consistent step times** (no infrastructure issues).
|
||||
|
||||
#### Why It Matters
|
||||
- Cheapest evaluation signal — available in real-time from W&B.
|
||||
- Critical for the **30-minute quality check** workflow.
|
||||
- At later training stages (when loss becomes meaningful), trajectory shape
|
||||
can predict final model quality.
|
||||
|
||||
#### Context: How This Evolves
|
||||
The team's experience shows evaluation signals change during a project:
|
||||
- **Early stage**: Loss may be flat or meaningless → focus on SSIM & visual
|
||||
inspection instead.
|
||||
- **Mid stage**: Loss starts decreasing → trajectory shape becomes useful.
|
||||
- **Late stage**: Loss is meaningful → can compare trajectories across runs.
|
||||
|
||||
This dynamic is a key insight from the team's workflow: don't over-rely on
|
||||
loss early; don't ignore it late.
|
||||
|
||||
---
|
||||
|
||||
### Grad Norm Stability
|
||||
|
||||
**Category**: Training health diagnostic
|
||||
**Status**: ✅ Active (from W&B `grad_norm`)
|
||||
**Trust**: Medium — diagnostic, not quality metric
|
||||
|
||||
#### What It Measures
|
||||
The magnitude of gradients during training. Stable grad norms indicate
|
||||
healthy optimization. Spikes or NaN values indicate training instability.
|
||||
|
||||
#### Alert Thresholds
|
||||
| Condition | Meaning |
|
||||
|-----------|---------|
|
||||
| Stable ~0.3–0.5 | Normal training |
|
||||
| Single spike > 3× average | Possible bad batch, monitor |
|
||||
| NaN or Inf | 🔴 Training has diverged — stop run |
|
||||
| Increasing trend | Learning rate may be too high |
|
||||
|
||||
---
|
||||
|
||||
## External Benchmarks
|
||||
|
||||
### GameWorld Score Benchmark (Matrix-Game)
|
||||
|
||||
**Category**: Multi-dimensional evaluation framework for interactive world models
|
||||
**Status**: 🟡 External — not implemented in-repo
|
||||
**Source**: [Matrix-Game 1.0 benchmark](https://github.com/SkyworkAI/Matrix-Game), used in [Matrix-Game 2.0 paper](https://arxiv.org/abs/2508.13009)
|
||||
|
||||
#### What It Measures
|
||||
A comprehensive benchmark examining **four critical capabilities**:
|
||||
|
||||
| Dimension | What It Evaluates | Example Signals |
|
||||
|-----------|-------------------|-----------------|
|
||||
| **Visual quality** | Frame-level realism, absence of artifacts | Color fidelity, sharpness, coherence |
|
||||
| **Temporal quality** | Smoothness across frames, motion consistency | Jitter, flickering, temporal aliasing |
|
||||
| **Action controllability** | Response to input actions (keyboard/mouse) | Action delay, correctness, smoothness |
|
||||
| **Physical rule understanding** | Adherence to physics (gravity, collision) | Object persistence, plausible motion |
|
||||
|
||||
#### Context from Matrix-Game 2.0
|
||||
- Evaluation uses **597-frame composite action sequences** over 32 Minecraft
|
||||
scenes and 16 wild scenes.
|
||||
- Action controllability assessment is **Minecraft-specific** — cannot be
|
||||
directly applied to wild/general scenes.
|
||||
- The paper notes that models that "collapse" to static frames can
|
||||
paradoxically score higher on consistency metrics — beware of this confound.
|
||||
|
||||
#### Relevance to FastVideo
|
||||
- Matrix-Game 2.0 is built on SkyReels-V2/Wan2.1 architecture — **same model
|
||||
family as FastVideo**.
|
||||
- Their distillation uses DMD-based Self-Forcing — **same technique** as our
|
||||
`self_forcing_distillation_pipeline.py`.
|
||||
- GameWorld Score dimensions are a useful framework for thinking about world
|
||||
model quality even outside gaming contexts.
|
||||
|
||||
---
|
||||
|
||||
## Human Preference Evaluation
|
||||
|
||||
**Category**: Gold-standard quality assessment
|
||||
**Status**: 🔴 Manual process — no automated implementation
|
||||
**Priority**: **Highest** — this is the most important evaluation signal
|
||||
**Trust**: Highest — but expensive
|
||||
|
||||
### What It Measures
|
||||
Human evaluators compare generated videos and rate them on dimensions like:
|
||||
- Overall quality and realism
|
||||
- Temporal coherence and smoothness
|
||||
- Prompt adherence / action correctness
|
||||
- Absence of artifacts
|
||||
|
||||
#### Why It's the Most Important Metric
|
||||
All automated metrics are **proxies** for human judgment. They can be gamed
|
||||
or may miss artifacts that humans easily notice. Human preference is the
|
||||
ultimate ground truth for video generation quality.
|
||||
|
||||
#### Cost & Practicality
|
||||
| Approach | Cost | Scale | When to Use |
|
||||
|----------|------|-------|-------------|
|
||||
| Internal team review | Low | ~10–50 videos | Every major checkpoint |
|
||||
| Crowdsource (MTurk, Scale) | Medium | 100+ videos | Pre-release validation |
|
||||
| A/B preference test | Medium | Pairs | Comparing two model versions |
|
||||
|
||||
#### Recommended Protocol
|
||||
1. Sample 10–20 videos from the model at a checkpoint.
|
||||
2. Include diverse prompts (easy + hard, short + long).
|
||||
3. Have 2–3 evaluators score each video 1–5 on: quality, coherence, fidelity.
|
||||
4. Record scores in the experiment journal.
|
||||
|
||||
---
|
||||
|
||||
## Metrics NOT Used
|
||||
|
||||
| Metric | Reason |
|
||||
|--------|--------|
|
||||
| ~~CLIP-Score~~ | Not used by the team. Measures text-image alignment using CLIP embeddings, but not well-suited for video temporal quality. |
|
||||
| Inception Score (IS) | Less informative than FVD for video; primarily an image metric. |
|
||||
| PSNR | Pixel-level metric; less perceptually meaningful than SSIM/LPIPS. |
|
||||
|
||||
---
|
||||
|
||||
## Adding a New Metric
|
||||
|
||||
Follow the SOP: `.agents/workflows/evaluation-development.md`
|
||||
|
||||
1. Prototype in `.agents/exploration/`
|
||||
2. Validate on known-good and known-bad samples
|
||||
3. Add to this registry
|
||||
4. Update the `evaluate-video-quality` skill
|
||||
@@ -0,0 +1,21 @@
|
||||
# Experiment Journal
|
||||
|
||||
Living log of all experiments. Each entry captures what was tried, the result,
|
||||
and any insights. Newest entries go at the top.
|
||||
|
||||
_No experiments logged yet. Use the `log-experiment` skill to add entries._
|
||||
|
||||
<!-- TEMPLATE — copy and fill for each new experiment:
|
||||
|
||||
## [YYYY-MM-DD] Experiment: <name>
|
||||
- **Hypothesis**: <what you expected to learn>
|
||||
- **Config**: model=..., lr=..., sp_size=..., gpus=..., script=...
|
||||
- **W&B run**: <run_id or URL>
|
||||
- **Duration**: <total wall time>
|
||||
- **Key metrics**: loss=..., step_time=..., grad_norm=...
|
||||
- **Checkpoint**: <path>
|
||||
- **Insight**: <what was learned>
|
||||
- **Status**: running | completed | failed | abandoned
|
||||
- **Related lessons**: `.agents/lessons/<filename>.md`
|
||||
|
||||
-->
|
||||
@@ -0,0 +1,5 @@
|
||||
{"name": "codebase-map", "description": "High-level structural index of the FastVideo-WorldModel repository", "path": "codebase-map/README.md", "status": "ready", "trust": "high"}
|
||||
{"name": "evaluation-registry", "description": "Catalog of all evaluation metrics with detailed explanations, implementation status, and usage guides", "path": "evaluation-registry/README.md", "status": "draft", "trust": "medium"}
|
||||
{"name": "experiment-journal", "description": "Living log of all experiments with hypotheses, configs, metrics, and insights", "path": "experiment-journal/README.md", "status": "draft", "trust": "medium"}
|
||||
{"name": "related-work", "description": "Index of related papers, repos, and blog posts with structured comparisons to FastVideo", "path": "related-work/README.md", "status": "draft", "trust": "low"}
|
||||
{"name": "dreamverse-integration", "description": "Consolidated knowledge base for the FastVideo public API refactor (PRs 0-17), LTX-2 streaming server upstream, Dreamverse migration from FastVideo-internal, and NVFP4 quantization landing", "path": "dreamverse-integration/README.md", "status": "ready", "trust": "high"}
|
||||
@@ -0,0 +1,34 @@
|
||||
# Related Work Index
|
||||
|
||||
Each file in this directory is a structured summary of a related paper, repo,
|
||||
or blog post relevant to FastVideo-WorldModel training.
|
||||
|
||||
## File Format
|
||||
|
||||
Each file is named `<slug>.md` and follows this structure:
|
||||
|
||||
```markdown
|
||||
---
|
||||
title: <paper/repo title>
|
||||
source: <URL or citation>
|
||||
type: paper | repo | blog
|
||||
date_indexed: <ISO-8601>
|
||||
tags: [world-model, distillation, evaluation, reward-shaping, ...]
|
||||
---
|
||||
|
||||
## Summary
|
||||
<1-2 paragraph summary of the work.>
|
||||
|
||||
## Key Differences from FastVideo
|
||||
- <Bullet points comparing their approach to ours.>
|
||||
|
||||
## Actionable Insights
|
||||
- <What we could adopt or adapt.>
|
||||
```
|
||||
|
||||
## How to Add New Entries
|
||||
|
||||
Use the `index-related-work` skill, or manually create a file following the
|
||||
template above.
|
||||
|
||||
_No related work indexed yet._
|
||||
@@ -0,0 +1,76 @@
|
||||
# Agent Onboarding — FastVideo-WorldModel
|
||||
|
||||
Welcome, agent. This is the **master onboarding** guide. Follow the steps below,
|
||||
then check if a **domain-specific onboarding** exists for your task.
|
||||
|
||||
## Domain-Specific Onboarding
|
||||
|
||||
If your task falls into one of these areas, read the specialized guide **after**
|
||||
completing the general steps below:
|
||||
|
||||
| Domain | Guide | When to Use |
|
||||
|--------|-------|-------------|
|
||||
| **WorldModel Training** | `worldmodel-training/README.md` | Training, finetuning, distillation, experiment management |
|
||||
|
||||
---
|
||||
|
||||
## Step 1: Understand the Codebase
|
||||
|
||||
Read these files to build your context:
|
||||
|
||||
| Priority | File | What you learn |
|
||||
|----------|------|----------------|
|
||||
| 1 | `AGENTS.md` | Coding guidelines, build/test commands, PR conventions |
|
||||
| 2 | `docs/design/overview.md` | Architecture: models, pipelines, configs, registry |
|
||||
| 3 | `fastvideo/train/` | Refactored training framework (YAML-driven, modular methods/models/callbacks) |
|
||||
| 4 | `docs/training/overview.md` | Training data flow and preprocessing |
|
||||
| 5 | `docs/training/finetune.md` | Training arguments, parallelism, LoRA, validation |
|
||||
| 6 | `docs/contributing/coding_agents.md` | How to add model pipelines with agent assistance |
|
||||
|
||||
## Step 2: Discover Available Resources
|
||||
|
||||
Read these two index files to see what skills and memory modules exist:
|
||||
|
||||
- **`.agents/skills/index.jsonl`** — catalog of all agent skills (name + description)
|
||||
- **`.agents/memory/index.jsonl`** — catalog of all memory modules (name + description)
|
||||
|
||||
Each entry has a `path` field pointing to the full content. Only load the
|
||||
full README.md for modules relevant to your current task.
|
||||
|
||||
## Step 3: Check for Existing Skills & SOPs
|
||||
|
||||
Before writing new code or procedures:
|
||||
|
||||
1. **Skills**: Read `.agents/skills/index.jsonl` — find a matching skill by description.
|
||||
2. **Workflows/SOPs**: Browse `.agents/workflows/` — step-by-step procedures for common tasks.
|
||||
3. **Lessons**: Browse `.agents/lessons/` — known pitfalls and their fixes.
|
||||
|
||||
If a skill or SOP exists for your task, **use it**. If not, you are in **exploration mode** — see Step 4.
|
||||
|
||||
## Step 4: Exploration Mode
|
||||
|
||||
If no existing skill/SOP covers your task:
|
||||
|
||||
1. Document your progress in `.agents/exploration/<topic>.md` using the template in `.agents/exploration/README.md`.
|
||||
2. At the end of your session, reflect:
|
||||
- **What worked** → propose a new skill or SOP in the exploration log.
|
||||
- **What failed** → create a lesson in `.agents/lessons/`.
|
||||
3. Flag the exploration log for human review.
|
||||
|
||||
## Quick Reference
|
||||
|
||||
```
|
||||
.agents/
|
||||
├── ONBOARDING.md ← you are here
|
||||
├── STATUS.md ← dashboard: completeness & trust of all components
|
||||
├── skills/ ← reusable agent skills
|
||||
├── workflows/ ← SOPs and procedures
|
||||
├── memory/ ← persistent context (folder per topic + index.jsonl)
|
||||
│ ├── index.jsonl
|
||||
│ ├── codebase-map/
|
||||
│ ├── experiment-journal/
|
||||
│ ├── evaluation-registry/
|
||||
│ └── related-work/
|
||||
├── lessons/ ← mistakes and fixes
|
||||
└── exploration/ ← draft procedures
|
||||
```
|
||||
@@ -0,0 +1,305 @@
|
||||
# WorldModel Training — Agent Onboarding
|
||||
|
||||
Specialized onboarding for agents working on FastVideo-WorldModel training,
|
||||
distillation, and evaluation. Read the master onboarding (`.agents/onboarding/README.md`)
|
||||
first, then come here.
|
||||
|
||||
---
|
||||
|
||||
## Domain Context
|
||||
|
||||
FastVideo-WorldModel trains **interactive world models** — video generation systems
|
||||
that respond to user actions (keyboard/mouse) in real-time. The architecture is
|
||||
based on **Wan2.1** (SkyReels-V2) DiT models with causal attention for
|
||||
auto-regressive streaming generation.
|
||||
|
||||
**Key techniques you will work with:**
|
||||
- Full finetuning and LoRA on Wan / LTX-2 / MatrixGame models
|
||||
- DMD-based distillation (few-step generation)
|
||||
- Self-Forcing distillation (causal streaming)
|
||||
- Diffusion-Forcing SFT (DFSFT) for causal models
|
||||
- VSA (Variable Sparsity Acceleration) for efficient training
|
||||
|
||||
---
|
||||
|
||||
## Training Code: Two Generations
|
||||
|
||||
### New modular framework: `fastvideo/train/` (preferred)
|
||||
|
||||
The refactored training code uses a **YAML-only config-driven** architecture
|
||||
with composable methods, per-role models, and a callback system. All new
|
||||
training work should use this framework.
|
||||
|
||||
### Legacy pipelines: `fastvideo/training/` (deprecated)
|
||||
|
||||
The old monolithic pipeline classes (`WanTrainingPipeline`,
|
||||
`DistillationPipeline`, etc.) still exist but are being phased out. The new
|
||||
framework imports select utilities from `fastvideo/training/` for backward
|
||||
compatibility (EMA, gradient clipping, checkpoint wrappers).
|
||||
|
||||
---
|
||||
|
||||
## Essential Reading (Training-Specific)
|
||||
|
||||
Read these **in order** before touching any training code:
|
||||
|
||||
| # | File | What You Learn |
|
||||
|---|------|----------------|
|
||||
| 1 | `docs/training/overview.md` | Training data flow: raw video → text embeddings + video latents → training |
|
||||
| 2 | `docs/training/finetune.md` | Training arguments, parallelism (SP/TP), LoRA, validation settings |
|
||||
| 3 | `docs/training/data_preprocess.md` | How to preprocess datasets into the expected format |
|
||||
| 4 | `docs/design/overview.md` | Architecture: models, pipelines, configs, registry |
|
||||
|
||||
---
|
||||
|
||||
## New Training Framework (`fastvideo/train/`)
|
||||
|
||||
### Architecture Overview
|
||||
|
||||
```
|
||||
fastvideo/train/
|
||||
├── __init__.py → exports Trainer
|
||||
├── trainer.py → main training loop coordinator
|
||||
├── entrypoint/
|
||||
│ ├── train.py → YAML-only training entrypoint
|
||||
│ └── dcp_to_diffusers.py → checkpoint conversion utility
|
||||
├── methods/ → training algorithms (TrainingMethod ABC)
|
||||
│ ├── base.py → TrainingMethod base class
|
||||
│ ├── fine_tuning/
|
||||
│ │ ├── finetune.py → FineTuneMethod (supervised finetuning)
|
||||
│ │ └── dfsft.py → DiffusionForcingSFTMethod (causal)
|
||||
│ ├── distribution_matching/
|
||||
│ │ ├── dmd2.py → DMD2Method (distribution matching distill)
|
||||
│ │ └── self_forcing.py → SelfForcingMethod (causal streaming)
|
||||
│ ├── knowledge_distillation/ → (stub, not yet implemented)
|
||||
│ └── consistency_model/ → (stub, not yet implemented)
|
||||
├── models/ → per-role model instances
|
||||
│ ├── base.py → ModelBase & CausalModelBase (ABC)
|
||||
│ └── wan/
|
||||
│ ├── wan.py → WanModel (non-causal)
|
||||
│ └── wan_causal.py → WanCausalModel (causal streaming)
|
||||
├── callbacks/ → training hooks & monitoring
|
||||
│ ├── callback.py → Callback base class + CallbackDict
|
||||
│ ├── grad_clip.py → GradNormClipCallback
|
||||
│ ├── ema.py → EMACallback (shadow weights)
|
||||
│ └── validation.py → ValidationCallback (sampling + eval)
|
||||
└── utils/ → configuration, building, checkpointing
|
||||
├── builder.py → build_from_config() (config → runtime)
|
||||
├── checkpoint.py → CheckpointManager (DCP-based)
|
||||
├── config.py → load_run_config() (YAML → RunConfig)
|
||||
├── training_config.py → TypedConfig dataclasses
|
||||
├── optimizer.py → build_optimizer_and_scheduler()
|
||||
├── instantiate.py → resolve_target() + instantiate()
|
||||
├── tracking.py → build_tracker() (W&B, etc.)
|
||||
├── dataloader.py → dataloader utilities
|
||||
├── module_state.py → apply_trainable()
|
||||
└── moduleloader.py → load_module_from_path()
|
||||
```
|
||||
|
||||
### Key Concepts
|
||||
|
||||
**TrainingMethod** (`methods/base.py`): Abstract base class for all training
|
||||
algorithms. Owns role models (student, teacher, critic), manages checkpoint
|
||||
state, and defines the training step interface.
|
||||
|
||||
**ModelBase** (`models/base.py`): Per-role model wrapper. Each role (student,
|
||||
teacher, critic) gets its own `ModelBase` instance owning a `transformer` and
|
||||
`noise_scheduler`. `CausalModelBase` extends this for streaming models.
|
||||
|
||||
**Callback system** (`callbacks/`): Composable hooks for gradient clipping,
|
||||
EMA, validation, etc. Configured via YAML, dispatched by `CallbackDict`.
|
||||
|
||||
**Config system** (`utils/config.py`, `utils/training_config.py`): YAML files
|
||||
are parsed into typed `RunConfig` dataclass trees. Models and methods use
|
||||
`_target_` fields for instantiation (similar to Hydra).
|
||||
|
||||
### Training Flow
|
||||
|
||||
```
|
||||
run_training_from_config(config_path)
|
||||
→ load_run_config() # YAML → RunConfig
|
||||
→ init_distributed() # TP/SP setup
|
||||
→ build_from_config() # instantiate models, method, dataloader
|
||||
→ Trainer.run() # main loop:
|
||||
├─ callbacks.on_train_start()
|
||||
├─ checkpoint_manager.maybe_resume()
|
||||
├─ for step in range(max_steps):
|
||||
│ ├─ method.single_train_step(batch)
|
||||
│ ├─ method.backward()
|
||||
│ ├─ callbacks.on_before_optimizer_step()
|
||||
│ ├─ method.optimizers_schedulers_step()
|
||||
│ ├─ tracker.log(metrics, step)
|
||||
│ ├─ callbacks.on_training_step_end()
|
||||
│ └─ checkpoint_manager.maybe_save(step)
|
||||
├─ callbacks.on_train_end()
|
||||
└─ checkpoint_manager.save_final()
|
||||
```
|
||||
|
||||
### Training Methods
|
||||
|
||||
| Method | Class | Use Case |
|
||||
|--------|-------|----------|
|
||||
| **FineTune** | `FineTuneMethod` | Single-role supervised finetuning |
|
||||
| **DFSFT** | `DiffusionForcingSFTMethod` | Diffusion-forcing SFT with inhomogeneous timesteps |
|
||||
| **DMD2** | `DMD2Method` | Multi-role distribution matching distillation (student + teacher + critic) |
|
||||
| **Self-Forcing** | `SelfForcingMethod` | Extends DMD2 for causal student rollouts |
|
||||
|
||||
### Launching Training (New Framework)
|
||||
|
||||
Training is launched via `torchrun` with a single YAML config:
|
||||
|
||||
```bash
|
||||
torchrun --nproc_per_node <N_GPUS> \
|
||||
-m fastvideo.train.entrypoint.train \
|
||||
--config examples/train/<config>.yaml
|
||||
```
|
||||
|
||||
### Example YAML Configs
|
||||
|
||||
| Config | Method | Description |
|
||||
|--------|--------|-------------|
|
||||
| `examples/train/finetune_wan2.1_t2v_1.3B_vsa_phase3.4_0.9sparsity.yaml` | FineTune | Wan 1.3B finetuning with VSA sparsity |
|
||||
| `examples/train/distill_wan2.1_t2v_1.3B_dmd2.yaml` | DMD2 | Wan 1.3B distillation (student + teacher + critic) |
|
||||
| `examples/train/dfsft_wan_causal_t2v_1.3B.yaml` | DFSFT | Causal Wan 1.3B diffusion-forcing SFT |
|
||||
| `examples/train/self_forcing_wan_causal_t2v_1.3B.yaml` | Self-Forcing | Causal streaming distillation |
|
||||
|
||||
### Checkpointing (New Framework)
|
||||
|
||||
**CheckpointManager** (`utils/checkpoint.py`) saves via `torch.distributed.checkpoint`:
|
||||
|
||||
```
|
||||
output_dir/
|
||||
└─ checkpoint-{step}/
|
||||
├─ dcp/ # DCP state dict
|
||||
├─ config.json # resolved training config
|
||||
└─ .fastvideo_metadata.json
|
||||
```
|
||||
|
||||
Checkpoint state includes: role model weights, per-role optimizers/schedulers,
|
||||
CUDA RNG state, and callback state (e.g., EMA shadow weights).
|
||||
|
||||
### Config Structure
|
||||
|
||||
A YAML config defines the full training pipeline:
|
||||
|
||||
```yaml
|
||||
models:
|
||||
student:
|
||||
_target_: fastvideo.train.models.wan.WanModel
|
||||
model_path: ...
|
||||
trainable: true
|
||||
teacher: # optional, for distillation
|
||||
_target_: fastvideo.train.models.wan.WanModel
|
||||
model_path: ...
|
||||
trainable: false
|
||||
|
||||
method:
|
||||
_target_: fastvideo.train.methods.fine_tuning.FineTuneMethod
|
||||
# method-specific params...
|
||||
|
||||
training:
|
||||
distributed: { num_gpus: 8, tp_size: 1, sp_size: 8 }
|
||||
data: { data_path: ..., batch_size: 1 }
|
||||
optimizer: { lr: 1e-5, lr_scheduler: constant_with_warmup }
|
||||
loop: { max_train_steps: 1000 }
|
||||
checkpoint: { output_dir: ./outputs }
|
||||
tracker: { trackers: [wandb], project_name: ... }
|
||||
|
||||
callbacks:
|
||||
grad_clip:
|
||||
_target_: fastvideo.train.callbacks.GradNormClipCallback
|
||||
max_grad_norm: 1.0
|
||||
validation:
|
||||
_target_: fastvideo.train.callbacks.ValidationCallback
|
||||
validation_steps: 100
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Legacy Training Pipelines (`fastvideo/training/`)
|
||||
|
||||
> **Note:** Use the new `fastvideo/train/` framework for new work. This section
|
||||
> is retained for reference on existing pipelines not yet migrated.
|
||||
|
||||
| Pipeline | Entrypoint | Use Case |
|
||||
|----------|-----------|----------|
|
||||
| Wan T2V finetune | `fastvideo/training/wan_training_pipeline.py` | Standard text-to-video finetune / LoRA |
|
||||
| Wan I2V finetune | `fastvideo/training/wan_i2v_training_pipeline.py` | Image-to-video (first frame conditioned) |
|
||||
| MatrixGame finetune | `fastvideo/training/matrixgame_training_pipeline.py` | Action-conditioned world model |
|
||||
| MatrixGame AR diffusion | `fastvideo/training/matrixgame_ar_diffusion_pipeline.py` | AR diffusion-forcing training |
|
||||
| MatrixGame ODE-init | `fastvideo/training/matrixgame_ode_causal_pipeline.py` | ODE-trajectory init |
|
||||
| MatrixGame self-forcing distill | `fastvideo/training/matrixgame_self_forcing_distillation_pipeline.py` | Self-forcing distillation |
|
||||
| LTX-2 finetune | `fastvideo/training/ltx2_training_pipeline.py` | LTX-2 architecture finetuning |
|
||||
| Wan DMD distillation | `fastvideo/training/wan_distillation_pipeline.py` | Few-step distillation via DMD |
|
||||
| Self-Forcing distill | `fastvideo/training/wan_self_forcing_distillation_pipeline.py` | Causal streaming distillation |
|
||||
|
||||
---
|
||||
|
||||
## Key Infrastructure
|
||||
|
||||
### W&B Integration
|
||||
- **Tracker**: `fastvideo/training/trackers.py` — `WandbTracker` class
|
||||
- **New framework tracker**: `fastvideo/train/utils/tracking.py` — `build_tracker()`
|
||||
- **Env vars**: `WANDB_API_KEY`, `WANDB_BASE_URL`, `WANDB_MODE`
|
||||
|
||||
### Parallelism
|
||||
- **SP** (Sequence Parallel): splits video frames across GPUs — `sp_size: N`
|
||||
- **TP** (Tensor Parallel): splits model layers across GPUs — `tp_size: N`
|
||||
- Typical configs: SP=2–8, TP=1–2
|
||||
|
||||
---
|
||||
|
||||
## Evaluation (for training runs)
|
||||
|
||||
Read `.agents/memory/evaluation-registry/README.md` for the full metric catalog.
|
||||
|
||||
**Quick summary for training agents:**
|
||||
| Metric | When to Use | Trust |
|
||||
|--------|-------------|-------|
|
||||
| **Loss trajectory** | Every run, real-time from W&B | Medium |
|
||||
| **SSIM** | When comparing against reference outputs | High |
|
||||
| **FVD** | For benchmarking model quality (`benchmarks/fvd/`) | High |
|
||||
| **LPIPS** | LoRA merge validation | Medium |
|
||||
| **Human preference** | Major checkpoints | Highest |
|
||||
|
||||
---
|
||||
|
||||
## Common Workflows
|
||||
|
||||
| Task | Skill / SOP |
|
||||
|------|-------------|
|
||||
| Launch a training run | `.agents/skills/launch-experiment/SKILL.md` |
|
||||
| Monitor a running experiment | `.agents/skills/monitor-experiment/SKILL.md` |
|
||||
| Summarize final results | `.agents/skills/summarize-run/SKILL.md` |
|
||||
| Full experiment lifecycle | `.agents/workflows/experiment-lifecycle.md` |
|
||||
| Capture lessons from failures | `.agents/workflows/lesson-capture.md` |
|
||||
|
||||
---
|
||||
|
||||
## World Model–Specific Concepts
|
||||
|
||||
### Action Injection (MatrixGame)
|
||||
The MatrixGame pipeline adds **action modules** to each DiT block, enabling
|
||||
frame-level mouse/keyboard input conditioning. The action sequence is injected
|
||||
per-frame alongside the latent video tokens.
|
||||
|
||||
### Causal Architecture
|
||||
For streaming generation, the model uses **causal attention** (each frame only
|
||||
attends to previous frames). This enables auto-regressive chunk-by-chunk
|
||||
generation — critical for real-time interactive world models.
|
||||
|
||||
### Self-Forcing Distillation
|
||||
A **data-free** distillation method where the student model is trained to
|
||||
generate coherent video sequences by being forced to use its own previous
|
||||
outputs (rather than ground-truth) as context. This produces models robust to
|
||||
their own error accumulation during long auto-regressive generation.
|
||||
|
||||
### DMD Distillation (Distribution Matching Distillation)
|
||||
Reduces inference steps from ~50 to 3–4 by training a student model to match
|
||||
the output distribution of the teacher model. Uses a critic network to estimate
|
||||
distribution divergence.
|
||||
|
||||
### Diffusion-Forcing SFT (DFSFT)
|
||||
Supervised finetuning with **inhomogeneous timesteps** across chunks — each
|
||||
chunk in a causal sequence can have a different noise level, training the model
|
||||
to handle mixed-fidelity contexts.
|
||||
Executable
+96
@@ -0,0 +1,96 @@
|
||||
#!/usr/bin/env bash
|
||||
# Sync .agents/skills/ into .claude/skills/ via per-skill symlinks.
|
||||
#
|
||||
# Why: Claude Code only scans .claude/skills/ and ~/.claude/skills/ for
|
||||
# user-invocable skills (no skillsPath config exists — see
|
||||
# https://code.claude.com/docs/en/skills.md). This repo's skills live
|
||||
# in .agents/skills/ so they travel with the repo and stay under git.
|
||||
# Run this once after cloning (or after adding/removing a skill) to
|
||||
# expose them to Claude Code without maintaining a parallel tree.
|
||||
#
|
||||
# Usage:
|
||||
# .agents/scripts/sync-skills.sh
|
||||
#
|
||||
# Idempotent and safe to re-run. Prunes stale symlinks whose source
|
||||
# has been removed from .agents/skills/. Leaves hand-written
|
||||
# .claude/skills/<name>/ directories untouched (only symlinks are
|
||||
# managed).
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
REPO_ROOT="$(git -C "$(dirname "$0")" rev-parse --show-toplevel)"
|
||||
SRC_DIR="$REPO_ROOT/.agents/skills"
|
||||
DST_DIR="$REPO_ROOT/.claude/skills"
|
||||
|
||||
if [[ ! -d "$SRC_DIR" ]]; then
|
||||
echo "Error: $SRC_DIR does not exist." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
mkdir -p "$DST_DIR"
|
||||
|
||||
linked=0
|
||||
unchanged=0
|
||||
skipped=0
|
||||
pruned=0
|
||||
|
||||
link_skill() {
|
||||
local name="$1"
|
||||
local src="$SRC_DIR/$name"
|
||||
local dst="$DST_DIR/$name"
|
||||
# Relative target keeps symlinks portable across clones.
|
||||
local rel="../../.agents/skills/$name"
|
||||
|
||||
if [[ -L "$dst" ]]; then
|
||||
if [[ "$(readlink "$dst")" == "$rel" ]]; then
|
||||
unchanged=$((unchanged + 1))
|
||||
return
|
||||
fi
|
||||
rm "$dst"
|
||||
elif [[ -e "$dst" ]]; then
|
||||
echo "Skipped (not a symlink): .claude/skills/$name" >&2
|
||||
skipped=$((skipped + 1))
|
||||
return
|
||||
fi
|
||||
|
||||
ln -s "$rel" "$dst"
|
||||
echo "Linked: .claude/skills/$name -> $rel"
|
||||
linked=$((linked + 1))
|
||||
}
|
||||
|
||||
prune_stale() {
|
||||
local link="$1"
|
||||
local target
|
||||
target="$(readlink "$link")"
|
||||
case "$target" in
|
||||
../../.agents/skills/*) ;;
|
||||
*) return ;;
|
||||
esac
|
||||
local name="${target##*/}"
|
||||
if [[ ! -d "$SRC_DIR/$name" ]]; then
|
||||
rm "$link"
|
||||
echo "Pruned stale: .claude/skills/$(basename "$link")"
|
||||
pruned=$((pruned + 1))
|
||||
fi
|
||||
}
|
||||
|
||||
for src in "$SRC_DIR"/*/; do
|
||||
[[ -d "$src" ]] || continue
|
||||
name="$(basename "$src")"
|
||||
# Only treat directories that actually contain a SKILL.md as skills.
|
||||
[[ -f "$src/SKILL.md" ]] || continue
|
||||
link_skill "$name"
|
||||
done
|
||||
|
||||
shopt -s nullglob
|
||||
for link in "$DST_DIR"/*; do
|
||||
[[ -L "$link" ]] || continue
|
||||
prune_stale "$link"
|
||||
done
|
||||
shopt -u nullglob
|
||||
|
||||
printf "\nSummary: %d linked, %d unchanged, %d pruned" "$linked" "$unchanged" "$pruned"
|
||||
if [[ "$skipped" -gt 0 ]]; then
|
||||
printf ", %d skipped (non-symlink collision)" "$skipped"
|
||||
fi
|
||||
printf "\n"
|
||||
@@ -0,0 +1,57 @@
|
||||
---
|
||||
name: <skill-name>
|
||||
description: <one-line description — Codex uses this for implicit invocation matching>
|
||||
---
|
||||
|
||||
# <Skill Name>
|
||||
|
||||
## Purpose
|
||||
<Why this skill exists and when to use it.>
|
||||
|
||||
## Prerequisites
|
||||
- <What must be true before using this skill>
|
||||
|
||||
## Inputs
|
||||
| Parameter | Required | Description |
|
||||
|-----------|----------|-------------|
|
||||
| `param1` | Yes | ... |
|
||||
|
||||
## Steps
|
||||
|
||||
1. **Step 1 title**
|
||||
- Detail...
|
||||
|
||||
2. **Step 2 title**
|
||||
- Detail...
|
||||
|
||||
## Outputs
|
||||
- <What this skill produces>
|
||||
|
||||
## Example Usage
|
||||
|
||||
```
|
||||
<Example invocation or prompt snippet>
|
||||
```
|
||||
|
||||
## References
|
||||
- <Links to relevant files in the codebase>
|
||||
|
||||
---
|
||||
|
||||
## Folder Structure
|
||||
|
||||
Each skill lives in its own directory under `.agents/skills/`:
|
||||
|
||||
```
|
||||
.agents/skills/<skill-name>/
|
||||
├── SKILL.md # Required: instructions + metadata (this file)
|
||||
├── scripts/ # Optional: executable helper scripts
|
||||
├── references/ # Optional: documentation, papers
|
||||
└── assets/ # Optional: templates, resources
|
||||
```
|
||||
|
||||
After creating a new skill, add an entry to `.agents/skills/index.jsonl`:
|
||||
|
||||
```json
|
||||
{"name": "<skill-name>", "description": "<description>", "path": "<skill-name>/SKILL.md", "status": "draft", "trust": "low"}
|
||||
```
|
||||
@@ -0,0 +1,173 @@
|
||||
---
|
||||
name: add-model-01-prep
|
||||
description: Use at the start of a FastVideo model port to gather required inputs, inspect/download HF weights, clone and install the official reference repo in the current environment, create a local_tests README skeleton, and produce a handoff before conversion or implementation.
|
||||
---
|
||||
|
||||
# Add Model Prep
|
||||
|
||||
## Goal
|
||||
|
||||
Prepare external assets and the shared parity-test environment for a FastVideo
|
||||
model port. Stop before writing conversion scripts, model components, pipeline
|
||||
code, registry entries, or executable parity tests.
|
||||
|
||||
## Ask First
|
||||
|
||||
Ask once, then proceed if the HF token is already exported:
|
||||
|
||||
```text
|
||||
Before prep: (1) official reference repo or Diffusers pipeline URL, (2) HF repo
|
||||
id or local weights path and whether it has a root model_index.json, (3) target
|
||||
model_family, (4) workload types, (5) which token env var is exported:
|
||||
HF_TOKEN, HUGGINGFACE_HUB_TOKEN, or HF_API_KEY, (6) may I stage clone and
|
||||
weights under the FastVideo repo root, and (7) may I install official reference
|
||||
dependencies into the current FastVideo conda/env for parity tests?
|
||||
```
|
||||
|
||||
Useful optional inputs: `pipeline_class`, `reference_dir`, `hf_revision`,
|
||||
`official_revision`, `reuse_hints`, `download_scope`.
|
||||
|
||||
## Rules
|
||||
|
||||
- Follow `../add-model/shared/common_rules.md` for token/auth safety, state files,
|
||||
escape hatches, and skip/pass semantics.
|
||||
- Run from the FastVideo repo root.
|
||||
- Use repo-relative defaults: `<ReferenceDir>/`,
|
||||
`official_weights/<model_family>/`, `converted_weights/<model_family>/`.
|
||||
- Install official reference deps into the current FastVideo environment, not a
|
||||
new venv/conda env, so parity tests run both implementations with one shared
|
||||
numeric stack.
|
||||
- If the reference is a Diffusers class/package instead of a cloneable repo,
|
||||
record import path and version instead of cloning.
|
||||
- Prep may create only the local-test README and `PORT_STATUS.md` skeletons;
|
||||
executable `.py` parity tests belong to `../add-model-02-parity/SKILL.md`.
|
||||
|
||||
## Escape Hatches
|
||||
|
||||
Follow `../add-model/shared/common_rules.md`. Prep-specific ask cases include
|
||||
overwriting an existing clone or weight directory, installing untrusted/private
|
||||
deps, choosing between incompatible official references, large downloads outside
|
||||
the agreed scope, or missing gated-repo auth setup by env var name.
|
||||
|
||||
## Workflow
|
||||
|
||||
1. Verify the repo:
|
||||
|
||||
```bash
|
||||
git rev-parse --show-toplevel
|
||||
```
|
||||
|
||||
Expected markers: `fastvideo/`, `scripts/checkpoint_conversion/`,
|
||||
`scripts/huggingface/download_hf.py`, `fastvideo/registry.py`.
|
||||
|
||||
2. Inspect HF or local weight layout:
|
||||
|
||||
```bash
|
||||
python ".agents/skills/add-model-01-prep/scripts/inspect_hf_layout.py" \
|
||||
"Org/Model" \
|
||||
--revision "<revision>" \
|
||||
--json
|
||||
```
|
||||
|
||||
For a local path, replace `Org/Model` with `/path/to/weights`. Record
|
||||
`source_layout`, `needs_conversion`, `model_index_class`, and
|
||||
`components_seen`.
|
||||
|
||||
3. Download HF weights if needed:
|
||||
|
||||
```bash
|
||||
python ".agents/skills/add-model-01-prep/scripts/download_hf_weights.py" \
|
||||
"Org/Model" \
|
||||
"official_weights/<model_family>" \
|
||||
--revision "<revision>"
|
||||
```
|
||||
|
||||
For selected files, repeat `--file-name`. For partial snapshots, repeat
|
||||
`--allow-pattern` or `--ignore-pattern`. If the user provided a local path,
|
||||
record it instead of copying large weights by default.
|
||||
|
||||
4. Clone the official reference repo if applicable:
|
||||
|
||||
```bash
|
||||
python ".agents/skills/add-model-01-prep/scripts/clone_reference_repo.py" \
|
||||
"<official_repo_url>" \
|
||||
"<ReferenceDir>" \
|
||||
--branch "<tag-or-branch>" \
|
||||
--commit "<commit-sha>" \
|
||||
--update-gitignore
|
||||
```
|
||||
|
||||
Omit `--branch`, `--commit`, or `--update-gitignore` when not needed. The
|
||||
helper refuses to overwrite existing paths and prints remote/HEAD instead.
|
||||
|
||||
5. Keep prep assets ignored. Ensure `.gitignore` includes relevant entries:
|
||||
|
||||
```gitignore
|
||||
/<ReferenceDir>/
|
||||
/official_weights/
|
||||
/converted_weights/
|
||||
```
|
||||
|
||||
6. Follow the official repo's setup instructions in the current environment.
|
||||
Inspect dependency files and README install docs before installing anything:
|
||||
|
||||
- `README*`, install docs, or model-card instructions.
|
||||
- `requirements*.txt`, `pyproject.toml`, `setup.py`, `environment.yml`.
|
||||
|
||||
Use the current FastVideo conda/env. Do not create a new env even if upstream
|
||||
docs recommend one; translate the needed install commands into the active env.
|
||||
Prefer editable/no-deps first so the official source is importable without
|
||||
changing shared pins:
|
||||
|
||||
```bash
|
||||
uv pip install --no-deps -e ./<ReferenceDir>
|
||||
```
|
||||
|
||||
Then install only missing official deps needed for parity imports. Stop before
|
||||
installing requirements that would change FastVideo's core stack. If upstream
|
||||
requires private/non-PyPI deps, record that parity needs a local stub helper
|
||||
rather than pretending setup is complete.
|
||||
|
||||
7. Create the model-family local test skeleton and top-level port state file:
|
||||
|
||||
```bash
|
||||
mkdir -p tests/local_tests/<model_family>
|
||||
cp ".agents/skills/add-model-01-prep/templates/local_tests_readme.md" \
|
||||
tests/local_tests/<model_family>/README.md
|
||||
cp ".agents/skills/add-model-01-prep/templates/port_status.md" \
|
||||
tests/local_tests/<model_family>/PORT_STATUS.md
|
||||
```
|
||||
|
||||
Edit every placeholder in the README and `PORT_STATUS.md`. The README gives
|
||||
later review agents enough information to reproduce the shared environment and
|
||||
run/review parity work:
|
||||
|
||||
- official code URL or import path, local clone path, and commit/version;
|
||||
- HF URL or local weight path, revision, access notes, and token env var name
|
||||
only;
|
||||
- commands already run and any blocked official dependency installs;
|
||||
- shared-env install commands to re-run without changing core pins;
|
||||
- expected local parity test paths and pytest commands;
|
||||
- private-dependency stubs or known setup gaps;
|
||||
- PR/review notes explaining which parity tests are required before handoff.
|
||||
|
||||
Do not include raw tokens, absolute cache paths that are not repo-reproducible,
|
||||
or large generated outputs. If prep is blocked before imports work, still create
|
||||
the README with `official_env_status=blocked` and the exact blocker.
|
||||
|
||||
`PORT_STATUS.md` must follow `../add-model/contracts/port_state.md`. Record open
|
||||
questions and prep issues immediately, using stable IDs such as `Q001` and
|
||||
`I001`. Keep resolved questions/issues in the table with a resolution instead of
|
||||
deleting them.
|
||||
|
||||
## Handoff
|
||||
|
||||
End with the canonical prep handoff contract from
|
||||
`../add-model/contracts/prep_handoff.md` and update the shared state files before
|
||||
handoff.
|
||||
|
||||
## Helper Scripts
|
||||
|
||||
- `scripts/inspect_hf_layout.py`: classify HF/local layout.
|
||||
- `scripts/download_hf_weights.py`: download HF snapshot or selected files.
|
||||
- `scripts/clone_reference_repo.py`: clone reference repo safely.
|
||||
@@ -0,0 +1,123 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Clone an official reference repo without overwriting existing paths."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import subprocess
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Clone a reference repo for FastVideo parity tests."
|
||||
)
|
||||
parser.add_argument("repo_url", help="Official reference repository URL")
|
||||
parser.add_argument("target_dir", help="Directory to clone into")
|
||||
parser.add_argument("--branch", help="Branch or tag to clone")
|
||||
parser.add_argument("--commit", help="Commit SHA to check out after clone")
|
||||
parser.add_argument(
|
||||
"--update-gitignore",
|
||||
action="store_true",
|
||||
help="Add the target directory to .gitignore if missing",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--gitignore",
|
||||
default=".gitignore",
|
||||
help="Path to gitignore file when --update-gitignore is used",
|
||||
)
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def run(command: list[str], check: bool = True) -> subprocess.CompletedProcess[str]:
|
||||
return subprocess.run(
|
||||
command,
|
||||
check=check,
|
||||
text=True,
|
||||
capture_output=True,
|
||||
)
|
||||
|
||||
|
||||
def print_existing_repo_info(target: Path) -> int:
|
||||
print(f"target_exists: {target}")
|
||||
if not (target / ".git").exists():
|
||||
print("error: target exists but is not a git repo", file=sys.stderr)
|
||||
return 1
|
||||
|
||||
remote = run(["git", "-C", str(target), "remote", "-v"], check=False)
|
||||
head = run(["git", "-C", str(target), "rev-parse", "HEAD"], check=False)
|
||||
if remote.stdout:
|
||||
print("remote_v:")
|
||||
print(remote.stdout.rstrip())
|
||||
if head.stdout:
|
||||
print(f"head: {head.stdout.strip()}")
|
||||
print("not_overwritten: true")
|
||||
return 0
|
||||
|
||||
|
||||
def gitignore_entry_for(target: Path) -> str:
|
||||
root = Path.cwd().resolve()
|
||||
resolved = target.resolve()
|
||||
try:
|
||||
relative = resolved.relative_to(root)
|
||||
except ValueError as exc:
|
||||
raise ValueError(
|
||||
"--update-gitignore requires target_dir to be under the current directory"
|
||||
) from exc
|
||||
|
||||
text = relative.as_posix().rstrip("/")
|
||||
return "/" + text + "/"
|
||||
|
||||
|
||||
def update_gitignore(path: Path, target: Path) -> bool:
|
||||
entry = gitignore_entry_for(target)
|
||||
existing = path.read_text().splitlines() if path.exists() else []
|
||||
if entry in existing:
|
||||
return False
|
||||
|
||||
new_text = "\n".join(existing).rstrip("\n")
|
||||
if new_text:
|
||||
new_text += "\n"
|
||||
new_text += entry + "\n"
|
||||
path.write_text(new_text)
|
||||
return True
|
||||
|
||||
|
||||
def main() -> int:
|
||||
args = parse_args()
|
||||
target = Path(args.target_dir)
|
||||
|
||||
if target.exists():
|
||||
return print_existing_repo_info(target)
|
||||
|
||||
command = ["git", "clone", "--depth", "1"]
|
||||
if args.branch:
|
||||
command.extend(["--branch", args.branch])
|
||||
command.extend([args.repo_url, str(target)])
|
||||
|
||||
try:
|
||||
run(command)
|
||||
if args.commit:
|
||||
run(["git", "-C", str(target), "fetch", "--depth", "1", "origin", args.commit])
|
||||
run(["git", "-C", str(target), "checkout", args.commit])
|
||||
except subprocess.CalledProcessError as exc:
|
||||
if exc.stdout:
|
||||
print(exc.stdout, end="")
|
||||
if exc.stderr:
|
||||
print(exc.stderr, end="", file=sys.stderr)
|
||||
return exc.returncode
|
||||
|
||||
head = run(["git", "-C", str(target), "rev-parse", "HEAD"])
|
||||
print(f"cloned: {target}")
|
||||
print(f"head: {head.stdout.strip()}")
|
||||
|
||||
if args.update_gitignore:
|
||||
changed = update_gitignore(Path(args.gitignore), target)
|
||||
print(f"gitignore_updated: {str(changed).lower()}")
|
||||
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -0,0 +1,105 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Download HF weights using the standard FastVideo token env vars."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import os
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
HF_TOKEN_ENV_KEYS = ("HF_TOKEN", "HUGGINGFACE_HUB_TOKEN", "HF_API_KEY")
|
||||
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Download a HF model snapshot or selected files into a local directory."
|
||||
)
|
||||
parser.add_argument("repo_id", help="HF repo id, for example Org/Model")
|
||||
parser.add_argument("local_dir", help="Destination directory")
|
||||
parser.add_argument("--repo-type", default="model", help="HF repo type (default: model)")
|
||||
parser.add_argument("--revision", help="HF branch, tag, or commit")
|
||||
parser.add_argument(
|
||||
"--file-name",
|
||||
action="append",
|
||||
default=[],
|
||||
help="Download one file; may be repeated. If omitted, download full snapshot.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--allow-pattern",
|
||||
action="append",
|
||||
default=[],
|
||||
help="Snapshot allow pattern; may be repeated. Ignored when --file-name is used.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--ignore-pattern",
|
||||
action="append",
|
||||
default=[],
|
||||
help="Snapshot ignore pattern; may be repeated. Ignored when --file-name is used.",
|
||||
)
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def resolve_token() -> tuple[str | None, str | None]:
|
||||
for key in HF_TOKEN_ENV_KEYS:
|
||||
value = os.environ.get(key)
|
||||
if value:
|
||||
return key, value
|
||||
return None, None
|
||||
|
||||
|
||||
def main() -> int:
|
||||
args = parse_args()
|
||||
token_env, token = resolve_token()
|
||||
local_dir = Path(args.local_dir).expanduser()
|
||||
|
||||
if token_env:
|
||||
print(f"token_env: {token_env}")
|
||||
else:
|
||||
print("token_env: none", file=sys.stderr)
|
||||
|
||||
try:
|
||||
if local_dir.exists() and not local_dir.is_dir():
|
||||
print(
|
||||
f"error: destination exists and is not a directory: {local_dir}",
|
||||
file=sys.stderr,
|
||||
)
|
||||
return 1
|
||||
local_dir.mkdir(parents=True, exist_ok=True)
|
||||
if args.file_name:
|
||||
from huggingface_hub import hf_hub_download
|
||||
|
||||
for file_name in args.file_name:
|
||||
path = hf_hub_download(
|
||||
repo_id=args.repo_id,
|
||||
filename=file_name,
|
||||
repo_type=args.repo_type,
|
||||
revision=args.revision,
|
||||
local_dir=str(local_dir),
|
||||
token=token,
|
||||
)
|
||||
print(f"downloaded_file: {path}")
|
||||
else:
|
||||
from huggingface_hub import snapshot_download
|
||||
|
||||
path = snapshot_download(
|
||||
repo_id=args.repo_id,
|
||||
repo_type=args.repo_type,
|
||||
revision=args.revision,
|
||||
local_dir=str(local_dir),
|
||||
token=token,
|
||||
allow_patterns=args.allow_pattern or None,
|
||||
ignore_patterns=args.ignore_pattern or None,
|
||||
)
|
||||
print(f"downloaded_snapshot: {path}")
|
||||
except Exception as exc: # noqa: BLE001 - CLI should print concise failures.
|
||||
print(f"error: {exc}", file=sys.stderr)
|
||||
return 1
|
||||
|
||||
print(f"local_dir: {local_dir.resolve()}")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -0,0 +1,264 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Inspect a Hugging Face repo or local weight directory layout."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
|
||||
HF_TOKEN_ENV_KEYS = ("HF_TOKEN", "HUGGINGFACE_HUB_TOKEN", "HF_API_KEY")
|
||||
RAW_WEIGHT_SUFFIXES = (".safetensors", ".pt", ".pth", ".ckpt", ".bin")
|
||||
KNOWN_COMPONENTS = {
|
||||
"audio_vae",
|
||||
"conditioner",
|
||||
"feature_extractor",
|
||||
"image_encoder",
|
||||
"scheduler",
|
||||
"text_encoder",
|
||||
"text_encoder_2",
|
||||
"tokenizer",
|
||||
"tokenizer_2",
|
||||
"transformer",
|
||||
"transformer_2",
|
||||
"unet",
|
||||
"upsampler",
|
||||
"vae",
|
||||
"vocoder",
|
||||
}
|
||||
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Classify a HF repo or local directory as Diffusers, raw, custom, or unknown."
|
||||
)
|
||||
parser.add_argument("source", help="HF repo id or local weights directory")
|
||||
parser.add_argument("--repo-type", default="model", help="HF repo type (default: model)")
|
||||
parser.add_argument("--revision", help="HF revision to inspect")
|
||||
parser.add_argument(
|
||||
"--max-local-files",
|
||||
type=int,
|
||||
default=20000,
|
||||
help="Maximum local files to scan recursively (default: 20000)",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--sample-limit",
|
||||
type=int,
|
||||
default=80,
|
||||
help="Number of file paths to print in human output (default: 80)",
|
||||
)
|
||||
parser.add_argument("--json", action="store_true", help="Emit JSON only")
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def resolve_token() -> tuple[str | None, str | None]:
|
||||
for key in HF_TOKEN_ENV_KEYS:
|
||||
value = os.environ.get(key)
|
||||
if value:
|
||||
return key, value
|
||||
return None, None
|
||||
|
||||
|
||||
def load_local_files(root: Path, max_files: int) -> tuple[list[str], bool]:
|
||||
files: list[str] = []
|
||||
truncated = False
|
||||
for path in root.rglob("*"):
|
||||
if not path.is_file():
|
||||
continue
|
||||
files.append(path.relative_to(root).as_posix())
|
||||
if len(files) >= max_files:
|
||||
truncated = True
|
||||
break
|
||||
return sorted(files), truncated
|
||||
|
||||
|
||||
def load_local_model_index(root: Path) -> tuple[dict[str, Any] | None, str | None]:
|
||||
index_path = root / "model_index.json"
|
||||
if not index_path.is_file():
|
||||
return None, None
|
||||
try:
|
||||
return json.loads(index_path.read_text()), None
|
||||
except Exception as exc: # noqa: BLE001 - surface malformed JSON clearly.
|
||||
return None, f"failed to parse local model_index.json: {exc}"
|
||||
|
||||
|
||||
def load_remote_files(
|
||||
repo_id: str,
|
||||
repo_type: str,
|
||||
revision: str | None,
|
||||
token: str | None,
|
||||
) -> list[str]:
|
||||
from huggingface_hub import list_repo_files
|
||||
|
||||
return sorted(
|
||||
list_repo_files(
|
||||
repo_id,
|
||||
repo_type=repo_type,
|
||||
revision=revision,
|
||||
token=token,
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
def load_remote_model_index(
|
||||
repo_id: str,
|
||||
repo_type: str,
|
||||
revision: str | None,
|
||||
token: str | None,
|
||||
) -> tuple[dict[str, Any] | None, str | None]:
|
||||
from huggingface_hub import hf_hub_download
|
||||
|
||||
try:
|
||||
path = hf_hub_download(
|
||||
repo_id=repo_id,
|
||||
filename="model_index.json",
|
||||
repo_type=repo_type,
|
||||
revision=revision,
|
||||
token=token,
|
||||
)
|
||||
except Exception as exc: # noqa: BLE001 - missing/inaccessible file is data.
|
||||
return None, f"failed to download model_index.json: {exc}"
|
||||
|
||||
try:
|
||||
return json.loads(Path(path).read_text()), None
|
||||
except Exception as exc: # noqa: BLE001 - surface malformed JSON clearly.
|
||||
return None, f"failed to parse remote model_index.json: {exc}"
|
||||
|
||||
|
||||
def root_file_names(files: list[str]) -> set[str]:
|
||||
return {name for name in files if "/" not in name}
|
||||
|
||||
|
||||
def component_names(files: list[str], model_index: dict[str, Any] | None) -> list[str]:
|
||||
components: set[str] = set()
|
||||
for name in files:
|
||||
parts = name.split("/", 1)
|
||||
if len(parts) != 2:
|
||||
continue
|
||||
top, rest = parts
|
||||
if top in KNOWN_COMPONENTS or rest == "config.json":
|
||||
components.add(top)
|
||||
|
||||
if model_index:
|
||||
for key, value in model_index.items():
|
||||
if key.startswith("_"):
|
||||
continue
|
||||
if isinstance(value, list) and len(value) == 2:
|
||||
components.add(key)
|
||||
|
||||
return sorted(components)
|
||||
|
||||
|
||||
def classify_layout(
|
||||
files: list[str],
|
||||
model_index: dict[str, Any] | None,
|
||||
components: list[str],
|
||||
) -> tuple[str, str]:
|
||||
roots = root_file_names(files)
|
||||
raw_weight_files = [name for name in roots if name.endswith(RAW_WEIGHT_SUFFIXES)]
|
||||
has_model_index = "model_index.json" in roots or model_index is not None
|
||||
|
||||
if has_model_index and components:
|
||||
return "diffusers", "no"
|
||||
if has_model_index:
|
||||
return "custom", "unknown"
|
||||
if raw_weight_files:
|
||||
return "raw_official", "yes"
|
||||
if any(name.endswith(RAW_WEIGHT_SUFFIXES) for name in files):
|
||||
return "custom", "yes"
|
||||
return "unknown", "unknown"
|
||||
|
||||
|
||||
def build_result(args: argparse.Namespace) -> dict[str, Any]:
|
||||
token_env, token = resolve_token()
|
||||
source_path = Path(args.source).expanduser()
|
||||
is_local = source_path.exists()
|
||||
|
||||
if is_local:
|
||||
root = source_path.resolve()
|
||||
if not root.is_dir():
|
||||
raise ValueError(f"local source is not a directory: {root}")
|
||||
files, truncated = load_local_files(root, args.max_local_files)
|
||||
model_index, model_index_error = load_local_model_index(root)
|
||||
source_kind = "local"
|
||||
source = str(root)
|
||||
else:
|
||||
files = load_remote_files(args.source, args.repo_type, args.revision, token)
|
||||
truncated = False
|
||||
model_index, model_index_error = load_remote_model_index(
|
||||
args.source,
|
||||
args.repo_type,
|
||||
args.revision,
|
||||
token,
|
||||
)
|
||||
source_kind = "hf"
|
||||
source = args.source
|
||||
|
||||
components = component_names(files, model_index)
|
||||
source_layout, needs_conversion = classify_layout(files, model_index, components)
|
||||
|
||||
return {
|
||||
"source": source,
|
||||
"source_kind": source_kind,
|
||||
"repo_type": None if is_local else args.repo_type,
|
||||
"revision": args.revision,
|
||||
"token_env": token_env,
|
||||
"source_layout": source_layout,
|
||||
"needs_conversion": needs_conversion,
|
||||
"model_index_class": (model_index or {}).get("_class_name"),
|
||||
"model_index_diffusers_version": (model_index or {}).get("_diffusers_version"),
|
||||
"model_index_error": model_index_error,
|
||||
"components_seen": components,
|
||||
"file_count": len(files),
|
||||
"file_scan_truncated": truncated,
|
||||
"files_sample": files[: args.sample_limit],
|
||||
}
|
||||
|
||||
|
||||
def print_human(result: dict[str, Any]) -> None:
|
||||
for key in (
|
||||
"source",
|
||||
"source_kind",
|
||||
"repo_type",
|
||||
"revision",
|
||||
"token_env",
|
||||
"source_layout",
|
||||
"needs_conversion",
|
||||
"model_index_class",
|
||||
"model_index_diffusers_version",
|
||||
"model_index_error",
|
||||
"file_count",
|
||||
"file_scan_truncated",
|
||||
):
|
||||
value = result.get(key)
|
||||
if value is not None:
|
||||
print(f"{key}: {value}")
|
||||
|
||||
components = result["components_seen"]
|
||||
print("components_seen: " + (", ".join(components) if components else "none"))
|
||||
print("files_sample:")
|
||||
for name in result["files_sample"]:
|
||||
print(f" {name}")
|
||||
|
||||
|
||||
def main() -> int:
|
||||
args = parse_args()
|
||||
try:
|
||||
result = build_result(args)
|
||||
except Exception as exc: # noqa: BLE001 - CLI should print concise failures.
|
||||
print(f"error: {exc}", file=sys.stderr)
|
||||
return 1
|
||||
|
||||
if args.json:
|
||||
print(json.dumps(result, indent=2, sort_keys=True))
|
||||
else:
|
||||
print_human(result)
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -0,0 +1,122 @@
|
||||
# <Model Family> Local Tests
|
||||
|
||||
Local-only parity and smoke tests for the `<model_family>` FastVideo port. These
|
||||
tests compare FastVideo against the official reference implementation and are
|
||||
not expected to run in CI unless explicitly promoted later.
|
||||
|
||||
Port progress, open questions, issues, and handoff notes live in
|
||||
`tests/local_tests/<model_family>/PORT_STATUS.md`.
|
||||
|
||||
## Reference Assets
|
||||
|
||||
| Field | Value |
|
||||
|---|---|
|
||||
| Model family | `<model_family>` |
|
||||
| Workload types | `<T2V/I2V/V2V/T2I/or compatibility shim with rationale>` |
|
||||
| Official reference | `<url or import path>` |
|
||||
| Local reference dir | `<ReferenceDir or none>` |
|
||||
| Official commit/version | `<sha, tag, package version, or unknown>` |
|
||||
| HF weights | `<HF repo id/url or local path>` |
|
||||
| HF revision | `<revision or default>` |
|
||||
| Local weights dir | `<official_weights/model_family or local path>` |
|
||||
| Source layout | `<diffusers/raw_official/monolithic/separate_components/mixed/custom/unknown>` |
|
||||
| Needs conversion | `<yes/no/unknown>` |
|
||||
|
||||
Do not write token values in this file. Use only the token env var name:
|
||||
`<HF_TOKEN or HUGGINGFACE_HUB_TOKEN or HF_API_KEY>`.
|
||||
|
||||
## Shared Environment Setup
|
||||
|
||||
Run from the FastVideo repo root in the same conda/env used for FastVideo.
|
||||
Do not create a separate upstream environment for parity tests.
|
||||
|
||||
```bash
|
||||
# Official reference source, if cloneable.
|
||||
python ".agents/skills/add-model-01-prep/scripts/clone_reference_repo.py" \
|
||||
"<official_repo_url>" \
|
||||
"<ReferenceDir>" \
|
||||
--commit "<commit-sha>" \
|
||||
--update-gitignore
|
||||
|
||||
# Editable install without changing shared core pins.
|
||||
uv pip install --no-deps -e ./<ReferenceDir>
|
||||
|
||||
# Additional official deps installed or required for imports:
|
||||
# <package list or none>
|
||||
```
|
||||
|
||||
Do not change core dependency versions (`torch`, `diffusers`, `transformers`,
|
||||
`flash-attn`, `triton`, CUDA packages) without explicit approval.
|
||||
|
||||
## Official Environment Status
|
||||
|
||||
```text
|
||||
dependency_changes: <none | installed no-deps editable | installed official deps in current env | blocked on user>
|
||||
official_env_status: <imports_ok | private_deps_need_stubs | blocked>
|
||||
private_dep_stubs: <none or tests/local_tests/helpers/<model_family>_upstream.py>
|
||||
blocked_on: <none or exact blocker>
|
||||
```
|
||||
|
||||
## Weight Setup
|
||||
|
||||
```bash
|
||||
python ".agents/skills/add-model-01-prep/scripts/download_hf_weights.py" \
|
||||
"<Org/Model>" \
|
||||
"official_weights/<model_family>" \
|
||||
--revision "<revision>"
|
||||
```
|
||||
|
||||
If weights are local-only, record the local path and do not copy large files into
|
||||
the repository.
|
||||
|
||||
## Prototype And Conversion Artifacts
|
||||
|
||||
State-dict key/shape dumps are generated after FastVideo native prototypes exist
|
||||
and are used to build the conversion mapping.
|
||||
|
||||
```text
|
||||
official_key_dumps:
|
||||
<component>: converted_weights/<model_family>/_mapping/<component>_official_keys.json
|
||||
fastvideo_key_dumps:
|
||||
<component>: converted_weights/<model_family>/_mapping/<component>_fastvideo_keys.json
|
||||
conversion_script: scripts/checkpoint_conversion/<model_family>_to_diffusers.py
|
||||
conversion_source_layout: <diffusers | separate_components | monolithic | mixed | custom>
|
||||
converted_weights_dir: converted_weights/<model_family>
|
||||
strict_load_status: <not_run | pass | pass_with_documented_exclusions | blocked>
|
||||
```
|
||||
|
||||
For monolithic official checkpoints, record the component prefix split here. For
|
||||
example, a single checkpoint may contain transformer, VAE/pretransform,
|
||||
conditioner, and scheduler/vocoder keys that the conversion script writes into
|
||||
separate FastVideo component subfolders.
|
||||
|
||||
## Expected Parity Tests
|
||||
|
||||
Planned local tests for this family:
|
||||
|
||||
| Component | Official files / args | Test | Concerns | Status |
|
||||
|---|---|---|---|---|
|
||||
| `<component>` | `<definition path; instantiation path + args>` | `tests/local_tests/<bucket>/test_<model_family>_<component>_parity.py` | `<prototype or setup concerns>` | `<planned/scaffold_skip/debug_red/non_skip_pass/blocked>` |
|
||||
| `pipeline` | `<official pipeline call>` | `tests/local_tests/pipelines/test_<model_family>_pipeline_parity.py` | `<pipeline concerns>` | `<planned/scaffold_skip/debug_red/non_skip_pass/blocked>` |
|
||||
|
||||
Include reused components in this table. Reuse is accepted only after the
|
||||
FastVideo component definition and official instantiation arguments have both
|
||||
been checked and the component parity test passes non-skip.
|
||||
|
||||
Run the relevant tests with:
|
||||
|
||||
```bash
|
||||
pytest tests/local_tests/<bucket>/test_<model_family>_<component>_parity.py -v -s
|
||||
pytest tests/local_tests/pipelines/test_<model_family>_pipeline_parity.py -v -s
|
||||
```
|
||||
|
||||
## Review Notes
|
||||
|
||||
- Required before handoff: non-skip PASS for each required component parity
|
||||
test, including reused components that own weights or numerical behavior.
|
||||
- Pipeline parity may start as a scaffold, but final handoff requires non-skip
|
||||
PASS or an explicit blocker accepted through the escape-hatch process.
|
||||
- User decisions and pause points are tracked as `E###` rows in
|
||||
`PORT_STATUS.md`; do not rely on chat history for escape-hatch context.
|
||||
- Review agents should verify this README's setup commands still match the PR,
|
||||
then run the listed parity tests or report the exact blocker.
|
||||
@@ -0,0 +1,69 @@
|
||||
# <Model Family> Port Status
|
||||
|
||||
## Summary
|
||||
|
||||
- model_family: `<model_family>`
|
||||
- workload_types: `<T2V/I2V/V2V/T2I/or compatibility shim with rationale>`
|
||||
- official_ref: `<url or import path>`
|
||||
- official_ref_dir: `<ReferenceDir or none>`
|
||||
- hf_weights_path: `<HF repo id/url or local path>`
|
||||
- local_weights_dir: `<official_weights/model_family or local path>`
|
||||
- source_layout: `<diffusers/raw_official/monolithic/separate_components/mixed/custom/unknown>`
|
||||
- local_tests_readme: `tests/local_tests/<model_family>/README.md`
|
||||
|
||||
## Current Phase
|
||||
|
||||
- phase: `prep`
|
||||
- status: `in_progress`
|
||||
- owner: `prep`
|
||||
- last_updated: `<YYYY-MM-DD>`
|
||||
|
||||
## Component Matrix
|
||||
|
||||
| Component | Type | Reuse/Port | Official Definition | Official Instantiation | FastVideo Target | Prototype | Conversion | Parity | Open Issues |
|
||||
|---|---|---|---|---|---|---|---|---|---|
|
||||
| `<component>` | `<dit/vae/encoder/generic>` | `<unknown/reuse/port>` | `<path + symbols>` | `<path + args>` | `<target files>` | `<not_started/in_progress/pass/blocked>` | `<not_started/pass/blocked>` | `<not_started/scaffold_skip/debug_red/non_skip_pass/blocked>` | `<none or IDs>` |
|
||||
|
||||
## Conversion State
|
||||
|
||||
- conversion_script: `scripts/checkpoint_conversion/<model_family>_to_diffusers.py`
|
||||
- converted_weights_dir: `converted_weights/<model_family>`
|
||||
- source_layout: `<diffusers/separate_components/monolithic/mixed/custom/unknown>`
|
||||
- strict_load_status: `not_run`
|
||||
- passthrough_components: `<none or list>`
|
||||
- retry_history: `<none>`
|
||||
|
||||
## Parity Commands
|
||||
|
||||
| Scope | Command | Last Result | Notes |
|
||||
|---|---|---|---|
|
||||
| component | `pytest tests/local_tests/<bucket>/test_<model_family>_<component>_parity.py -v -s` | `not_run` | `<notes>` |
|
||||
| pipeline | `pytest tests/local_tests/pipelines/test_<model_family>_pipeline_parity.py -v -s` | `not_run` | `<notes>` |
|
||||
|
||||
## Open Questions
|
||||
|
||||
| ID | Question | Owner | Needed By Phase | Status | Resolution |
|
||||
|---|---|---|---|---|---|
|
||||
| Q001 | `<question>` | `<owner>` | `<phase>` | `<open/resolved>` | `<resolution or blank>` |
|
||||
|
||||
## Issues And Blockers
|
||||
|
||||
| ID | Phase | Component | Severity | Issue | Evidence | Owner | Status | Resolution |
|
||||
|---|---|---|---|---|---|---|---|---|
|
||||
| I001 | `<phase>` | `<component or all>` | `<low/medium/high/blocker>` | `<issue>` | `<logs/paths/commands>` | `<owner>` | `<open/resolved>` | `<resolution or blank>` |
|
||||
|
||||
## Escape Hatches
|
||||
|
||||
| ID | Phase | Decision Type | Question | Recommended Option | Status | Resolution |
|
||||
|---|---|---|---|---|---|---|
|
||||
| E001 | `<phase>` | `<scope/dependency/auth/cost/destructive/ambiguity/blocker>` | `<one precise question>` | `<safe recommended option>` | `<open/resolved>` | `<resolution or blank>` |
|
||||
|
||||
## Decisions
|
||||
|
||||
| Date | Decision | Rationale | Impact |
|
||||
|---|---|---|---|
|
||||
| `<YYYY-MM-DD>` | `<decision>` | `<why>` | `<affected components/phases>` |
|
||||
|
||||
## Handoff Notes
|
||||
|
||||
- `<short notes for the next agent>`
|
||||
@@ -0,0 +1,228 @@
|
||||
---
|
||||
name: add-model-02-parity
|
||||
description: Use during /add-model after reference/architecture study to scaffold and later activate local FastVideo component parity tests. Emphasizes early test creation, official-reference loading, standardized FastVideo loading, and non-skip handoff gates.
|
||||
---
|
||||
|
||||
# Add Model Parity
|
||||
|
||||
## Goal
|
||||
|
||||
Create parity tests as early as possible in a FastVideo port. The first pass can
|
||||
land before conversion or component implementation as an executable scaffold;
|
||||
handoff is blocked until the same tests become non-skip PASS with real weights.
|
||||
|
||||
## When To Run
|
||||
|
||||
Follow `../add-model/shared/common_rules.md` for token/auth safety, state files,
|
||||
escape hatches, and skip/pass semantics.
|
||||
|
||||
Run immediately after `/add-model` Phase 1 has identified:
|
||||
|
||||
- official component classes and call signatures;
|
||||
- FastVideo target component buckets/classes/configs;
|
||||
- local reference clone or import path from `add-model-01-prep`;
|
||||
- local raw or Diffusers weight path;
|
||||
- `official_env_status=imports_ok`, or private deps that will be stubbed
|
||||
locally in tests;
|
||||
- `local_tests_readme` documenting setup and planned review/test commands;
|
||||
- expected component inputs and output tensors.
|
||||
|
||||
Do not wait for all FastVideo components to be implemented. Write the tests
|
||||
first, then let component-porting subagents make them pass.
|
||||
|
||||
## Outputs
|
||||
|
||||
- One component parity test per required component, including reused components:
|
||||
`tests/local_tests/<bucket>/test_<family>_<component>_parity.py`.
|
||||
- Optional helper for upstream private deps:
|
||||
`tests/local_tests/helpers/<family>_upstream.py`.
|
||||
- Pipeline parity is owned later by `../add-model-09-pipeline/SKILL.md` after all
|
||||
component parity tests pass non-skip.
|
||||
- A parity status block for the `/add-model` parity verification phase.
|
||||
|
||||
## Early Scaffold Rules
|
||||
|
||||
- A scaffold may skip while the FastVideo class, converted weights, or official
|
||||
import is missing.
|
||||
- A scaffold must already encode the real official load path, FastVideo load
|
||||
path, deterministic inputs, expected output extraction, and tolerance target.
|
||||
- Each parity test must declare its coverage scope in the file docstring or a
|
||||
module constant: `production_loader`, `implementation_subcomponent`, or `both`.
|
||||
Implementation/subcomponent parity may bypass production loaders deliberately,
|
||||
but final handoff still needs production-loader coverage somewhere before the
|
||||
pipeline depends on that component.
|
||||
- Official reference imports must run in the current FastVideo environment; do
|
||||
not create or assume a separate upstream venv/conda env.
|
||||
- A scaffold is not evidence of correctness. It becomes evidence only after a
|
||||
local non-skip PASS.
|
||||
- Prefer env-var path overrides with repo-relative defaults.
|
||||
- Keep tests local-only under `tests/local_tests/`; package/CI quality tests are
|
||||
added later.
|
||||
- Update shared state files as described in
|
||||
`../add-model/shared/common_rules.md` whenever adding or activating parity
|
||||
tests.
|
||||
|
||||
## Component Template
|
||||
|
||||
Copy `templates/component_parity_test.py` and fill every `TODO` marker. The
|
||||
template is distilled from:
|
||||
|
||||
- `tests/local_tests/transformers/test_ltx2.py`
|
||||
- `tests/local_tests/transformers/test_gamecraft_parity.py`
|
||||
- `tests/local_tests/encoders/test_ltx2_gemma_parity.py`
|
||||
- `tests/local_tests/vaes/test_oobleck_vae_parity.py`
|
||||
- `tests/local_tests/sd35/test_sd35_component_parity.py`
|
||||
|
||||
The template supports three states:
|
||||
|
||||
| State | Meaning |
|
||||
|---|---|
|
||||
| Scaffold skip | Test is committed early, but official import, FastVideo class, or weights are not available yet. |
|
||||
| Debug red | Both sides load and the test fails numerically. This is useful: porting can chase the first drift. |
|
||||
| Non-skip pass | Required before `/add-model` handoff. |
|
||||
|
||||
## Subagent Dispatch Pattern
|
||||
|
||||
After Phase 1, dispatch one parity subagent per component before or alongside
|
||||
component implementation:
|
||||
|
||||
```text
|
||||
Create a local parity test scaffold for <family> <component>.
|
||||
|
||||
Use the prep handoff:
|
||||
- official_ref_dir/import: <...>
|
||||
- local_weights_dir: <...>
|
||||
- source_layout: <...>
|
||||
- needs_conversion: <yes/no>
|
||||
- official_env_status: <imports_ok | private_deps_need_stubs>
|
||||
- local_tests_readme: tests/local_tests/<model_family>/README.md
|
||||
- port_state_file: tests/local_tests/<model_family>/PORT_STATUS.md
|
||||
- official_definition_files: <paths + classes/functions>
|
||||
- official_instantiation_files: <paths + factory/pipeline/config call sites + args>
|
||||
- concerns_or_unknowns: <known ambiguous inputs, outputs, deps, or args>
|
||||
|
||||
The complete per-component packet must match
|
||||
`../add-model/contracts/component_context.md`.
|
||||
|
||||
Read the official component call path and the planned FastVideo component API.
|
||||
Add tests/local_tests/<bucket>/test_<family>_<component>_parity.py based on
|
||||
add-model-02-parity/templates/component_parity_test.py.
|
||||
|
||||
The scaffold must load the official model with real weights when available,
|
||||
load the FastVideo model through the standardized config/class/loader path when
|
||||
available, create deterministic inputs, compare concrete outputs, and skip only
|
||||
when a dependency is genuinely missing. Do not make an unconditional skip or a
|
||||
shape-only test.
|
||||
```
|
||||
|
||||
## FastVideo Load Patterns
|
||||
|
||||
Pick the narrowest load path that matches the component:
|
||||
|
||||
| Component | Preferred FastVideo load path |
|
||||
|---|---|
|
||||
| DiT / transformer | Bucket config + model class, or `TransformerLoader` when testing converted Diffusers component dirs. |
|
||||
| VAE | VAE class `from_pretrained(...)` when implemented, or bucket config + class for local converted dirs. |
|
||||
| Text/image encoder | Bucket config + model class; pass HF subpaths from `local_weights_dir` or converted component dirs. |
|
||||
| Scheduler/conditioner | Native class/config plus exact official kwargs. |
|
||||
|
||||
For early scaffolds, an import of the planned FastVideo class may be inside a
|
||||
helper that calls `pytest.skip` if the class does not exist yet. Replace that
|
||||
skip with a real import once the component PR adds the class.
|
||||
|
||||
Direct class/config construction is allowed for implementation or subcomponent
|
||||
parity, such as connector-only encoder checks or official monolithic-checkpoint
|
||||
mapping tests. Label that scope explicitly and add separate production-loader
|
||||
coverage when converted component dirs are available.
|
||||
|
||||
## Official Load Patterns
|
||||
|
||||
- Clone/reference repo path: add its source dir to `sys.path` before imports.
|
||||
- HF/Diffusers reference: import only inside the test, not production code.
|
||||
- Private deps: add a helper under `tests/local_tests/helpers/` to install
|
||||
stubs before importing upstream modules; do not rely on an external upstream
|
||||
environment.
|
||||
- Gated HF repos: resolve `HF_TOKEN`, `HUGGINGFACE_HUB_TOKEN`, or `HF_API_KEY`
|
||||
under the token rules in `../add-model/shared/common_rules.md`.
|
||||
|
||||
## Non-Skip Activation Checklist
|
||||
|
||||
Before `/add-model` handoff, each scaffolded test must be activated:
|
||||
|
||||
```text
|
||||
[ ] Official side imports and loads real weights.
|
||||
[ ] FastVideo side imports and loads the converted or original weights.
|
||||
[ ] Test executes at least one real forward call on both sides.
|
||||
[ ] Test compares output tensors, not only shapes or state-dict keys.
|
||||
[ ] Local pytest output contains PASSED, not SKIPPED or XFAIL.
|
||||
[ ] Tolerance is justified for the component scope and kernel alignment.
|
||||
```
|
||||
|
||||
## Component Parity Details
|
||||
|
||||
Reference imports:
|
||||
|
||||
- Import from `official_ref_dir` or the recorded package/import path.
|
||||
- If upstream has private deps, add a helper under
|
||||
`tests/local_tests/helpers/<family>_upstream.py` that installs minimal stubs
|
||||
before importing upstream modules.
|
||||
- Common stubs: identity compile/op-registration decorators, CP world size set to
|
||||
1, identity scatter/gather, and test-friendly custom-op kernels.
|
||||
- Stub decorators that register `torch.ops.<ns>.<op>` must preserve the
|
||||
`torch.library` registration side-effect. Identity decorators alone are not
|
||||
enough.
|
||||
- Delete stub helpers and every `install_stubs()` call as soon as the real deps
|
||||
become required installs. No-op shims are dead code.
|
||||
|
||||
Kernel and wrapper pitfalls:
|
||||
|
||||
- If parity routes flash-attn GQA through SDPA, expand KV heads manually on the
|
||||
SDPA side with `repeat_interleave` along the head axis.
|
||||
- If upstream VAE `decode()` denormalizes internally but FastVideo/Diffusers
|
||||
expects pre-denormalized latents, apply `z = z * std + mean` only on the
|
||||
FastVideo side in the parity test.
|
||||
- Per-channel VAE `latents_mean` / `latents_std` must be reshaped explicitly,
|
||||
e.g. `.view(1, z_dim, 1, 1, 1)` for 5D video latents.
|
||||
|
||||
Tolerance guide:
|
||||
|
||||
| Scope | Start `atol` / `rtol` | Notes |
|
||||
|---|---|---|
|
||||
| Single block, same kernel | `1e-4` / `1e-4` | Tight default. |
|
||||
| Full DiT, aligned kernels | `1e-2` / `1e-2` | Cross-layer accumulation. |
|
||||
| Full DiT, cross-kernel bf16 | `0.1` / `0.1` | Also require abs-mean drift below 5% and per-modality diagnostics. |
|
||||
| VAE decode fp32 | `5e-2` / `5e-2` | After normalization alignment. |
|
||||
| Encoder wrapper around same HF class | `1e-3` / `1e-3` | Should be near-zero. |
|
||||
|
||||
Element-wise `assert_close` alone is not enough for deep full-DiT parity. Also
|
||||
log global abs-mean drift and per-modality summaries.
|
||||
|
||||
Useful local commands:
|
||||
|
||||
```bash
|
||||
pytest tests/local_tests/<bucket>/test_<family>_*parity*.py -v -s
|
||||
pytest tests/local_tests -k "<family> and parity" -v -s
|
||||
```
|
||||
|
||||
## Escape Hatches
|
||||
|
||||
Follow `../add-model/shared/common_rules.md`. Parity-specific ask cases include
|
||||
private dependency approval, choosing between incompatible official references,
|
||||
accepting a shape-only substitute, or loosening required tolerances.
|
||||
|
||||
## Pipeline Parity
|
||||
|
||||
Pipeline parity is later than component parity because it needs stages, presets,
|
||||
registry wiring, converted weights, and green component parity. Record official
|
||||
pipeline call notes in `local_tests_readme`, but do not treat pipeline parity as
|
||||
owned by this skill.
|
||||
|
||||
Use `../add-model-09-pipeline/SKILL.md` and its
|
||||
`templates/pipeline_parity_test.py` for pipeline parity scaffolding and
|
||||
debugging. Compare denoised latents or decoded media, not just successful
|
||||
generation.
|
||||
|
||||
## Handoff Status Block
|
||||
|
||||
Return `../add-model/contracts/parity_status.md` to `/add-model` and update the
|
||||
shared state files before handoff.
|
||||
@@ -0,0 +1,201 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
"""Component parity scaffold for <FAMILY> <COMPONENT>.
|
||||
|
||||
This file is intended to be created early in a port. It may skip until the
|
||||
official reference, FastVideo class, and real weights are available, but it must
|
||||
never become an unconditional skip or shape-only test.
|
||||
|
||||
Fill every TODO before considering this test active.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import importlib
|
||||
import os
|
||||
from pathlib import Path
|
||||
import sys
|
||||
|
||||
import pytest
|
||||
import torch
|
||||
from torch.testing import assert_close
|
||||
|
||||
|
||||
os.environ.setdefault("MASTER_ADDR", "localhost")
|
||||
os.environ.setdefault("MASTER_PORT", "29519")
|
||||
os.environ.setdefault("DISABLE_SP", "1")
|
||||
os.environ.setdefault("FASTVIDEO_ATTENTION_BACKEND", "TORCH_SDPA")
|
||||
|
||||
REPO_ROOT = Path(__file__).resolve().parents[3]
|
||||
FAMILY = "<family>" # TODO: snake_case family name.
|
||||
COMPONENT = "<component>" # TODO: transformer | vae | encoder | conditioner | ...
|
||||
PARITY_SCOPE = "implementation_subcomponent" # TODO: production_loader | implementation_subcomponent | both
|
||||
OFFICIAL_MODULE = "<official.module>" # TODO: e.g. "ltx_core.model.transformer".
|
||||
OFFICIAL_CLASS = "<OfficialClass>" # TODO: official class/factory name.
|
||||
FASTVIDEO_CONFIG_MODULE = "fastvideo.configs.models.<bucket>" # TODO.
|
||||
FASTVIDEO_CONFIG_CLASS = "<FastVideoConfig>" # TODO.
|
||||
FASTVIDEO_MODEL_MODULE = "fastvideo.models.<bucket>.<module>" # TODO.
|
||||
FASTVIDEO_MODEL_CLASS = "<FastVideoModel>" # TODO.
|
||||
|
||||
OFFICIAL_REF_DIR = Path(
|
||||
os.getenv("<FAMILY_UPPER>_OFFICIAL_REF_DIR", REPO_ROOT / "<ReferenceDir>")
|
||||
)
|
||||
LOCAL_WEIGHTS_DIR = Path(
|
||||
os.getenv("<FAMILY_UPPER>_LOCAL_WEIGHTS_DIR", REPO_ROOT / "official_weights" / FAMILY)
|
||||
)
|
||||
CONVERTED_WEIGHTS_DIR = Path(
|
||||
os.getenv("<FAMILY_UPPER>_CONVERTED_WEIGHTS_DIR", REPO_ROOT / "converted_weights" / FAMILY)
|
||||
)
|
||||
|
||||
|
||||
def _resolve_hf_token() -> str | None:
|
||||
for key in ("HF_TOKEN", "HUGGINGFACE_HUB_TOKEN", "HF_API_KEY"):
|
||||
value = os.environ.get(key)
|
||||
if value:
|
||||
return value
|
||||
return None
|
||||
|
||||
|
||||
def _add_official_to_path() -> None:
|
||||
"""Add the official source path before importing upstream modules."""
|
||||
# TODO: adjust for the official repo layout. Common examples:
|
||||
# OFFICIAL_REF_DIR / "src"
|
||||
# OFFICIAL_REF_DIR / "packages" / "<pkg>" / "src"
|
||||
# OFFICIAL_REF_DIR
|
||||
official_src = OFFICIAL_REF_DIR / "src"
|
||||
if not official_src.exists():
|
||||
official_src = OFFICIAL_REF_DIR
|
||||
if official_src.exists() and str(official_src) not in sys.path:
|
||||
sys.path.insert(0, str(official_src))
|
||||
|
||||
|
||||
def _import_or_skip(module_name: str, attr_name: str | None = None):
|
||||
if "<" in module_name or (attr_name is not None and "<" in attr_name):
|
||||
pytest.skip(f"Template import placeholder not filled: {module_name}.{attr_name}")
|
||||
try:
|
||||
module = importlib.import_module(module_name)
|
||||
except Exception as exc: # noqa: BLE001 - local parity should skip missing refs.
|
||||
pytest.skip(f"Cannot import {module_name}: {exc}")
|
||||
if attr_name is None:
|
||||
return module
|
||||
try:
|
||||
return getattr(module, attr_name)
|
||||
except AttributeError:
|
||||
pytest.skip(f"{module_name} has no attribute {attr_name}")
|
||||
|
||||
|
||||
def _load_official_model(device: torch.device, dtype: torch.dtype) -> torch.nn.Module:
|
||||
"""Load the official component with real weights."""
|
||||
_add_official_to_path()
|
||||
if not OFFICIAL_REF_DIR.exists():
|
||||
pytest.skip(f"Official reference missing: {OFFICIAL_REF_DIR}")
|
||||
if not LOCAL_WEIGHTS_DIR.exists():
|
||||
pytest.skip(f"Local weights missing: {LOCAL_WEIGHTS_DIR}")
|
||||
|
||||
# TODO: import official class/factory and load real weights strictly.
|
||||
# Examples in-tree:
|
||||
# - LTX2: SingleGPUModelBuilder(...).build(device=device, dtype=dtype)
|
||||
# - GameCraft: torch.load(...)["module"] -> official_model.load_state_dict(...)
|
||||
# - Oobleck: create_model_from_config(config) + ckpt state_dict
|
||||
OfficialClass = _import_or_skip(OFFICIAL_MODULE, OFFICIAL_CLASS)
|
||||
model = OfficialClass() # TODO: pass official config kwargs.
|
||||
state_dict = {} # TODO: load official state dict from LOCAL_WEIGHTS_DIR.
|
||||
missing, unexpected = model.load_state_dict(state_dict, strict=True)
|
||||
assert not missing and not unexpected, (
|
||||
f"official load mismatch missing={missing[:5]} unexpected={unexpected[:5]}"
|
||||
)
|
||||
return model.to(device=device, dtype=dtype).eval()
|
||||
|
||||
|
||||
def _load_fastvideo_model(device: torch.device, dtype: torch.dtype) -> torch.nn.Module:
|
||||
"""Load the FastVideo component with the same tensor content."""
|
||||
if not CONVERTED_WEIGHTS_DIR.exists() and not LOCAL_WEIGHTS_DIR.exists():
|
||||
pytest.skip(
|
||||
f"No FastVideo loadable weights: {CONVERTED_WEIGHTS_DIR} or {LOCAL_WEIGHTS_DIR}"
|
||||
)
|
||||
|
||||
# TODO: replace with the bucket-specific FastVideo config/class/loader.
|
||||
# DiT examples:
|
||||
# from fastvideo.configs.models.dits import <Config>
|
||||
# from fastvideo.models.dits.<module> import <Model>
|
||||
# VAE examples:
|
||||
# from fastvideo.models.vaes.<module> import <VAE>
|
||||
# model = <VAE>.from_pretrained(...)
|
||||
FastVideoConfig = _import_or_skip(FASTVIDEO_CONFIG_MODULE, FASTVIDEO_CONFIG_CLASS)
|
||||
FastVideoModel = _import_or_skip(FASTVIDEO_MODEL_MODULE, FASTVIDEO_MODEL_CLASS)
|
||||
|
||||
config = FastVideoConfig()
|
||||
model = FastVideoModel(config=config)
|
||||
state_dict = {} # TODO: load converted or directly mapped state dict.
|
||||
missing, unexpected = model.load_state_dict(state_dict, strict=True)
|
||||
assert not missing and not unexpected, (
|
||||
f"FastVideo load mismatch missing={missing[:5]} unexpected={unexpected[:5]}"
|
||||
)
|
||||
return model.to(device=device, dtype=dtype).eval()
|
||||
|
||||
|
||||
def _make_inputs(device: torch.device, dtype: torch.dtype) -> dict[str, torch.Tensor]:
|
||||
"""Create deterministic inputs matching the official component call."""
|
||||
torch.manual_seed(0)
|
||||
# TODO: replace with component-specific tensors and metadata.
|
||||
return {
|
||||
"hidden_states": torch.randn(1, 4, 16, device=device, dtype=dtype),
|
||||
"timestep": torch.tensor([10], device=device),
|
||||
}
|
||||
|
||||
|
||||
def _run_official(model: torch.nn.Module, inputs: dict[str, torch.Tensor]) -> torch.Tensor:
|
||||
"""Run official component and return the tensor to compare."""
|
||||
with torch.inference_mode():
|
||||
output = model(**inputs) # TODO: adapt official call signature.
|
||||
if isinstance(output, dict):
|
||||
sample = output.get("sample")
|
||||
output = sample if sample is not None else output.get("x")
|
||||
elif hasattr(output, "sample"):
|
||||
output = output.sample
|
||||
elif isinstance(output, tuple):
|
||||
output = output[0]
|
||||
assert torch.is_tensor(output), f"official output is not tensor: {type(output)}"
|
||||
return output.detach().float().cpu()
|
||||
|
||||
|
||||
def _run_fastvideo(model: torch.nn.Module, inputs: dict[str, torch.Tensor]) -> torch.Tensor:
|
||||
"""Run FastVideo component and return the tensor to compare."""
|
||||
with torch.inference_mode():
|
||||
output = model(**inputs) # TODO: adapt FastVideo call signature.
|
||||
if isinstance(output, dict):
|
||||
sample = output.get("sample")
|
||||
output = sample if sample is not None else output.get("x")
|
||||
elif hasattr(output, "sample"):
|
||||
output = output.sample
|
||||
elif isinstance(output, tuple):
|
||||
output = output[0]
|
||||
assert torch.is_tensor(output), f"FastVideo output is not tensor: {type(output)}"
|
||||
return output.detach().float().cpu()
|
||||
|
||||
|
||||
@pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA required for this parity test.")
|
||||
def test_component_parity():
|
||||
"""Compare official and FastVideo outputs on identical inputs."""
|
||||
device = torch.device("cuda:0")
|
||||
dtype = torch.bfloat16
|
||||
|
||||
official = _load_official_model(device, dtype)
|
||||
fastvideo = _load_fastvideo_model(device, dtype)
|
||||
inputs = _make_inputs(device, dtype)
|
||||
|
||||
official_out = _run_official(official, inputs)
|
||||
fastvideo_out = _run_fastvideo(fastvideo, inputs)
|
||||
|
||||
assert official_out.shape == fastvideo_out.shape
|
||||
diff = (official_out - fastvideo_out).abs()
|
||||
print(
|
||||
f"official abs_mean={official_out.abs().mean().item():.6f} "
|
||||
f"fastvideo abs_mean={fastvideo_out.abs().mean().item():.6f} "
|
||||
f"diff_max={diff.max().item():.6f} diff_mean={diff.mean().item():.6f}"
|
||||
)
|
||||
|
||||
# TODO: pick tolerance by scope:
|
||||
# - single block / same kernel: 1e-4
|
||||
# - full DiT aligned kernels: 1e-2
|
||||
# - full DiT cross-kernel bf16: 1e-1 + abs_mean drift check
|
||||
# - VAE decode fp32: 5e-2 after normalization alignment
|
||||
assert_close(fastvideo_out, official_out, atol=1e-4, rtol=1e-4)
|
||||
@@ -0,0 +1,110 @@
|
||||
---
|
||||
name: add-model-03-port-dit
|
||||
description: Use during /add-model Phase 4 or Phase 6 to prototype or parity-debug one FastVideo-native DiT/transformer component.
|
||||
---
|
||||
|
||||
# Add Model Port DiT
|
||||
|
||||
## Goal
|
||||
|
||||
Prototype or parity-debug one diffusion transformer in FastVideo-native code.
|
||||
This skill is for one component only; do not work on the VAE, encoders,
|
||||
pipeline, or unrelated conversion code unless the current component cannot load
|
||||
without a minimal fix there.
|
||||
|
||||
## Inputs
|
||||
|
||||
Follow `../add-model/shared/component_skill_common.md` and require the complete
|
||||
packet from `../add-model/contracts/component_context.md`.
|
||||
|
||||
DiT-specific packet fields:
|
||||
|
||||
- `component`: transformer or DiT name.
|
||||
- `parity_test`: `tests/local_tests/<bucket>/test_<family>_<component>_parity.py`.
|
||||
- `weights`: converted transformer dir or local official path.
|
||||
- `target_files`: `fastvideo/models/dits/<family>.py` and
|
||||
`fastvideo/configs/models/dits/<family>.py`.
|
||||
|
||||
## Modes
|
||||
|
||||
Use the common prototype and parity-debug modes from
|
||||
`../add-model/shared/component_skill_common.md`.
|
||||
|
||||
DiT-specific prototype concerns include ambiguous official flags, shape
|
||||
mismatches, missing FastVideo layer equivalents, and dedicated output heads.
|
||||
|
||||
## Reuse Proof
|
||||
|
||||
Apply the shared reuse proof. DiT-specific comparison must include attention
|
||||
algorithm, positional embeddings, RoPE/patching, timestep/guidance embeddings,
|
||||
scaling constants, dtype casts, state-dict names, and every output head.
|
||||
|
||||
## Existing FastVideo Patterns
|
||||
|
||||
- Base class: `fastvideo/models/dits/base.py::BaseDiT`.
|
||||
- Config bases: `DiTConfig` and `DiTArchConfig` in
|
||||
`fastvideo/configs/models/dits/base.py`.
|
||||
- Use the matching DiT config bucket. Wrong bucket inheritance can typecheck but
|
||||
fail during pipeline wiring.
|
||||
- Config export: add the config to
|
||||
`fastvideo/configs/models/dits/__init__.py`.
|
||||
- Registry discovery: set `EntryClass = <ClassName>` in the model file.
|
||||
- Loader path: `TransformerLoader` reads `transformer/config.json`, calls
|
||||
`dit_config.update_model_arch(config)`, resolves `_class_name` through
|
||||
`ModelRegistry`, and constructs the class with `config` and `hf_config`.
|
||||
- Reference examples: `stable_audio.py`, `wanvideo.py`, `sd3.py`, `longcat.py`,
|
||||
and `ltx2.py`.
|
||||
- Layer guidance: `fastvideo/layers/AGENTS.md`.
|
||||
|
||||
## Implementation Rules
|
||||
|
||||
- Use FastVideo-native layers by default: `ReplicatedLinear` for DiT hot-path
|
||||
linears, `DistributedAttention` for standard full-sequence attention, and
|
||||
`LocalAttention` for local/window attention or simple single-GPU parity paths.
|
||||
- Raw SDPA is acceptable for cross-modality flat streams when no FastVideo
|
||||
distributed equivalent exists; document the SP gap in the module docstring.
|
||||
- Mirror official tensor contracts exactly: latent packing, patch ordering,
|
||||
timestep embedding scale, RoPE/positional embedding, guidance embedding,
|
||||
cross-attention context order, output head order, and dtype casts.
|
||||
- Preserve all output heads that the official DiT emits. Do not silently drop
|
||||
audio, depth, pose, mask, or auxiliary heads.
|
||||
- Put architecture fields on `DiTArchConfig`; keep inference steps, CFG scales,
|
||||
FPS, flow shift, and sampling defaults out of the arch config.
|
||||
- Define `_fsdp_shard_conditions`, `_compile_conditions`,
|
||||
`param_names_mapping`, and `reverse_param_names_mapping` where needed.
|
||||
- Follow the production import boundary in
|
||||
`../add-model/shared/common_rules.md`.
|
||||
|
||||
## Prototype Checks
|
||||
|
||||
Follow the shared prototype success criteria. A useful one-off check is:
|
||||
|
||||
```bash
|
||||
python - <<'PY'
|
||||
# Import the target config/class, instantiate with random weights, and print
|
||||
# state_dict names/shapes for the conversion mapping.
|
||||
PY
|
||||
```
|
||||
|
||||
## Parity-Debug Loop
|
||||
|
||||
Run the shared parity-debug loop. The component test command is:
|
||||
|
||||
```bash
|
||||
pytest <parity_test> -v -s
|
||||
```
|
||||
|
||||
For numerical drift, narrow the first divergent block with per-block hooks or
|
||||
intermediate tensor comparisons before changing layers.
|
||||
|
||||
## Escape Hatches
|
||||
|
||||
Follow `../add-model/shared/common_rules.md` and the component-specific guidance
|
||||
in `../add-model/shared/component_skill_common.md`. DiT-specific ask cases include
|
||||
dropping an output head/modality, accepting an unsupported kernel/private op, or
|
||||
choosing between incompatible official transformer definitions.
|
||||
|
||||
## Handoff
|
||||
|
||||
Return `../add-model/contracts/component_skill_handoff.md` following the common
|
||||
handoff rules in `../add-model/shared/component_skill_common.md`.
|
||||
@@ -0,0 +1,100 @@
|
||||
---
|
||||
name: add-model-04-port-vae
|
||||
description: Use during /add-model Phase 4 or Phase 6 to prototype or parity-debug one FastVideo-native VAE component.
|
||||
---
|
||||
|
||||
# Add Model Port VAE
|
||||
|
||||
## Goal
|
||||
|
||||
Prototype or parity-debug one VAE or autoencoder in FastVideo-native code. This
|
||||
skill covers video, image, and audio VAEs.
|
||||
|
||||
## Inputs
|
||||
|
||||
Follow `../add-model/shared/component_skill_common.md` and require the complete
|
||||
packet from `../add-model/contracts/component_context.md`.
|
||||
|
||||
VAE-specific packet fields:
|
||||
|
||||
- `component`: VAE or autoencoder name.
|
||||
- `parity_test`: `tests/local_tests/vaes/test_<family>_<component>_parity.py`.
|
||||
- `weights`: converted VAE dir, HF subfolder, or local official path.
|
||||
- `target_files`: `fastvideo/models/vaes/<arch_or_family>.py` and
|
||||
`fastvideo/configs/models/vaes/<arch_or_family>.py`.
|
||||
|
||||
## Modes
|
||||
|
||||
Use the common prototype and parity-debug modes from
|
||||
`../add-model/shared/component_skill_common.md`.
|
||||
|
||||
VAE-specific prototype concerns include latent normalization, stochastic
|
||||
posterior behavior, tiling incompatibility, temporal/spatial/audio layout, and
|
||||
decode output containers.
|
||||
|
||||
## Reuse Proof
|
||||
|
||||
Apply the shared reuse proof. VAE-specific comparison must include latent layout,
|
||||
temporal/spatial/audio compression, scaling factor, mean/std normalization,
|
||||
posterior behavior, encode/decode output objects, tiling flags, and cropping.
|
||||
|
||||
## Existing FastVideo Patterns
|
||||
|
||||
- Shared tiling wrapper: `fastvideo/models/vaes/common.py::ParallelTiledVAE`.
|
||||
- Config bases: `VAEConfig` and `VAEArchConfig` in
|
||||
`fastvideo/configs/models/vaes/base.py`.
|
||||
- Use the matching VAE config bucket. Wrong bucket inheritance can typecheck but
|
||||
fail during pipeline wiring.
|
||||
- Config export: add the config to
|
||||
`fastvideo/configs/models/vaes/__init__.py`.
|
||||
- Registry discovery: set `EntryClass = <ClassName>` in the model file.
|
||||
- Loader path: VAE loaders resolve `_class_name` through `ModelRegistry` and
|
||||
load converted component weights from the VAE subdir.
|
||||
- Reference examples: `oobleck.py`, `autoencoder_kl.py`, `wanvae.py`,
|
||||
`ltx2vae.py`, and `gamecraftvae.py`.
|
||||
- Layer guidance: `fastvideo/layers/AGENTS.md`.
|
||||
|
||||
## Implementation Rules
|
||||
|
||||
- Name reusable VAE architectures by architecture (`oobleck.py`,
|
||||
`autoencoder_kl.py`); name family-specific VAEs by family.
|
||||
- Match official encode/decode contracts exactly: input layout, latent layout,
|
||||
temporal/spatial/audio compression, scaling factor, mean/std normalization,
|
||||
posterior sampling behavior, decode output object, and frame/sample cropping.
|
||||
- Compare deterministic outputs in parity: decode outputs, encode mean/mode, or
|
||||
round-trip tensors. Do not compare stochastic samples unless the RNG path is
|
||||
explicitly controlled.
|
||||
- Use FastVideo tiling only when it preserves official numerics for the tested
|
||||
shape; disable it in config for audio or unsupported dimensions.
|
||||
- Put architecture constants on `VAEArchConfig`; put `load_encoder`,
|
||||
`load_decoder`, tiling, dtype, and pretrained path fields on `VAEConfig`.
|
||||
- Follow the production import boundary in
|
||||
`../add-model/shared/common_rules.md`.
|
||||
|
||||
## Prototype Checks
|
||||
|
||||
Follow the shared prototype success criteria.
|
||||
|
||||
## Parity-Debug Loop
|
||||
|
||||
Run the shared parity-debug loop. The component test command is:
|
||||
|
||||
```bash
|
||||
pytest <parity_test> -v -s
|
||||
```
|
||||
|
||||
For numerical drift, check normalization, latent scaling, posterior mode vs
|
||||
sample, channel order, and temporal/spatial/audio cropping before changing
|
||||
layers.
|
||||
|
||||
## Escape Hatches
|
||||
|
||||
Follow `../add-model/shared/common_rules.md` and the component-specific guidance
|
||||
in `../add-model/shared/component_skill_common.md`. VAE-specific ask cases include
|
||||
dropping an encode/decode path, accepting an unsupported private op, or choosing
|
||||
between incompatible official VAE definitions.
|
||||
|
||||
## Handoff
|
||||
|
||||
Return `../add-model/contracts/component_skill_handoff.md` following the common
|
||||
handoff rules in `../add-model/shared/component_skill_common.md`.
|
||||
@@ -0,0 +1,119 @@
|
||||
---
|
||||
name: add-model-05-port-encoder
|
||||
description: Use during /add-model Phase 4 or Phase 6 to prototype or parity-debug one FastVideo-native text, image, audio, or compound encoder component.
|
||||
---
|
||||
|
||||
# Add Model Port Encoder
|
||||
|
||||
## Goal
|
||||
|
||||
Prototype or parity-debug one encoder or encoder-like conditioner in
|
||||
FastVideo-native code. Use this for text encoders, image encoders, audio
|
||||
encoders, and compound conditioners that fit the encoder config/loader bucket.
|
||||
|
||||
## Inputs
|
||||
|
||||
Follow `../add-model/shared/component_skill_common.md` and require the complete
|
||||
packet from `../add-model/contracts/component_context.md`.
|
||||
|
||||
Encoder-specific packet fields:
|
||||
|
||||
- `component`: encoder or encoder-like conditioner name.
|
||||
- `parity_test`: `tests/local_tests/encoders/test_<family>_<component>_parity.py`.
|
||||
- `weights`: converted encoder dir, HF subfolder, or external HF id.
|
||||
- `target_files`: `fastvideo/models/encoders/<arch_or_family>.py` and
|
||||
`fastvideo/configs/models/encoders/<arch_or_family>.py`.
|
||||
|
||||
## Modes
|
||||
|
||||
Use the common prototype and parity-debug modes from
|
||||
`../add-model/shared/component_skill_common.md`.
|
||||
|
||||
Encoder-specific prototype concerns include tokenizer kwargs, hidden-state
|
||||
extraction, output packing, connector order, and external/passthrough weight
|
||||
needs.
|
||||
|
||||
## Reuse Proof
|
||||
|
||||
Apply the shared reuse proof. Encoder-specific comparison must include tokenizer
|
||||
contracts, hidden-state extraction, masks, positional IDs, output packing,
|
||||
connector/projection ordering, passthrough paths, and returned dataclass shape.
|
||||
|
||||
## Existing FastVideo Patterns
|
||||
|
||||
- Base classes: `TextEncoder` and `ImageEncoder` in
|
||||
`fastvideo/models/encoders/base.py`.
|
||||
- Output type: `BaseEncoderOutput`.
|
||||
- Config bases: `TextEncoderConfig`, `ImageEncoderConfig`,
|
||||
`TextEncoderArchConfig`, and `ImageEncoderArchConfig` in
|
||||
`fastvideo/configs/models/encoders/base.py`.
|
||||
- Use the matching encoder config bucket. Wrong bucket inheritance can typecheck
|
||||
but fail during pipeline wiring.
|
||||
- Config export: add the config to
|
||||
`fastvideo/configs/models/encoders/__init__.py`.
|
||||
- Registry discovery: set `EntryClass = <ClassName>` or a list of class names in
|
||||
the model file.
|
||||
- Reference examples: native `t5.py`, `clip.py`, `siglip.py`, `llama.py`,
|
||||
`qwen2_5.py`, `gemma.py`, and compound `stable_audio_conditioner.py`.
|
||||
- Layer guidance: `fastvideo/layers/AGENTS.md`.
|
||||
|
||||
## Implementation Rules
|
||||
|
||||
- Reuse tokenizers and pure data utilities when needed, but do not add runtime
|
||||
third-party model-class imports as a placeholder for a component that owns
|
||||
weights or numerical behavior.
|
||||
- For LLM-style encoders, follow existing tensor-parallel patterns such as
|
||||
`QKVParallelLinear`, `MergedColumnParallelLinear`, `RowParallelLinear`,
|
||||
`VocabParallelEmbedding`, and `RMSNorm` when matching native examples.
|
||||
- Match official hidden-state extraction exactly: layer index, pooled output,
|
||||
attention mask dtype, padding side, truncation, special tokens, final norm,
|
||||
output_hidden_states, and returned tuple/dataclass shape.
|
||||
- For connector or conditioner modules, preserve sub-conditioner order and the
|
||||
exact packing of cross-attention tokens, masks, and global conditioning.
|
||||
- Put tokenizer kwargs and architecture constants on the arch config when they
|
||||
affect numerical behavior.
|
||||
- If an external HF encoder is explicitly accepted as a lazy wrapper, keep it
|
||||
isolated, document why it is not a native port, and still require parity for
|
||||
the wrapper's output contract.
|
||||
|
||||
Hybrid external-HF encoder checklist:
|
||||
|
||||
- Put external model folders in passthrough subfolders such as
|
||||
`text_encoder/<external_name>/`, or record a root `model_index.json` path field
|
||||
that the loader resolves to a local directory.
|
||||
- Keep external model parameters out of the FastVideo-owned state-dict surface
|
||||
when the external model is loaded lazily from its own HF files.
|
||||
- Convert and strict-check only the FastVideo-owned connector/projection weights;
|
||||
document external model weights as passthrough.
|
||||
- Add parity for the wrapper's final output contract and, when useful, a narrower
|
||||
connector-only parity test that labels its scope as
|
||||
`implementation_subcomponent`.
|
||||
- Verify the production loader resolves the same external path used by the
|
||||
pipeline, not just the direct class used in the parity test.
|
||||
|
||||
## Prototype Checks
|
||||
|
||||
Follow the shared prototype success criteria.
|
||||
|
||||
## Parity-Debug Loop
|
||||
|
||||
Run the shared parity-debug loop. The component test command is:
|
||||
|
||||
```bash
|
||||
pytest <parity_test> -v -s
|
||||
```
|
||||
|
||||
For numerical drift, check tokenization, masks, hidden-state selection,
|
||||
positional IDs, dtype/autocast, and output packing before changing layers.
|
||||
|
||||
## Escape Hatches
|
||||
|
||||
Follow `../add-model/shared/common_rules.md` and the component-specific guidance
|
||||
in `../add-model/shared/component_skill_common.md`. Encoder-specific ask cases
|
||||
include accepting private model-code execution, choosing between incompatible
|
||||
tokenizer/encoder references, or dropping a required conditioning stream.
|
||||
|
||||
## Handoff
|
||||
|
||||
Return `../add-model/contracts/component_skill_handoff.md` following the common
|
||||
handoff rules in `../add-model/shared/component_skill_common.md`.
|
||||
@@ -0,0 +1,111 @@
|
||||
---
|
||||
name: add-model-06-port-generic
|
||||
description: Use during /add-model Phase 4 or Phase 6 to prototype or parity-debug one non-DiT, non-VAE, non-encoder FastVideo component.
|
||||
---
|
||||
|
||||
# Add Model Port Generic
|
||||
|
||||
## Goal
|
||||
|
||||
Prototype or parity-debug one scheduler, conditioner, upsampler, vocoder,
|
||||
adapter, preprocessor, or unknown component in FastVideo-native code.
|
||||
|
||||
## Inputs
|
||||
|
||||
Follow `../add-model/shared/component_skill_common.md` and require the complete
|
||||
packet from `../add-model/contracts/component_context.md`.
|
||||
|
||||
Generic-component packet fields:
|
||||
|
||||
- `component`: component name.
|
||||
- `component_type`: scheduler, conditioner, upsampler, vocoder, adapter,
|
||||
preprocessor, or unknown.
|
||||
- `parity_test`: `tests/local_tests/<bucket>/test_<family>_<component>_parity.py`.
|
||||
- `weights`: converted component dir, HF subfolder, or none.
|
||||
- `target_files`: matching `fastvideo/models/` and `fastvideo/configs/models/`
|
||||
bucket files when applicable.
|
||||
|
||||
## Modes
|
||||
|
||||
Use the common prototype and parity-debug modes from
|
||||
`../add-model/shared/component_skill_common.md`.
|
||||
|
||||
Generic-component prototype concerns include stateless/stateful ambiguity,
|
||||
missing loader buckets, source prefixes, mutable scheduler state, and output
|
||||
container shape.
|
||||
|
||||
## Reuse Proof
|
||||
|
||||
Apply the shared reuse proof. Generic-component comparison must include mutable
|
||||
state, scaling constants, scheduler/conditioner semantics, output containers, and
|
||||
whether the component owns state or is stateless.
|
||||
|
||||
## Existing FastVideo Patterns
|
||||
|
||||
- Schedulers live under `fastvideo/models/schedulers/` and expose `EntryClass`.
|
||||
- Upsamplers use `fastvideo/models/upsamplers/` plus configs under
|
||||
`fastvideo/configs/models/upsamplers/`; see `hunyuan15.py`.
|
||||
- Vocoders and audio-specific modules can live under `fastvideo/models/audio/`
|
||||
with configs under `fastvideo/configs/models/audio/`; see `ltx2_audio_vae.py`.
|
||||
- Compound conditioners may fit the encoder bucket when the pipeline loader uses
|
||||
`ConditionerLoader`; see `stable_audio_conditioner.py`.
|
||||
- Registry discovery uses `EntryClass`; config bucket exports are required when
|
||||
pipeline configs import them by bucket.
|
||||
- Use the narrowest matching config bucket. Wrong bucket inheritance can typecheck
|
||||
but fail during pipeline wiring.
|
||||
- Layer guidance: `fastvideo/layers/AGENTS.md`.
|
||||
|
||||
## Bucket Decision
|
||||
|
||||
- If the component is a transformer/DiT, stop and use `add-model-03-port-dit`.
|
||||
- If the component is a VAE/autoencoder, stop and use `add-model-04-port-vae`.
|
||||
- If the component is a text/image/audio encoder or encoder-like conditioner,
|
||||
stop and use `add-model-05-port-encoder` unless the loader requires a different
|
||||
bucket.
|
||||
- Otherwise choose the narrowest existing bucket. Add a new bucket only when no
|
||||
existing loader/config shape can represent the component without misleading
|
||||
names or unsafe runtime behavior.
|
||||
|
||||
## Implementation Rules
|
||||
|
||||
- Match official behavior, not just shapes: constructor args, default values,
|
||||
runtime flags, RNG use, dtype/autocast, scaling constants, masks, and output
|
||||
containers all matter.
|
||||
- Keep the implementation minimal and native. Do not keep a runtime import of
|
||||
the official implementation as the production component.
|
||||
- For schedulers, compare timesteps, sigmas/noise levels, step outputs, shift
|
||||
handling, prediction type, and any mutable internal state.
|
||||
- For upsamplers, compare resize mode, align_corners, residual branches,
|
||||
causal padding, normalization, and exact target-shape behavior.
|
||||
- For vocoders/audio components, compare waveform shape, sample-rate contract,
|
||||
channel order, hop length, normalization, and dtype.
|
||||
- If private upstream deps are required only for tests, keep stubs under
|
||||
`tests/local_tests/helpers/` and do not import them from production code.
|
||||
|
||||
## Prototype Checks
|
||||
|
||||
Follow the shared prototype success criteria.
|
||||
|
||||
## Parity-Debug Loop
|
||||
|
||||
Run the shared parity-debug loop. The component test command is:
|
||||
|
||||
```bash
|
||||
pytest <parity_test> -v -s
|
||||
```
|
||||
|
||||
For numerical drift, add targeted intermediate comparisons in the test to
|
||||
identify the first divergent operation.
|
||||
|
||||
## Escape Hatches
|
||||
|
||||
Follow `../add-model/shared/common_rules.md` and the component-specific guidance
|
||||
in `../add-model/shared/component_skill_common.md`. Generic-component ask cases
|
||||
include creating a new loader bucket, accepting an unsupported private op,
|
||||
choosing between incompatible official definitions, or dropping a required
|
||||
component.
|
||||
|
||||
## Handoff
|
||||
|
||||
Return `../add-model/contracts/component_skill_handoff.md` following the common
|
||||
handoff rules in `../add-model/shared/component_skill_common.md`.
|
||||
@@ -0,0 +1,186 @@
|
||||
---
|
||||
name: add-model-07-conversion
|
||||
description: Use during /add-model Phase 5 to write and verify a FastVideo checkpoint conversion script after native component prototypes expose FastVideo state-dict keys/shapes.
|
||||
---
|
||||
|
||||
# Add Model Conversion
|
||||
|
||||
## Goal
|
||||
|
||||
Convert official weights into a FastVideo-loadable component layout after Phase 4
|
||||
native prototypes exist. The conversion script owns parameter mapping, component
|
||||
splitting, passthrough assets, config emission, and strict-load verification.
|
||||
|
||||
## Inputs
|
||||
|
||||
Follow `../add-model/shared/common_rules.md` for token/auth safety, state files,
|
||||
escape hatches, production boundaries, and skip/pass semantics.
|
||||
|
||||
Require the initial request from
|
||||
`../add-model/contracts/conversion_request.md`.
|
||||
|
||||
If the FastVideo key/shape dump is missing, return to `/add-model` Phase 4. Do
|
||||
not write a final mapping against an unimplemented component.
|
||||
|
||||
For Phase 6 retry requests from component skills, also require the retry shape
|
||||
from `../add-model/contracts/conversion_request.md`.
|
||||
|
||||
## Output
|
||||
|
||||
- `scripts/checkpoint_conversion/<family>_to_diffusers.py`.
|
||||
- `converted_weights/<family>/` with `model_index.json` and per-component
|
||||
subfolders.
|
||||
- Updated `tests/local_tests/<model_family>/README.md` with conversion command,
|
||||
source layout, output path, and strict-load status.
|
||||
- Updated `tests/local_tests/<model_family>/PORT_STATUS.md` with conversion
|
||||
state, retry history, open questions, and issues/blockers.
|
||||
|
||||
## Reference Scripts
|
||||
|
||||
- `scripts/checkpoint_conversion/convert_ltx2_weights.py`: component prefix
|
||||
splitting, metadata config extraction, passthrough Gemma/tokenizer assets, and
|
||||
optional component-only output.
|
||||
- `scripts/checkpoint_conversion/stable_audio_to_diffusers.py`: monolithic
|
||||
`model.safetensors` split into transformer/VAE/conditioner, plus copied
|
||||
passthrough subfolders. Use this shape for single-checkpoint official repos.
|
||||
- `scripts/checkpoint_conversion/convert_gamecraft_full.py`: separate official
|
||||
sources for transformer, VAE, text encoders, tokenizers, scheduler, and root
|
||||
`model_index.json`.
|
||||
- `scripts/checkpoint_conversion/longcat_to_fastvideo.py`: fused QKV/KV split,
|
||||
renamed native transformer weights, and copied existing Diffusers components.
|
||||
- `scripts/checkpoint_conversion/pt_to_safetensors.py`: simple `.pt` extraction
|
||||
helper for nested checkpoint dictionaries.
|
||||
|
||||
## Source Layout Decision
|
||||
|
||||
Choose exactly one primary layout:
|
||||
|
||||
| Layout | Conversion behavior |
|
||||
|---|---|
|
||||
| `diffusers` | Usually no tensor remap; verify configs/classes and copy or update `_class_name` only when needed. |
|
||||
| `raw_official` | Convert a raw official checkpoint file or directory. Choose explicit component ownership before writing output. |
|
||||
| `separate_components` | Convert/copy each component from its own file or directory. |
|
||||
| `monolithic` | Load one model checkpoint and split state dict by authoritative prefixes into component buckets. |
|
||||
| `mixed` | Convert some components and copy passthrough components such as tokenizers, text encoders, schedulers, or already-Diffusers VAE dirs. |
|
||||
| `custom` | Document why none of the above fits before writing conversion code. |
|
||||
|
||||
Monolithic checkpoints need explicit prefix ownership. For example, Stable Audio
|
||||
uses one `model.safetensors` with DiT, pretransform/VAE, and conditioner keys;
|
||||
the converter splits those keys into FastVideo component subfolders and writes
|
||||
per-component configs.
|
||||
|
||||
## Script Shape
|
||||
|
||||
Start from `templates/family_to_diffusers.py` or the closest reference script.
|
||||
Keep the script explicit and reviewable:
|
||||
|
||||
- `COMPONENT_SPECS` or `COMPONENT_PREFIXES` declares component ownership.
|
||||
- `PARAM_NAME_MAP` declares key renames.
|
||||
- `SKIP_PATTERNS` declares intentionally dropped training-only keys.
|
||||
- tensor split/fuse helpers are named by operation, e.g. `split_qkv`.
|
||||
- `build_component_configs(...)` writes loader-compatible config files. Most
|
||||
model components use `config.json`; schedulers use `scheduler_config.json`.
|
||||
- `build_model_index(...)` writes a root `model_index.json` matching the target
|
||||
FastVideo pipeline and component classes.
|
||||
- verification reports missing, unexpected, skipped, unchanged, renamed, and
|
||||
shape-mismatched keys.
|
||||
|
||||
`model_index.json` library tokens must match FastVideo loaders:
|
||||
|
||||
- standard native DiT/VAE/audio/vocoder/upsampler components loaded by existing
|
||||
Diffusers-style loaders usually use `"diffusers"` with a FastVideo
|
||||
`_class_name` in the component `config.json`;
|
||||
- text encoders, tokenizers, image encoders, processors, and feature extractors
|
||||
usually use `"transformers"`;
|
||||
- `conditioner` currently expects `"fastvideo"`;
|
||||
- use fully qualified `"fastvideo.<module>"` only when intentionally relying on
|
||||
the custom fastvideo-library escape path;
|
||||
- do not write bare `"fastvideo"` for transformer, VAE, or other loaders that
|
||||
expect `"diffusers"` unless the loader explicitly expects it.
|
||||
|
||||
## Mapping Rules
|
||||
|
||||
- Use Phase 4 key/shape dumps to derive mappings. Do not guess from official key
|
||||
names alone.
|
||||
- Preserve each component's official file paths, parity test path, and prototype
|
||||
concerns in comments or structured constants near the mapping that uses them.
|
||||
- Every official inference parameter should be mapped, copied through, or listed
|
||||
as intentionally skipped with a reason.
|
||||
- Every FastVideo prototype parameter should receive a tensor or be listed as an
|
||||
intentional external/passthrough parameter.
|
||||
- Shape matches are necessary but not sufficient; check semantic pairing for
|
||||
Q/K/V, gate/up/down, norm scale/bias, LoRA/base, and modality-specific heads.
|
||||
- If official and FastVideo fuse or split tensors differently, convert tensors in
|
||||
the script rather than changing production code to match checkpoint quirks.
|
||||
|
||||
## Verification
|
||||
|
||||
Run conversion locally, then verify before returning to Phase 6. For retry
|
||||
requests, update the mapping, rerun conversion, and refresh only the implicated
|
||||
converted component when safe; otherwise rerun the full conversion.
|
||||
|
||||
```bash
|
||||
python scripts/checkpoint_conversion/<family>_to_diffusers.py \
|
||||
--src <official_weights> \
|
||||
--revision <hf_revision> \
|
||||
--dst converted_weights/<model_family>
|
||||
```
|
||||
|
||||
Omit `--revision` for local sources or when prep recorded `default` / `none`.
|
||||
|
||||
Minimum output layout:
|
||||
|
||||
```text
|
||||
converted_weights/<family>/
|
||||
model_index.json
|
||||
transformer/config.json
|
||||
transformer/*.safetensors
|
||||
vae/config.json
|
||||
vae/*.safetensors
|
||||
scheduler/scheduler_config.json as needed
|
||||
text_encoder/... as needed
|
||||
```
|
||||
|
||||
Required checks:
|
||||
|
||||
- `model_index.json` exists and lists every required component.
|
||||
- Each converted component has the config filename its loader expects and
|
||||
safetensors weights when it owns weights. Scheduler dirs require
|
||||
`scheduler_config.json`; most other native model dirs use `config.json`.
|
||||
- Weight filenames may vary by loader: transformer and VAE loaders glob all
|
||||
`*.safetensors`; text encoders may load `*.safetensors`, `*.bin`, and
|
||||
sometimes `*.pt`; `conditioner` currently expects
|
||||
`diffusion_pytorch_model.safetensors`. Use the loader's actual accepted layout
|
||||
rather than assuming one global filename.
|
||||
- Passthrough components are copied or referenced deliberately.
|
||||
- Each emitted component config validates through the same path production
|
||||
loaders use. Instantiate the relevant config and call `update_model_arch(...)`
|
||||
or `update_model_config(...)` with the emitted JSON so unknown keys fail during
|
||||
conversion, not at pipeline load time.
|
||||
- Record production loader strictness for every stateful component. If the loader
|
||||
intentionally uses non-strict loading, add explicit missing/unexpected-key
|
||||
assertions in the parity test and document exactly which keys are allowed.
|
||||
- Each new FastVideo component strict-loads converted weights where its production
|
||||
loader is strict. If strict loading is impossible, record the exact allowed
|
||||
missing/unexpected keys and why they are not inference weights.
|
||||
- Retry fixes include the original component parity evidence and the new
|
||||
strict-load result in `local_tests_readme` so the component subagent can resume
|
||||
without rediscovering context.
|
||||
- `local_tests_readme` records the command, output directory, and strict-load
|
||||
result.
|
||||
|
||||
Do not chase numerical parity in this skill except to identify a conversion
|
||||
mapping bug. Long parity-debug loops belong to `/add-model` Phase 6.
|
||||
|
||||
## Escape Hatches
|
||||
|
||||
Follow `../add-model/shared/common_rules.md`. Conversion-specific ask cases
|
||||
include selecting between incompatible official checkpoints, publishing/uploading
|
||||
weights, overwriting an existing converted repo not created by this run,
|
||||
accepting non-strict missing inference weights, or dropping a component/output
|
||||
from scope.
|
||||
|
||||
## Handoff
|
||||
|
||||
Return `../add-model/contracts/conversion_handoff.md` and update the shared state
|
||||
files before handoff.
|
||||
@@ -0,0 +1,312 @@
|
||||
#!/usr/bin/env python3
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
"""Convert <model_family> official weights to a FastVideo Diffusers-style tree.
|
||||
|
||||
This template supports both separate component sources and a monolithic pipeline
|
||||
checkpoint that must be split by component prefix. Replace every TODO before
|
||||
using it for a real port.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import shutil
|
||||
from collections import OrderedDict
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
import torch
|
||||
from safetensors import safe_open
|
||||
from safetensors.torch import load_file, save_file
|
||||
|
||||
try:
|
||||
from huggingface_hub import snapshot_download
|
||||
except ImportError: # pragma: no cover - optional local conversion dependency
|
||||
snapshot_download = None
|
||||
|
||||
|
||||
# TODO: fill with authoritative component prefixes for monolithic checkpoints.
|
||||
# Example: {"model.model.": "transformer", "pretransform.model.": "vae"}
|
||||
COMPONENT_PREFIXES: dict[str, str] = {}
|
||||
|
||||
# TODO: fill with component-specific source paths for separate-component repos.
|
||||
# Example: {"transformer": "transformer/model.safetensors", "vae": "vae/"}
|
||||
SEPARATE_COMPONENT_PATHS: dict[str, str] = {}
|
||||
|
||||
# TODO: copy passthrough dirs that are already loadable by FastVideo/Diffusers.
|
||||
PASSTHROUGH_SUBFOLDERS: tuple[str, ...] = ("tokenizer", "scheduler")
|
||||
|
||||
# TODO: add regex renames derived from Phase 4 key/shape dumps.
|
||||
PARAM_NAME_MAP: dict[str, str] = {}
|
||||
|
||||
# TODO: include training-only or dynamically-computed keys that must not load.
|
||||
SKIP_PATTERNS: tuple[str, ...] = ()
|
||||
|
||||
|
||||
def _hf_token() -> str | None:
|
||||
return (
|
||||
os.environ.get("HF_TOKEN") or os.environ.get("HUGGINGFACE_HUB_TOKEN")
|
||||
or os.environ.get("HF_API_KEY")
|
||||
)
|
||||
|
||||
|
||||
def resolve_src(src: str, revision: str | None) -> Path:
|
||||
if os.path.exists(src):
|
||||
return Path(src)
|
||||
if snapshot_download is None:
|
||||
raise RuntimeError("huggingface_hub is required when --src is a repo id")
|
||||
return Path(snapshot_download(repo_id=src, revision=revision, token=_hf_token()))
|
||||
|
||||
|
||||
def load_checkpoint(path: Path) -> dict[str, torch.Tensor]:
|
||||
if path.is_dir():
|
||||
weights: dict[str, torch.Tensor] = {}
|
||||
for shard in sorted(path.glob("*.safetensors")):
|
||||
weights.update(load_file(str(shard)))
|
||||
if weights:
|
||||
return weights
|
||||
raise FileNotFoundError(f"No safetensors found in {path}")
|
||||
|
||||
if path.suffix == ".safetensors":
|
||||
return load_file(str(path))
|
||||
|
||||
checkpoint = torch.load(path, map_location="cpu", weights_only=True)
|
||||
if isinstance(checkpoint, dict):
|
||||
for key in ("state_dict", "model_state_dict", "model", "module", "ema"):
|
||||
if key in checkpoint and isinstance(checkpoint[key], dict):
|
||||
return checkpoint[key]
|
||||
return checkpoint
|
||||
raise TypeError(f"Unsupported checkpoint type: {type(checkpoint)!r}")
|
||||
|
||||
|
||||
def should_skip_key(key: str) -> bool:
|
||||
return any(re.search(pattern, key) for pattern in SKIP_PATTERNS)
|
||||
|
||||
|
||||
def apply_mapping(key: str) -> str | None:
|
||||
if should_skip_key(key):
|
||||
return None
|
||||
for pattern, replacement in PARAM_NAME_MAP.items():
|
||||
if re.match(pattern, key):
|
||||
return re.sub(pattern, replacement, key)
|
||||
return key
|
||||
|
||||
|
||||
def split_monolithic(
|
||||
state: dict[str, torch.Tensor],
|
||||
) -> dict[str, OrderedDict[str, torch.Tensor]]:
|
||||
components: dict[str, OrderedDict[str, torch.Tensor]] = {
|
||||
name: OrderedDict() for name in set(COMPONENT_PREFIXES.values())
|
||||
}
|
||||
intentionally_skipped: list[str] = []
|
||||
unowned: list[str] = []
|
||||
for key, value in state.items():
|
||||
if should_skip_key(key):
|
||||
intentionally_skipped.append(key)
|
||||
continue
|
||||
for prefix, component in COMPONENT_PREFIXES.items():
|
||||
if key.startswith(prefix):
|
||||
mapped = apply_mapping(key[len(prefix):])
|
||||
if mapped is not None:
|
||||
components[component][mapped] = value
|
||||
break
|
||||
else:
|
||||
unowned.append(key)
|
||||
if unowned:
|
||||
sample = ", ".join(unowned[:10])
|
||||
raise ValueError(
|
||||
f"Unowned monolithic keys: {len(unowned)}. "
|
||||
f"Add COMPONENT_PREFIXES or SKIP_PATTERNS entries. Sample: {sample}"
|
||||
)
|
||||
if intentionally_skipped:
|
||||
print(f"Intentionally skipped {len(intentionally_skipped)} keys")
|
||||
return {name: weights for name, weights in components.items() if weights}
|
||||
|
||||
|
||||
def load_separate_components(src_dir: Path) -> dict[str, OrderedDict[str, torch.Tensor]]:
|
||||
components: dict[str, OrderedDict[str, torch.Tensor]] = {}
|
||||
for component, rel_path in SEPARATE_COMPONENT_PATHS.items():
|
||||
state = load_checkpoint(src_dir / rel_path)
|
||||
converted: OrderedDict[str, torch.Tensor] = OrderedDict()
|
||||
for key, value in state.items():
|
||||
mapped = apply_mapping(key)
|
||||
if mapped is not None:
|
||||
converted[mapped] = value
|
||||
components[component] = converted
|
||||
return components
|
||||
|
||||
|
||||
def build_component_configs(_src_dir: Path) -> dict[str, dict[str, Any]]:
|
||||
# TODO: emit config content accepted by FastVideo loaders. Most components use
|
||||
# config.json; schedulers use scheduler_config.json.
|
||||
return {
|
||||
"transformer": {"_class_name": "<FastVideoTransformerClass>"},
|
||||
"vae": {"_class_name": "<FastVideoVAEClass>"},
|
||||
}
|
||||
|
||||
|
||||
def config_filename(component: str) -> str:
|
||||
if component == "scheduler":
|
||||
return "scheduler_config.json"
|
||||
return "config.json"
|
||||
|
||||
|
||||
def source_label(src: str) -> str:
|
||||
if os.path.exists(src):
|
||||
return Path(src).name
|
||||
return src
|
||||
|
||||
|
||||
def build_model_index(
|
||||
src: str,
|
||||
revision: str | None,
|
||||
available_components: set[str],
|
||||
) -> dict[str, Any]:
|
||||
# TODO: match the target pipeline and every required component.
|
||||
index: dict[str, Any] = {
|
||||
"_class_name": "<FastVideoPipelineClass>",
|
||||
"_diffusers_version": "0.30.0",
|
||||
"_fastvideo_converted_from": source_label(src),
|
||||
# Existing transformer/VAE loaders expect "diffusers" even when
|
||||
# _class_name names a FastVideo-native class registered in FastVideo.
|
||||
"transformer": ["diffusers", "<FastVideoTransformerClass>"],
|
||||
"vae": ["diffusers", "<FastVideoVAEClass>"],
|
||||
}
|
||||
if revision:
|
||||
index["_fastvideo_converted_revision"] = revision
|
||||
return {
|
||||
key: value
|
||||
for key, value in index.items()
|
||||
if key.startswith("_") or key in available_components
|
||||
}
|
||||
|
||||
|
||||
def validate_component_configs(configs: dict[str, dict[str, Any]]) -> None:
|
||||
# TODO: instantiate each FastVideo config and call update_model_arch(...) or
|
||||
# update_model_config(...) with this JSON so unknown emitted keys fail here.
|
||||
placeholder_configs = [
|
||||
name for name, config in configs.items() if "<" in json.dumps(config)
|
||||
]
|
||||
if placeholder_configs:
|
||||
raise ValueError(f"Replace config placeholders for: {placeholder_configs}")
|
||||
|
||||
|
||||
def verify_conversion(
|
||||
dst_dir: Path,
|
||||
components: dict[str, OrderedDict[str, torch.Tensor]],
|
||||
) -> None:
|
||||
del dst_dir, components
|
||||
# TODO: load each emitted stateful component through its production loader and
|
||||
# assert strict load, or document exact allowed missing/unexpected keys.
|
||||
raise NotImplementedError(
|
||||
"Implement production config validation and strict-load checks"
|
||||
)
|
||||
|
||||
|
||||
def write_component(
|
||||
dst_dir: Path,
|
||||
name: str,
|
||||
state: dict[str, torch.Tensor],
|
||||
config: dict[str, Any] | None,
|
||||
) -> None:
|
||||
component_dir = dst_dir / name
|
||||
if component_dir.exists() and any(component_dir.iterdir()):
|
||||
shutil.rmtree(component_dir)
|
||||
component_dir.mkdir(parents=True, exist_ok=True)
|
||||
save_file(
|
||||
dict(state), str(component_dir / "diffusion_pytorch_model.safetensors")
|
||||
)
|
||||
if config is not None:
|
||||
config_path = component_dir / config_filename(name)
|
||||
with config_path.open("w", encoding="utf-8") as f:
|
||||
json.dump(config, f, indent=2)
|
||||
f.write("\n")
|
||||
print(f"Wrote {name}: {len(state)} tensors")
|
||||
|
||||
|
||||
def copy_passthrough(src_dir: Path, dst_dir: Path) -> list[str]:
|
||||
copied: list[str] = []
|
||||
for subfolder in PASSTHROUGH_SUBFOLDERS:
|
||||
src = src_dir / subfolder
|
||||
if not src.is_dir():
|
||||
continue
|
||||
dst = dst_dir / subfolder
|
||||
if dst.exists():
|
||||
shutil.rmtree(dst)
|
||||
shutil.copytree(src, dst)
|
||||
copied.append(subfolder)
|
||||
print(f"Copied {subfolder}/")
|
||||
return copied
|
||||
|
||||
|
||||
def default_monolithic_checkpoint(src_path: Path) -> Path:
|
||||
if src_path.is_file():
|
||||
return src_path
|
||||
return src_path / "model.safetensors"
|
||||
|
||||
|
||||
def convert(
|
||||
src: str,
|
||||
dst: str,
|
||||
layout: str,
|
||||
revision: str | None,
|
||||
) -> None:
|
||||
src_path = resolve_src(src, revision)
|
||||
dst_dir = Path(dst)
|
||||
dst_dir.mkdir(parents=True, exist_ok=True)
|
||||
model_index_path = dst_dir / "model_index.json"
|
||||
|
||||
if layout in {"monolithic", "raw_official"}:
|
||||
# TODO: replace model.safetensors with the official monolithic file name.
|
||||
components = split_monolithic(
|
||||
load_checkpoint(default_monolithic_checkpoint(src_path))
|
||||
)
|
||||
elif layout in {"separate_components", "mixed"}:
|
||||
if not src_path.is_dir():
|
||||
raise ValueError(f"{layout} layout requires a source directory: {src_path}")
|
||||
components = load_separate_components(src_path)
|
||||
else:
|
||||
raise ValueError(f"Unsupported template layout: {layout}")
|
||||
|
||||
copied = (
|
||||
copy_passthrough(src_path, dst_dir) if src_path.is_dir() else []
|
||||
)
|
||||
configs = build_component_configs(src_path if src_path.is_dir() else src_path.parent)
|
||||
validate_component_configs(configs)
|
||||
for name, state in components.items():
|
||||
write_component(dst_dir, name, state, configs.get(name))
|
||||
|
||||
available = set(components) | set(copied)
|
||||
with model_index_path.open("w", encoding="utf-8") as f:
|
||||
json.dump(build_model_index(src, revision, available), f, indent=2)
|
||||
f.write("\n")
|
||||
print(f"Wrote {dst_dir / 'model_index.json'}")
|
||||
verify_conversion(dst_dir, components)
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument(
|
||||
"--src", required=True, help="HF repo id, local dir, or checkpoint path"
|
||||
)
|
||||
parser.add_argument("--revision", help="HF branch, tag, or commit for repo sources")
|
||||
parser.add_argument(
|
||||
"--dst",
|
||||
required=True,
|
||||
help="Output converted_weights/<model_family> directory",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--layout",
|
||||
choices=("raw_official", "monolithic", "separate_components", "mixed"),
|
||||
required=True,
|
||||
help="Official source layout",
|
||||
)
|
||||
args = parser.parse_args()
|
||||
convert(args.src, args.dst, args.layout, args.revision)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,222 @@
|
||||
---
|
||||
name: add-model-08-trace
|
||||
description: Use during /add-model Phase 6 when component parity has failed and root cause requires layer-by-layer divergence analysis. Instruments both the official reference and FastVideo port with forward hooks to find the first numerical divergence point.
|
||||
---
|
||||
|
||||
# Add-Model Trace
|
||||
|
||||
## Manual Invocation
|
||||
|
||||
Load this skill when `/add-model` Phase 6 component parity has failed and the
|
||||
root cause requires layer-by-layer divergence analysis. This skill is not
|
||||
auto-fired. The calling subagent (DiT, VAE, encoder, or generic port skill)
|
||||
loads it when its standard parity-debug loop hits a wall and cannot isolate
|
||||
the divergence from end-to-end tensor comparisons alone.
|
||||
|
||||
Do not load this skill for first-pass parity failures. Try weight-diff and
|
||||
end-to-end tensor comparison first. Load this skill only when those do not
|
||||
isolate the cause.
|
||||
|
||||
## Goal
|
||||
|
||||
Find the first numerical divergence point between FastVideo's port and the
|
||||
official reference, layer by layer, by instrumenting both sides at matching
|
||||
tensor boundaries. The investigation must leave zero source residue in
|
||||
production code when it closes.
|
||||
|
||||
## When To Run
|
||||
|
||||
After a component parity test FAILS at a bf16-noise-realistic tolerance AND
|
||||
the calling subagent's first-pass debug (weight-diff, end-to-end tensor
|
||||
compare) does not isolate the cause.
|
||||
|
||||
Required inputs before starting:
|
||||
|
||||
- A working FastVideo loader for the component under investigation.
|
||||
- A working official loader, typically via
|
||||
`tests/local_tests/helpers/<family>_upstream.py::load_upstream_<component>`.
|
||||
- Shared deterministic test inputs (same tensors on both sides).
|
||||
- The component parity test file path and its current failure output.
|
||||
|
||||
## Hard Rules: Instrumentation Hierarchy
|
||||
|
||||
Apply these in priority order. Use the highest-priority method that works for
|
||||
the target site.
|
||||
|
||||
### (1) Forward hooks (PREFERRED)
|
||||
|
||||
`module.register_forward_hook(...)` and `register_forward_pre_hook(...)`.
|
||||
Always within `try/finally` with `handle.remove()`. Zero source residue.
|
||||
|
||||
```python
|
||||
handle = module.register_forward_hook(fn)
|
||||
try:
|
||||
output = model(inputs)
|
||||
finally:
|
||||
handle.remove()
|
||||
```
|
||||
|
||||
### (2) Runtime monkey-patch (PREFERRED over source edits)
|
||||
|
||||
`module.attr = wrapped_func` or `cls.method = wrapped_method`, restored via
|
||||
`try/finally` (save original first). Use for free functions and non-Module
|
||||
sites such as activation functions (`swiglu`, `apply_rotary_emb`) that cannot
|
||||
be hooked as `nn.Module` submodules.
|
||||
|
||||
```python
|
||||
original = cls.method
|
||||
cls.method = wrapped
|
||||
try:
|
||||
output = model(inputs)
|
||||
finally:
|
||||
cls.method = original
|
||||
```
|
||||
|
||||
### (3) Source edits in FastVideo's own code
|
||||
|
||||
Only when (1) and (2) are insufficient. Track all edits within a single named
|
||||
`git stash` boundary OR a temporary branch. Run `git diff` before closing the
|
||||
investigation to confirm the stash or branch is clean. The cleanup gate
|
||||
enforces this.
|
||||
|
||||
### (4) Source edits in official repo source
|
||||
|
||||
Allowed if EITHER:
|
||||
|
||||
- (a) The official repo is a git-tracked clone (e.g. `daVinci-MagiHuman/` at
|
||||
the repo root): use `git diff` in the clone path to verify cleanup.
|
||||
- (b) It's installed editable (`pip install -e .`): use `git diff` in the
|
||||
editable source path to verify cleanup.
|
||||
|
||||
If the official repo is installed non-editable in site-packages: back up the
|
||||
target file (`cp original.py original.py.trace-backup`) before editing, then
|
||||
restore from backup at the end (or `pip install --force-reinstall <pkg>`).
|
||||
The cleanup gate verifies via diff-against-backup or zero-diff-in-clone.
|
||||
|
||||
## Logging Contract
|
||||
|
||||
One log file per side. Paths:
|
||||
|
||||
```
|
||||
/tmp/opencode/<family>_<component>_up_layers.log
|
||||
/tmp/opencode/<family>_<component>_fv_layers.log
|
||||
```
|
||||
|
||||
Format: one line per captured tensor, space-separated:
|
||||
|
||||
```
|
||||
<name> <shape> <abs_mean> <sum> <min> <max>
|
||||
```
|
||||
|
||||
Example:
|
||||
|
||||
```
|
||||
block[00] (1,512,1024) 0.012345 6.3210 -0.4321 0.4321
|
||||
```
|
||||
|
||||
Keep the format diff-friendly. Running `diff /tmp/opencode/x_up.log
|
||||
/tmp/opencode/x_fv.log` should highlight the first divergent line directly.
|
||||
Retain side-by-side stdout output alongside the per-side files for human
|
||||
review.
|
||||
|
||||
## Drill-Down Loop
|
||||
|
||||
**Initial run:** attach hooks to every top-level block (`model.block.layers[i]`
|
||||
or equivalent). Identify the first block index `NN` where abs_mean relative
|
||||
drift exceeds 0.5% compared to the previous block.
|
||||
|
||||
**Drill run:** set `<FAMILY>_DEBUG_DRILL_LAYER=NN` and re-run. The script
|
||||
attaches submodule hooks inside block `NN`: attention output, mlp.pre_norm,
|
||||
mlp.up_gate_proj, mlp.down_proj input (via pre-hook) and output, mlp output,
|
||||
attn_post_norm (if present), mlp_post_norm (if present).
|
||||
|
||||
**Iterate:** if the drill run points to a free function (e.g. an activation
|
||||
not wrapped in an `nn.Module`), switch to a monkey-patch (method 2) to
|
||||
intercept its output via the next module's pre-hook.
|
||||
|
||||
The loop ends when the first divergent submodule is identified with a
|
||||
file:line citation in the official source.
|
||||
|
||||
## Hypothesis Toggles
|
||||
|
||||
Use env-var-gated monkey-patches to A/B test suspect implementations without
|
||||
source edits. Pattern: `<FAMILY>_DEBUG_PATCH_<HYPOTHESIS>=1`.
|
||||
|
||||
Example from the magi-human investigation:
|
||||
|
||||
```
|
||||
MAGI_DEBUG_PATCH_LINEAR=1
|
||||
```
|
||||
|
||||
This patched `PackedExpertLinear.forward` to mirror upstream's
|
||||
`_BF16ComputeLinear` explicit-cast pattern, isolating a dtype-cast difference
|
||||
as the root cause.
|
||||
|
||||
Document all toggles in the script docstring. Each toggle must:
|
||||
|
||||
- save the original before patching;
|
||||
- restore the original in a `try/finally` block;
|
||||
- print a `[debug] Patched <ClassName>.<method>` line to stdout when active.
|
||||
|
||||
## Cleanup Gate
|
||||
|
||||
The calling agent MUST report `[cleanup-gate] PASS` on all five items before
|
||||
handoff. Do not hand off with any item unresolved.
|
||||
|
||||
1. `git diff` in the FastVideo repo: empty. No stray prints, hooks, or
|
||||
monkey-patches in production code.
|
||||
2. `git diff` in the official-repo clone (if used): empty. For non-editable
|
||||
site-packages installs: `diff original.py original.py.trace-backup` is
|
||||
empty OR `pip install --force-reinstall <pkg>` succeeded and the installed
|
||||
file matches the original.
|
||||
3. `git stash list`: only the named investigation stash (or empty). No
|
||||
unnamed stashes left from this session.
|
||||
4. No new untracked files outside `/tmp/opencode/` (logs) and the existing
|
||||
debug script directory (`tests/local_tests/transformers/` or equivalent).
|
||||
5. `mypy` clean on any production files touched during the investigation.
|
||||
|
||||
## Escape Hatches
|
||||
|
||||
Escalate to the calling bucket skill when:
|
||||
|
||||
- A forward hook on an official module raises because of a custom `forward`
|
||||
signature or varlen handler args that the hook closure cannot satisfy. The
|
||||
bucket skill has component-specific knowledge to work around this.
|
||||
- The first divergent layer is `block[0]`, meaning the divergence is in the
|
||||
adapter, modality dispatcher, coordinate embedding, or packing step before
|
||||
any block runs. Check those sites first; the bug is not in attention or MLP.
|
||||
- Per-block drift is never zero anywhere across all blocks. This usually means
|
||||
the inputs are not bit-identical between sides. Verify with a state-dict
|
||||
compare (weight-diff script) AND confirm the input tensors are the same
|
||||
object or have identical values before the forward call.
|
||||
|
||||
## Handoff
|
||||
|
||||
Return to the calling subagent with:
|
||||
|
||||
- File paths to per-side logs (`/tmp/opencode/<family>_<component>_{up,fv}_layers.log`).
|
||||
- The identified first divergent layer or submodule name.
|
||||
- The upstream file:line citation where the divergence originates.
|
||||
- Hypothesis verdict if an A/B toggle was used (e.g. "PATCH_LINEAR=1 closes
|
||||
the gap, confirming dtype-cast difference in PackedExpertLinear").
|
||||
- Cleanup-gate status: `[cleanup-gate] PASS` or a list of unresolved items.
|
||||
|
||||
The calling agent uses this to scope the production fix in the FastVideo
|
||||
component file.
|
||||
|
||||
## References
|
||||
|
||||
- `templates/block_trace_debug.py` in this skill directory: the canonical
|
||||
template this skill generalizes.
|
||||
- `tests/local_tests/transformers/_debug_magi_human_block_parity.py` in the
|
||||
FastVideo3 repo: the worked magi-human example this skill was extracted from.
|
||||
- `add-model/SKILL.md` Phase 6: the calling context for this skill.
|
||||
- `add-model-03-port-dit/SKILL.md`, `add-model-04-port-vae/SKILL.md`,
|
||||
`add-model-05-port-encoder/SKILL.md`, `add-model-06-port-generic/SKILL.md`:
|
||||
bucket-specific debug language and component-specific escape-hatch knowledge.
|
||||
|
||||
## Changelog
|
||||
|
||||
| Date | Change |
|
||||
|---|---|
|
||||
| 2026-05-01 | Initial skill extracted from `_debug_magi_human_block_parity.py` pattern. |
|
||||
@@ -0,0 +1,356 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
"""Per-block divergence debugger template for FastVideo model ports.
|
||||
|
||||
Run directly (not a pytest test):
|
||||
python tests/local_tests/transformers/_debug_<family>_<component>_parity.py
|
||||
|
||||
Generalizes: tests/local_tests/transformers/_debug_magi_human_block_parity.py
|
||||
|
||||
Fill FAMILY, COMPONENT, and the two loader functions. Run once for the initial
|
||||
drift table, then set <FAMILY>_DEBUG_DRILL_LAYER=NN to drill into submodules.
|
||||
Add <FAMILY>_DEBUG_PATCH_<HYPOTHESIS>=1 to A/B test a suspect implementation.
|
||||
|
||||
CLEANUP: all hooks removed in try/finally; monkey-patches restored in
|
||||
try/finally; source edits tracked in a named git stash. Zero source residue.
|
||||
See add-model-08-trace/SKILL.md for the full cleanup gate checklist.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import gc
|
||||
import os
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
import torch
|
||||
|
||||
FAMILY: str = "<family>" # e.g. "magi_human", "ltx2", "wan"
|
||||
COMPONENT: str = "<component>" # e.g. "dit", "vae", "encoder"
|
||||
DRILL_LAYER_ENV: str = "<FAMILY>_DEBUG_DRILL_LAYER"
|
||||
HYPOTHESIS_ENV: str = "<FAMILY>_DEBUG_PATCH_<HYPOTHESIS>"
|
||||
REL_THRESHOLD: float = 0.005 # 0.5% abs_mean drift flags a block as divergent
|
||||
LOG_DIR: Path = Path("/tmp/opencode")
|
||||
|
||||
REPO_ROOT = Path(__file__).resolve().parents[3]
|
||||
sys.path.insert(0, str(REPO_ROOT))
|
||||
|
||||
|
||||
def load_official(device: torch.device) -> torch.nn.Module:
|
||||
"""Load the official upstream model. TODO: implement for your family.
|
||||
|
||||
Example (magi-human):
|
||||
from tests.local_tests.helpers.magi_human_upstream import install_stubs, load_upstream_dit
|
||||
install_stubs()
|
||||
return load_upstream_dit(base_shard_dir, device=device, dtype=None)
|
||||
"""
|
||||
raise NotImplementedError(f"Fill load_official() for {FAMILY}/{COMPONENT}.")
|
||||
|
||||
|
||||
def load_fastvideo(device: torch.device) -> torch.nn.Module:
|
||||
"""Load the FastVideo-native model. TODO: implement for your family.
|
||||
|
||||
Example (magi-human):
|
||||
from fastvideo.configs.models.dits.magi_human import MagiHumanVideoConfig
|
||||
from fastvideo.models.dits.magi_human import MagiHumanDiT
|
||||
from safetensors.torch import load_file; import glob
|
||||
fv = MagiHumanDiT(MagiHumanVideoConfig())
|
||||
state = {}
|
||||
for shard in sorted(glob.glob(str(transformer_dir / "*.safetensors"))): state.update(load_file(shard))
|
||||
fv.load_state_dict(state, strict=False); return fv.to(device).eval()
|
||||
"""
|
||||
raise NotImplementedError(f"Fill load_fastvideo() for {FAMILY}/{COMPONENT}.")
|
||||
|
||||
|
||||
def build_inputs(device: torch.device) -> dict[str, Any]:
|
||||
"""Return deterministic inputs shared by both sides. TODO: replace.
|
||||
|
||||
Both sides must receive the SAME tensors (clone before each forward call).
|
||||
Non-identical inputs cause non-zero drift everywhere.
|
||||
"""
|
||||
torch.manual_seed(0)
|
||||
return {"x": torch.randn(64, 1024, dtype=torch.bfloat16, device=device)}
|
||||
|
||||
|
||||
def _stat(name: str, t: torch.Tensor) -> dict:
|
||||
f = t.detach().float()
|
||||
return {
|
||||
"name": name,
|
||||
"shape": tuple(t.shape),
|
||||
"abs_mean": f.abs().mean().item(),
|
||||
"sum": f.sum().item(),
|
||||
"min": f.min().item(),
|
||||
"max": f.max().item(),
|
||||
}
|
||||
|
||||
|
||||
def _attach_block_hooks(
|
||||
model: torch.nn.Module,
|
||||
label: str,
|
||||
log: list[dict],
|
||||
tensors: dict[str, torch.Tensor] | None = None,
|
||||
drill_layer: int | None = None,
|
||||
) -> list[Any]:
|
||||
"""Return hook handles. Caller MUST remove them in try/finally."""
|
||||
handles: list[Any] = []
|
||||
|
||||
def _hook(name: str):
|
||||
def fn(_module, _inputs, outputs):
|
||||
t = outputs[0] if isinstance(outputs, tuple) else outputs
|
||||
if not torch.is_tensor(t):
|
||||
return
|
||||
log.append({"side": label, **_stat(name, t)})
|
||||
if tensors is not None:
|
||||
tensors[name] = t.detach().float().cpu()
|
||||
return fn
|
||||
|
||||
def _pre_hook(name: str):
|
||||
# Pre-hooks observe a free function's output by intercepting the next
|
||||
# module's input (useful when the activation is not an nn.Module).
|
||||
def fn(_module, inputs):
|
||||
t = inputs[0] if isinstance(inputs, tuple) else inputs
|
||||
if not torch.is_tensor(t):
|
||||
return
|
||||
key = f"{name}<in>"
|
||||
log.append({"side": label, **_stat(key, t)})
|
||||
if tensors is not None:
|
||||
tensors[key] = t.detach().float().cpu()
|
||||
return fn
|
||||
|
||||
# TODO: adapt attribute paths to your model. Remove adapter block if absent.
|
||||
if hasattr(model, "adapter"):
|
||||
handles.append(model.adapter.register_forward_hook(_hook("adapter")))
|
||||
|
||||
# TODO: adapt model.block.layers to your block container.
|
||||
# Alternatives: model.transformer.layers, model.blocks, model.layers
|
||||
block_layers = model.block.layers # type: ignore[attr-defined]
|
||||
for i, layer in enumerate(block_layers):
|
||||
handles.append(layer.register_forward_hook(_hook(f"block[{i:02d}]")))
|
||||
if drill_layer is not None and i == drill_layer:
|
||||
tag = f"L{i:02d}"
|
||||
# TODO: adapt submodule names to your layer's attributes.
|
||||
# magi-human uses: attention, mlp.pre_norm, mlp.up_gate_proj,
|
||||
# mlp.down_proj (pre+post), mlp, attn_post_norm, mlp_post_norm.
|
||||
if hasattr(layer, "attention"):
|
||||
handles.append(
|
||||
layer.attention.register_forward_hook(_hook(f"{tag}.attention"))
|
||||
)
|
||||
if hasattr(layer, "mlp"):
|
||||
mlp = layer.mlp
|
||||
if hasattr(mlp, "pre_norm"):
|
||||
handles.append(
|
||||
mlp.pre_norm.register_forward_hook(_hook(f"{tag}.mlp.pre_norm"))
|
||||
)
|
||||
if hasattr(mlp, "up_gate_proj"):
|
||||
handles.append(
|
||||
mlp.up_gate_proj.register_forward_hook(
|
||||
_hook(f"{tag}.mlp.up_gate_proj")
|
||||
)
|
||||
)
|
||||
if hasattr(mlp, "down_proj"):
|
||||
handles.append(
|
||||
mlp.down_proj.register_forward_pre_hook(
|
||||
_pre_hook(f"{tag}.mlp.down_proj")
|
||||
)
|
||||
)
|
||||
handles.append(
|
||||
mlp.down_proj.register_forward_hook(_hook(f"{tag}.mlp.down_proj"))
|
||||
)
|
||||
handles.append(mlp.register_forward_hook(_hook(f"{tag}.mlp")))
|
||||
if hasattr(layer, "attn_post_norm"):
|
||||
handles.append(
|
||||
layer.attn_post_norm.register_forward_hook(
|
||||
_hook(f"{tag}.attn_post_norm")
|
||||
)
|
||||
)
|
||||
if hasattr(layer, "mlp_post_norm"):
|
||||
handles.append(
|
||||
layer.mlp_post_norm.register_forward_hook(
|
||||
_hook(f"{tag}.mlp_post_norm")
|
||||
)
|
||||
)
|
||||
return handles
|
||||
|
||||
|
||||
def _apply_hypothesis_patch() -> bool:
|
||||
"""Apply an optional monkey-patch gated by HYPOTHESIS_ENV. TODO: implement.
|
||||
|
||||
Pattern: save original on the class, patch, restore in _restore_hypothesis_patch().
|
||||
"""
|
||||
if os.getenv(HYPOTHESIS_ENV) != "1":
|
||||
return False
|
||||
# TODO: import FastVideo class, save original, apply patch.
|
||||
print(f"[debug] Hypothesis patch {HYPOTHESIS_ENV}=1 applied.")
|
||||
return True
|
||||
|
||||
|
||||
def _restore_hypothesis_patch() -> None:
|
||||
if os.getenv(HYPOTHESIS_ENV) != "1":
|
||||
return
|
||||
# TODO: restore original, e.g.: _mod.TargetClass.method = _mod._ORIGINAL_METHOD
|
||||
|
||||
|
||||
def _write_log(entries: list[dict], path: Path) -> None:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
with open(path, "w") as f:
|
||||
for e in entries:
|
||||
f.write(
|
||||
f"{e['name']} {e['shape']} "
|
||||
f"{e['abs_mean']:.8f} {e['sum']:.4f} "
|
||||
f"{e['min']:.6f} {e['max']:.6f}\n"
|
||||
)
|
||||
|
||||
|
||||
def _sort_key(name: str, drill_layer: int) -> tuple:
|
||||
if name == "adapter":
|
||||
return (0, "")
|
||||
if name.startswith(f"L{drill_layer:02d}."):
|
||||
sub_order = {
|
||||
"attention": 0, "attn_post_norm": 1, "mlp.pre_norm": 2,
|
||||
"mlp.up_gate_proj": 3, "mlp.down_proj<in>": 4,
|
||||
"mlp.down_proj": 5, "mlp": 6, "mlp_post_norm": 7,
|
||||
}.get(name.split(".", 1)[1], 9)
|
||||
return (1, f"block[{drill_layer:02d}]", sub_order)
|
||||
if name.startswith("block["):
|
||||
return (1, name, 99)
|
||||
return (2, name, 0)
|
||||
|
||||
|
||||
def _print_table(by_name: dict[str, dict], drill_layer: int) -> int | None:
|
||||
hdr = (
|
||||
f"{'name':<18} {'up_shape':<22} {'up_absmean':>12} {'fv_absmean':>12} "
|
||||
f"{'absmean_diff':>14} {'rel%':>8} {'up_sum':>14} {'fv_sum':>14} {'sum_diff':>12}"
|
||||
)
|
||||
print(f"\n{hdr}\n{'-' * len(hdr)}")
|
||||
first_div: int | None = None
|
||||
for name in sorted(by_name.keys(), key=lambda n: _sort_key(n, drill_layer)):
|
||||
d = by_name[name]
|
||||
up, fv = d.get("up"), d.get("fv")
|
||||
if up is None or fv is None:
|
||||
continue
|
||||
am_diff = abs(up["abs_mean"] - fv["abs_mean"])
|
||||
am_rel = am_diff / max(up["abs_mean"], 1e-9)
|
||||
sum_diff = abs(up["sum"] - fv["sum"])
|
||||
flag = ""
|
||||
if name.startswith("block[") and am_rel > REL_THRESHOLD:
|
||||
flag = " <<< DIVERGE"
|
||||
if first_div is None:
|
||||
first_div = int(name[len("block["):-1])
|
||||
print(
|
||||
f"{name:<18} {str(up['shape']):<22} {up['abs_mean']:>12.6f} "
|
||||
f"{fv['abs_mean']:>12.6f} {am_diff:>14.6f} {am_rel * 100:>7.3f}% "
|
||||
f"{up['sum']:>14.4f} {fv['sum']:>14.4f} {sum_diff:>12.4f}{flag}"
|
||||
)
|
||||
return first_div
|
||||
|
||||
|
||||
def _print_elementwise(up_t: dict[str, torch.Tensor], fv_t: dict[str, torch.Tensor], drill_layer: int) -> None:
|
||||
common = set(up_t.keys()) & set(fv_t.keys())
|
||||
if not common:
|
||||
return
|
||||
hdr = f"{'name':<30} {'shape':<22} {'diff_max':>12} {'diff_mean':>12} {'diff_rel%':>10}"
|
||||
print(f"\nElement-wise diffs for drilled L{drill_layer:02d} submodules:\n{hdr}\n{'-' * len(hdr)}")
|
||||
for name in sorted(common):
|
||||
a, b = up_t[name], fv_t[name]
|
||||
if a.shape != b.shape:
|
||||
continue
|
||||
diff = (a - b).abs()
|
||||
rel = (diff.mean().item() / max(a.abs().mean().item(), 1e-9)) * 100
|
||||
print(
|
||||
f"{name:<30} {str(tuple(a.shape)):<22} "
|
||||
f"{diff.max().item():>12.6f} {diff.mean().item():>12.6f} {rel:>9.4f}%"
|
||||
)
|
||||
|
||||
|
||||
def main() -> None:
|
||||
if not torch.cuda.is_available():
|
||||
print("Need CUDA. Skipping.")
|
||||
return
|
||||
|
||||
# TODO: add precondition checks (official clone present, weights available).
|
||||
|
||||
drill_layer = int(os.getenv(DRILL_LAYER_ENV, "0"))
|
||||
device = torch.device("cuda:0")
|
||||
patched = _apply_hypothesis_patch()
|
||||
try:
|
||||
inputs = build_inputs(device)
|
||||
|
||||
print("Loading official model...")
|
||||
official = load_official(device)
|
||||
up_log: list[dict] = []
|
||||
up_t: dict[str, torch.Tensor] = {}
|
||||
up_handles = _attach_block_hooks(official, "up", up_log, up_t, drill_layer)
|
||||
print("Running official forward (with hooks)...")
|
||||
try:
|
||||
with torch.inference_mode():
|
||||
# TODO: adapt forward call signature to your component.
|
||||
ref_out = official(**{k: v.clone() for k, v in inputs.items()})
|
||||
if isinstance(ref_out, dict):
|
||||
sample = ref_out.get("sample")
|
||||
ref_out = sample if sample is not None else ref_out.get("x")
|
||||
elif hasattr(ref_out, "sample"):
|
||||
ref_out = ref_out.sample
|
||||
elif isinstance(ref_out, tuple):
|
||||
ref_out = ref_out[0]
|
||||
assert torch.is_tensor(ref_out), f"official output is not tensor: {type(ref_out)}"
|
||||
ref_out = ref_out.detach().float().cpu()
|
||||
finally:
|
||||
for h in up_handles:
|
||||
h.remove()
|
||||
del official
|
||||
gc.collect()
|
||||
torch.cuda.empty_cache()
|
||||
|
||||
print("Loading FastVideo model...")
|
||||
fv = load_fastvideo(device)
|
||||
fv_log: list[dict] = []
|
||||
fv_t: dict[str, torch.Tensor] = {}
|
||||
fv_handles = _attach_block_hooks(fv, "fv", fv_log, fv_t, drill_layer)
|
||||
print("Running FastVideo forward (with hooks)...")
|
||||
try:
|
||||
with torch.inference_mode():
|
||||
# TODO: adapt forward call signature to your component.
|
||||
fv_out = fv(**{k: v.clone() for k, v in inputs.items()})
|
||||
if isinstance(fv_out, dict):
|
||||
sample = fv_out.get("sample")
|
||||
fv_out = sample if sample is not None else fv_out.get("x")
|
||||
elif hasattr(fv_out, "sample"):
|
||||
fv_out = fv_out.sample
|
||||
elif isinstance(fv_out, tuple):
|
||||
fv_out = fv_out[0]
|
||||
assert torch.is_tensor(fv_out), f"FastVideo output is not tensor: {type(fv_out)}"
|
||||
fv_out = fv_out.detach().float().cpu()
|
||||
finally:
|
||||
for h in fv_handles:
|
||||
h.remove()
|
||||
finally:
|
||||
_restore_hypothesis_patch()
|
||||
|
||||
LOG_DIR.mkdir(parents=True, exist_ok=True)
|
||||
up_path = LOG_DIR / f"{FAMILY}_{COMPONENT}_up_layers.log"
|
||||
fv_path = LOG_DIR / f"{FAMILY}_{COMPONENT}_fv_layers.log"
|
||||
_write_log(up_log, up_path)
|
||||
_write_log(fv_log, fv_path)
|
||||
print(f"\nLogs: {up_path} {fv_path}\nDiff: diff {up_path} {fv_path}")
|
||||
|
||||
by_name: dict[str, dict] = {}
|
||||
for entry in up_log + fv_log:
|
||||
by_name.setdefault(entry["name"], {})[entry["side"]] = entry
|
||||
first_div = _print_table(by_name, drill_layer)
|
||||
|
||||
print()
|
||||
if first_div is not None:
|
||||
print(f"First block exceeding {REL_THRESHOLD * 100:.2f}% drift: block[{first_div:02d}]")
|
||||
print(f"Re-run with {DRILL_LAYER_ENV}={first_div} to drill submodules.")
|
||||
else:
|
||||
print(f"No block exceeded {REL_THRESHOLD * 100:.2f}% -- divergence is amortized or pre-block.")
|
||||
|
||||
diff = (ref_out - fv_out).abs()
|
||||
print(f"\nFinal ref_abs={ref_out.abs().mean():.6f} fv_abs={fv_out.abs().mean():.6f} "
|
||||
f"diff_max={diff.max():.6f} diff_mean={diff.mean():.6f}")
|
||||
_print_elementwise(up_t, fv_t, drill_layer)
|
||||
if patched:
|
||||
print(f"\n[debug] Hypothesis {HYPOTHESIS_ENV}=1 was active this run.")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,248 @@
|
||||
---
|
||||
name: add-model-09-pipeline
|
||||
description: Use during /add-model Phase 7 after all required component parity tests pass to define FastVideo pipeline wiring, configs, presets, registry entries, examples, smoke tests, and pipeline parity tests.
|
||||
---
|
||||
|
||||
# Add Model Pipeline
|
||||
|
||||
## Goal
|
||||
|
||||
Implement and verify the end-to-end FastVideo pipeline after the native
|
||||
components and converted weights have passed non-skip component parity. This
|
||||
skill owns pipeline class/stage wiring, pipeline configs, presets, registry
|
||||
entries, examples, smoke tests, and pipeline parity-debug.
|
||||
|
||||
FastVideo has one pipeline architecture: stage-based composition through
|
||||
`ComposedPipelineBase`. Add or specialize stages only when existing stages cannot
|
||||
represent the official behavior safely.
|
||||
|
||||
## Hard Gate
|
||||
|
||||
Do not start pipeline work until every required component, including reused
|
||||
components, has a non-skip local parity PASS.
|
||||
|
||||
If any component row is missing, skipped, red, or blocked, return to `/add-model`
|
||||
Phase 6. Pipeline parity cannot distinguish stage wiring mistakes from broken
|
||||
component numerics when component parity is still unresolved.
|
||||
|
||||
## Inputs
|
||||
|
||||
Follow `../add-model/shared/common_rules.md` for token/auth safety, state files,
|
||||
escape hatches, production boundaries, and skip/pass semantics.
|
||||
|
||||
Require a complete packet matching
|
||||
`../add-model/contracts/pipeline_context.md`.
|
||||
|
||||
The packet must include:
|
||||
|
||||
- official pipeline files and official call/default sources;
|
||||
- workload types, input/output modalities, and output contract;
|
||||
- converted or source `model_index.json` path;
|
||||
- component parity rows, all `non_skip_pass`;
|
||||
- target FastVideo pipeline/config/preset/registry/example/test paths;
|
||||
- `local_tests_readme` and `port_state_file` paths.
|
||||
|
||||
## Outputs
|
||||
|
||||
- Pipeline package under `fastvideo/pipelines/basic/<family>/`.
|
||||
- Pipeline config under `fastvideo/configs/pipelines/<family>.py` or a documented
|
||||
family-local config file when that matches existing project style.
|
||||
- Presets under `fastvideo/pipelines/basic/<family>/presets.py`.
|
||||
- Registry updates in `fastvideo/registry.py`.
|
||||
- Basic example under `examples/inference/basic/basic_<family>*.py`.
|
||||
- Local smoke and parity tests under `tests/local_tests/pipelines/`.
|
||||
- Updated `tests/local_tests/<model_family>/README.md`.
|
||||
- Updated `tests/local_tests/<model_family>/PORT_STATUS.md`.
|
||||
- Handoff matching `../add-model/contracts/pipeline_handoff.md`.
|
||||
|
||||
## Mode: Pipeline Definition
|
||||
|
||||
Use this mode first.
|
||||
|
||||
1. Read the official pipeline call path before editing FastVideo code.
|
||||
2. Compare official defaults against the planned FastVideo config and presets:
|
||||
steps, CFG scales, secondary CFG, flow shift, schedulers, sigmas, seed/RNG,
|
||||
resolution, frames, FPS, duration, VAE scaling, decode slicing, negative
|
||||
prompt defaults, and output heads.
|
||||
3. Create or update the pipeline class with `_required_config_modules` matching
|
||||
the emitted `model_index.json` and `ComposedPipelineBase.load_modules`.
|
||||
Runtime pipeline resolution is exact: `model_index.json["_class_name"]` must
|
||||
match a registered `EntryClass.__name__`, or a wrapper/alias class in
|
||||
`EntryClass`. Registry detectors do not select the executable pipeline class.
|
||||
4. Add new public generation kwargs to `fastvideo/api/sampling_param.py` before
|
||||
examples or presets use them. `SamplingParam.update()` ignores unknown keys
|
||||
except for logging, and preset defaults apply only to declared fields. Add CLI
|
||||
args when the option should be available from command-line entrypoints.
|
||||
5. Put loader-time changes in `load_modules()` or earlier, not
|
||||
`initialize_pipeline()`. `ComposedPipelineBase.__init__` loads modules before
|
||||
`post_init()` calls `initialize_pipeline()`, so process-global flags, loader
|
||||
path rewrites, dtype overrides, and tokenizer path changes needed for loading
|
||||
cannot be introduced there.
|
||||
6. Use `self.get_module("transformer_2", None)` and similar optional accessors
|
||||
for truly optional modules. Do not hard-require optional modules by accident.
|
||||
7. Avoid mutating class-level `_required_config_modules` in custom code. If a
|
||||
pipeline needs dynamic modules, copy the list to an instance-owned value or
|
||||
pass `required_config_modules` explicitly so one pipeline instance cannot leak
|
||||
module requirements into another.
|
||||
8. Create the stage chain in official execution order. Prefer existing shared
|
||||
stages for standard text encoding, timestep preparation, latent preparation,
|
||||
denoising, and decoding.
|
||||
9. Add model-specific stages only for family-specific behavior that does not fit
|
||||
the shared stage contracts.
|
||||
10. Add pipeline config classes for wiring and runtime defaults. Do not duplicate
|
||||
component architecture fields unless a loader requires them in the subconfig.
|
||||
Family-local config files such as
|
||||
`fastvideo/pipelines/basic/<family>/pipeline_configs.py` are valid only when
|
||||
`fastvideo/registry.py` imports and registers the classes explicitly.
|
||||
11. Add `InferencePreset` objects with `model_family`, `name`, `version`,
|
||||
`defaults`, optional validation-only `stage_schemas`, and an `ALL_PRESETS`
|
||||
tuple. `stage_schemas` validates user-facing `stage_overrides` names; it does
|
||||
not drive `create_pipeline_stages()` execution.
|
||||
12. Register config classes and presets in `fastvideo/registry.py`: add
|
||||
`register_configs(...)`, import the family's `ALL_PRESETS`, and append it to
|
||||
`_register_presets()`. Detectors should cover HF paths and `_class_name`
|
||||
strings for config/preset lookup, but not as a replacement for exact pipeline
|
||||
class-name resolution.
|
||||
13. Add a basic example with a user-story docstring and normal file-path inputs
|
||||
for image, audio, or video references. Keep orchestration glue in the
|
||||
pipeline or a helper, not in the example.
|
||||
14. Add a separate smoke test
|
||||
`tests/local_tests/pipelines/test_<family>_pipeline_smoke.py` that proves
|
||||
imports, `EntryClass`, registry, presets, config defaults, and at least one
|
||||
real load/generate path when weights are local. Older local tests sometimes
|
||||
colocate smoke checks in parity files; new ports should use the separate file
|
||||
convention.
|
||||
15. Add or update pipeline parity test scaffolding with
|
||||
`templates/pipeline_parity_test.py`.
|
||||
16. Update `local_tests_readme` and `port_state_file` with commands, statuses,
|
||||
default sources, decisions, and blockers.
|
||||
|
||||
Production import boundaries are defined in
|
||||
`../add-model/shared/common_rules.md`.
|
||||
|
||||
## Mode: Pipeline Parity Debug
|
||||
|
||||
Run after pipeline definition and after smoke can execute far enough to load the
|
||||
pipeline. Loop until pipeline parity is a non-skip PASS or a precise blocker is
|
||||
returned.
|
||||
|
||||
Mandatory order:
|
||||
|
||||
```bash
|
||||
pytest tests/local_tests/pipelines/test_<family>_pipeline_smoke.py -v -s
|
||||
DISABLE_SP=1 pytest tests/local_tests/pipelines/test_<family>_pipeline_parity.py -v -s
|
||||
python examples/inference/basic/basic_<family>.py
|
||||
```
|
||||
|
||||
Pipeline parity must compare real outputs, not only successful generation:
|
||||
|
||||
- denoised latents when decode parity is expensive or nondeterministic;
|
||||
- decoded videos/images when visual output should be deterministic enough;
|
||||
- decoded waveform or audio features for audio pipelines;
|
||||
- separate video and audio targets for joint AV pipelines unless a validated
|
||||
joint metric exists.
|
||||
|
||||
Debug pipeline drift in this order:
|
||||
|
||||
1. Confirm both sides use the same component weights and component parity PASS
|
||||
results are still valid.
|
||||
2. Align official and FastVideo call arguments, presets, and default values.
|
||||
3. Align scheduler timesteps, sigmas/noise levels, prediction type, flow shift,
|
||||
guidance math, and secondary-guidance branches.
|
||||
4. Align RNG: initial latents/noise, generator device, seed, per-step noise, VAE
|
||||
sampling, and any official `+1 frame` or crop/slice behavior.
|
||||
5. Align conditioning: prompt templates, negative prompts, masks, image/audio
|
||||
preprocessing, modality packing, text truncation, and dtype/autocast.
|
||||
6. Align decode: latent scaling, per-channel mean/std, tiling flags, output
|
||||
channel order, sample rate, FPS, and final slicing.
|
||||
7. Add targeted stage-level diagnostics to identify the first divergent stage.
|
||||
|
||||
If the first divergence belongs to component implementation, strict loading, or
|
||||
conversion mapping, stop pipeline edits and return `next_step=return_to_phase_6`
|
||||
with the exact failing evidence. Do not patch conversion from this skill.
|
||||
|
||||
## Stage And Variant Rules
|
||||
|
||||
- Canonical video T2V order: `InputValidationStage`, `TextEncodingStage`,
|
||||
`ConditioningStage`, `TimestepPreparationStage`, `LatentPreparationStage`,
|
||||
`DenoisingStage`, `DecodingStage`.
|
||||
- Canonical video I2V delta adds image loading/encoding and image VAE encoding in
|
||||
the official order, commonly: `TextEncodingStage`, `ImageEncodingStage`,
|
||||
`ConditioningStage`, `TimestepPreparationStage`, `LatentPreparationStage`,
|
||||
`ImageVAEEncodingStage`, `DenoisingStage`, `DecodingStage`.
|
||||
- Treat `ConditioningStage` as default-present for Wan-style pipelines, but still
|
||||
follow the reference if another family truly skips or replaces it.
|
||||
- T2V video pipelines usually use validation, text encoding, conditioning,
|
||||
timestep preparation, latent preparation, denoising, and decoding.
|
||||
- I2V adds image loading/encoding and image-latent preparation according to the
|
||||
official pipeline, not by assuming CLIP or Wan-specific branches.
|
||||
- Pick image, audio, and video encoders from the reference. Do not assume CLIP or
|
||||
any other common encoder unless the reference uses it.
|
||||
- Cross-attention class names are not prescribed; match the family style and
|
||||
preserve the official tensor contract.
|
||||
- `WorkloadType` currently has no `T2A`, `A2A`, or `AV` values. Until that enum
|
||||
is extended, audio-only pipelines may register with `WorkloadType.T2V` and
|
||||
preset `workload_type="t2v"` as a compatibility shim, but must document the
|
||||
rationale in code and `PORT_STATUS.md`.
|
||||
- Audio-only pipelines should not force real video semantics into presets. Use
|
||||
minimal video-shaped placeholders such as small `height`/`width` and
|
||||
`num_frames=1` only when shared `VideoGenerator`/validation paths require them,
|
||||
and document that the real output is audio.
|
||||
- Record modality-specific shape knobs and output contract in the pipeline
|
||||
handoff: video uses `height`, `width`, `num_frames`, and `fps`; audio uses
|
||||
`audio_seconds` and `sampling_rate`; joint AV records both plus whether output
|
||||
is muxed or paired files.
|
||||
- Use sibling pipeline classes/configs when required modules, HF repo layout,
|
||||
stage chains, or inputs differ materially.
|
||||
- Use one kwargs-driven pipeline class only when variants share weights,
|
||||
modules, stage chain, and safe call semantics.
|
||||
- Split later if components diverge, workload tags require separate discovery,
|
||||
signatures become unsafe, or stage branches become substantial.
|
||||
- If the DiT branches on `added_kv_proj_dim`, document the T2V/I2V split.
|
||||
- If the reference uses `transformer_2`, `boundary_ratio`, `guidance_scale_2`, or
|
||||
DMD step lists, keep those on config, presets, or stages deliberately.
|
||||
- Support every official output head in scope. If a head is out of scope, record
|
||||
explicit user approval in `PORT_STATUS.md`.
|
||||
|
||||
Pipeline verification order:
|
||||
|
||||
```bash
|
||||
pytest tests/local_tests/pipelines/test_<family>_pipeline_smoke.py -v -s
|
||||
DISABLE_SP=1 pytest -v -s tests/local_tests/pipelines/test_<family>_pipeline_parity.py
|
||||
python examples/inference/basic/basic_<family>.py
|
||||
```
|
||||
|
||||
Smoke tests prove loadability only. They are not a substitute for numerical
|
||||
component or pipeline parity.
|
||||
|
||||
## Escape Hatches
|
||||
|
||||
Follow `../add-model/shared/common_rules.md`. Pipeline-specific ask cases include
|
||||
dropping a public mode, modality, or output head; adding a new workload enum;
|
||||
changing official defaults for user-facing behavior; accepting a known pipeline
|
||||
parity blocker; running GPU-heavy quality work outside the agreed scope; or
|
||||
publishing/uploading generated references or converted weights.
|
||||
|
||||
## Handoff
|
||||
|
||||
Return `../add-model/contracts/pipeline_handoff.md` and update the shared state
|
||||
files before handoff.
|
||||
|
||||
Do not hand back a green pipeline if smoke or parity skipped locally. A skip is a
|
||||
setup gap, not a pass.
|
||||
|
||||
## References
|
||||
|
||||
- `fastvideo/pipelines/composed_pipeline_base.py` for module loading and stage
|
||||
execution.
|
||||
- `fastvideo/pipelines/basic/wan/` for standard video T2V/I2V/DMD variants.
|
||||
- `fastvideo/pipelines/basic/stable_audio/` for audio-specific stage composition.
|
||||
- `fastvideo/configs/pipelines/stable_audio.py` and
|
||||
`fastvideo/pipelines/basic/stable_audio/presets.py` for config/preset shape.
|
||||
- `fastvideo/registry.py` for `register_configs(...)` and preset registration.
|
||||
- `tests/local_tests/pipelines/test_gamecraft_pipeline_parity.py` for latent
|
||||
parity structure.
|
||||
- `tests/local_tests/pipelines/test_stable_audio_pipeline_parity.py` for audio
|
||||
parity structure.
|
||||
- `tests/local_tests/pipelines/test_stable_audio_pipeline_smoke.py` for no-GPU
|
||||
import/registry/preset preflight shape.
|
||||
@@ -0,0 +1,153 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
"""Pipeline parity scaffold for TODO_MODEL_FAMILY.
|
||||
|
||||
Copy this file to
|
||||
`tests/local_tests/pipelines/test_<family>_pipeline_parity.py` and replace every
|
||||
TODO before treating it as an executable scaffold.
|
||||
|
||||
The filled test should compare denoised latents, decoded media, audio waveform,
|
||||
or another concrete output from the official pipeline against FastVideo. A
|
||||
successful generation without tensor/media comparison is not parity.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
import pytest
|
||||
import torch
|
||||
from torch.testing import assert_close
|
||||
|
||||
_REPO_ROOT = Path(__file__).resolve().parents[3]
|
||||
_MODEL_FAMILY = "TODO_MODEL_FAMILY"
|
||||
_OFFICIAL_REF_ENV = "TODO_OFFICIAL_REF_PATH"
|
||||
_OFFICIAL_REF_DEFAULT = _REPO_ROOT / "TODO_OFFICIAL_REF_DIR"
|
||||
_FASTVIDEO_MODEL_ENV = "TODO_FASTVIDEO_MODEL_PATH"
|
||||
_FASTVIDEO_MODEL_DEFAULT = _REPO_ROOT / "converted_weights" / _MODEL_FAMILY
|
||||
|
||||
|
||||
def _path_from_env(env_name: str, default: Path) -> Path:
|
||||
return Path(os.getenv(env_name, str(default))).expanduser()
|
||||
|
||||
|
||||
def _add_official_to_path() -> Path:
|
||||
official_path = _path_from_env(_OFFICIAL_REF_ENV, _OFFICIAL_REF_DEFAULT)
|
||||
if not official_path.exists():
|
||||
pytest.skip(f"Official reference not found at {official_path}")
|
||||
if str(official_path) not in sys.path:
|
||||
sys.path.insert(0, str(official_path))
|
||||
return official_path
|
||||
|
||||
|
||||
def _log_tensor_stats(label: str, tensor: torch.Tensor) -> None:
|
||||
value = tensor.detach().float()
|
||||
print(
|
||||
f"[{_MODEL_FAMILY} PIPELINE] {label}: shape={tuple(tensor.shape)} "
|
||||
f"dtype={tensor.dtype} device={tensor.device} "
|
||||
f"min={value.min().item():.6f} max={value.max().item():.6f} "
|
||||
f"mean={value.mean().item():.6f} std={value.std().item():.6f}"
|
||||
)
|
||||
|
||||
|
||||
def _extract_tensor(output: Any, key: str) -> torch.Tensor:
|
||||
if isinstance(output, dict):
|
||||
value = output.get(key)
|
||||
else:
|
||||
value = getattr(output, key, None)
|
||||
if value is None:
|
||||
raise AssertionError(f"Pipeline output did not contain {key!r}")
|
||||
if not torch.is_tensor(value):
|
||||
try:
|
||||
import numpy as np
|
||||
value = torch.from_numpy(np.asarray(value))
|
||||
except Exception as exc: # pragma: no cover - scaffold guard
|
||||
raise AssertionError(f"Could not convert {key!r} to tensor") from exc
|
||||
return value.detach().float().cpu()
|
||||
|
||||
|
||||
def _run_official_pipeline(
|
||||
official_path: Path,
|
||||
params: dict[str, Any],
|
||||
device: torch.device,
|
||||
) -> Any:
|
||||
del official_path, params, device
|
||||
pytest.skip(
|
||||
"TODO: import the official pipeline/factory, load official weights, "
|
||||
"run with params, and return the comparison target."
|
||||
)
|
||||
|
||||
|
||||
def _run_fastvideo_pipeline(model_path: Path, params: dict[str, Any]) -> Any:
|
||||
from fastvideo import VideoGenerator
|
||||
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
str(model_path),
|
||||
num_gpus=1,
|
||||
use_fsdp_inference=False,
|
||||
dit_cpu_offload=False,
|
||||
vae_cpu_offload=False,
|
||||
text_encoder_cpu_offload=False,
|
||||
)
|
||||
try:
|
||||
return generator.generate_video(
|
||||
prompt=params["prompt"],
|
||||
negative_prompt=params.get("negative_prompt"),
|
||||
output_path=f"outputs_{_MODEL_FAMILY}/pipeline_parity",
|
||||
save_video=False,
|
||||
height=params.get("height"),
|
||||
width=params.get("width"),
|
||||
num_frames=params.get("num_frames"),
|
||||
fps=params.get("fps"),
|
||||
num_inference_steps=params["num_inference_steps"],
|
||||
guidance_scale=params.get("guidance_scale"),
|
||||
seed=params["seed"],
|
||||
)
|
||||
finally:
|
||||
generator.shutdown()
|
||||
|
||||
|
||||
@pytest.mark.skipif(
|
||||
not torch.cuda.is_available(),
|
||||
reason="TODO_MODEL_FAMILY pipeline parity requires CUDA.",
|
||||
)
|
||||
def test_todo_model_family_pipeline_official_parity() -> None:
|
||||
official_path = _add_official_to_path()
|
||||
fastvideo_model_path = _path_from_env(
|
||||
_FASTVIDEO_MODEL_ENV,
|
||||
_FASTVIDEO_MODEL_DEFAULT,
|
||||
)
|
||||
if not fastvideo_model_path.exists():
|
||||
pytest.skip(f"FastVideo model path not found at {fastvideo_model_path}")
|
||||
|
||||
device = torch.device("cuda:0")
|
||||
params = {
|
||||
"prompt": "TODO: stable parity prompt",
|
||||
"negative_prompt": "",
|
||||
"height": 64,
|
||||
"width": 64,
|
||||
"num_frames": 9,
|
||||
"fps": 8,
|
||||
"num_inference_steps": 4,
|
||||
"guidance_scale": 1.0,
|
||||
"seed": 0,
|
||||
}
|
||||
|
||||
official_output = _run_official_pipeline(official_path, params, device)
|
||||
fastvideo_output = _run_fastvideo_pipeline(fastvideo_model_path, params)
|
||||
|
||||
comparison_key = "TODO_COMPARISON_KEY"
|
||||
official_tensor = _extract_tensor(official_output, comparison_key)
|
||||
fastvideo_tensor = _extract_tensor(fastvideo_output, comparison_key)
|
||||
|
||||
_log_tensor_stats("official", official_tensor)
|
||||
_log_tensor_stats("fastvideo", fastvideo_tensor)
|
||||
assert official_tensor.shape == fastvideo_tensor.shape
|
||||
|
||||
diff = (official_tensor - fastvideo_tensor).abs()
|
||||
print(
|
||||
f"diff max={diff.max().item():.6f} "
|
||||
f"mean={diff.mean().item():.6f} median={diff.median().item():.6f}"
|
||||
)
|
||||
assert_close(fastvideo_tensor, official_tensor, atol=1e-2, rtol=1e-2)
|
||||
@@ -0,0 +1,141 @@
|
||||
---
|
||||
name: add-model-10-pr-review
|
||||
description: Review rubric for FastVideo PRs that add or modify model families, variants, first-class components, checkpoint conversion, pipelines, parity coverage, or generated-media quality baselines. Use when reviewing a PR whose diff touches fastvideo/models/, fastvideo/pipelines/basic/, fastvideo/registry.py, scripts/checkpoint_conversion/, fastvideo/tests/ssim/, or related model-port surfaces. Pairs with review-pr-link as a project-scoped review pass; produces findings, not fixes.
|
||||
---
|
||||
|
||||
# Add-Model PR Review
|
||||
|
||||
Use this skill when a reviewed PR appears to add, port, or substantially modify
|
||||
a FastVideo model family, model variant, first-class model component,
|
||||
checkpoint conversion, model pipeline, or local parity coverage.
|
||||
|
||||
This is a review skill, not an implementation workflow. Do not run `/add-model`
|
||||
or start writing missing port code during review. Use the add-model skill stack
|
||||
as a rubric for findings.
|
||||
|
||||
## Trigger Paths
|
||||
|
||||
Trigger this skill if `git diff --name-only <base>...HEAD` includes any of:
|
||||
|
||||
- `fastvideo/models/dits/`, `fastvideo/configs/models/dits/`
|
||||
- `fastvideo/models/vaes/`, `fastvideo/configs/models/vaes/`
|
||||
- `fastvideo/models/encoders/`, `fastvideo/configs/models/encoders/`
|
||||
- `fastvideo/models/schedulers/`, `fastvideo/configs/models/schedulers/`
|
||||
- `fastvideo/models/upsamplers/`, `fastvideo/configs/models/upsamplers/`
|
||||
- `fastvideo/models/audio/`, `fastvideo/configs/models/audio/`
|
||||
- `fastvideo/pipelines/basic/`, `fastvideo/configs/pipelines/`
|
||||
- `fastvideo/registry.py`, `fastvideo/api/sampling_param.py`
|
||||
- `scripts/checkpoint_conversion/`
|
||||
- `examples/inference/basic/`
|
||||
- `tests/local_tests/`, especially component or pipeline parity tests
|
||||
- `fastvideo/tests/ssim/` or other quality-regression tests for generated media
|
||||
|
||||
Also trigger when the PR title/body claims a new model, model variant, VAE,
|
||||
encoder, scheduler, conditioner, pipeline, conversion script, or generated-media
|
||||
quality baseline even if the path list is incomplete.
|
||||
|
||||
## Review Inputs
|
||||
|
||||
Read these add-model references as review checklists:
|
||||
|
||||
- `../add-model/SKILL.md`: phase gates and final handoff requirements.
|
||||
- `../add-model/shared/common_rules.md`: token/auth safety, production import
|
||||
boundaries, state files, and skip/pass semantics.
|
||||
- `../add-model/contracts/final_handoff.md`: final evidence expected from a
|
||||
complete port.
|
||||
- `../add-model/contracts/component_context.md` and
|
||||
`../add-model/contracts/component_skill_handoff.md`: component evidence and
|
||||
parity-debug expectations.
|
||||
- `../add-model/contracts/conversion_request.md` and
|
||||
`../add-model/contracts/conversion_handoff.md`: conversion evidence,
|
||||
strict-load status, config validation, and retry context.
|
||||
- `../add-model/contracts/pipeline_context.md` and
|
||||
`../add-model/contracts/pipeline_handoff.md`: pipeline class/stage/config/
|
||||
preset/registry/example evidence.
|
||||
|
||||
Then read only the satellite skill(s) that match touched areas:
|
||||
|
||||
- DiT/transformer changes: `../add-model-03-port-dit/SKILL.md`.
|
||||
- VAE changes: `../add-model-04-port-vae/SKILL.md`.
|
||||
- Encoder/conditioner changes: `../add-model-05-port-encoder/SKILL.md`.
|
||||
- Scheduler/upsampler/vocoder/other components:
|
||||
`../add-model-06-port-generic/SKILL.md`.
|
||||
- Component parity tests: `../add-model-02-parity/SKILL.md`.
|
||||
- Checkpoint conversion: `../add-model-07-conversion/SKILL.md`.
|
||||
- Pipeline/config/presets/registry/examples:
|
||||
`../add-model-09-pipeline/SKILL.md`.
|
||||
- Prep/state docs: `../add-model-01-prep/SKILL.md`.
|
||||
|
||||
## Required Review Lanes
|
||||
|
||||
For a full model-family or model-variant PR, cover all lanes. For a
|
||||
component-only PR, cover the component, conversion/parity as applicable, and the
|
||||
documented downstream consumer.
|
||||
|
||||
1. Scope and source-of-truth lane:
|
||||
Verify the PR clearly identifies the official reference, weights/revision,
|
||||
supported variants, modalities, output heads, and any approved scope cuts.
|
||||
|
||||
2. Component lane:
|
||||
Verify each required component is FastVideo-native or has a documented and
|
||||
accepted lazy-wrapper exception. Check bucket/config inheritance, `EntryClass`,
|
||||
state-dict surface, reused-component evidence, and output heads.
|
||||
|
||||
3. Conversion lane:
|
||||
Verify mappings are derived from prototype key/shape dumps, source layout is
|
||||
supported, skipped keys are intentional, emitted configs validate through
|
||||
production paths, component strict-load status is recorded, `model_index.json`
|
||||
library tokens match loaders, and revisions are pinned when converting from
|
||||
HF.
|
||||
|
||||
4. Component parity lane:
|
||||
Verify local parity tests exist for every required component, including reused
|
||||
components. Scaffolds may skip in CI, but the PR must provide local non-skip
|
||||
PASS evidence or an explicit accepted blocker.
|
||||
|
||||
5. Pipeline lane:
|
||||
Verify stage order, required modules, `_class_name` / `EntryClass.__name__`
|
||||
resolution, config defaults, presets, `SamplingParam` fields, registry
|
||||
registration, examples, smoke tests, and pipeline parity.
|
||||
|
||||
6. Quality and evidence lane:
|
||||
Verify media quality regression is added or explicitly deferred, examples run,
|
||||
generated outputs are non-corrupt, `tests/local_tests/<family>/README.md` and
|
||||
`PORT_STATUS.md` are current, and final blockers are surfaced in the review.
|
||||
|
||||
## Findings To Prioritize
|
||||
|
||||
Prioritize review findings in this order:
|
||||
|
||||
- Missing or skipped required component parity without accepted blocker.
|
||||
- Pipeline parity/smoke/example missing or skipped for a pipeline PR.
|
||||
- Conversion emits unloadable or unvalidated configs/weights.
|
||||
- Wrong `model_index.json` `_class_name`, component library token, or registry
|
||||
class resolution.
|
||||
- Runtime diffusers/transformers model-class imports for components that own
|
||||
weights or numerical behavior.
|
||||
- Dropped modalities, output heads, variants, or conditioning streams without
|
||||
explicit approval.
|
||||
- Reused FastVideo component lacks exact definition/instantiation proof or
|
||||
non-skip parity.
|
||||
- Public generation kwargs/preset defaults missing from `SamplingParam`.
|
||||
- Tests only check shapes, importability, or successful generation without
|
||||
numerical/media comparison.
|
||||
- Tokens, credentials, reference clones, staged weights, or generated bulk assets
|
||||
committed to the PR.
|
||||
|
||||
## Output Format
|
||||
|
||||
Write normal code-review findings first, ordered by severity. Include file and
|
||||
line references from the PR diff when possible.
|
||||
|
||||
Use this phrasing for missing add-model evidence:
|
||||
|
||||
```text
|
||||
This PR does not satisfy the add-model <component|conversion|pipeline|final>
|
||||
gate because <specific required evidence> is missing. The risk is <runtime load,
|
||||
numerical parity, dropped output, registry resolution, etc.>.
|
||||
```
|
||||
|
||||
Keep the summary short. Mention which lanes were reviewed and which could not be
|
||||
verified because assets, GPU time, or external credentials were unavailable.
|
||||
@@ -0,0 +1,26 @@
|
||||
# add-model skill review backlog (historical)
|
||||
|
||||
Updated 2026-04-30 after the phase-based `/add-model` rewrite.
|
||||
|
||||
All skill-text items from the prior review were incorporated into the current
|
||||
split skill stack under `.agents/skills/add-model*`. This file is kept
|
||||
only for codebase-owner follow-ups that are not blockers for the skill workflow.
|
||||
|
||||
### 1. Audit `wan_to_diffusers.py` usage
|
||||
|
||||
`SKILL.md` now treats `scripts/checkpoint_conversion/wan_to_diffusers.py` as a
|
||||
legacy regex-reference file, not a conversion-script template.
|
||||
|
||||
Open codebase question: is this module still imported by live code? If yes,
|
||||
document the caller near the script or in developer docs. If no, delete it in a
|
||||
separate cleanup PR.
|
||||
|
||||
### 2. Decide Whether To Add Audio Workload Enums
|
||||
|
||||
The current pipeline skill documents the repository's compatibility workaround:
|
||||
until `WorkloadType` grows audio values, audio-only pipelines may register as
|
||||
`T2V` with explicit rationale and minimal video-shaped placeholders when shared
|
||||
`VideoGenerator` paths require them.
|
||||
|
||||
Open codebase question: should `WorkloadType` be extended now with audio and
|
||||
joint AV variants, or should the first audio pipeline PR own that enum change?
|
||||
@@ -0,0 +1,435 @@
|
||||
---
|
||||
name: add-model
|
||||
description: Manual /add-model workflow for implementing a FastVideo model or first-class component port after add-model-01-prep has staged reference code and weights. Organizes the port into numbered phases with conversion rules, component policies, parity gates, and handoff checks.
|
||||
---
|
||||
|
||||
# Add Model
|
||||
|
||||
## Manual Invocation
|
||||
|
||||
This skill is for explicit `/add-model` use only. Do not auto-start it from a
|
||||
casual model-port mention. The setup-only workflow is
|
||||
`../add-model-01-prep/SKILL.md`.
|
||||
|
||||
## Goal
|
||||
|
||||
Port a new FastVideo model family, model variant, or first-class reusable
|
||||
component so it can be loaded through FastVideo's native model, config, stage,
|
||||
registry, preset, and test infrastructure.
|
||||
|
||||
FastVideo has one pipeline architecture: stage-based composition via
|
||||
`ComposedPipelineBase`. Vary the stages and modules, not the architecture.
|
||||
|
||||
## Scope Shapes
|
||||
|
||||
Use this skill for either shape:
|
||||
|
||||
| Shape | Required output |
|
||||
|---|---|
|
||||
| Full model family or variant | Native components, conversion if needed, pipeline config/class, presets, registry, smoke test, local parity tests, example, quality regression. |
|
||||
| First-class component contribution | Native component class/config, bucket export, component parity test, and a documented downstream pipeline that will consume it. Skip pipeline/preset/registry rows only when the contribution is intentionally component-only. |
|
||||
|
||||
If upstream ships many variants, lock scope before coding. "Base model" means
|
||||
checkpoint variant, not a modality subset. If the base checkpoint produces
|
||||
audio, pose, depth, masks, or other output heads, either support those outputs
|
||||
or get explicit user agreement to drop them.
|
||||
|
||||
## Required Input
|
||||
|
||||
Start from an `add-model-01-prep` handoff, or equivalent fields matching
|
||||
`contracts/prep_handoff.md`.
|
||||
|
||||
Before Phase 0, read the shared rules and all relevant schemas:
|
||||
|
||||
- `shared/common_rules.md`
|
||||
- `contracts/prep_handoff.md`
|
||||
- `contracts/port_state.md`
|
||||
- `contracts/escape_hatch.md`
|
||||
- `contracts/component_context.md`
|
||||
- `contracts/parity_status.md`
|
||||
- `contracts/conversion_request.md`
|
||||
- `contracts/conversion_handoff.md`
|
||||
- `contracts/component_skill_handoff.md`
|
||||
- `contracts/pipeline_context.md`
|
||||
- `contracts/pipeline_handoff.md`
|
||||
- `contracts/final_handoff.md`
|
||||
|
||||
## Hard Rules
|
||||
|
||||
- Follow `shared/common_rules.md` for token/auth safety, state files, escape
|
||||
hatches, production import boundaries, and skip/pass semantics.
|
||||
- If the prep handoff is missing or ambiguous, stop and run
|
||||
`../add-model-01-prep/SKILL.md`.
|
||||
- If a needed component is not ported, do not ship the pipeline that needs it.
|
||||
- Wan is grandfathered for missing local parity; do not copy its missing-test
|
||||
precedent for new work.
|
||||
|
||||
## Escape Hatches
|
||||
|
||||
Follow `shared/common_rules.md` and `contracts/escape_hatch.md`. The main
|
||||
orchestrator should ask only when no phase skill can safely continue under the
|
||||
shared rules.
|
||||
|
||||
## Files Map
|
||||
|
||||
| Area | Paths |
|
||||
|---|---|
|
||||
| DiT | `fastvideo/models/dits/<family>.py`, `fastvideo/configs/models/dits/<family>.py`, bucket `__init__.py`. |
|
||||
| VAE | `fastvideo/models/vaes/<arch_or_family>.py`, `fastvideo/configs/models/vaes/<arch_or_family>.py`, bucket `__init__.py`. Name by shared arch when reusable (`oobleck.py`, `autoencoder_kl.py`), otherwise by family (`wanvae.py`). |
|
||||
| Encoder / conditioner / scheduler / upsampler | Native class/config in the matching `fastvideo/models/<bucket>/` and `fastvideo/configs/models/<bucket>/` bucket. |
|
||||
| Lazy loader wrapper | Optional `fastvideo/models/<bucket>/<family>_loader.py` or similar thin `nn.Module` wrapper when a component is fetched from an external HF repo and should be hidden from host-pipeline state-dict matching. |
|
||||
| Conversion | `scripts/checkpoint_conversion/<family>_to_diffusers.py` only when `needs_conversion=yes`. |
|
||||
| Pipeline | `fastvideo/pipelines/basic/<family>/<family>_pipeline.py` plus sibling files for variants whose components or required modules differ. |
|
||||
| Pipeline config | `fastvideo/configs/pipelines/<family>.py` or `fastvideo/pipelines/basic/<family>/pipeline_configs.py`. |
|
||||
| Stages | `fastvideo/pipelines/basic/<family>/stages/` only for model-specific stage subclasses. |
|
||||
| Presets / registry | `fastvideo/pipelines/basic/<family>/presets.py`, `fastvideo/registry.py`. |
|
||||
| Tests | Component parity under `tests/local_tests/<bucket>/`; pipeline smoke/parity under `tests/local_tests/pipelines/`; CI-backed quality tests under `fastvideo/tests/`. |
|
||||
| Example | `examples/inference/basic/basic_<family>*.py`, one per public mode/variant. |
|
||||
|
||||
## Phase 0: Scope And Handoff Gate
|
||||
|
||||
1. Validate every required handoff field.
|
||||
2. Resolve `needs_conversion=unknown` before component work:
|
||||
|
||||
```bash
|
||||
python ".agents/skills/add-model-01-prep/scripts/inspect_hf_layout.py" \
|
||||
"<hf-or-local-path>" \
|
||||
--json
|
||||
```
|
||||
|
||||
3. List first-PR scope across both axes:
|
||||
- Variant axis: base, distill, SR/refine, causal, DMD, I2V, V2V, etc.
|
||||
- Modality axis: video, image, audio, pose, depth, masks, text, etc.
|
||||
4. For component-only work, explicitly name the downstream full-pipeline PR or
|
||||
planned consumer.
|
||||
5. Confirm `official_env_status` is `imports_ok` or
|
||||
`private_deps_need_stubs`. If it is `blocked`, return to
|
||||
`../add-model-01-prep/SKILL.md` before parity scaffolding.
|
||||
6. Confirm `local_tests_readme` exists and records official setup, HF weights,
|
||||
dependency changes, and planned parity commands for reviewers.
|
||||
7. Confirm `port_state_file` exists, follows `contracts/port_state.md`, and has
|
||||
rows for open questions/issues found during prep.
|
||||
8. If there are multiple official implementations, choose the one whose
|
||||
architecture matches the published weights. A blessed library port can be a
|
||||
better parity reference than a highly configurable research repo; document
|
||||
the choice in tests.
|
||||
|
||||
## Phase 1: Reference And Architecture Study
|
||||
|
||||
Read the official pipeline call path before writing code.
|
||||
|
||||
Record:
|
||||
|
||||
- Required modules from `model_index.json` or equivalent: transformer, VAE,
|
||||
text encoders, tokenizers, scheduler, image encoders, audio VAE, vocoder,
|
||||
conditioners, upsamplers.
|
||||
- Input/output modalities and every dedicated DiT output head.
|
||||
- Text/image/audio encoding flow, latent shape, dtype, scaling, packing,
|
||||
scheduler/timestep math, guidance math, VAE normalization, and decode flow.
|
||||
- Whether the official code relies on private deps, custom ops, or special
|
||||
kernels that parity tests must stub.
|
||||
|
||||
Arch config rule:
|
||||
|
||||
- `ArchConfig` fields must match the emitted per-component config, especially
|
||||
`transformer/config.json`, one-to-one.
|
||||
- Pipeline knobs do not belong on the DiT arch config: inference steps, CFG
|
||||
scales, flow shift, FPS, VAE stride, text target length, data-proxy knobs,
|
||||
eval defaults, and sampling defaults go on `PipelineConfig`, presets, or
|
||||
stages.
|
||||
- If the HF repo is raw or has empty configs, synthesize
|
||||
`transformer/config.json` from the official Python model-config class, not
|
||||
from data/eval config classes.
|
||||
|
||||
## Phase 2: Early Parity Scaffolding
|
||||
|
||||
Create component parity tests before or alongside implementation. Use
|
||||
`../add-model-02-parity/SKILL.md` and its `templates/component_parity_test.py`.
|
||||
The official reference must import in the current FastVideo environment, or the
|
||||
prep handoff must identify private deps that will be stubbed locally for tests.
|
||||
Use `local_tests_readme` as the reviewer-facing source for setup commands and
|
||||
update its planned test table as parity scaffolds are added.
|
||||
|
||||
This phase is early by design:
|
||||
|
||||
- Official loading can be implemented from the reference study.
|
||||
- FastVideo loading can target planned standardized class/config/loader paths.
|
||||
- Tests may initially skip because the FastVideo class or converted weights do
|
||||
not exist yet.
|
||||
- The scaffold must still contain real official loading, deterministic inputs,
|
||||
output extraction, and concrete tensor comparisons. No unconditional skips,
|
||||
no shape-only tests.
|
||||
|
||||
Use subagents here: dispatch one parity-test subagent per required component,
|
||||
including components that may be reused. Their output becomes the red/skip
|
||||
target that porting or reuse-verification subagents make pass later.
|
||||
|
||||
## Phase 3: Reuse Gate And Component Dispatch
|
||||
|
||||
Build a component inventory before implementation:
|
||||
|
||||
| Field | Meaning |
|
||||
|---|---|
|
||||
| Component | transformer, VAE, text encoder, image encoder, scheduler, conditioner, upsampler, vocoder, etc. |
|
||||
| Official definition | Repo-relative source file, class/function name, and relevant line/range if known. |
|
||||
| Official instantiation | Repo-relative pipeline/config/factory call site plus constructor args and runtime flags. |
|
||||
| FastVideo target | Existing class to reuse or new bucket/file/config to add. |
|
||||
| Parity test | Required local test path, including reused components. |
|
||||
| Status | `reuse_pending`, `reuse_proven`, `port_pending`, `non_skip_pass`, or `blocked`. |
|
||||
|
||||
Reuse is allowed only from the checked-out FastVideo tree. Do not wait for or
|
||||
depend on an open PR adding a native class; add the native port directly in this
|
||||
PR if the current tree cannot be reused.
|
||||
|
||||
Reuse decision:
|
||||
|
||||
1. Record exact official definition and instantiation evidence for every
|
||||
component.
|
||||
2. If an existing FastVideo class and config match both definition and
|
||||
instantiation, pass that reused target to the bucket-specific skill in
|
||||
`mode=prototype` and require reuse evidence plus key/shape dumps.
|
||||
3. If either definition or instantiation differs, port the component directly as
|
||||
FastVideo-native code through the bucket-specific skill.
|
||||
4. Reused components still require non-skip component parity against the exact
|
||||
official instantiation used by the target pipeline.
|
||||
|
||||
Porting subagent dispatch:
|
||||
|
||||
- Dispatch one subagent per component after Phase 2 parity scaffolds exist.
|
||||
- Use `../add-model-03-port-dit/SKILL.md` for DiTs/transformers.
|
||||
- Use `../add-model-04-port-vae/SKILL.md` for VAEs.
|
||||
- Use `../add-model-05-port-encoder/SKILL.md` for text, image, audio, or compound
|
||||
encoders/conditioners that fit the encoder config bucket.
|
||||
- Use `../add-model-06-port-generic/SKILL.md` for schedulers, upsamplers,
|
||||
vocoders, adapters, preprocessors, or unknown components.
|
||||
- Each subagent owns one component only and must loop on that component's local
|
||||
parity test until it produces a non-skip PASS or returns a precise blocker.
|
||||
|
||||
Every component subagent must receive a complete packet matching
|
||||
`contracts/component_context.md`. If any required path is unknown, pass `unknown`
|
||||
plus the exact search already performed. Do not silently omit ambiguous official
|
||||
files or prototype concerns.
|
||||
|
||||
Bucket, layer, and attention rules live in the bucket-specific skills and
|
||||
`fastvideo/layers/AGENTS.md`.
|
||||
|
||||
## Phase 4: Native Component Prototype
|
||||
|
||||
Conversion needs a FastVideo state-dict surface. Use the Phase 3
|
||||
bucket-specific skill in `mode=prototype` for every required component, including
|
||||
reused components.
|
||||
|
||||
Prototype success criteria:
|
||||
|
||||
- the FastVideo-native or reused class/config can import and instantiate with the
|
||||
exact official architecture args;
|
||||
- official and FastVideo key/shape dumps exist for every stateful component;
|
||||
- `local_tests_readme` and `port_state_file` record prototype status and concerns;
|
||||
- the returned handoff matches `contracts/component_skill_handoff.md`.
|
||||
|
||||
Do not chase numerical parity in Phase 4. Prototype mode ends when conversion has
|
||||
the key/shape surface it needs, or when the component skill returns a precise
|
||||
blocker or escape hatch.
|
||||
|
||||
## Phase 5: Param Mapping And Weight Conversion
|
||||
|
||||
Use `../add-model-07-conversion/SKILL.md` after Phase 4 prototypes exist.
|
||||
Send a request matching `contracts/conversion_request.md`; consume the returned
|
||||
`contracts/conversion_handoff.md` update before Phase 6.
|
||||
|
||||
Use the prep handoff's `needs_conversion` value:
|
||||
|
||||
- `no`: verify the source already has the component layout FastVideo loaders can
|
||||
consume, then record any passthrough components.
|
||||
- `yes`: write `scripts/checkpoint_conversion/<family>_to_diffusers.py` and
|
||||
output `converted_weights/<family>/`.
|
||||
- `unknown`: return to Phase 0.
|
||||
|
||||
The conversion skill owns source-layout handling, mapping derivation, config and
|
||||
`model_index.json` emission, passthrough assets, strict-load verification, and
|
||||
Phase 6 retry requests. Component skills must not patch conversion scripts or
|
||||
converted weights ad hoc.
|
||||
|
||||
## Phase 6: Component Parity Debug
|
||||
|
||||
This is the expected expensive loop. Dispatch one subagent per required
|
||||
component, including reused components, using the bucket-specific skill in
|
||||
`mode=parity-debug`.
|
||||
|
||||
Each subagent gets:
|
||||
|
||||
- the complete component context packet from Phase 3/4;
|
||||
- updated conversion mapping notes and strict-load result from Phase 5;
|
||||
- any prototype concerns or unknowns that were not resolved before conversion.
|
||||
|
||||
The bucket-specific skills own parity-debug tactics. If a failure belongs to
|
||||
conversion, route it through `../add-model-07-conversion/SKILL.md` with a retry
|
||||
request matching `contracts/conversion_request.md`, then resume the component
|
||||
skill with the updated conversion handoff.
|
||||
|
||||
Phase 6 ends only when every required component handoff reports
|
||||
`parity_status=non_skip_pass`, or when a precise blocker or escape hatch is
|
||||
recorded in `port_state_file`.
|
||||
|
||||
## Phase 7: Pipeline, Stages, And Variants
|
||||
|
||||
Do not start Phase 7 until every required component, reused or ported, has a
|
||||
non-skip local parity PASS from Phase 6. If any component parity test is still
|
||||
`scaffold_skip`, `debug_red`, `blocked`, or missing, resume Phase 6 first.
|
||||
|
||||
Use `../add-model-09-pipeline/SKILL.md` for pipeline definition and parity-debug.
|
||||
Send a complete packet matching `contracts/pipeline_context.md`; consume the
|
||||
returned `contracts/pipeline_handoff.md` before moving to quality regression or
|
||||
final handoff.
|
||||
|
||||
The pipeline skill owns:
|
||||
|
||||
- pipeline class, stage chain, and optional model-specific stages;
|
||||
- pipeline config, presets, registry updates, and examples;
|
||||
- official args/defaults/presets comparison before setting FastVideo defaults;
|
||||
- pipeline smoke and parity tests;
|
||||
- continuous pipeline parity-debug until non-skip PASS or precise blocker;
|
||||
- updates to `local_tests_readme` and `port_state_file`.
|
||||
|
||||
The pipeline handoff must explicitly cover stage order, variants, modality and
|
||||
output-head handling, config/preset/registry/example status, smoke/parity tests,
|
||||
and any return-to-Phase-6 evidence.
|
||||
|
||||
## Phase 8: PipelineConfig, Presets, Registry, Examples
|
||||
|
||||
This phase is implemented through `../add-model-09-pipeline/SKILL.md` after the
|
||||
Phase 7 component-parity gate passes. Accept the pipeline handoff only if it
|
||||
covers configs, presets, registry detection/exact class resolution, examples,
|
||||
new `SamplingParam` fields for public kwargs/defaults, and local smoke/parity
|
||||
status. Detailed rules live in `../add-model-09-pipeline/SKILL.md`.
|
||||
|
||||
## Phase 9: Parity Activation And Local Verification
|
||||
|
||||
Local parity is author-run, not CI-enforced. CI may only run package-level
|
||||
quality tests later. Before handoff, Phase 2 scaffolds must be activated into
|
||||
non-skip PASS results.
|
||||
|
||||
Order is mandatory:
|
||||
|
||||
1. Run conversion if needed.
|
||||
2. Run component parity for every required component, including reused ones.
|
||||
3. Run pipeline smoke.
|
||||
4. Run pipeline parity.
|
||||
5. Run the basic example.
|
||||
|
||||
If pipeline smoke or parity points back to component implementation,
|
||||
strict-load, or conversion mapping, return to Phase 6 or Phase 5 rather than
|
||||
patching around the issue in the pipeline.
|
||||
|
||||
Skip policy:
|
||||
|
||||
- Follow `shared/common_rules.md`: a committed local test may skip for absent
|
||||
clones/weights, but a local skip is not a verified pass.
|
||||
|
||||
Use the commands and tolerance guidance from `../add-model-02-parity/SKILL.md` for
|
||||
component checks and from `../add-model-09-pipeline/SKILL.md` for pipeline smoke,
|
||||
pipeline parity, and examples. Record exact commands, status, and blockers in
|
||||
`local_tests_readme` and `port_state_file`.
|
||||
|
||||
## Phase 10: Quality Regression
|
||||
|
||||
Video outputs:
|
||||
|
||||
- Add `fastvideo/tests/ssim/test_<family>_similarity.py` when output video
|
||||
quality must be preserved.
|
||||
- Seed references through `seed-ssim-references` after the test exists.
|
||||
|
||||
Audio outputs:
|
||||
|
||||
- SSIM does not apply. Use an audio-specific regression metric such as
|
||||
mel-spectrogram L1, multi-resolution STFT, CLAP cosine, or a project-approved
|
||||
learned metric.
|
||||
- Document the metric and hardware/runtime assumptions in the test.
|
||||
|
||||
Joint AV outputs:
|
||||
|
||||
- Keep video and audio regression checks separate unless there is a validated
|
||||
joint metric.
|
||||
|
||||
## Phase 11: Post-Parity Review And Handoff
|
||||
|
||||
After parity is green, run a hot-path review before handoff:
|
||||
|
||||
- Hoist constant tensor allocations out of sampler/denoising loops.
|
||||
- Replace per-step `randn_like` churn with preallocated buffers plus
|
||||
`.normal_()` when safe.
|
||||
- Move `torch.backends.*` flag changes to one-shot setup/load paths.
|
||||
- Delete `batch.extra` writes that nothing reads.
|
||||
- Derive magic constants from configs when possible.
|
||||
|
||||
Pre-handoff checklist:
|
||||
|
||||
```text
|
||||
[ ] Prep handoff is complete and committed nowhere with token values.
|
||||
[ ] Conversion was run if needed and output loads with real weights.
|
||||
[ ] Every required component, reused or newly ported, has a non-skip local parity PASS.
|
||||
[ ] `local_tests_readme` lists every component parity test, command, status, and blocker if any.
|
||||
[ ] `port_state_file` has every open question/issue either resolved or listed as an explicit blocker.
|
||||
[ ] Any `next_step=ask_user` has a matching `escape_hatch` block and `E###` row.
|
||||
[ ] Pipeline smoke has a non-skip local PASS.
|
||||
[ ] Pipeline parity has a non-skip local PASS against the official reference.
|
||||
[ ] Basic example runs and writes a non-corrupt output.
|
||||
[ ] Video SSIM or audio-specific quality regression is added or explicitly deferred.
|
||||
[ ] Runtime production code has no diffusers/transformers model-class imports.
|
||||
[ ] Production comments are WHY-focused; examples have user-story docstrings.
|
||||
[ ] Post-parity hot-path pass is complete.
|
||||
```
|
||||
|
||||
Ask before deleting any reference clone or staged weights created by
|
||||
`add-model-01-prep`. Leave `.gitignore` entries so future parity assets stay
|
||||
untracked. Never commit the clone, weights, `.env`, credentials, or anything
|
||||
matching `*secret*`.
|
||||
|
||||
## References
|
||||
|
||||
- `../add-model-01-prep/SKILL.md` for user-input collection, HF inspection,
|
||||
weight staging, reference cloning, and setup handoff.
|
||||
- `contracts/` for canonical handoff schemas used by prep, parity, conversion,
|
||||
component porting, escape hatches, and final handoff.
|
||||
- `../add-model-02-parity/SKILL.md` for early component parity scaffolds and
|
||||
activation templates.
|
||||
- `../add-model-07-conversion/SKILL.md` for Phase 5 mapping, conversion scripts,
|
||||
monolithic checkpoint splitting, and strict-load checks.
|
||||
- `../add-model-03-port-dit/SKILL.md`, `../add-model-04-port-vae/SKILL.md`,
|
||||
`../add-model-05-port-encoder/SKILL.md`, and
|
||||
`../add-model-06-port-generic/SKILL.md` for component subagent implementation
|
||||
and parity-debug loops.
|
||||
- `../add-model-09-pipeline/SKILL.md` for pipeline definition, config/preset/
|
||||
registry/example wiring, smoke tests, and pipeline parity-debug.
|
||||
- `fastvideo/layers/AGENTS.md` for native layer selection and state-dict surface
|
||||
guidance.
|
||||
- `docs/contributing/coding_agents.md` for narrative context.
|
||||
- `docs/design/overview.md` for pipeline/config/registry architecture.
|
||||
- `fastvideo/pipelines/basic/wan/` for standard T2V/I2V/DMD/Causal variants.
|
||||
- `fastvideo/pipelines/basic/ltx2/` for non-standard stages and audio/video
|
||||
patterns.
|
||||
- `tests/local_tests/pipelines/test_gamecraft_pipeline_parity.py` for pipeline
|
||||
parity shape.
|
||||
- `tests/local_tests/transformers/test_ltx2.py`,
|
||||
`tests/local_tests/vaes/test_ltx2_vae.py`, and
|
||||
`tests/local_tests/encoders/test_ltx2_gemma_parity.py` for component parity.
|
||||
- `scripts/checkpoint_conversion/convert_ltx2_weights.py` for modern conversion
|
||||
script shape.
|
||||
- `scripts/checkpoint_conversion/wan_to_diffusers.py` for legacy regex mapping
|
||||
reference only.
|
||||
- `REVIEW.md` is historical; its decisions are incorporated here as of
|
||||
2026-04-30.
|
||||
|
||||
## Changelog
|
||||
|
||||
| Date | Change |
|
||||
|---|---|
|
||||
| 2026-04-24 | Initial FastVideo add-model workflow. |
|
||||
| 2026-04-30 | Split external setup into `add-model-01-prep`. |
|
||||
| 2026-04-30 | Rewrote as manual `/add-model` phase workflow and incorporated `REVIEW.md` decisions. |
|
||||
| 2026-04-30 | Extracted early parity scaffolding into `add-model-02-parity` and moved it before conversion/component implementation. |
|
||||
| 2026-04-30 | Added component reuse proof gate, bucket-specific porting skills, and parity PASS requirement for reused components. |
|
||||
| 2026-04-30 | Split prototype, conversion, and parity-debug phases; added conversion skill for monolithic and separate checkpoint layouts. |
|
||||
| 2026-04-30 | Extracted handoff schemas into `contracts/` for shared use across skills. |
|
||||
| 2026-04-30 | Added pipeline skill contract and Phase 7 component-parity gate. |
|
||||
| 2026-04-30 | Added escape-hatch contract for user decisions and `ask_user` handoffs. |
|
||||
@@ -0,0 +1,29 @@
|
||||
# Add Model Contracts
|
||||
|
||||
Canonical handoff schemas for the `/add-model` workflow. When a skill needs to
|
||||
send or receive structured context, use these files instead of inventing a local
|
||||
schema.
|
||||
|
||||
| Contract | Use |
|
||||
|---|---|
|
||||
| `prep_handoff.md` | `add-model-01-prep` output and `/add-model` Phase 0 input. |
|
||||
| `port_state.md` | Per-port `PORT_STATUS.md` file tracking progress, open questions, and issues. |
|
||||
| `escape_hatch.md` | Shared pause-and-ask schema for user decisions the workflow cannot safely choose. |
|
||||
| `component_context.md` | Per-component packet passed to parity, prototype, conversion, and parity-debug subagents. |
|
||||
| `parity_status.md` | `add-model-02-parity` scaffold/activation status returned to `/add-model`. |
|
||||
| `conversion_request.md` | Phase 5 conversion input and Phase 6 conversion retry request. |
|
||||
| `conversion_handoff.md` | `add-model-07-conversion` output back to `/add-model` and component subagents. |
|
||||
| `component_skill_handoff.md` | Component porting skill output in prototype or parity-debug mode. |
|
||||
| `pipeline_context.md` | Phase 7 packet passed to `add-model-09-pipeline` after component parity is green. |
|
||||
| `pipeline_handoff.md` | `add-model-09-pipeline` output back to `/add-model` after pipeline definition or parity-debug. |
|
||||
| `final_handoff.md` | Final `/add-model` pre-handoff checklist summary. |
|
||||
|
||||
Rules:
|
||||
|
||||
- Do not omit required fields. Use `unknown` plus the search already performed
|
||||
when the value is not known yet.
|
||||
- Do not include raw token values. Use env var names only.
|
||||
- Keep model-specific mapping details in conversion scripts and the local tests
|
||||
README/status notes, not in generic skill docs.
|
||||
- Use `next_step=ask_user` only with an `escape_hatch` block matching
|
||||
`escape_hatch.md`.
|
||||
@@ -0,0 +1,55 @@
|
||||
# Component Context Contract
|
||||
|
||||
Canonical per-component packet passed from `/add-model` to parity, prototype,
|
||||
conversion, and parity-debug subagents.
|
||||
|
||||
```text
|
||||
component_context:
|
||||
model_family: <snake_case>
|
||||
component: <name>
|
||||
component_type: <dit|vae|encoder|scheduler|conditioner|upsampler|vocoder|generic>
|
||||
mode: parity-scaffold | prototype | parity-debug
|
||||
official_ref_dir: <path or import path>
|
||||
official_definition_files:
|
||||
- path: <repo-relative or absolute path in official repo>
|
||||
symbols: <class/function names>
|
||||
notes: <layer graph, output contract, state-dict owner>
|
||||
official_instantiation_files:
|
||||
- path: <repo-relative or absolute path in official repo>
|
||||
symbols: <factory/pipeline/config names>
|
||||
args: <constructor args, config values, runtime flags>
|
||||
official_weight_source: <checkpoint file, subfolder, prefix, or passthrough source>
|
||||
fastvideo_target_files:
|
||||
- fastvideo/models/<bucket>/<file>.py
|
||||
- fastvideo/configs/models/<bucket>/<file>.py
|
||||
local_tests_readme: tests/local_tests/<model_family>/README.md
|
||||
port_state_file: tests/local_tests/<model_family>/PORT_STATUS.md
|
||||
parity_test: tests/local_tests/<bucket>/test_<family>_<component>_parity.py
|
||||
prototype_key_dumps:
|
||||
official: converted_weights/<family>/_mapping/<component>_official_keys.json | planned | unknown
|
||||
fastvideo: converted_weights/<family>/_mapping/<component>_fastvideo_keys.json | planned | unknown
|
||||
conversion:
|
||||
script: scripts/checkpoint_conversion/<family>_to_diffusers.py | not_created | not_needed | unknown
|
||||
converted_component_dir: converted_weights/<family>/<component> | not_created | not_needed | unknown
|
||||
model_index_library: <diffusers|transformers|fastvideo|fastvideo.*|unknown|none>
|
||||
config_file: <config.json|scheduler_config.json|none|unknown>
|
||||
mapping_notes: <key prefixes, split/fuse concerns, skipped keys, not_created, not_needed, or unknown>
|
||||
production_loader_strictness: <strict|non_strict_with_allowed_keys|stateless|unknown>
|
||||
strict_load: <not_run | pass | pass_with_documented_exclusions | blocked>
|
||||
concerns_or_unknowns:
|
||||
- <prototype mismatch, ambiguous arg, missing op, dtype concern, output head, etc.>
|
||||
```
|
||||
|
||||
Rules:
|
||||
|
||||
- If any required path is unknown, pass `unknown` plus the exact search already
|
||||
performed.
|
||||
- Do not silently omit ambiguous official files, instantiation args, or prototype
|
||||
concerns.
|
||||
- For reused components, still fill every field and set `fastvideo_target_files`
|
||||
to the reused class/config.
|
||||
- In `mode=parity-scaffold`, prototype and conversion fields may be `planned`,
|
||||
`not_created`, `not_needed`, or `unknown`; do not invent paths or statuses that
|
||||
do not exist yet.
|
||||
- Update `port_state_file` when concerns, issues, conversion status, or parity
|
||||
status change.
|
||||
@@ -0,0 +1,41 @@
|
||||
# Component Skill Handoff Contract
|
||||
|
||||
Returned by `add-model-03-port-dit`, `add-model-04-port-vae`,
|
||||
`add-model-05-port-encoder`, and `add-model-06-port-generic`.
|
||||
|
||||
```text
|
||||
component: <name>
|
||||
mode: prototype | parity-debug
|
||||
files_changed: <model/config/export/test/readme paths>
|
||||
official_files_used: <definition files, instantiation files>
|
||||
prototype_key_dumps: <official path, fastvideo path, or none>
|
||||
port_state_file: tests/local_tests/<model_family>/PORT_STATUS.md
|
||||
concerns_or_unknowns: <remaining or newly discovered concerns>
|
||||
parity_test: <path>
|
||||
parity_status: scaffold_skip | debug_red | non_skip_pass | blocked
|
||||
production_loader_strictness: strict | non_strict_with_allowed_keys | stateless
|
||||
strict_load: pass | pass_with_documented_exclusions | blocked | not_run
|
||||
pytest_output: <command + short result>
|
||||
blocker: <none or exact missing dependency/weights/numeric mismatch>
|
||||
conversion_retry_request: <none or failing keys/shapes/prefixes/evidence for add-model-07-conversion>
|
||||
readme_updated: yes | no
|
||||
next_step: phase_5_conversion | phase_5_conversion_retry | phase_6_continue | ask_user | blocked
|
||||
escape_hatch: <none or block matching contracts/escape_hatch.md>
|
||||
```
|
||||
|
||||
Rules:
|
||||
|
||||
- In `mode=prototype`, parity may be `scaffold_skip` or `blocked`; key dumps are
|
||||
the required artifact. Successful prototype handoff should use
|
||||
`next_step=phase_5_conversion`.
|
||||
- In `mode=parity-debug`, final success requires `parity_status=non_skip_pass`.
|
||||
- If conversion is implicated, return `conversion_retry_request` and do not edit
|
||||
conversion scripts or converted weights directly.
|
||||
- If production loading is non-strict, list allowed missing/unexpected keys in
|
||||
the parity test or handoff and mark `strict_load=pass_with_documented_exclusions`.
|
||||
- Update `port_state_file` before returning: component row, open questions,
|
||||
issues/blockers, decisions, and handoff notes.
|
||||
- Return an `escape_hatch` only for user decisions, not for normal component
|
||||
implementation or parity-debug failures.
|
||||
- Use `next_step=ask_user` only with an `escape_hatch` block and a matching
|
||||
`PORT_STATUS.md` row.
|
||||
@@ -0,0 +1,41 @@
|
||||
# Conversion Handoff Contract
|
||||
|
||||
Returned by `../add-model-07-conversion/SKILL.md` to `/add-model` and component
|
||||
parity-debug subagents.
|
||||
|
||||
```text
|
||||
conversion_script: scripts/checkpoint_conversion/<family>_to_diffusers.py
|
||||
source_layout: <diffusers|raw_official|separate_components|monolithic|mixed|custom>
|
||||
converted_weights_dir: converted_weights/<model_family>
|
||||
port_state_file: tests/local_tests/<model_family>/PORT_STATUS.md
|
||||
components_written: <list>
|
||||
passthrough_components: <list>
|
||||
strict_load: pass | pass_with_documented_exclusions | blocked
|
||||
component_context_updates:
|
||||
- component: <name>
|
||||
converted_component_dir: <path>
|
||||
model_index_library: <diffusers|transformers|fastvideo|fastvideo.*>
|
||||
config_file: <path or none>
|
||||
config_validation: pass | blocked | not_applicable
|
||||
mapping_notes: <prefixes, split/fuse ops, skipped keys>
|
||||
production_loader_strictness: strict | non_strict_with_allowed_keys | stateless
|
||||
strict_load: pass | pass_with_documented_exclusions | blocked | not_run
|
||||
retry_resolved: <yes | no | not_a_retry>
|
||||
concerns_or_unknowns: <remaining list>
|
||||
blocked_on: <none or exact blocker>
|
||||
next_step: phase_6_component_parity_debug | ask_user
|
||||
escape_hatch: <none or block matching contracts/escape_hatch.md>
|
||||
```
|
||||
|
||||
Rules:
|
||||
|
||||
- Include strict-load evidence for every stateful converted component.
|
||||
- If a component intentionally loads non-strictly, list the exact missing or
|
||||
unexpected keys and why they are safe.
|
||||
- Include the actual `model_index.json` library token, config filename, and config
|
||||
validation result for every emitted component.
|
||||
- Preserve retry evidence so the requesting component subagent can resume with
|
||||
updated context.
|
||||
- Keep `port_state_file` synchronized with `component_context_updates`.
|
||||
- Use `next_step=ask_user` only with an `escape_hatch` block and a matching
|
||||
`PORT_STATUS.md` row.
|
||||
@@ -0,0 +1,54 @@
|
||||
# Conversion Request Contract
|
||||
|
||||
Consumed by `../add-model-07-conversion/SKILL.md` in Phase 5 and during Phase 6
|
||||
conversion retries.
|
||||
|
||||
Initial conversion request:
|
||||
|
||||
```text
|
||||
model_family: <snake_case>
|
||||
source_layout: diffusers | raw_official | monolithic | separate_components | mixed | custom
|
||||
official_weights: <HF repo, local dir, or checkpoint file>
|
||||
hf_revision: <revision | default | none>
|
||||
converted_weights_dir: converted_weights/<model_family>
|
||||
local_tests_readme: tests/local_tests/<model_family>/README.md
|
||||
port_state_file: tests/local_tests/<model_family>/PORT_STATUS.md
|
||||
components:
|
||||
- name: <transformer|vae|text_encoder|conditioner|scheduler|...>
|
||||
component_type: <dit|vae|encoder|scheduler|conditioner|upsampler|vocoder|generic>
|
||||
official_definition_files: <paths + symbols>
|
||||
official_instantiation_files: <paths + call sites + args>
|
||||
official_weight_source: <checkpoint file, prefix, subfolder, or passthrough source>
|
||||
official_keys: <path to official key/shape dump>
|
||||
fastvideo_keys: <path to FastVideo prototype key/shape dump>
|
||||
fastvideo_class: <class name>
|
||||
model_index_library: <diffusers|transformers|fastvideo|fastvideo.*>
|
||||
config_filename: <config.json|scheduler_config.json|none>
|
||||
production_loader_strictness: <strict|non_strict_with_allowed_keys|stateless>
|
||||
source_prefix_or_path: <prefix or path>
|
||||
parity_test: <component parity test path>
|
||||
prototype_concerns_or_unknowns: <short list>
|
||||
```
|
||||
|
||||
Retry request from a component skill:
|
||||
|
||||
```text
|
||||
conversion_retry_request:
|
||||
component: <name>
|
||||
parity_test: <path>
|
||||
failing_keys: <official and FastVideo keys, if known>
|
||||
expected_actual_shapes: <expected vs actual shapes, if known>
|
||||
source_prefix_or_path: <prefix/path implicated by the failure>
|
||||
evidence: <strict-load error, first divergent tensor, parity log excerpt>
|
||||
suspected_fix: <rename | split | fuse | skip | component bucket | config | unknown>
|
||||
```
|
||||
|
||||
Rules:
|
||||
|
||||
- Phase 5 conversion requires Phase 4 official/FastVideo key dumps.
|
||||
- Component skills must use the retry request instead of editing conversion
|
||||
scripts or converted weights directly.
|
||||
- Conversion must update `port_state_file` with conversion status, retry history,
|
||||
strict-load status, new issues, and resolved issues.
|
||||
- Conversion must validate emitted config keys through the production config
|
||||
update path and record the config filename expected by each loader.
|
||||
@@ -0,0 +1,51 @@
|
||||
# Escape Hatch Contract
|
||||
|
||||
Canonical pause-and-ask schema for `/add-model` skills. Use this when the next
|
||||
action requires user input instead of autonomous debugging.
|
||||
|
||||
```text
|
||||
escape_hatch:
|
||||
needs_user_input: yes | no
|
||||
decision_type: scope | dependency | auth | cost | destructive | ambiguity | blocker
|
||||
question: <one precise question>
|
||||
recommended_option: <safe recommended choice>
|
||||
options:
|
||||
- <option + consequence>
|
||||
safe_default: <what the agent will do after approval, or none>
|
||||
blocked_until_answered: yes | no
|
||||
state_snapshot:
|
||||
phase: <phase or skill mode>
|
||||
files_changed:
|
||||
- <paths>
|
||||
command_or_test: <last relevant command, or not_run>
|
||||
evidence: <short logs, paths, error text, or blocker ID>
|
||||
```
|
||||
|
||||
Use `needs_user_input=no` when the handoff is green or the next step is already
|
||||
specified by the workflow.
|
||||
|
||||
Ask the user only for decisions the workflow cannot safely choose:
|
||||
|
||||
- product or PR scope changes, including dropping a modality, output head, or
|
||||
variant;
|
||||
- core dependency changes, version pin changes, or installing untrusted/private
|
||||
dependencies;
|
||||
- auth setup for gated repos, using env var names only and never token values;
|
||||
- large downloads, publishing weights, SSIM/reference uploads, or GPU-heavy work
|
||||
where cost/runtime approval is needed;
|
||||
- destructive file/git operations, overwriting existing clones/weights, or
|
||||
deleting staged assets;
|
||||
- ambiguous official sources of truth with incompatible behavior;
|
||||
- accepting a known blocker, loosening parity/quality tolerances, or shipping
|
||||
without required non-skip parity.
|
||||
|
||||
Do not ask for normal recoverable failures:
|
||||
|
||||
- missing imports, missing local paths, skipped tests, failing parity, conversion
|
||||
mapping errors, strict-load failures, format/lint failures, or implementation
|
||||
bugs covered by the skill workflow.
|
||||
|
||||
Before returning `next_step=ask_user`, update
|
||||
`tests/local_tests/<model_family>/PORT_STATUS.md` with the blocker/question ID,
|
||||
include the exact evidence, and provide one recommended option plus at most three
|
||||
alternatives.
|
||||
@@ -0,0 +1,38 @@
|
||||
# Final Handoff Contract
|
||||
|
||||
Completed by `/add-model` before handing work back to the user or opening a PR.
|
||||
|
||||
```text
|
||||
final_handoff:
|
||||
prep_handoff_complete: yes | no
|
||||
conversion_status: not_needed | pass | blocked
|
||||
components:
|
||||
- name: <component>
|
||||
reuse_or_port: reused | ported
|
||||
parity_test: <path>
|
||||
parity_status: non_skip_pass | blocked
|
||||
concerns_or_unknowns: <none or list>
|
||||
pipeline_smoke: pass | blocked | not_run
|
||||
pipeline_parity: pass | blocked | not_run
|
||||
example_status: pass | blocked | not_run
|
||||
quality_regression: added | deferred_with_reason | not_applicable
|
||||
local_tests_readme: tests/local_tests/<model_family>/README.md
|
||||
port_state_file: tests/local_tests/<model_family>/PORT_STATUS.md
|
||||
token_values_committed: no
|
||||
runtime_third_party_model_imports: none | listed_with_rationale
|
||||
blockers: <none or list>
|
||||
escape_hatch: <none or block matching contracts/escape_hatch.md>
|
||||
```
|
||||
|
||||
Required before handoff:
|
||||
|
||||
- Every required component, reused or ported, has non-skip local parity PASS.
|
||||
- Pipeline smoke and pipeline parity are non-skip PASS, or a blocker is explicit.
|
||||
- Basic example runs and writes a non-corrupt output.
|
||||
- `local_tests_readme` lists every component parity command/status/blocker.
|
||||
- `port_state_file` has no unresolved blocker that is omitted from the final
|
||||
response or PR notes.
|
||||
- No raw HF token values, credentials, `.env`, reference clone, or staged weight
|
||||
blobs are committed.
|
||||
- If final handoff is blocked on user input, include an `escape_hatch` block and
|
||||
matching `PORT_STATUS.md` row.
|
||||
@@ -0,0 +1,44 @@
|
||||
# Parity Status Contract
|
||||
|
||||
Returned by `../add-model-02-parity/SKILL.md` to `/add-model` and later updated by
|
||||
component parity-debug subagents.
|
||||
|
||||
```text
|
||||
component_parity:
|
||||
- component: <name>
|
||||
test: tests/local_tests/<bucket>/test_<family>_<component>_parity.py
|
||||
status: scaffold_skip | debug_red | non_skip_pass | blocked
|
||||
missing: <none | fastvideo_class | converted_weights | official_import | ...>
|
||||
coverage_scope: production_loader | implementation_subcomponent | both
|
||||
official_definition_files: <paths>
|
||||
official_instantiation_files: <paths>
|
||||
concerns_or_unknowns: <short list>
|
||||
pipeline_parity:
|
||||
test: <path or not-created>
|
||||
status: not_started | scaffold_skip | debug_red | non_skip_pass | blocked
|
||||
local_tests_readme: tests/local_tests/<model_family>/README.md
|
||||
port_state_file: tests/local_tests/<model_family>/PORT_STATUS.md
|
||||
notes: <short list>
|
||||
escape_hatch: <none or block matching contracts/escape_hatch.md>
|
||||
```
|
||||
|
||||
Status meanings:
|
||||
|
||||
- `scaffold_skip`: test is present but skips for a specific missing dependency,
|
||||
FastVideo class, or weights.
|
||||
- `debug_red`: both sides load and the test fails numerically.
|
||||
- `non_skip_pass`: required before final handoff for every required component,
|
||||
including reused components.
|
||||
- `blocked`: a precise missing dependency, weight, official call path, or
|
||||
component/conversion regression prevents local activation.
|
||||
|
||||
Coverage meanings:
|
||||
|
||||
- `production_loader`: FastVideo side loads through the same loader/path used by
|
||||
a pipeline.
|
||||
- `implementation_subcomponent`: FastVideo side constructs classes or remaps
|
||||
tensors directly to isolate implementation behavior.
|
||||
- `both`: the test covers both production loading and implementation behavior.
|
||||
|
||||
Use `escape_hatch` only when blocked status requires a user decision. Normal
|
||||
skips or red parity should be debugged by the workflow without asking.
|
||||
@@ -0,0 +1,82 @@
|
||||
# Pipeline Context Contract
|
||||
|
||||
Canonical packet passed from `/add-model` to `add-model-09-pipeline` for pipeline
|
||||
definition and pipeline parity-debug work.
|
||||
|
||||
```text
|
||||
pipeline_context:
|
||||
model_family: <snake_case>
|
||||
mode: pipeline-definition | pipeline-parity-debug
|
||||
workload_types:
|
||||
- <T2V|I2V|V2V|T2I|compatibility-shim-with-rationale>
|
||||
modalities:
|
||||
inputs: <text/image/video/audio/pose/depth/mask/etc.>
|
||||
outputs: <video/image/audio/joint-av/latents/etc.>
|
||||
official_ref_dir: <path or import path>
|
||||
official_pipeline_files:
|
||||
- path: <repo-relative or absolute path in official repo>
|
||||
symbols: <pipeline/factory/sample functions>
|
||||
notes: <stage order, mutable state, output contract>
|
||||
official_call:
|
||||
command_or_api: <official CLI, Python call, or package entrypoint>
|
||||
args_and_defaults: <height, width, frames, fps, duration, steps, CFG, scheduler, seeds, etc.>
|
||||
preset_source: <model card, config file, official script, or unknown>
|
||||
scheduler_and_rng: <timestep/sigma/noise/generator behavior>
|
||||
output_contract: <decoded media, denoised latents, waveform, dict keys, etc.>
|
||||
model_index:
|
||||
class_name: <FastVideo pipeline class name to emit in model_index.json>
|
||||
entry_class_names: <registered EntryClass.__name__ values that must include class_name>
|
||||
required_modules: <text_encoder, tokenizer, vae, transformer, scheduler, etc.>
|
||||
passthrough_modules: <tokenizer, scheduler, processor, external HF dirs, or none>
|
||||
sampling_param:
|
||||
new_fields: <none or list of public kwargs/preset defaults to add to SamplingParam>
|
||||
cli_fields: <none or list of fields that need CLI args>
|
||||
placeholder_fields: <none or video-shaped compatibility placeholders with rationale>
|
||||
components:
|
||||
- name: <component>
|
||||
component_type: <dit|vae|encoder|scheduler|conditioner|upsampler|vocoder|generic>
|
||||
parity_test: tests/local_tests/<bucket>/test_<family>_<component>_parity.py
|
||||
parity_status: non_skip_pass
|
||||
fastvideo_target_files: <model/config/export files>
|
||||
converted_component_dir: converted_weights/<family>/<component>
|
||||
conversion:
|
||||
converted_weights_dir: converted_weights/<family>
|
||||
source_layout: <diffusers|raw_official|monolithic|separate_components|mixed|custom>
|
||||
model_index_path: converted_weights/<family>/model_index.json
|
||||
fastvideo_targets:
|
||||
pipeline_files:
|
||||
- fastvideo/pipelines/basic/<family>/<family>_pipeline.py
|
||||
stage_files:
|
||||
- fastvideo/pipelines/basic/<family>/stages/<stage>.py
|
||||
pipeline_config_files:
|
||||
- fastvideo/configs/pipelines/<family>.py
|
||||
preset_file: fastvideo/pipelines/basic/<family>/presets.py
|
||||
registry_file: fastvideo/registry.py
|
||||
example_files:
|
||||
- examples/inference/basic/basic_<family>.py
|
||||
smoke_test: tests/local_tests/pipelines/test_<family>_pipeline_smoke.py
|
||||
parity_test: tests/local_tests/pipelines/test_<family>_pipeline_parity.py
|
||||
local_tests_readme: tests/local_tests/<model_family>/README.md
|
||||
port_state_file: tests/local_tests/<model_family>/PORT_STATUS.md
|
||||
concerns_or_unknowns:
|
||||
- <pipeline branch, unsupported workload, output head, preset ambiguity, etc.>
|
||||
```
|
||||
|
||||
Rules:
|
||||
|
||||
- Start only after every required component, reused or ported, has
|
||||
`parity_status=non_skip_pass`. If any row is missing or skipped, return to
|
||||
`/add-model` Phase 6.
|
||||
- Record official call arguments and default sources before writing FastVideo
|
||||
presets. Do not invent inference defaults from memory.
|
||||
- `model_index.class_name` must match a registered pipeline `EntryClass.__name__`;
|
||||
registry detectors are not sufficient for executable pipeline resolution.
|
||||
- Every public generation kwarg or preset default must be represented in
|
||||
`SamplingParam`, or documented as an intentional internal-only field.
|
||||
- `T2A`, `A2A`, and `AV` may be used only after `WorkloadType` supports them;
|
||||
otherwise record the compatibility shim and rationale explicitly.
|
||||
- Keep token values out of the packet. Use only token environment variable names.
|
||||
- If a target path is unknown, use `unknown` plus the exact search already
|
||||
performed.
|
||||
- Update `local_tests_readme` and `port_state_file` whenever pipeline smoke,
|
||||
parity, presets, registry, examples, or blockers change.
|
||||
@@ -0,0 +1,71 @@
|
||||
# Pipeline Handoff Contract
|
||||
|
||||
Returned by `add-model-09-pipeline` to `/add-model` after pipeline definition or
|
||||
pipeline parity-debug work.
|
||||
|
||||
```text
|
||||
pipeline_handoff:
|
||||
model_family: <snake_case>
|
||||
mode: pipeline-definition | pipeline-parity-debug
|
||||
files_changed:
|
||||
- <pipeline/config/preset/registry/stage/example/test/readme/status paths>
|
||||
official_files_used:
|
||||
- <definition/call/default source paths>
|
||||
required_config_modules:
|
||||
emitted: <list from pipeline class>
|
||||
model_index: <list from converted or source model_index.json>
|
||||
status: match | mismatch | blocked
|
||||
pipeline_class_resolution:
|
||||
model_index_class_name: <_class_name>
|
||||
entry_class_names: <registered EntryClass.__name__ values>
|
||||
status: exact_match | alias_added | blocked
|
||||
sampling_param:
|
||||
fields_added: <none or list>
|
||||
cli_fields_added: <none or list>
|
||||
unknown_kwargs_checked: yes | no | blocked
|
||||
stage_chain:
|
||||
- <stage names in execution order>
|
||||
pipeline_config:
|
||||
file: <path>
|
||||
classes: <class names>
|
||||
official_defaults_checked: yes | no | blocked
|
||||
presets:
|
||||
file: <path>
|
||||
names: <preset names>
|
||||
status: pass | blocked | not_run
|
||||
registry:
|
||||
status: pass | blocked | not_run
|
||||
detectors: <HF paths and model_index _class_name strings covered>
|
||||
smoke_test:
|
||||
path: tests/local_tests/pipelines/test_<family>_pipeline_smoke.py
|
||||
status: non_skip_pass | blocked | not_run
|
||||
pytest_output: <command + short result>
|
||||
pipeline_parity:
|
||||
path: tests/local_tests/pipelines/test_<family>_pipeline_parity.py
|
||||
status: scaffold_skip | debug_red | non_skip_pass | blocked
|
||||
pytest_output: <command + short result>
|
||||
comparison_target: <latents|decoded video|audio|joint outputs>
|
||||
example:
|
||||
path: examples/inference/basic/basic_<family>.py
|
||||
status: pass | blocked | not_run
|
||||
output: <path or none>
|
||||
readme_updated: yes | no
|
||||
port_state_updated: yes | no
|
||||
blockers: <none or exact blocker list>
|
||||
next_step: phase_10_quality_regression | return_to_phase_6 | ask_user
|
||||
escape_hatch: <none or block matching contracts/escape_hatch.md>
|
||||
```
|
||||
|
||||
Rules:
|
||||
|
||||
- `pipeline-definition` may return with parity still `scaffold_skip` only if the
|
||||
exact missing dependency, weight, or call-path blocker is recorded.
|
||||
- Final `/add-model` handoff requires `smoke_test.status=non_skip_pass` and
|
||||
`pipeline_parity.status=non_skip_pass`, unless the user explicitly accepts a
|
||||
documented blocker.
|
||||
- If parity failure traces to a component, conversion, or strict-load issue,
|
||||
return `next_step=return_to_phase_6` and include the exact failing evidence.
|
||||
- Keep `local_tests_readme` and `port_state_file` synchronized with this
|
||||
handoff before returning.
|
||||
- Use `next_step=ask_user` only with an `escape_hatch` block and a matching
|
||||
`PORT_STATUS.md` row.
|
||||
@@ -0,0 +1,89 @@
|
||||
# Port State Contract
|
||||
|
||||
Canonical per-port state file created during prep and updated by every
|
||||
`/add-model` phase.
|
||||
|
||||
Path:
|
||||
|
||||
```text
|
||||
tests/local_tests/<model_family>/PORT_STATUS.md
|
||||
```
|
||||
|
||||
Purpose:
|
||||
|
||||
- Single source of truth for resumable port progress.
|
||||
- Tracks component status, conversion status, parity status, open questions,
|
||||
blockers, escape hatches, and issue history.
|
||||
- Lets review agents run the same setup/tests without reconstructing handoffs
|
||||
from conversation history.
|
||||
|
||||
Required sections:
|
||||
|
||||
```text
|
||||
# <Model Family> Port Status
|
||||
|
||||
## Summary
|
||||
- model_family:
|
||||
- workload_types:
|
||||
- official_ref:
|
||||
- official_ref_dir:
|
||||
- hf_weights_path:
|
||||
- local_weights_dir:
|
||||
- source_layout:
|
||||
- local_tests_readme:
|
||||
|
||||
## Current Phase
|
||||
- phase:
|
||||
- status: not_started | in_progress | blocked | complete
|
||||
- owner: orchestrator | prep | parity | conversion | component:<name> | pipeline
|
||||
- last_updated:
|
||||
|
||||
## Component Matrix
|
||||
| Component | Type | Reuse/Port | Official Definition | Official Instantiation | FastVideo Target | Prototype | Conversion | Parity | Open Issues |
|
||||
|---|---|---|---|---|---|---|---|---|---|
|
||||
|
||||
## Conversion State
|
||||
- conversion_script:
|
||||
- converted_weights_dir:
|
||||
- source_layout:
|
||||
- strict_load_status:
|
||||
- passthrough_components:
|
||||
- retry_history:
|
||||
|
||||
## Parity Commands
|
||||
| Scope | Command | Last Result | Notes |
|
||||
|---|---|---|---|
|
||||
|
||||
## Open Questions
|
||||
| ID | Question | Owner | Needed By Phase | Status | Resolution |
|
||||
|---|---|---|---|---|---|
|
||||
|
||||
## Issues And Blockers
|
||||
| ID | Phase | Component | Severity | Issue | Evidence | Owner | Status | Resolution |
|
||||
|---|---|---|---|---|---|---|---|---|
|
||||
|
||||
## Escape Hatches
|
||||
| ID | Phase | Decision Type | Question | Recommended Option | Status | Resolution |
|
||||
|---|---|---|---|---|---|---|
|
||||
|
||||
## Decisions
|
||||
| Date | Decision | Rationale | Impact |
|
||||
|---|---|---|---|
|
||||
|
||||
## Handoff Notes
|
||||
- <short notes for the next agent>
|
||||
```
|
||||
|
||||
Rules:
|
||||
|
||||
- Update this file whenever a phase starts, blocks, resolves an issue, or hands
|
||||
off to another skill.
|
||||
- Record open questions and issues immediately. Do not leave blockers only in
|
||||
chat history or subagent responses.
|
||||
- Use stable IDs: `Q001`, `Q002`, `I001`, `I002`, etc.
|
||||
- Use stable escape-hatch IDs: `E001`, `E002`, etc. Link them from handoff
|
||||
`escape_hatch.state_snapshot.evidence` when returning `next_step=ask_user`.
|
||||
- Do not include raw token values, machine-local cache internals, or large output
|
||||
dumps. Use repo-relative paths when possible.
|
||||
- If a question or issue is resolved, keep the row and fill `Resolution` instead
|
||||
of deleting it.
|
||||
@@ -0,0 +1,42 @@
|
||||
# Prep Handoff Contract
|
||||
|
||||
Produced by `../add-model-01-prep/SKILL.md` and consumed by `/add-model` Phase 0.
|
||||
|
||||
```text
|
||||
model_family: <snake_case>
|
||||
workload_types: <T2V/I2V/V2V/T2I/or compatibility shim with rationale>
|
||||
official_ref: <url or import path>
|
||||
official_ref_dir: <ReferenceDir or none>
|
||||
official_ref_commit: <sha or unknown>
|
||||
hf_weights_path: <HF id or local path>
|
||||
hf_revision: <revision or default>
|
||||
local_weights_dir: official_weights/<model_family> or <local path>
|
||||
source_layout: diffusers | raw_official | monolithic | separate_components | mixed | custom | unknown
|
||||
model_index_class: <_class_name or none>
|
||||
components_seen: <components>
|
||||
needs_conversion: yes | no | unknown
|
||||
hf_token_env: <env var name only>
|
||||
dependency_changes: none | installed no-deps editable | installed official deps in current env | blocked on user
|
||||
official_env_status: imports_ok | private_deps_need_stubs | blocked
|
||||
local_tests_readme: tests/local_tests/<model_family>/README.md
|
||||
port_state_file: tests/local_tests/<model_family>/PORT_STATUS.md
|
||||
gitignore_entries_added: <list>
|
||||
next_step: add-model | ask_user
|
||||
open_questions: <short list>
|
||||
escape_hatch: <none or block matching contracts/escape_hatch.md>
|
||||
```
|
||||
|
||||
Validation:
|
||||
|
||||
- `official_env_status` must be `imports_ok` or `private_deps_need_stubs` before
|
||||
component parity scaffolding.
|
||||
- `local_tests_readme` must exist and describe official setup, HF weights,
|
||||
dependency changes, planned parity commands, and review notes.
|
||||
- `port_state_file` must exist and follow `contracts/port_state.md`.
|
||||
- Prep does not go directly to conversion; `/add-model` must run component
|
||||
prototype/key-dump Phase 4 before Phase 5 conversion.
|
||||
- `T2A`, `A2A`, and `AV` may be used only after `WorkloadType` supports them;
|
||||
otherwise record the compatibility shim and rationale explicitly.
|
||||
- Never include HF token values.
|
||||
- Use `next_step=ask_user` only with an `escape_hatch` block and a matching
|
||||
`PORT_STATUS.md` row.
|
||||
@@ -0,0 +1,96 @@
|
||||
# Shared Add-Model Rules
|
||||
|
||||
These rules apply to every `add-model` related skill: prep, parity, conversion,
|
||||
component porting, pipeline, and the main `/add-model` orchestrator.
|
||||
|
||||
## Token And Auth Safety
|
||||
|
||||
- Never accept, print, echo, log, hard-code, or commit raw HF token values.
|
||||
- Refer only to token environment variable names: `HF_TOKEN`,
|
||||
`HUGGINGFACE_HUB_TOKEN`, or `HF_API_KEY`.
|
||||
- Scripts may read those environment variables but must not print their values.
|
||||
- Ask for auth setup only by env var name. Do not ask the user to paste a token.
|
||||
- Read scope is needed for gated repos during conversion/load. Write scope is
|
||||
needed for publishing converted weights or seeding generated references.
|
||||
|
||||
## Shared State Files
|
||||
|
||||
- `tests/local_tests/<model_family>/README.md` is the reviewer-facing setup and
|
||||
verification log. Keep it current with setup commands, dependency blockers,
|
||||
parity commands, conversion commands, and pass/blocker status.
|
||||
- `tests/local_tests/<model_family>/PORT_STATUS.md` is the per-port state file.
|
||||
It must follow `../contracts/port_state.md` and keep stable `Q###`, `I###`, and
|
||||
`E###` IDs.
|
||||
- Keep resolved questions/issues in `PORT_STATUS.md` with the resolution instead
|
||||
of deleting them.
|
||||
- Before returning a handoff, update both state files when the skill changed
|
||||
setup, tests, conversion, parity status, blockers, or decisions.
|
||||
- Do not include raw tokens, non-reproducible absolute cache paths, large
|
||||
generated outputs, `.env`, credentials, or anything matching `*secret*`.
|
||||
|
||||
## Escape Hatches
|
||||
|
||||
Continue autonomously for recoverable setup, implementation, conversion,
|
||||
strict-load, smoke, parity-debug, lint, or test failures. Stop and ask the user
|
||||
only when the next action requires a product, cost, safety, auth, dependency, or
|
||||
scope decision the workflow cannot safely choose.
|
||||
|
||||
Use `../contracts/escape_hatch.md` whenever returning `next_step=ask_user`.
|
||||
|
||||
Ask for user input only for:
|
||||
|
||||
- scope changes, such as dropping a modality, output head, variant, component, or
|
||||
public mode;
|
||||
- core dependency changes, untrusted/private dependency installs, or version pin
|
||||
changes;
|
||||
- auth setup for gated repos, using env var names only;
|
||||
- large downloads, publishing weights, SSIM/reference uploads, or GPU-heavy work
|
||||
where cost/runtime approval is needed;
|
||||
- destructive file/git operations, overwriting existing clones/weights, or
|
||||
deleting staged assets;
|
||||
- incompatible official sources of truth where no reference can be chosen from
|
||||
published weights and docs;
|
||||
- accepting a blocker, loosening tolerances, using shape-only substitutes, or
|
||||
shipping without required non-skip parity.
|
||||
|
||||
Do not ask for normal recoverable failures: missing imports, missing local paths,
|
||||
skipped tests, failing parity, conversion mapping bugs, strict-load errors,
|
||||
format/lint failures, smoke failures, registry import issues, example failures,
|
||||
or implementation bugs covered by the phase workflow.
|
||||
|
||||
Before asking:
|
||||
|
||||
- update `PORT_STATUS.md` with an `E###` escape-hatch row plus any linked `Q###`
|
||||
or `I###` row;
|
||||
- include exact evidence: command, path, short error text, parity or strict-load
|
||||
excerpt, or blocker ID;
|
||||
- provide one recommended option and at most three alternatives;
|
||||
- set the relevant handoff `next_step=ask_user` and include the `escape_hatch`
|
||||
block.
|
||||
|
||||
Skill-specific escape-hatch sections may add extra examples, but they must not
|
||||
weaken these shared rules.
|
||||
|
||||
## Production Boundary
|
||||
|
||||
- No runtime `from diffusers import <model class>` or
|
||||
`from transformers import <model class>` in `fastvideo/` production code.
|
||||
- Components that own weights or numerical behavior must be FastVideo-native
|
||||
unless the user explicitly accepts a documented lazy-wrapper exception.
|
||||
- Allowed third-party runtime exceptions are tokenizers and pure data utilities
|
||||
when they match existing project patterns.
|
||||
- Tests may import diffusers/transformers as parity references.
|
||||
- Production comments explain why, not what or provenance. Avoid narrative
|
||||
comments like `vendored from`, `matches upstream`, `REVIEW`, or session-history
|
||||
commentary.
|
||||
|
||||
## Verification Semantics
|
||||
|
||||
- A committed local test may skip when clones, weights, or private deps are absent
|
||||
so CI and other contributors are not blocked.
|
||||
- On the porter's machine, a skip is not a pass. Fix the missing import, weights,
|
||||
or path before claiming verification.
|
||||
- New ports require local non-skip parity for required components and pipeline
|
||||
parity when a pipeline is in scope.
|
||||
- Smoke tests prove loadability only. They are not a substitute for numerical
|
||||
component or pipeline parity.
|
||||
@@ -0,0 +1,117 @@
|
||||
# Component Skill Common Instructions
|
||||
|
||||
These instructions apply to `add-model-03-port-dit`, `add-model-04-port-vae`,
|
||||
`add-model-05-port-encoder`, and `add-model-06-port-generic`. Bucket-specific skills
|
||||
add target paths, implementation patterns, drift checks, and scope questions.
|
||||
|
||||
## Required Context
|
||||
|
||||
Require the complete packet from `../contracts/component_context.md`.
|
||||
|
||||
Do not start if the official definition files, official instantiation files, or
|
||||
parity test path are missing. Ask the `/add-model` orchestrator for the complete
|
||||
component context packet instead of rediscovering broad scope silently.
|
||||
|
||||
If the parity scaffold is missing, create it first with
|
||||
`../../add-model-02-parity/templates/component_parity_test.py`.
|
||||
|
||||
## Prototype Mode
|
||||
|
||||
Prototype mode runs before conversion:
|
||||
|
||||
- implement or prove reuse for the minimal native component, config, export, and
|
||||
`EntryClass` surface needed by the relevant loader;
|
||||
- instantiate with random weights using the exact official architecture args, or
|
||||
instantiate/document stateless components with no weights;
|
||||
- dump official and FastVideo `state_dict()` names/shapes for every stateful
|
||||
component so conversion can derive mappings from real surfaces;
|
||||
- return concerns discovered during prototype work, such as ambiguous official
|
||||
flags, shape mismatches, private ops, missing loader buckets, passthrough
|
||||
weights, or output heads;
|
||||
- update `local_tests_readme` with prototype status and key-dump paths;
|
||||
- update `port_state_file` with prototype status, open questions, issues, and
|
||||
handoff notes;
|
||||
- do not chase numerical parity and do not block on converted weights.
|
||||
|
||||
Prototype mode succeeds when the component imports, instantiates with official
|
||||
args, and required key/shape dumps exist. Converted weights and parity PASS are
|
||||
not required yet.
|
||||
|
||||
## Parity-Debug Mode
|
||||
|
||||
Parity-debug mode runs after conversion:
|
||||
|
||||
- strict-load converted weights through the same path the pipeline will use, or
|
||||
document that the component is stateless or an approved passthrough;
|
||||
- use `conversion_context` and `concerns_or_unknowns` to decide whether a failure
|
||||
belongs to mapping, loading, implementation, tokenization, normalization,
|
||||
scheduler semantics, or the parity test;
|
||||
- run only the component parity test first with `pytest <parity_test> -v -s`;
|
||||
- if it skips, fix the missing official import, FastVideo class, tokenizer,
|
||||
converted weights, or path;
|
||||
- if it fails numerically, add targeted intermediate comparisons to identify the
|
||||
first divergent operation or tensor;
|
||||
- update component implementation only when the failure is a component
|
||||
layer/config/forward/contract bug;
|
||||
- update `local_tests_readme` with the command, result, and blocker or PASS;
|
||||
- update `port_state_file` with parity status, resolved/new issues, open
|
||||
questions, and handoff notes;
|
||||
- keep iterating until the test is a non-skip PASS or return a precise blocker.
|
||||
|
||||
## Conversion Boundary
|
||||
|
||||
Component skills must not patch conversion scripts or converted weights ad hoc.
|
||||
If the first drift or strict-load failure points to wrong keys, missing tensors,
|
||||
shape mismatches, component prefixes, split/fuse logic, skipped-key policy, or
|
||||
config emission, return a conversion retry request for
|
||||
`../../add-model-07-conversion/SKILL.md` matching
|
||||
`../contracts/conversion_request.md`.
|
||||
Resume parity-debug only after conversion returns an updated handoff.
|
||||
|
||||
## Reuse Proof
|
||||
|
||||
When `fastvideo_target_files` point to existing FastVideo code instead of a new
|
||||
port:
|
||||
|
||||
- compare the official definition against the FastVideo target: graph/operation
|
||||
structure, parameter or state shapes, normalization, activation, positional or
|
||||
temporal behavior, scaling constants, dtype behavior, state-dict names, output
|
||||
containers, and output tensors;
|
||||
- compare the official instantiation against the FastVideo config and loader
|
||||
args: constructor args, config values, defaults, variant flags, optional
|
||||
submodules, checkpoint metadata, tokenizer/media paths, and loader path;
|
||||
- treat a matching class instantiated with different args as not reusable;
|
||||
- record reuse evidence in `local_tests_readme` and keep the reused component in
|
||||
`prototype_key_dumps` when it owns state so conversion and parity-debug use the
|
||||
same surface;
|
||||
- still run parity-debug to a non-skip PASS. If mismatch is found, return the
|
||||
concern so `/add-model` can switch the component to a native port.
|
||||
|
||||
## Handoff
|
||||
|
||||
Return `../contracts/component_skill_handoff.md`.
|
||||
|
||||
Mode-specific expectations:
|
||||
|
||||
- In `mode=prototype`, `parity_status` may be `scaffold_skip` or `blocked`; key
|
||||
dumps are the required artifact and successful prototype handoff should use
|
||||
`next_step=phase_5_conversion`.
|
||||
- In `mode=parity-debug`, final success requires
|
||||
`parity_status=non_skip_pass`.
|
||||
- If conversion is implicated, return `conversion_retry_request` and leave
|
||||
conversion edits to `add-model-07-conversion`.
|
||||
- If production loading is non-strict, list allowed missing/unexpected keys in
|
||||
the parity test or handoff and mark
|
||||
`strict_load=pass_with_documented_exclusions`.
|
||||
|
||||
## Escape Hatches
|
||||
|
||||
Follow `common_rules.md`. Do not ask for normal prototype or parity-debug
|
||||
failures such as missing imports, tokenizer/path issues, red parity,
|
||||
strict-load failures, key mismatches, shape mismatches, or implementation bugs.
|
||||
Return conversion retry requests or precise blockers as directed by the workflow.
|
||||
|
||||
Ask only when component work requires a scope or safety decision, such as
|
||||
dropping a required stream/output/path, changing core dependencies, accepting
|
||||
private model code or unsupported private ops, choosing between incompatible
|
||||
official definitions, creating a new loader bucket, or loosening required parity.
|
||||
@@ -0,0 +1,333 @@
|
||||
---
|
||||
name: decompose-pipeline-pr
|
||||
description: Decompose an oversized FastVideo pipeline PR into a stack of independently-reviewable PRs. Tiers the diff by blast radius (invisible / dead code / cross-cutting infra / activation), produces a branch graph and worktree bootstrap, drafts the AGENTS.md manifest, flags missing tests on cross-cutting infra changes, and extracts lessons from the PR body.
|
||||
---
|
||||
|
||||
# Decompose Pipeline PR
|
||||
|
||||
## Purpose
|
||||
|
||||
When a PR adds a new pipeline (or first-class component port) and crosses
|
||||
~3,000 LOC, single-shot review converges to rubber-stamping. This skill
|
||||
decomposes such a PR into a stack of independently-reviewable PRs without
|
||||
disturbing `main`.
|
||||
|
||||
It is the inverse of `add-model`: where `add-model` walks adding a new
|
||||
pipeline as a fresh PR, this skill walks decomposing an existing oversized
|
||||
pipeline PR.
|
||||
|
||||
**Worked example:** PR #1280 (daVinci-MagiHuman, 9,812 LOC, 56 files) →
|
||||
2 prerequisite PRs off main + 8-PR stack:
|
||||
- #1293 `will/activation-trace` (prerequisite)
|
||||
- #1294 `will/loader-infra` (prerequisite)
|
||||
- #1295 (1/8) housekeeping
|
||||
- #1296 (2/8) t5gemma encoder
|
||||
- #1297 (3/8) DiT
|
||||
- #1298 (4/8) pipeline stages
|
||||
- #1299 (5/8) pipeline orchestrator
|
||||
- #1300 (6/8) provenance (AGENTS.md, JOURNAL.md, lessons)
|
||||
- #1301 (7/8) conversion scripts
|
||||
- #1302 (8/8) registry activation
|
||||
|
||||
## Prerequisites
|
||||
|
||||
- Open PR number on `hao-ai-lab/FastVideo` (or any FastVideo fork)
|
||||
- `gh` CLI authenticated against the target remote
|
||||
- Local git worktree support (`git worktree`)
|
||||
- Git config `user.name` / `user.email` set
|
||||
- Pre-commit installed (`pre-commit install --hook-type pre-commit --hook-type commit-msg`)
|
||||
- The target PR's branch fetched locally as `origin/<feature-branch>`
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Required | Description |
|
||||
|-----------|----------|-------------|
|
||||
| PR number or URL | Yes | E.g. `1280` or `https://github.com/hao-ai-lab/FastVideo/pull/1280` |
|
||||
| Max desired PR size | No | Defaults to ~2,500 LOC of code per stack PR (excluding generated/journal files) |
|
||||
| Output dir | No | Defaults to `.agents/exploration/decompose-<pr-number>.md` |
|
||||
|
||||
## Steps
|
||||
|
||||
### 1. Verify ground truth (do not trust `gh pr diff --name-only`)
|
||||
|
||||
`gh pr diff <N> --name-only` has been observed to emit phantom file entries.
|
||||
Always cross-check against the authoritative `git diff`:
|
||||
|
||||
```bash
|
||||
git fetch origin pull/<N>/head:<feature-branch>
|
||||
git diff origin/main..origin/<feature-branch> --name-status > /tmp/pr-<N>-files.txt
|
||||
git diff origin/main..origin/<feature-branch> --stat
|
||||
```
|
||||
|
||||
Use the `--name-status` output as the authoritative file list. If it
|
||||
disagrees with `gh pr diff --name-only`, trust the git diff.
|
||||
|
||||
### 2. Tier the diff by blast radius
|
||||
|
||||
Classify every changed file into one of four tiers:
|
||||
|
||||
| Tier | Description | Examples |
|
||||
|---|---|---|
|
||||
| **Tier 0 — Invisible** | Lint/style/CI configs that don't affect runtime | `.gitignore`, `pyproject.toml` (codespell only), `.agents/skills/index.jsonl` stubs |
|
||||
| **Tier 1 — Dead code** | New files in their own dirs; aggregator one-liners | `fastvideo/models/dits/<new>/`, `fastvideo/pipelines/basic/<new>/`, `examples/inference/basic/basic_<new>*.py`, `tests/local_tests/<new>/`, `__init__.py` exports |
|
||||
| **Tier 2 — Cross-cutting infra** | Modifications to files used by every pipeline | See protected-paths list below |
|
||||
| **Tier 3 — Activation switch** | `register_configs(...)` calls + the example scripts that demo them | `fastvideo/registry.py` |
|
||||
|
||||
**FastVideo Tier 2 protected paths:**
|
||||
```
|
||||
fastvideo/utils.py
|
||||
fastvideo/pipelines/composed_pipeline_base.py
|
||||
fastvideo/models/loader/component_loader.py
|
||||
fastvideo/configs/models/dits/__init__.py
|
||||
fastvideo/configs/models/encoders/__init__.py
|
||||
fastvideo/configs/models/vaes/__init__.py
|
||||
fastvideo/envs.py
|
||||
fastvideo/fastvideo_args.py
|
||||
fastvideo/distributed/**
|
||||
fastvideo/layers/**
|
||||
fastvideo/attention/**
|
||||
fastvideo/registry.py # treat as Tier 3 if change is the activation
|
||||
```
|
||||
|
||||
Tier 3 detection (mechanical):
|
||||
```bash
|
||||
git diff origin/main..origin/<feature-branch> -- fastvideo/registry.py | \
|
||||
grep -E "^\+.*register_configs\("
|
||||
```
|
||||
|
||||
If `registry.py` only contains `register_configs` additions, treat it as
|
||||
Tier 3. If it modifies existing behavior, treat it as Tier 2 (rare).
|
||||
|
||||
### 3. Identify reusable Tier-1 components
|
||||
|
||||
Within Tier 1, look for sub-trees that are **not** model-specific and could
|
||||
land separately:
|
||||
|
||||
- Encoders matching a known multi-model base (T5/T5-Gemma/Llama/Gemma/CLIP variants)
|
||||
- New stage classes that subclass shared bases without referencing the new model
|
||||
- Hook/profiler/debug infra under `fastvideo/hooks/`
|
||||
- New helpers that have no model-specific dependencies
|
||||
|
||||
These get split into their own PRs (e.g. PR 4 `t5gemma-encoder` in the
|
||||
MagiHuman example).
|
||||
|
||||
### 4. Hunt for missing test coverage on Tier 2 changes
|
||||
|
||||
For every Tier-2 file modified, check whether the original PR added unit
|
||||
tests for the new behavior:
|
||||
|
||||
```bash
|
||||
for f in <list-of-tier-2-files>; do
|
||||
echo "=== Tests for $f ==="
|
||||
git diff origin/main..origin/<feature-branch> -- \
|
||||
"$(echo $f | sed 's|fastvideo/|fastvideo/tests/|; s|\.py|*|')"
|
||||
done
|
||||
```
|
||||
|
||||
If a Tier-2 PR has no accompanying tests, **emit a "must-add tests" list**
|
||||
with a sketch of the case grid. Tier-2 PRs do not ship without those tests.
|
||||
|
||||
The MagiHuman example required this for PR-B (`utils.py`): the original PR
|
||||
shipped no `test_utils_loader.py`, so the decomposition added 9 unit-test
|
||||
cases covering the umbrella-detector boundary, the optional-component-dirs
|
||||
relaxation, and regression coverage on every existing 2-segment HF id.
|
||||
|
||||
### 5. Build the dependency DAG and topo-sort
|
||||
|
||||
Edges:
|
||||
- Tier 2 infra → Tier 1 code that imports it
|
||||
- Reusable Tier 1 components → model-specific Tier 1 code that uses them
|
||||
(encoder before DiT before pipeline)
|
||||
- Tier 1 → Tier 3 (activation always last)
|
||||
- Tier 0 has no dependents (lands first as a freebie)
|
||||
|
||||
Topo-sort produces the stack ordering. Pull Tier-2 PRs **out of the stack**
|
||||
when they have no model-specific dependency — they should land off main
|
||||
with their own focused review, not buried in a model port.
|
||||
|
||||
Render as a tree (markdown):
|
||||
|
||||
```
|
||||
main
|
||||
├─ <prereq-A>
|
||||
│ └─ <prereq-B>
|
||||
│ ├─ <stack-01-housekeeping>
|
||||
│ │ └─ <stack-02-encoder>
|
||||
│ │ └─ <stack-03-dit>
|
||||
│ │ └─ ...
|
||||
│ │ └─ <stack-N-activate>
|
||||
│ └─ (parallel) <skill-pr> off main
|
||||
```
|
||||
|
||||
### 6. Detect mis-shelved docs and debug scratch
|
||||
|
||||
Two categories to flag:
|
||||
|
||||
- **Mis-shelved docs**: Markdown files under `tests/local_tests/` are
|
||||
journals, not tests. Flag for relocation to the package dir as
|
||||
`JOURNAL.md`.
|
||||
- **Debug scratch**: files starting with `_debug_`, `_scratch_`, or
|
||||
`_explore_`. Flag for drop (do not carry into any output PR).
|
||||
|
||||
For MagiHuman: `tests/local_tests/magi-human.md` → relocate. Two
|
||||
`_debug_magi_human_*.py` files → drop.
|
||||
|
||||
### 7. Author the AGENTS.md manifest skeleton
|
||||
|
||||
For the new pipeline package, generate a 6-section `AGENTS.md` scaffold
|
||||
with the file table pre-populated from the diff:
|
||||
|
||||
1. **Manifest** — file table by role
|
||||
2. **Parity invariants** — load-bearing rules with one-paragraph each + lesson refs
|
||||
3. **Cross-refs** — "If you change X, re-run Y" matrix
|
||||
4. **Run book** — single pytest command + prereqs (HF tokens, GPU, wall-time)
|
||||
5. **Open questions** — known issues (e.g. tolerance carve-outs)
|
||||
6. **Provenance** — PR table with branch names and source SHA
|
||||
|
||||
The provenance section is filled incrementally during stack execution and
|
||||
finalized in the activation PR.
|
||||
|
||||
### 8. Extract lessons from the PR body
|
||||
|
||||
Scan the PR body for sections titled "Key implementation work", "Bug hunt",
|
||||
"Lessons", or sentences with patterns like "took N waves to localize",
|
||||
"silent regression", "investigation revealed". Each becomes a candidate
|
||||
`.agents/lessons/<YYYY-MM-DD>_<slug>.md` draft.
|
||||
|
||||
Lessons MUST follow the existing template in
|
||||
`.agents/lessons/README.md`:
|
||||
- YAML frontmatter: `date`, `experiment`, `category`, `severity`
|
||||
- Sections: What Happened, Root Cause, Fix / Workaround, Prevention
|
||||
- Filename: `<YYYY-MM-DD>_<short-slug>.md`
|
||||
|
||||
Lessons co-locate with the code they concern: a conversion-script lesson
|
||||
lands in the same PR as the conversion script, not in the docs PR.
|
||||
|
||||
### 9. Emit the commit-footer convention
|
||||
|
||||
Every commit in the stack ends with:
|
||||
|
||||
```
|
||||
<Feature>-Stack: N/M
|
||||
```
|
||||
|
||||
E.g. `Magi-Stack: 5/8`. Use the package directory name as the feature key.
|
||||
After all PRs squash-merge, `git log --grep='^<Feature>-Stack:'` reconstructs
|
||||
the lineage even if PR numbers later get renumbered.
|
||||
|
||||
### 10. Produce the worktree bootstrap
|
||||
|
||||
Generate a runnable bash script:
|
||||
|
||||
```bash
|
||||
#!/bin/bash
|
||||
set -euo pipefail
|
||||
|
||||
REPO=/home/<user>/FastVideo
|
||||
WORKTREE=/home/<user>/FastVideoMagi # NB: directory name must be a valid
|
||||
# Python identifier (no hyphens) so
|
||||
# mypy doesn't choke
|
||||
SOURCE_PR=<N>
|
||||
SOURCE_BRANCH=will/<feature>
|
||||
SOURCE_SHA=$(git -C "$REPO" rev-parse "origin/$SOURCE_BRANCH")
|
||||
|
||||
git -C "$REPO" fetch origin main:main
|
||||
git -C "$REPO" fetch "origin/$SOURCE_BRANCH"
|
||||
git -C "$REPO" worktree add "$WORKTREE" origin/main
|
||||
|
||||
# Capture baseline for provenance
|
||||
mkdir -p "$REPO/.agents/exploration"
|
||||
cat > "$REPO/.agents/exploration/<feature>-baseline-${SOURCE_SHA:0:8}.txt" <<EOF
|
||||
Source PR: <repo>#$SOURCE_PR
|
||||
Source SHA: $SOURCE_SHA
|
||||
Authoritative file count: $(git -C "$REPO" diff origin/main..origin/$SOURCE_BRANCH --name-only | wc -l)
|
||||
Date captured: $(date -u +%Y-%m-%dT%H:%M:%SZ)
|
||||
EOF
|
||||
```
|
||||
|
||||
### 11. Author preserve via `git checkout`, not `cherry-pick`
|
||||
|
||||
For each stack PR:
|
||||
|
||||
```bash
|
||||
git -C "$WORKTREE" switch -c <new-branch> <base-branch>
|
||||
git -C "$WORKTREE" checkout origin/<source-branch> -- <file1> <file2> ...
|
||||
git -C "$WORKTREE" commit -m "[<scope>]: <subject>
|
||||
|
||||
<body>
|
||||
|
||||
<Feature>-Stack: N/M"
|
||||
git -C "$WORKTREE" push -u origin <new-branch>
|
||||
gh pr create --base <base-branch> --head <new-branch> --title "..." --body "$(cat <<EOF ... EOF)"
|
||||
```
|
||||
|
||||
Notes:
|
||||
- `git checkout origin/<source> -- <files>` extracts only the named files,
|
||||
preserving the diff. The original PR's author is **not** preserved on the
|
||||
new commit (it's authored by whoever runs the script). Reference the
|
||||
original PR + source SHA in every commit body and PR description for
|
||||
authorship attribution.
|
||||
- **Never use `git cherry-pick`** for this workflow — cherry-pick applies
|
||||
whole commits, which mixes concerns across PR boundaries.
|
||||
|
||||
## Outputs
|
||||
|
||||
The skill produces:
|
||||
|
||||
1. A markdown decomposition plan (`.agents/exploration/decompose-<pr>.md`)
|
||||
2. A proposed branch graph
|
||||
3. A worktree-bootstrap script
|
||||
4. Per-PR file allocation lists (under `/tmp/<feature>-stack/`)
|
||||
5. AGENTS.md scaffolds for any new pipeline packages
|
||||
6. Draft lesson files (placed alongside the PR that owns the code they concern)
|
||||
7. A finalized provenance table for the package AGENTS.md
|
||||
|
||||
## Anti-Patterns
|
||||
|
||||
The skill should warn against:
|
||||
|
||||
- **"Just rebase the megaPR into smaller commits."** Doesn't help review;
|
||||
reviewer still sees one PR.
|
||||
- **Co-locating tests under the new package.** FastVideo's convention is
|
||||
by-kind under `fastvideo/tests/` and `tests/local_tests/<family>/`. Don't
|
||||
invent a new layout per pipeline.
|
||||
- **Splitting Tier 2 changes into "one file per PR."** Tier 2 PRs are
|
||||
about semantic units (e.g., "loader umbrella + optional component dirs"
|
||||
together because they jointly define the new diffusers-format contract),
|
||||
not file-count.
|
||||
- **Landing the activation switch first** ("just register, the code can
|
||||
be empty"). The skill enforces activation-last so every intermediate
|
||||
state is dead code, not broken code.
|
||||
- **Trusting `gh pr diff --name-only`.** Cross-check against
|
||||
`git diff origin/main..origin/<feature-branch> --name-status` —
|
||||
`gh`'s output has been observed to include phantom entries.
|
||||
- **Worktree dir names with hyphens.** mypy interprets them as invalid
|
||||
Python package names and refuses to run. Use CamelCase or underscores.
|
||||
- **Skipping the lesson-extraction step.** PR bodies contain the most
|
||||
expensive learnings of the original implementation. Losing them to a
|
||||
squash-merge is the silent decay of institutional knowledge.
|
||||
|
||||
## Example Usage
|
||||
|
||||
```
|
||||
User: split PR 1280
|
||||
Agent: [invokes decompose-pipeline-pr]
|
||||
→ produces .agents/exploration/decompose-1280.md with:
|
||||
- tiered file table (56 files: 3 tier-0, 35 tier-1, 9 tier-2,
|
||||
9 tier-3)
|
||||
- branch graph (PR-A + PR-B + 8-PR stack)
|
||||
- worktree bootstrap script
|
||||
- per-PR file lists
|
||||
- AGENTS.md scaffold for fastvideo/pipelines/basic/magi_human/
|
||||
- 3 draft lessons extracted from the PR body
|
||||
→ asks user to confirm before opening branches
|
||||
```
|
||||
|
||||
## References
|
||||
|
||||
- The MagiHuman decomposition (worked example):
|
||||
`fastvideo/pipelines/basic/magi_human/AGENTS.md` (after PR #1302 merges)
|
||||
- Existing skill: `.agents/skills/add-model/SKILL.md` (the inverse — adding
|
||||
a new pipeline as a fresh PR)
|
||||
- Lesson template: `.agents/lessons/README.md`
|
||||
- Skill template: `.agents/skills/SKILL_TEMPLATE.md`
|
||||
@@ -0,0 +1,185 @@
|
||||
# dreamverse-deploy — redeploy migrated Dreamverse on a chosen GPU
|
||||
|
||||
**Scope:** project (lives in this repo at `.agents/skills/dreamverse-deploy/`)
|
||||
|
||||
**When to use:** you want to (re)launch the migrated `apps/dreamverse/` backend
|
||||
+ frontend on this dev node, pinned to a specific physical GPU. Tears down
|
||||
any existing deploy on the same ports first, then boots fresh and waits for
|
||||
both `/readyz` and the FE root to return 200.
|
||||
|
||||
**Pairs with:** [`integration-plan.md`](../../memory/dreamverse-integration/integration-plan.md)
|
||||
"Local GPU4 verification hook" + [`decisions-log.md D-19`](../../memory/dreamverse-integration/decisions-log.md#d-19).
|
||||
|
||||
## Prerequisites
|
||||
|
||||
- Working tree on a branch that has `apps/dreamverse/` (e.g. `will/dreamverse-monorepo`)
|
||||
- Local conda env at `~/miniconda3/envs/fv-main/` with `flashinfer-python`,
|
||||
`cerebras-cloud-sdk`, `openai` installed (override the default path with
|
||||
`DREAMVERSE_PYTHON=/path/to/python`)
|
||||
- `~/.env` exporting `CEREBRAS_API_KEY`, `GROQ_API_KEY`, etc.
|
||||
- pnpm installed at `/home/william5lin/.local/share/pnpm/pnpm` (or in `$PATH`)
|
||||
- `gcc-13` + `g++-13` at `/usr/bin/` (workaround for nvcc gcc-15 rejection)
|
||||
- **Recommended:** native ffmpeg env file at `apps/dreamverse/scripts/ffmpeg-env.sh`
|
||||
(built once via `bash apps/dreamverse/scripts/install_native_ffmpeg.sh`).
|
||||
When present, the deploy sources it inside the backend setsid block so the
|
||||
worker spawns ffmpeg from `$HOME/opt/ffmpeg-native/bin/ffmpeg` (LTO + libx264
|
||||
+ native arch) instead of the system `/usr/bin/ffmpeg`. When missing, the
|
||||
deploy falls back to system ffmpeg with a warning. Set
|
||||
`DREAMVERSE_REQUIRE_NATIVE_FFMPEG=true` to make the missing env file a hard
|
||||
failure.
|
||||
|
||||
If any required prereq is missing, the script fails fast with a clear message.
|
||||
|
||||
## Usage
|
||||
|
||||
```bash
|
||||
# Deploy on GPU 4 with default ports (backend 8009, FE 5274) — torch.compile
|
||||
# and warmup are both OFF by default so first-segment cold start is ~45s
|
||||
# instead of ~3-4min.
|
||||
./.agents/skills/dreamverse-deploy/scripts/dreamverse-deploy.sh 4
|
||||
|
||||
# Deploy on GPU 6 with custom ports
|
||||
./.agents/skills/dreamverse-deploy/scripts/dreamverse-deploy.sh 6 8089 5275
|
||||
|
||||
# Deploy on GPU 0 with warmup enabled
|
||||
./.agents/skills/dreamverse-deploy/scripts/dreamverse-deploy.sh --warmup 0
|
||||
|
||||
# Deploy with torch.compile enabled (max-autotune; first segment ~3-4min,
|
||||
# subsequent segments save ~3s — only worth it for benchmarking)
|
||||
./.agents/skills/dreamverse-deploy/scripts/dreamverse-deploy.sh --torch-compile 4
|
||||
|
||||
# Deploy with both warmup AND torch.compile enabled
|
||||
./.agents/skills/dreamverse-deploy/scripts/dreamverse-deploy.sh --warmup --torch-compile 4
|
||||
|
||||
# Flags can appear before, between, or after positional args
|
||||
./.agents/skills/dreamverse-deploy/scripts/dreamverse-deploy.sh 4 8089 5275 --warmup
|
||||
```
|
||||
|
||||
### Arguments
|
||||
|
||||
| Position | Name | Default | Notes |
|
||||
|---|---|---|---|
|
||||
| 1 | `GPU` | (required) | Physical GPU index, e.g. `4` |
|
||||
| 2 | `BACKEND_PORT` | `8009` | TCP port for the FastAPI server |
|
||||
| 3 | `FRONTEND_PORT` | `5274` | TCP port for the Next.js dev server |
|
||||
|
||||
### Flags
|
||||
|
||||
| Flag | Default | Notes |
|
||||
|---|---|---|
|
||||
| `--warmup` / `--no-warmup` | off | Run GPU warmup at boot (~minutes). Overrides `DREAMVERSE_WARMUP` |
|
||||
| `--torch-compile` / `--no-torch-compile` | off | Enable max-autotune `torch.compile`. First segment ~3-4min when on, ~45s when off. Overrides `DREAMVERSE_TORCH_COMPILE` |
|
||||
| `--nvenc` / `--no-nvenc` | off | Use `h264_nvenc` hardware encoder instead of `libx264` software. Eliminates ~1100ms/segment of CPU encoding cost (raises realtime ratio from ~0.78x → ≥1.0x, eliminating inter-segment buffer-drain stutter). Requires native ffmpeg built with `--enable-nvenc` (the install script's default since the NVENC update). Hard-fails up-front if the binary is missing or lacks NVENC. Overrides `DREAMVERSE_NVENC` |
|
||||
| `-h` / `--help` | — | Show usage |
|
||||
|
||||
Flags can appear in any position relative to the positional args. Explicit flag values always win over env-var defaults.
|
||||
|
||||
### Environment variables (used when no flag is given)
|
||||
|
||||
| Var | Default | Purpose |
|
||||
|---|---|---|
|
||||
| `DREAMVERSE_WARMUP` | `false` | Same as `--warmup`/`--no-warmup`. Flag takes precedence |
|
||||
| `DREAMVERSE_TORCH_COMPILE` | `false` | Same as `--torch-compile`/`--no-torch-compile`. Flag takes precedence |
|
||||
| `DREAMVERSE_NVENC` | `false` | Same as `--nvenc`/`--no-nvenc`. Flag takes precedence |
|
||||
| `DREAMVERSE_PYTHON` | `~/miniconda3/envs/fv-main/bin/python` | Conda env python used for prereq probes (flashinfer import). The wrapper at `apps/dreamverse/scripts/dreamverse-server` still resolves python via the `.venv` symlink, which points at the same interpreter on this dev node |
|
||||
| `DREAMVERSE_REPO_ROOT` | git rev-parse | Repo root override |
|
||||
| `DREAMVERSE_LOG_DIR` | `/tmp/opencode/dreamverse-deploy` | Where to write `backend.log` / `frontend.log` |
|
||||
| `DREAMVERSE_REQUIRE_NATIVE_FFMPEG` | `false` | If `true`, fail when `$HOME/opt/ffmpeg-native/bin/ffmpeg` is absent |
|
||||
|
||||
## What it does
|
||||
|
||||
1. Validates prereqs.
|
||||
2. Kills any process on the target backend/frontend ports + waits for the
|
||||
target GPU to release memory (allows up to 30s for cleanup).
|
||||
3. Sources `~/.env`.
|
||||
4. Exports the env recipe required for boot:
|
||||
- `CUDA_VISIBLE_DEVICES=<gpu>`
|
||||
- `FASTVIDEO_ENABLE_DEVTOOLS=1`
|
||||
- `FASTVIDEO_ENABLE_STARTUP_WARMUP=<DREAMVERSE_WARMUP>`
|
||||
- `FASTVIDEO_GPU_COUNT=1`
|
||||
- `ENABLE_TORCH_COMPILE=<0|1 derived from DREAMVERSE_TORCH_COMPILE>`
|
||||
- `CC=/usr/bin/gcc-13 CXX=/usr/bin/g++-13 CUDAHOSTCXX=/usr/bin/g++-13`
|
||||
- `NVCC_PREPEND_FLAGS="-ccbin /usr/bin/gcc-13 -allow-unsupported-compiler"`
|
||||
- `FASTVIDEO_FFMPEG_BIN=$HOME/opt/ffmpeg-native/bin/ffmpeg` +
|
||||
`FASTVIDEO_VIDEO_CODEC=libx264` (when the native binary exists)
|
||||
5. Launches the backend via `apps/dreamverse/scripts/dreamverse-server` in a
|
||||
detached `setsid` session, captures PID.
|
||||
6. Polls `/readyz` until 200 (max 5 min).
|
||||
7. Launches the frontend via `pnpm run dev:devtools` in a detached session,
|
||||
captures PID.
|
||||
8. Polls FE `/` until 200 (max 60s).
|
||||
9. Prints URLs, PIDs, and log paths.
|
||||
|
||||
## What it does NOT do
|
||||
|
||||
- Does not modify `~/.env` or the FastVideo `.venv`.
|
||||
- Does not push code or commit anything.
|
||||
- Does not run Playwright. Use the e2e wrapper separately:
|
||||
```bash
|
||||
cd apps/dreamverse/web
|
||||
PLAYWRIGHT_SKIP_WEBSERVER=1 BACKEND_URL=http://127.0.0.1:8009 \
|
||||
PLAYWRIGHT_BASE_URL=http://127.0.0.1:5274 \
|
||||
NEXT_PUBLIC_INCLUDE_DEVTOOLS=1 \
|
||||
pnpm exec playwright test
|
||||
```
|
||||
The fast suite (8 specs, ~5s) runs by default; the long-running
|
||||
two-segment audio-continuation spec is gated behind
|
||||
`PLAYWRIGHT_LONG_RUNNING=1` (see below).
|
||||
|
||||
## Long-running e2e (paired with `--warmup --torch-compile`)
|
||||
|
||||
[`apps/dreamverse/web/e2e/long-running-segments.spec.ts`](../../../apps/dreamverse/web/e2e/long-running-segments.spec.ts)
|
||||
drives a real two-segment session through the FE, captures every WS
|
||||
frame, and asserts segments 1 AND 2 both reach `media_segment_complete`
|
||||
with at least one binary fMP4 chunk per segment — the canonical
|
||||
regression guard against the D-20 BrokenPipe pattern documented in
|
||||
[`decisions-log.md D-20`](../../memory/dreamverse-integration/decisions-log.md#d-20).
|
||||
Skipped by default. Enable with:
|
||||
|
||||
```bash
|
||||
./.agents/skills/dreamverse-deploy/scripts/dreamverse-deploy.sh \
|
||||
--warmup --torch-compile 4
|
||||
|
||||
cd apps/dreamverse/web
|
||||
PLAYWRIGHT_SKIP_WEBSERVER=1 \
|
||||
BACKEND_URL=http://127.0.0.1:8009 \
|
||||
PLAYWRIGHT_BASE_URL=http://127.0.0.1:5274 \
|
||||
NEXT_PUBLIC_INCLUDE_DEVTOOLS=1 \
|
||||
PLAYWRIGHT_LONG_RUNNING=1 \
|
||||
pnpm exec playwright test e2e/long-running-segments.spec.ts
|
||||
```
|
||||
|
||||
Expected runtime: ~7-9 minutes on a B200 (torch.compile max-autotune
|
||||
warm-up dominates the cold start; per-test timeout is 900s). The spec
|
||||
hard-fails on any WS `error`/`step_error` frame so the BrokenPipe
|
||||
regression surfaces with the actual ffmpeg/audio diagnostics rather
|
||||
than an opaque "test timed out".
|
||||
|
||||
## Teardown
|
||||
|
||||
Stop both services without redeploying:
|
||||
|
||||
```bash
|
||||
# Stop services on default ports (port-pattern based)
|
||||
./.agents/skills/dreamverse-deploy/scripts/dreamverse-deploy.sh --stop
|
||||
|
||||
# Stop AND nuke any process holding GPU N
|
||||
./.agents/skills/dreamverse-deploy/scripts/dreamverse-deploy.sh --stop 4
|
||||
```
|
||||
|
||||
The redeploy path (`<GPU>` mode) automatically nukes any process holding the
|
||||
target GPU before launching — including orphan `multiproc_executor` worker
|
||||
subprocesses left over from a parent backend that was killed without grace.
|
||||
This was the failure mode of an earlier naive port-only kill: parent dies,
|
||||
children survive, GPU stays full, next deploy OOMs.
|
||||
|
||||
## Notes
|
||||
|
||||
- The wrapper at `apps/dreamverse/scripts/dreamverse-server` is what makes
|
||||
the migrated `apps/dreamverse/server/main.py` run instead of the legacy
|
||||
conda-installed Dreamverse — see [decisions-log.md D-19](../../memory/dreamverse-integration/decisions-log.md#d-19) for why
|
||||
this matters.
|
||||
- The B200 / sm_100a NVCC flags are mandatory on this dev node because the
|
||||
conda toolchain ships gcc-15, which nvcc rejects. If you're on a machine
|
||||
with a supported native gcc, those exports are still safe (no-op when the
|
||||
paths don't exist; the script verifies them upfront).
|
||||
@@ -0,0 +1,466 @@
|
||||
#!/usr/bin/env bash
|
||||
# See ../SKILL.md for full usage.
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
is_pid_alive() {
|
||||
kill -0 "$1" 2>/dev/null
|
||||
}
|
||||
|
||||
terminate_pid() {
|
||||
local pid="$1"
|
||||
local label="${2:-pid=${pid}}"
|
||||
|
||||
[[ -n "${pid}" ]] && [[ "${pid}" != "$$" ]] || return 0
|
||||
is_pid_alive "${pid}" || return 0
|
||||
|
||||
kill "${pid}" 2>/dev/null || true
|
||||
for _ in $(seq 1 10); do
|
||||
is_pid_alive "${pid}" || return 0
|
||||
sleep 0.5
|
||||
done
|
||||
|
||||
if is_pid_alive "${pid}"; then
|
||||
kill -9 "${pid}" 2>/dev/null && echo " force-killed ${label}" || true
|
||||
fi
|
||||
}
|
||||
|
||||
terminate_pattern() {
|
||||
local pattern="$1"
|
||||
local pid
|
||||
|
||||
if ! command -v pgrep >/dev/null 2>&1; then
|
||||
pkill -TERM -f "${pattern}" 2>/dev/null || true
|
||||
sleep 2
|
||||
pkill -KILL -f "${pattern}" 2>/dev/null || true
|
||||
return 0
|
||||
fi
|
||||
|
||||
for pid in $(pgrep -f -- "${pattern}" 2>/dev/null || true); do
|
||||
terminate_pid "${pid}" "pattern='${pattern}' pid=${pid}"
|
||||
done
|
||||
}
|
||||
|
||||
list_port_pids() {
|
||||
local port="$1"
|
||||
|
||||
if command -v lsof >/dev/null 2>&1; then
|
||||
lsof -t -iTCP:"${port}" -sTCP:LISTEN 2>/dev/null || true
|
||||
return 0
|
||||
fi
|
||||
|
||||
ss -tlnp 2>/dev/null | awk -v port=":${port}" '
|
||||
$0 ~ port {
|
||||
while (match($0, /pid=[0-9]+/)) {
|
||||
print substr($0, RSTART + 4, RLENGTH - 4)
|
||||
$0 = substr($0, RSTART + RLENGTH)
|
||||
}
|
||||
}
|
||||
' || true
|
||||
}
|
||||
|
||||
if [[ "${1:-}" == "--stop" ]]; then
|
||||
for pat in 'apps/dreamverse/server/main.py' 'main.py --host 0.0.0.0 --port' 'next dev --port' 'next-server (v'; do
|
||||
terminate_pattern "${pat}"
|
||||
done
|
||||
if [[ -n "${2:-}" ]] && [[ "${2}" =~ ^[0-9]+$ ]]; then
|
||||
gpu_uuid="$(nvidia-smi --query-gpu=index,uuid --format=csv,noheader 2>/dev/null | awk -F', ' -v g="${2}" '$1==g {print $2}')"
|
||||
if [[ -n "${gpu_uuid}" ]]; then
|
||||
for pid in $(nvidia-smi --query-compute-apps=pid,gpu_uuid --format=csv,noheader 2>/dev/null \
|
||||
| awk -F', ' -v u="${gpu_uuid}" '$2==u {print $1}'); do
|
||||
terminate_pid "${pid}" "GPU${2} pid=${pid}"
|
||||
done
|
||||
fi
|
||||
fi
|
||||
sleep 2
|
||||
echo "stopped: ports may take a few seconds to free"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Args
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
usage() {
|
||||
cat <<USAGE
|
||||
Usage: $(basename "$0") [FLAGS] <GPU> [BACKEND_PORT] [FRONTEND_PORT]
|
||||
$(basename "$0") --stop [GPU]
|
||||
|
||||
Positional:
|
||||
GPU Physical GPU index (required), e.g. 4
|
||||
BACKEND_PORT default 8009
|
||||
FRONTEND_PORT default 5274
|
||||
|
||||
Flags (override env vars when both set):
|
||||
--warmup / --no-warmup run GPU warmup at boot (default off)
|
||||
--torch-compile / --no-torch-compile
|
||||
enable max-autotune torch.compile
|
||||
(default off — first segment ~3-4min
|
||||
when on, ~45s when off)
|
||||
--nvenc / --no-nvenc use h264_nvenc hardware encoder (default
|
||||
off — uses libx264 software encoder).
|
||||
Requires native ffmpeg built with NVENC.
|
||||
-h, --help show this help
|
||||
|
||||
Env overrides:
|
||||
DREAMVERSE_WARMUP 'true'|'false' (default false)
|
||||
DREAMVERSE_TORCH_COMPILE 'true'|'false' (default false)
|
||||
DREAMVERSE_NVENC 'true'|'false' (default false)
|
||||
DREAMVERSE_REPO_ROOT default: \$(git rev-parse --show-toplevel)
|
||||
DREAMVERSE_LOG_DIR default: /tmp/opencode/dreamverse-deploy
|
||||
DREAMVERSE_REQUIRE_NATIVE_FFMPEG 'true'|'false' (default false)
|
||||
USAGE
|
||||
}
|
||||
|
||||
WARMUP_OVERRIDE=""
|
||||
TORCH_COMPILE_OVERRIDE=""
|
||||
NVENC_OVERRIDE=""
|
||||
POSITIONAL=()
|
||||
while [[ $# -gt 0 ]]; do
|
||||
case "$1" in
|
||||
-h|--help) usage; exit 0 ;;
|
||||
--warmup) WARMUP_OVERRIDE=true; shift ;;
|
||||
--no-warmup) WARMUP_OVERRIDE=false; shift ;;
|
||||
--torch-compile) TORCH_COMPILE_OVERRIDE=true; shift ;;
|
||||
--no-torch-compile) TORCH_COMPILE_OVERRIDE=false; shift ;;
|
||||
--nvenc) NVENC_OVERRIDE=true; shift ;;
|
||||
--no-nvenc) NVENC_OVERRIDE=false; shift ;;
|
||||
--) shift; while [[ $# -gt 0 ]]; do POSITIONAL+=("$1"); shift; done ;;
|
||||
-*) echo "error: unknown flag '$1'" >&2; usage >&2; exit 2 ;;
|
||||
*) POSITIONAL+=("$1"); shift ;;
|
||||
esac
|
||||
done
|
||||
set -- "${POSITIONAL[@]+"${POSITIONAL[@]}"}"
|
||||
|
||||
if [[ $# -lt 1 ]]; then
|
||||
usage >&2
|
||||
exit 2
|
||||
fi
|
||||
|
||||
GPU="${1}"
|
||||
BACKEND_PORT="${2:-8009}"
|
||||
FRONTEND_PORT="${3:-5274}"
|
||||
|
||||
if ! [[ "${GPU}" =~ ^[0-9]+$ ]]; then
|
||||
echo "error: GPU must be a non-negative integer (got '${GPU}')" >&2
|
||||
exit 2
|
||||
fi
|
||||
|
||||
WARMUP="${WARMUP_OVERRIDE:-${DREAMVERSE_WARMUP:-false}}"
|
||||
case "${WARMUP}" in
|
||||
true|false) ;;
|
||||
*) echo "error: warmup must be 'true' or 'false' (got '${WARMUP}')" >&2; exit 2 ;;
|
||||
esac
|
||||
|
||||
TORCH_COMPILE="${TORCH_COMPILE_OVERRIDE:-${DREAMVERSE_TORCH_COMPILE:-false}}"
|
||||
case "${TORCH_COMPILE}" in
|
||||
true|false) ;;
|
||||
*) echo "error: torch-compile must be 'true' or 'false' (got '${TORCH_COMPILE}')" >&2; exit 2 ;;
|
||||
esac
|
||||
TORCH_COMPILE_FLAG=$([[ "${TORCH_COMPILE}" == "true" ]] && echo 1 || echo 0)
|
||||
|
||||
NVENC="${NVENC_OVERRIDE:-${DREAMVERSE_NVENC:-false}}"
|
||||
case "${NVENC}" in
|
||||
true|false) ;;
|
||||
*) echo "error: nvenc must be 'true' or 'false' (got '${NVENC}')" >&2; exit 2 ;;
|
||||
esac
|
||||
|
||||
REPO_ROOT="${DREAMVERSE_REPO_ROOT:-$(git rev-parse --show-toplevel 2>/dev/null || pwd)}"
|
||||
LOG_DIR="${DREAMVERSE_LOG_DIR:-/tmp/opencode/dreamverse-deploy}"
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Prereq checks
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
bail() { echo "error: $*" >&2; exit 3; }
|
||||
|
||||
[[ -d "${REPO_ROOT}/apps/dreamverse" ]] \
|
||||
|| bail "REPO_ROOT '${REPO_ROOT}' does not contain apps/dreamverse/. Are you on a migration branch?"
|
||||
[[ -x "${REPO_ROOT}/apps/dreamverse/scripts/dreamverse-server" ]] \
|
||||
|| bail "wrapper script missing or not executable: apps/dreamverse/scripts/dreamverse-server"
|
||||
|
||||
CONDA_ENV_PYTHON="${DREAMVERSE_PYTHON:-${HOME}/miniconda3/envs/fv-main/bin/python}"
|
||||
[[ -x "${CONDA_ENV_PYTHON}" ]] \
|
||||
|| bail "conda env python missing at ${CONDA_ENV_PYTHON} (set DREAMVERSE_PYTHON to override)"
|
||||
"${CONDA_ENV_PYTHON}" -c 'import flashinfer' 2>/dev/null \
|
||||
|| bail "flashinfer-python not installed in ${CONDA_ENV_PYTHON} (run: ${CONDA_ENV_PYTHON} -m pip install flashinfer-python --no-build-isolation)"
|
||||
|
||||
PNPM="${PNPM:-}"
|
||||
if [[ -n "${PNPM}" ]]; then
|
||||
PNPM_REQUESTED="${PNPM}"
|
||||
PNPM="$(command -v "${PNPM}" 2>/dev/null || true)"
|
||||
[[ -n "${PNPM}" ]] || bail "pnpm not executable or not in PATH: ${PNPM_REQUESTED} (set PNPM to override)"
|
||||
elif [[ -x "${HOME}/.local/share/pnpm/pnpm" ]]; then
|
||||
PNPM="${HOME}/.local/share/pnpm/pnpm"
|
||||
else
|
||||
PNPM="$(command -v pnpm 2>/dev/null || true)"
|
||||
fi
|
||||
[[ -n "${PNPM}" ]] && [[ -x "${PNPM}" ]] || bail "pnpm not found. Set PNPM, install at ${HOME}/.local/share/pnpm/pnpm, or add pnpm to PATH"
|
||||
|
||||
GCC13="$(command -v "${GCC13:-gcc-13}" 2>/dev/null || true)"
|
||||
GPP13="$(command -v "${GPP13:-g++-13}" 2>/dev/null || true)"
|
||||
[[ -n "${GCC13}" ]] && command -v "${GCC13}" >/dev/null 2>&1 \
|
||||
|| bail "gcc-13 not found or not executable (needed for nvcc workaround). Set GCC13 or install gcc-13 in PATH"
|
||||
[[ -n "${GPP13}" ]] && command -v "${GPP13}" >/dev/null 2>&1 \
|
||||
|| bail "g++-13 not found or not executable (needed for nvcc workaround). Set GPP13 or install g++-13 in PATH"
|
||||
|
||||
[[ -f "${HOME}/.env" ]] || echo "warn: ${HOME}/.env missing — provider API keys may be unset" >&2
|
||||
|
||||
NATIVE_FFMPEG_BIN="${HOME}/opt/ffmpeg-native/bin/ffmpeg"
|
||||
if [[ "${NVENC}" == "true" ]]; then
|
||||
NATIVE_VIDEO_CODEC=h264_nvenc
|
||||
else
|
||||
NATIVE_VIDEO_CODEC=libx264
|
||||
fi
|
||||
REQUIRE_NATIVE_FFMPEG="${DREAMVERSE_REQUIRE_NATIVE_FFMPEG:-false}"
|
||||
case "${REQUIRE_NATIVE_FFMPEG}" in
|
||||
true|false) ;;
|
||||
*) bail "DREAMVERSE_REQUIRE_NATIVE_FFMPEG must be 'true' or 'false' (got '${REQUIRE_NATIVE_FFMPEG}')" ;;
|
||||
esac
|
||||
if [[ -x "${NATIVE_FFMPEG_BIN}" ]]; then
|
||||
if [[ "${NVENC}" == "true" ]]; then
|
||||
encoder_list="$("${NATIVE_FFMPEG_BIN}" -hide_banner -encoders 2>/dev/null || true)"
|
||||
if [[ "${encoder_list}" != *h264_nvenc* ]]; then
|
||||
bail "--nvenc requested but ${NATIVE_FFMPEG_BIN} was not built with NVENC. Rebuild: bash apps/dreamverse/scripts/install_native_ffmpeg.sh (with ENABLE_NVENC=1, the default)"
|
||||
fi
|
||||
if ! "${NATIVE_FFMPEG_BIN}" -hide_banner -loglevel error -y \
|
||||
-f lavfi -i 'color=red:size=64x64:rate=24:duration=0.2' \
|
||||
-c:v h264_nvenc -f null - >/dev/null 2>&1; then
|
||||
bail "--nvenc requested but the GPU on this host has no NVENC silicon (probe failed: 'OpenEncodeSessionEx unsupported device'). Datacenter Blackwell (B200) and some H100 SKUs ship without NVENC; --nvenc only works on hosts with NVENC-capable GPUs (RTX 50-series, T4, A10, etc.)."
|
||||
fi
|
||||
fi
|
||||
echo " native ffmpeg: ${NATIVE_FFMPEG_BIN} (codec=${NATIVE_VIDEO_CODEC})"
|
||||
elif [[ "${REQUIRE_NATIVE_FFMPEG}" == "true" ]] || [[ "${NVENC}" == "true" ]]; then
|
||||
bail "${NATIVE_FFMPEG_BIN} missing (required by --nvenc or DREAMVERSE_REQUIRE_NATIVE_FFMPEG=true). Run: bash apps/dreamverse/scripts/install_native_ffmpeg.sh"
|
||||
else
|
||||
echo "warn: ${NATIVE_FFMPEG_BIN} missing — backend will fall back to system ffmpeg (\$(command -v ffmpeg))." >&2
|
||||
echo " Build native ffmpeg with: bash apps/dreamverse/scripts/install_native_ffmpeg.sh" >&2
|
||||
fi
|
||||
echo " python: ${CONDA_ENV_PYTHON}"
|
||||
|
||||
mkdir -p "${LOG_DIR}"
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Teardown anything on target ports
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
echo "[1/8] killing any existing deploy on ports ${BACKEND_PORT}/${FRONTEND_PORT} and GPU ${GPU}..."
|
||||
|
||||
kill_port_pid() {
|
||||
local port="$1"
|
||||
local pid
|
||||
|
||||
for pid in $(list_port_pids "${port}"); do
|
||||
terminate_pid "${pid}" "port=${port} pid=${pid}"
|
||||
done
|
||||
}
|
||||
|
||||
for pat in "main.py --host 0.0.0.0 --port ${BACKEND_PORT}" "next dev --port ${FRONTEND_PORT}" "NEXT_PUBLIC_INCLUDE_DEVTOOLS=1 next dev --port ${FRONTEND_PORT}"; do
|
||||
terminate_pattern "${pat}"
|
||||
done
|
||||
kill_port_pid "${BACKEND_PORT}"
|
||||
kill_port_pid "${FRONTEND_PORT}"
|
||||
|
||||
gpu_uuid="$(nvidia-smi --query-gpu=index,uuid --format=csv,noheader 2>/dev/null | awk -F', ' -v g="${GPU}" '$1==g {print $2}')"
|
||||
if [[ -n "${gpu_uuid}" ]]; then
|
||||
for pid in $(nvidia-smi --query-compute-apps=pid,gpu_uuid --format=csv,noheader 2>/dev/null \
|
||||
| awk -F', ' -v u="${gpu_uuid}" '$2==u {print $1}'); do
|
||||
if [[ -n "${pid}" ]] && [[ "${pid}" != "$$" ]]; then
|
||||
cmd="$(ps -p "${pid}" -o comm= 2>/dev/null || true)"
|
||||
terminate_pid "${pid}" "GPU${GPU} pid=${pid} (${cmd:-?})"
|
||||
fi
|
||||
done
|
||||
fi
|
||||
|
||||
for i in $(seq 1 30); do
|
||||
free_be=true
|
||||
free_fe=true
|
||||
ss -tln 2>/dev/null | grep -qE ":${BACKEND_PORT}\b" && free_be=false
|
||||
ss -tln 2>/dev/null | grep -qE ":${FRONTEND_PORT}\b" && free_fe=false
|
||||
gpu_mem="$(nvidia-smi --query-gpu=memory.used --format=csv,noheader,nounits 2>/dev/null | sed -n "$((GPU + 1))p" || echo 99999)"
|
||||
if "${free_be}" && "${free_fe}" && [[ "${gpu_mem}" -lt 1000 ]]; then
|
||||
break
|
||||
fi
|
||||
sleep 1
|
||||
done
|
||||
|
||||
gpu_mem="$(nvidia-smi --query-gpu=memory.used --format=csv,noheader,nounits 2>/dev/null | sed -n "$((GPU + 1))p" || echo 0)"
|
||||
echo " ports cleared; GPU${GPU} at ${gpu_mem} MiB"
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Launch backend
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
echo "[2/8] launching backend on GPU ${GPU} port ${BACKEND_PORT} (warmup=${WARMUP} torch_compile=${TORCH_COMPILE} nvenc=${NVENC})..."
|
||||
|
||||
backend_log="${LOG_DIR}/backend-gpu${GPU}.log"
|
||||
: > "${backend_log}"
|
||||
|
||||
setsid bash -c "
|
||||
set -a
|
||||
if [[ -f \"${HOME}/.env\" ]]; then
|
||||
source \"${HOME}/.env\"
|
||||
fi
|
||||
set +a
|
||||
if [[ -x \"${NATIVE_FFMPEG_BIN}\" ]]; then
|
||||
export FASTVIDEO_FFMPEG_BIN=\"${NATIVE_FFMPEG_BIN}\"
|
||||
export FASTVIDEO_VIDEO_CODEC=\"${NATIVE_VIDEO_CODEC}\"
|
||||
fi
|
||||
export DREAMVERSE_PYTHON=\"${CONDA_ENV_PYTHON}\"
|
||||
export CUDA_VISIBLE_DEVICES=${GPU}
|
||||
export FASTVIDEO_ENABLE_DEVTOOLS=1
|
||||
export FASTVIDEO_ENABLE_STARTUP_WARMUP=${WARMUP}
|
||||
export FASTVIDEO_GPU_COUNT=1
|
||||
export ENABLE_TORCH_COMPILE=${TORCH_COMPILE_FLAG}
|
||||
export CC=${GCC13}
|
||||
export CXX=${GPP13}
|
||||
export CUDAHOSTCXX=${GPP13}
|
||||
export NVCC_PREPEND_FLAGS=\"-ccbin ${GCC13} -allow-unsupported-compiler\"
|
||||
cd \"${REPO_ROOT}\"
|
||||
exec ./apps/dreamverse/scripts/dreamverse-server --host 0.0.0.0 --port ${BACKEND_PORT}
|
||||
" > "${backend_log}" 2>&1 < /dev/null &
|
||||
disown
|
||||
|
||||
# Wait briefly, then resolve actual python PID (the inner process, not the
|
||||
# wrapper bash).
|
||||
sleep 4
|
||||
backend_pid="$(pgrep -f "main.py --host 0.0.0.0 --port ${BACKEND_PORT}" | head -1 || true)"
|
||||
|
||||
if [[ -z "${backend_pid}" ]]; then
|
||||
echo "error: backend failed to spawn. Last 30 lines of log:" >&2
|
||||
tail -30 "${backend_log}" >&2
|
||||
exit 4
|
||||
fi
|
||||
|
||||
echo " backend pid=${backend_pid} log=${backend_log}"
|
||||
|
||||
# Poll /readyz. Deadline scales with warmup + torch.compile flags
|
||||
# because warmup runs two synthetic segments before /readyz=200, and
|
||||
# torch.compile max-autotune adds ~3-4min cold start to the first
|
||||
# segment. Empirical worst case (warmup=true, torch_compile=true):
|
||||
# ~7 min on B200; we budget 15 min for safety.
|
||||
if [[ "${WARMUP}" == "true" ]] && [[ "${TORCH_COMPILE}" == "true" ]]; then
|
||||
READYZ_BUDGET_SECONDS=900
|
||||
elif [[ "${WARMUP}" == "true" ]] || [[ "${TORCH_COMPILE}" == "true" ]]; then
|
||||
READYZ_BUDGET_SECONDS=480
|
||||
else
|
||||
READYZ_BUDGET_SECONDS=300
|
||||
fi
|
||||
READYZ_POLL_INTERVAL=6
|
||||
READYZ_MAX_ITERS=$(( READYZ_BUDGET_SECONDS / READYZ_POLL_INTERVAL ))
|
||||
|
||||
echo "[3/8] polling http://127.0.0.1:${BACKEND_PORT}/readyz (budget=${READYZ_BUDGET_SECONDS}s) ..."
|
||||
ready=0
|
||||
for i in $(seq 1 ${READYZ_MAX_ITERS}); do
|
||||
code="$(curl -s -o /dev/null -w '%{http_code}' --max-time 2 "http://127.0.0.1:${BACKEND_PORT}/readyz" 2>/dev/null || echo 000)"
|
||||
if [[ "${code}" == "200" ]]; then
|
||||
ready=1
|
||||
break
|
||||
fi
|
||||
if ! kill -0 "${backend_pid}" 2>/dev/null; then
|
||||
echo "error: backend pid ${backend_pid} died. Last 50 lines:" >&2
|
||||
tail -50 "${backend_log}" >&2
|
||||
exit 5
|
||||
fi
|
||||
sleep ${READYZ_POLL_INTERVAL}
|
||||
done
|
||||
|
||||
if [[ "${ready}" != "1" ]]; then
|
||||
echo "error: backend did not become /readyz=200 within ${READYZ_BUDGET_SECONDS}s. Last 50 lines:" >&2
|
||||
tail -50 "${backend_log}" >&2
|
||||
exit 5
|
||||
fi
|
||||
|
||||
echo "[4/8] backend /readyz OK"
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Launch frontend
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
echo "[5/8] launching frontend on port ${FRONTEND_PORT}..."
|
||||
|
||||
frontend_log="${LOG_DIR}/frontend-port${FRONTEND_PORT}.log"
|
||||
: > "${frontend_log}"
|
||||
|
||||
# Resolve dev script: dev:devtools forces port 5274 + devtools env. If the
|
||||
# requested port differs, run `next dev --port` directly with devtools env.
|
||||
fe_cmd="run dev:devtools"
|
||||
if [[ "${FRONTEND_PORT}" != "5274" ]]; then
|
||||
fe_cmd="exec next dev --port ${FRONTEND_PORT}"
|
||||
fi
|
||||
|
||||
setsid bash -c "
|
||||
cd \"${REPO_ROOT}/apps/dreamverse/web\"
|
||||
export NEXT_PUBLIC_INCLUDE_DEVTOOLS=1
|
||||
export BACKEND_URL=http://127.0.0.1:${BACKEND_PORT}
|
||||
export BACKEND_HOST=127.0.0.1
|
||||
export BACKEND_PORT=${BACKEND_PORT}
|
||||
exec '${PNPM}' ${fe_cmd}
|
||||
" > "${frontend_log}" 2>&1 < /dev/null &
|
||||
disown
|
||||
|
||||
sleep 4
|
||||
frontend_pid="$(pgrep -f "next dev --port ${FRONTEND_PORT}" | head -1 || true)"
|
||||
if [[ -z "${frontend_pid}" ]]; then
|
||||
echo "error: frontend failed to spawn. Last 30 lines:" >&2
|
||||
tail -30 "${frontend_log}" >&2
|
||||
exit 6
|
||||
fi
|
||||
|
||||
echo " frontend pid=${frontend_pid} log=${frontend_log}"
|
||||
|
||||
# Poll FE root
|
||||
echo "[6/8] polling http://127.0.0.1:${FRONTEND_PORT}/ ..."
|
||||
fe_ready=0
|
||||
for i in $(seq 1 30); do
|
||||
code="$(curl -s -o /dev/null -w '%{http_code}' --max-time 2 "http://127.0.0.1:${FRONTEND_PORT}/" 2>/dev/null || echo 000)"
|
||||
if [[ "${code}" == "200" ]]; then
|
||||
fe_ready=1
|
||||
break
|
||||
fi
|
||||
if ! kill -0 "${frontend_pid}" 2>/dev/null; then
|
||||
echo "error: frontend pid ${frontend_pid} died. Last 30 lines:" >&2
|
||||
tail -30 "${frontend_log}" >&2
|
||||
exit 7
|
||||
fi
|
||||
sleep 2
|
||||
done
|
||||
|
||||
if [[ "${fe_ready}" != "1" ]]; then
|
||||
echo "error: frontend did not respond 200 within 60s. Last 30 lines:" >&2
|
||||
tail -30 "${frontend_log}" >&2
|
||||
exit 7
|
||||
fi
|
||||
|
||||
echo "[7/8] frontend / OK"
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Print summary
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
cwd="$(readlink "/proc/${backend_pid}/cwd" 2>/dev/null || echo unknown)"
|
||||
gpu_mem_now="$(nvidia-smi --query-gpu=memory.used --format=csv,noheader,nounits 2>/dev/null | sed -n "$((GPU + 1))p" || echo 0)"
|
||||
ffmpeg_in_use="$(tr '\0' '\n' < "/proc/${backend_pid}/environ" 2>/dev/null | sed -n 's/^FASTVIDEO_FFMPEG_BIN=//p' | head -1)"
|
||||
[[ -z "${ffmpeg_in_use}" ]] && ffmpeg_in_use="$(command -v ffmpeg 2>/dev/null || echo '<not found>') (system fallback)"
|
||||
|
||||
cat <<SUMMARY
|
||||
[8/8] redeploy OK
|
||||
|
||||
Frontend : http://localhost:${FRONTEND_PORT} (PID ${frontend_pid})
|
||||
Backend : http://localhost:${BACKEND_PORT} (PID ${backend_pid})
|
||||
cwd=${cwd}
|
||||
gpu=${GPU} mem=${gpu_mem_now} MiB
|
||||
ffmpeg=${ffmpeg_in_use}
|
||||
|
||||
Logs : ${backend_log}
|
||||
${frontend_log}
|
||||
|
||||
Stop : ./.agents/skills/dreamverse-deploy/scripts/dreamverse-deploy.sh --stop
|
||||
|
||||
E2E : cd apps/dreamverse/web && \\
|
||||
PLAYWRIGHT_SKIP_WEBSERVER=1 \\
|
||||
BACKEND_URL=http://127.0.0.1:${BACKEND_PORT} \\
|
||||
PLAYWRIGHT_BASE_URL=http://127.0.0.1:${FRONTEND_PORT} \\
|
||||
NEXT_PUBLIC_INCLUDE_DEVTOOLS=1 \\
|
||||
pnpm exec playwright test
|
||||
SUMMARY
|
||||
@@ -0,0 +1,128 @@
|
||||
---
|
||||
name: evaluate-video-quality
|
||||
description: Evaluate generated video quality using available metrics (SSIM, loss trajectory, caption consistency)
|
||||
---
|
||||
|
||||
# Evaluate Video Quality
|
||||
|
||||
## Purpose
|
||||
Assess the quality of videos generated by a training run. Combines multiple
|
||||
signals to give a holistic quality assessment. This skill is **evolving** —
|
||||
new metrics will be added as they are developed.
|
||||
|
||||
## Prerequisites
|
||||
- Generated videos available locally or via W&B artifacts.
|
||||
- For SSIM: reference videos from official implementations.
|
||||
- For caption consistency: LLM access (optional, stub for now).
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Required | Description |
|
||||
|-----------|----------|-------------|
|
||||
| `video_paths` | Yes | List of paths to generated videos |
|
||||
| `reference_paths` | No | Paths to reference videos (for SSIM) |
|
||||
| `prompts` | No | Prompts used to generate videos (for caption check) |
|
||||
| `loss_summary` | No | Path to W&B summary JSON (for loss trajectory) |
|
||||
| `metrics` | No | Which metrics to run (default: all available) |
|
||||
|
||||
## Available Metrics
|
||||
|
||||
Check `.agents/memory/evaluation-registry/README.md` for the current catalog.
|
||||
|
||||
### SSIM (Active)
|
||||
|
||||
Leverages the existing infrastructure in `fastvideo/tests/ssim/`.
|
||||
|
||||
```bash
|
||||
pytest fastvideo/tests/ssim/ -vs --video-path <generated> --reference-path <reference>
|
||||
```
|
||||
|
||||
Or use the SSIM utility directly:
|
||||
|
||||
```python
|
||||
from fastvideo.tests.ssim.ssim_utils import compute_ssim
|
||||
score = compute_ssim(generated_video, reference_video)
|
||||
# score > 0.85 is typically "acceptable"
|
||||
```
|
||||
|
||||
**Interpretation**:
|
||||
| SSIM Range | Quality |
|
||||
|------------|---------|
|
||||
| > 0.90 | Excellent — very close to reference |
|
||||
| 0.80–0.90 | Good — acceptable for most uses |
|
||||
| 0.70–0.80 | Fair — noticeable differences |
|
||||
| < 0.70 | Poor — significant quality issues |
|
||||
|
||||
### Loss Trajectory (Active)
|
||||
|
||||
Analyze the loss curve shape from W&B summary:
|
||||
|
||||
```python
|
||||
import json
|
||||
with open(loss_summary_path) as f:
|
||||
summary = json.load(f)
|
||||
|
||||
final_loss = summary["train_loss"]
|
||||
runtime = summary["_runtime"]
|
||||
steps = summary["_step"]
|
||||
```
|
||||
|
||||
**Early-stage heuristics** (first 500 steps):
|
||||
- Loss should be decreasing (even slightly).
|
||||
- Grad norm should be stable (no wild oscillations).
|
||||
- If loss is flat or increasing, flag for review.
|
||||
|
||||
### Caption Consistency (Draft — Not Yet Calibrated)
|
||||
|
||||
Use an LLM to evaluate whether the video content matches the input prompt.
|
||||
|
||||
```
|
||||
Prompt: "A golden retriever playing in the snow"
|
||||
Video: <path>
|
||||
|
||||
Score the video on:
|
||||
1. Object presence (is there a golden retriever?)
|
||||
2. Action accuracy (is it playing?)
|
||||
3. Environment match (is there snow?)
|
||||
4. Overall coherence (does it look natural?)
|
||||
|
||||
Each 1-5, total /20.
|
||||
```
|
||||
|
||||
> ⚠️ This metric is in **draft** status. Results should not be treated as
|
||||
> ground truth until calibrated against human judgments.
|
||||
|
||||
## Steps
|
||||
|
||||
1. **Identify available metrics** — Check `.agents/memory/evaluation-registry/README.md`.
|
||||
2. **Run each metric** — Collect scores.
|
||||
3. **Aggregate** — Produce a combined quality report.
|
||||
4. **Log** — Update the experiment journal with quality results.
|
||||
|
||||
## Outputs
|
||||
|
||||
```markdown
|
||||
## Video Quality Report: <experiment_name>
|
||||
|
||||
| Metric | Score | Threshold | Status |
|
||||
|--------|-------|-----------|--------|
|
||||
| SSIM (avg) | 0.87 | > 0.80 | ✅ Pass |
|
||||
| Loss trajectory | decreasing | decreasing | ✅ Pass |
|
||||
| Caption consistency | 16/20 | > 14/20 | ✅ Pass |
|
||||
|
||||
### Per-Video Scores
|
||||
| Video | SSIM | Caption |
|
||||
|-------|------|---------|
|
||||
| video_001.mp4 | 0.89 | 17/20 |
|
||||
| video_002.mp4 | 0.85 | 15/20 |
|
||||
```
|
||||
|
||||
## References
|
||||
- `fastvideo/tests/ssim/` — SSIM test infrastructure
|
||||
- `fastvideo/tests/training/Vanilla/test_training_loss.py` — loss comparison
|
||||
- `.agents/memory/evaluation-registry/README.md` — metric catalog
|
||||
|
||||
## Changelog
|
||||
| Date | Change |
|
||||
|------|--------|
|
||||
| 2026-03-02 | Initial version with SSIM, loss trajectory, caption consistency stub |
|
||||
@@ -0,0 +1,9 @@
|
||||
{"name": "launch-experiment", "description": "Generate and execute a training launch command for FastVideo models", "path": "launch-experiment/SKILL.md", "status": "draft", "trust": "low"}
|
||||
{"name": "monitor-experiment", "description": "Poll a running W&B training run for progress and emit structured alerts", "path": "monitor-experiment/SKILL.md", "status": "draft", "trust": "low"}
|
||||
{"name": "summarize-run", "description": "Extract a W&B run summary into a structured experiment report", "path": "summarize-run/SKILL.md", "status": "draft", "trust": "low"}
|
||||
{"name": "log-experiment", "description": "Append or update an experiment entry in the experiment journal", "path": "log-experiment/SKILL.md", "status": "draft", "trust": "low"}
|
||||
{"name": "evaluate-video-quality", "description": "Evaluate generated video quality using available metrics (SSIM, loss trajectory, caption consistency)", "path": "evaluate-video-quality/SKILL.md", "status": "draft", "trust": "low"}
|
||||
{"name": "seed-ssim-references", "description": "Run a new or updated fastvideo/tests/ssim/ test on Modal, pull generated videos, and upload them to FastVideo/ssim-reference-videos so the test has a regression baseline", "path": "seed-ssim-references/SKILL.md", "status": "draft", "trust": "low"}
|
||||
{"name": "reseed-ssim-references", "description": "Re-seed (overwrite) HF reference videos for an existing fastvideo/tests/ssim/ test and a single model id on Modal L40S. Always backs up current refs first, regenerates on Modal, pauses for the user to eyeball before-vs-after, then uploads with --force scoped to --model-id. Sister skill to seed-ssim-references; use when intentional code change has invalidated existing refs", "path": "reseed-ssim-references/SKILL.md", "status": "draft", "trust": "low"}
|
||||
{"name": "decompose-pipeline-pr", "description": "Decompose an oversized FastVideo pipeline PR into a stack of independently-reviewable PRs. Tiers the diff by blast radius (invisible / dead code / cross-cutting infra / activation), produces a branch graph and worktree bootstrap, drafts the AGENTS.md manifest, flags missing tests on cross-cutting infra changes, and extracts lessons from the PR body. Worked example: PR #1280 daVinci-MagiHuman (9.8k LOC) decomposed into 10 stacked PRs.", "path": "decompose-pipeline-pr/SKILL.md", "status": "tested", "trust": "medium"}
|
||||
{"name": "add-model", "description": "Add a new model (or variant) to FastVideo: DiT + configs + pipeline + presets + registry + tests. Walks through FastVideo's single stage-based pipeline architecture with exact file paths and registration hooks.", "path": "add-model/SKILL.md", "status": "draft", "trust": "low"}
|
||||
@@ -0,0 +1,127 @@
|
||||
---
|
||||
name: launch-experiment
|
||||
description: Generate and execute a training launch command for FastVideo models
|
||||
---
|
||||
|
||||
# Launch Experiment
|
||||
|
||||
## Purpose
|
||||
Construct a fully-specified `torchrun` training command for a FastVideo model
|
||||
given a target pipeline, dataset, and hyperparameter overrides. This skill
|
||||
automates the boilerplate of setting environment variables, picking the right
|
||||
entrypoint, and applying defaults from the closest example script.
|
||||
|
||||
## Prerequisites
|
||||
- The repo is cloned and `fastvideo` is installed (`uv pip install -e ".[dev]"`).
|
||||
- Dataset is preprocessed (see `docs/training/data_preprocess.md`).
|
||||
- `WANDB_API_KEY` is set in the environment (or `WANDB_MODE=offline` for local).
|
||||
- GPU resources are available (multi-GPU requires NCCL).
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Required | Description |
|
||||
|-----------|----------|-------------|
|
||||
| `pipeline` | Yes | Training pipeline type: `finetune`, `distill-dmd`, `self-forcing`, `lora`, `consistency` |
|
||||
| `model` | Yes | Model family: `wan-t2v-1.3B`, `wan-i2v-14B`, `ltx2`, `matrixgame` |
|
||||
| `data_path` | Yes | Path to preprocessed dataset (parquet) |
|
||||
| `num_gpus` | Yes | Number of GPUs |
|
||||
| `overrides` | No | Dict of hyperparameter overrides (any CLI arg) |
|
||||
| `output_dir` | No | Output directory (default: `outputs/<model>_<pipeline>`) |
|
||||
| `run_name` | No | W&B run name (default: auto-generated) |
|
||||
|
||||
## Steps
|
||||
|
||||
### 1. Identify the training entrypoint
|
||||
|
||||
| Pipeline | Entrypoint |
|
||||
|----------|-----------|
|
||||
| `finetune` (Wan T2V) | `fastvideo/training/wan_training_pipeline.py` |
|
||||
| `finetune` (Wan I2V) | `fastvideo/training/wan_i2v_training_pipeline.py` |
|
||||
| `finetune` (LTX-2) | `fastvideo/training/ltx2_training_pipeline.py` |
|
||||
| `finetune` (MatrixGame) | `fastvideo/training/matrixgame_training_pipeline.py` |
|
||||
| `distill-dmd` | `fastvideo/training/wan_distillation_pipeline.py` |
|
||||
| `self-forcing` | `fastvideo/training/wan_self_forcing_distillation_pipeline.py` |
|
||||
|
||||
### 2. Resolve default hyperparameters
|
||||
|
||||
Find the closest example script in `examples/training/` for the model:
|
||||
|
||||
| Model | Example Script Directory |
|
||||
|-------|-------------------------|
|
||||
| `wan-t2v-1.3B` | `examples/training/finetune/wan_t2v_1.3B/crush_smol/` |
|
||||
| `wan-i2v-14B` | `examples/training/finetune/wan_i2v_14B_480p/crush_smol/` |
|
||||
| `ltx2` | `examples/training/finetune/ltx2/` |
|
||||
| `matrixgame` | `examples/training/finetune/MatrixGame2.0/` |
|
||||
| `distill-dmd` | `scripts/distill/v1_distill_dmd_wan.sh` |
|
||||
|
||||
Read the script to extract default values for:
|
||||
- `--learning_rate`, `--train_batch_size`, `--sp_size`, `--tp_size`
|
||||
- `--num_latent_t`, `--num_height`, `--num_width`, `--num_frames`
|
||||
- `--gradient_accumulation_steps`, `--max_train_steps`
|
||||
- `--mixed_precision`, `--weight_decay`, `--max_grad_norm`
|
||||
- `--validation_steps`, `--validation_sampling_steps`
|
||||
|
||||
### 3. Set environment variables
|
||||
|
||||
```bash
|
||||
export WANDB_API_KEY="${WANDB_API_KEY}"
|
||||
export WANDB_BASE_URL="https://api.wandb.ai"
|
||||
export FASTVIDEO_ATTENTION_BACKEND=FLASH_ATTN
|
||||
export TOKENIZERS_PARALLELISM=false
|
||||
export TRITON_CACHE_DIR=/tmp/triton_cache
|
||||
```
|
||||
|
||||
### 4. Construct the torchrun command
|
||||
|
||||
```bash
|
||||
torchrun --nnodes 1 --nproc_per_node <num_gpus> \
|
||||
<entrypoint> \
|
||||
--pretrained_model_name_or_path <model_hf_id> \
|
||||
--data_path "<data_path>" \
|
||||
--output_dir "<output_dir>" \
|
||||
--wandb_run_name "<run_name>" \
|
||||
--tracker_project_name "<project_name>" \
|
||||
--log_validation \
|
||||
<...all hyperparameters...>
|
||||
```
|
||||
|
||||
### 5. Log to experiment journal
|
||||
|
||||
After launching, append an entry to `.agents/memory/experiment-journal/README.md`:
|
||||
|
||||
```markdown
|
||||
## [YYYY-MM-DD] Experiment: <run_name>
|
||||
- **Hypothesis**: <user-provided or auto-generated>
|
||||
- **Config**: model=<model>, lr=<lr>, sp_size=<sp>, gpus=<n>, script=<entrypoint>
|
||||
- **W&B run**: <pending — will be updated by monitor skill>
|
||||
- **Status**: running
|
||||
```
|
||||
|
||||
## Outputs
|
||||
- A ready-to-execute shell command.
|
||||
- An experiment journal entry.
|
||||
|
||||
## Example Usage
|
||||
|
||||
```
|
||||
Launch a Wan T2V 1.3B finetune on 4 GPUs with lr=5e-5 and max_train_steps=1000:
|
||||
|
||||
pipeline: finetune
|
||||
model: wan-t2v-1.3B
|
||||
data_path: data/crush_smol_preprocessed/
|
||||
num_gpus: 4
|
||||
overrides:
|
||||
learning_rate: 5e-5
|
||||
max_train_steps: 1000
|
||||
```
|
||||
|
||||
## References
|
||||
- `examples/training/finetune/wan_t2v_1.3B/crush_smol/finetune_t2v.sh`
|
||||
- `scripts/distill/v1_distill_dmd_wan.sh`
|
||||
- `docs/training/finetune.md` (training arguments table)
|
||||
- `fastvideo/training/trackers.py` (tracker initialization)
|
||||
|
||||
## Changelog
|
||||
| Date | Change |
|
||||
|------|--------|
|
||||
| 2026-03-02 | Initial version |
|
||||
@@ -0,0 +1,87 @@
|
||||
---
|
||||
name: log-experiment
|
||||
description: Append or update an experiment entry in the experiment journal
|
||||
---
|
||||
|
||||
# Log Experiment
|
||||
|
||||
## Purpose
|
||||
Create or update an entry in `.agents/memory/experiment-journal/README.md` to maintain
|
||||
a living record of all experiments and their outcomes.
|
||||
|
||||
## Prerequisites
|
||||
- `.agents/memory/experiment-journal/README.md` exists.
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Required | Description |
|
||||
|-----------|----------|-------------|
|
||||
| `name` | Yes | Experiment name / identifier |
|
||||
| `hypothesis` | No | What you expected to learn |
|
||||
| `config` | Yes | Key config: model, lr, sp_size, gpus, script |
|
||||
| `wandb_run` | No | W&B run ID or URL |
|
||||
| `duration` | No | Total wall time |
|
||||
| `metrics` | No | Key metrics dict (loss, step_time, grad_norm) |
|
||||
| `checkpoint` | No | Path to checkpoint |
|
||||
| `insight` | No | What was learned |
|
||||
| `status` | Yes | `running`, `completed`, `failed`, `abandoned` |
|
||||
| `lessons` | No | Paths to related lesson files |
|
||||
|
||||
## Steps
|
||||
|
||||
### 1. Check for existing entry
|
||||
|
||||
Search `.agents/memory/experiment-journal/README.md` for an entry with the same name.
|
||||
If found, update it instead of creating a duplicate.
|
||||
|
||||
### 2. Format the entry
|
||||
|
||||
```markdown
|
||||
## [YYYY-MM-DD] Experiment: <name>
|
||||
- **Hypothesis**: <hypothesis or "N/A">
|
||||
- **Config**: model=<model>, lr=<lr>, sp_size=<sp>, gpus=<n>, script=<script>
|
||||
- **W&B run**: <wandb_run or "pending">
|
||||
- **Duration**: <duration or "in progress">
|
||||
- **Key metrics**: loss=<loss>, step_time=<step_time>, grad_norm=<grad_norm>
|
||||
- **Checkpoint**: <checkpoint or "N/A">
|
||||
- **Insight**: <insight or "pending">
|
||||
- **Status**: <status>
|
||||
- **Related lessons**: <lessons or "none">
|
||||
```
|
||||
|
||||
### 3. Insert at the top of the journal
|
||||
|
||||
New entries go at the top of the file (after the header), so the most recent
|
||||
experiments are always visible first.
|
||||
|
||||
### 4. Warn on duplicates
|
||||
|
||||
If a similar experiment name exists with `status: completed`, warn that this
|
||||
may be a repeat. If it's `status: running`, assume this is an update.
|
||||
|
||||
## Outputs
|
||||
- Updated `.agents/memory/experiment-journal/README.md`.
|
||||
|
||||
## Example Usage
|
||||
|
||||
```
|
||||
Log a completed experiment:
|
||||
|
||||
name: wan-t2v-finetune-lr5e5-sp4
|
||||
config: model=wan-t2v-1.3B, lr=5e-5, sp_size=4, gpus=4
|
||||
wandb_run: fastvideo/training/run_abc123
|
||||
duration: 2h 15m
|
||||
metrics: {loss: 0.065, step_time: 2.3, grad_norm: 0.35}
|
||||
checkpoint: outputs/wan_finetune/checkpoint-1000
|
||||
insight: LR 5e-5 converges 30% faster than 1e-5 with no quality loss
|
||||
status: completed
|
||||
```
|
||||
|
||||
## References
|
||||
- `.agents/memory/experiment-journal/README.md` — journal file
|
||||
- `.agents/workflows/experiment-lifecycle.md` — when to log
|
||||
|
||||
## Changelog
|
||||
| Date | Change |
|
||||
|------|--------|
|
||||
| 2026-03-02 | Initial version |
|
||||
@@ -0,0 +1,134 @@
|
||||
---
|
||||
name: monitor-experiment
|
||||
description: Poll a running W&B training run for progress and emit structured alerts
|
||||
---
|
||||
|
||||
# Monitor Experiment
|
||||
|
||||
## Purpose
|
||||
Continuously (or on-demand) check a running experiment's W&B metrics and emit
|
||||
alerts for anomalies. Supports the "30-minute quality check" paradigm: after
|
||||
the first 30 minutes of a long training run, produce a checkpoint quality
|
||||
report before committing more resources.
|
||||
|
||||
## Prerequisites
|
||||
- `WANDB_API_KEY` is set in the environment.
|
||||
- The experiment is actively logging to W&B (not in `WANDB_MODE=offline`).
|
||||
- For offline mode: read from local `wandb-summary.json` instead.
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Required | Description |
|
||||
|-----------|----------|-------------|
|
||||
| `run_id` | Yes* | W&B run ID (e.g., `entity/project/run_id`) |
|
||||
| `output_dir` | Yes* | Local output directory (for offline mode fallback) |
|
||||
| `poll_interval` | No | Seconds between polls (default: 60) |
|
||||
| `alert_on` | No | List of alert conditions to enable (default: all) |
|
||||
|
||||
\* One of `run_id` or `output_dir` is required.
|
||||
|
||||
## Steps
|
||||
|
||||
### 1. Connect to the run
|
||||
|
||||
**Online mode** (preferred):
|
||||
|
||||
```python
|
||||
import wandb
|
||||
api = wandb.Api()
|
||||
run = api.run("<run_id>")
|
||||
```
|
||||
|
||||
**Offline fallback**:
|
||||
|
||||
```python
|
||||
import json
|
||||
summary_path = f"{output_dir}/tracker/wandb/latest-run/files/wandb-summary.json"
|
||||
with open(summary_path) as f:
|
||||
summary = json.load(f)
|
||||
```
|
||||
|
||||
### 2. Track key metrics
|
||||
|
||||
| Metric | W&B Key | Description |
|
||||
|--------|---------|-------------|
|
||||
| Training loss | `train_loss` | Primary training loss |
|
||||
| Gradient norm | `grad_norm` | Gradient magnitude |
|
||||
| Step time | `step_time` | Wall-clock seconds per step |
|
||||
| Learning rate | `learning_rate` | Current LR |
|
||||
| Avg step time | `avg_step_time` | Running average step time |
|
||||
| Validation videos | `validation_videos_*` | Generated validation samples |
|
||||
|
||||
### 3. Evaluate alert conditions
|
||||
|
||||
| Alert | Condition | Severity |
|
||||
|-------|-----------|----------|
|
||||
| **Loss spike** | `current_loss > 3 × rolling_avg_loss` | 🔴 Critical |
|
||||
| **NaN/Inf gradient** | `grad_norm` is NaN or Inf | 🔴 Critical |
|
||||
| **Step time regression** | `step_time > 2 × baseline_step_time` | 🟡 Warning |
|
||||
| **No progress** | No new W&B logs for > 10 minutes | 🟡 Warning |
|
||||
| **Loss plateau** | Loss change < 1% over last 100 steps | 🟢 Info |
|
||||
|
||||
### 4. Emit structured status
|
||||
|
||||
Output format (agent-consumable):
|
||||
|
||||
```json
|
||||
{
|
||||
"run_id": "...",
|
||||
"step": 500,
|
||||
"metrics": {
|
||||
"train_loss": 0.078,
|
||||
"grad_norm": 0.41,
|
||||
"step_time": 2.5,
|
||||
"learning_rate": 1e-6
|
||||
},
|
||||
"alerts": [
|
||||
{"type": "loss_spike", "severity": "critical", "message": "Loss jumped to 0.45 (avg: 0.08)"}
|
||||
],
|
||||
"status": "running"
|
||||
}
|
||||
```
|
||||
|
||||
### 5. 30-Minute Quality Check
|
||||
|
||||
After the first 30 minutes of wall-clock time:
|
||||
1. Summarize the loss curve shape (decreasing? at what rate?).
|
||||
2. Check if validation videos have been generated.
|
||||
3. Report step count, loss at start vs. current, and estimated time to completion.
|
||||
4. Produce a go/no-go recommendation.
|
||||
|
||||
```markdown
|
||||
## 30-Minute Check: <run_name>
|
||||
- **Steps completed**: 150
|
||||
- **Loss**: 0.12 → 0.08 (↓ 33%)
|
||||
- **Grad norm**: stable at ~0.4
|
||||
- **Step time**: 2.5s/step (consistent)
|
||||
- **Validation videos**: 5 generated at step 100
|
||||
- **Recommendation**: ✅ Continue — loss is decreasing normally
|
||||
```
|
||||
|
||||
## Outputs
|
||||
- Structured JSON status updates.
|
||||
- Alert messages for anomalous conditions.
|
||||
- 30-minute checkpoint quality report.
|
||||
|
||||
## Example Usage
|
||||
|
||||
```
|
||||
Monitor W&B run "fastvideo/Wan_distillation/abc123":
|
||||
|
||||
run_id: fastvideo/Wan_distillation/abc123
|
||||
poll_interval: 120
|
||||
alert_on: [loss_spike, nan_gradient, step_time_regression]
|
||||
```
|
||||
|
||||
## References
|
||||
- `fastvideo/training/trackers.py` — `WandbTracker` implementation
|
||||
- `fastvideo/tests/training/Vanilla/test_training_loss.py` — how summaries are compared
|
||||
- `fastvideo/tests/training/Vanilla/a40_reference_wandb_summary.json` — reference summary format
|
||||
|
||||
## Changelog
|
||||
| Date | Change |
|
||||
|------|--------|
|
||||
| 2026-03-02 | Initial version |
|
||||
@@ -0,0 +1,343 @@
|
||||
---
|
||||
name: reseed-ssim-references
|
||||
description: Re-seed HF reference videos for a single existing SSIM test on Modal L40S. Always backs up current refs locally first, regenerates on Modal, pauses for the user to eyeball before-vs-after quality, then overwrites the targeted `<model_id>` subtree on `FastVideo/ssim-reference-videos` with `--force`. Use when an intentional code change (model port fix, attention backend swap, kernel upgrade, hyperparameter change) has invalidated existing refs and they need to be regenerated. Pairs with `seed-ssim-references`, which is for first-time seeding only.
|
||||
---
|
||||
|
||||
# Re-seed SSIM Reference Videos
|
||||
|
||||
## Purpose
|
||||
|
||||
Replace the existing SSIM reference videos for a single `(test_file, model_id)`
|
||||
pair on the HF dataset (`FastVideo/ssim-reference-videos`). This is **destructive**
|
||||
on HF — the old refs are overwritten — so the skill always:
|
||||
|
||||
1. Confirms intent with a one-liner the user has to type.
|
||||
2. Downloads the existing refs as a local, timestamped backup.
|
||||
3. Regenerates on Modal L40S (same code path that CI uses).
|
||||
4. Pauses for a side-by-side eyeball of backup vs new mp4s.
|
||||
5. Uploads with `--force`, scoped to the single `--model-id`.
|
||||
6. Reminds the user to keep the backup until the PR lands.
|
||||
|
||||
Pairs with `seed-ssim-references`, which is the inverse (first-time seeding
|
||||
only, refuses to overwrite). Re-seeding is intentionally a separate, more
|
||||
ceremonial operation because mistakenly clobbering production refs is much
|
||||
harder to recover from than failing closed.
|
||||
|
||||
## When to use
|
||||
|
||||
- An intentional code change (model port fix, kernel upgrade, attention
|
||||
backend swap, hyperparameter change in the test itself) has shifted the
|
||||
expected SSIM output and the existing refs no longer represent the new
|
||||
ground truth.
|
||||
- A test is failing in CI **for the right reason** (the new code is correct,
|
||||
the old refs are stale).
|
||||
|
||||
## When not to use
|
||||
|
||||
- A test is failing for the **wrong** reason (the port is buggy, not the
|
||||
refs). Fix the port; re-seeding hides the bug.
|
||||
- A brand-new test that has no refs on HF yet. Use `seed-ssim-references`.
|
||||
- "Just to clean up drift" without a concrete code change to point at. The
|
||||
PR description has to justify *why* refs changed; without a concrete
|
||||
change, there's nothing to write.
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Required | Description |
|
||||
|-----------|----------|-------------|
|
||||
| `test_file` | Yes | Path to the SSIM test, e.g. `fastvideo/tests/ssim/test_matrixgame_similarity.py`. Validated against `fastvideo/tests/ssim/test_*_similarity.py`. |
|
||||
| `model_id` | Yes | Single model id from the test's `*_MODEL_TO_PARAMS`, e.g. `Matrix-Game-2.0-Diffusers-Base`. Re-seed runs are **per model**. For multi-model tests, invoke the skill once per model. |
|
||||
| `intent_rationale` | Yes | One-line explanation of *why* refs are being regenerated (e.g. "Relax FA-2 head_size whitelist to include 80 — matrix_game now uses FLASH_ATTN instead of TORCH_SDPA"). Recorded in the backup directory and reused in the PR description. |
|
||||
|
||||
Hardcoded:
|
||||
|
||||
- Modal GPU: **L40S** (matches CI; re-seeding from another SKU produces refs
|
||||
that L40S CI cannot match).
|
||||
- Quality tier: **`default`**. `full_quality` is a separate, deliberate
|
||||
operation.
|
||||
- HF repo: `FastVideo/ssim-reference-videos` (override via
|
||||
`FASTVIDEO_SSIM_REFERENCE_HF_REPO`).
|
||||
- Device folder: `L40S_reference_videos`.
|
||||
|
||||
## Prerequisites
|
||||
|
||||
The user has confirmed:
|
||||
|
||||
- `modal` CLI authenticated.
|
||||
- `hf` CLI authenticated, **and** `HF_API_KEY` (or `HUGGINGFACE_HUB_TOKEN` /
|
||||
`HF_TOKEN`) exported with **write** access to
|
||||
`FastVideo/ssim-reference-videos`.
|
||||
- The current branch's code is the change that motivated the re-seed (i.e.
|
||||
`git rev-parse HEAD` is the commit that intentionally invalidated refs).
|
||||
|
||||
Fail fast if any of these are missing.
|
||||
|
||||
## Steps
|
||||
|
||||
### 1. Validate inputs and confirm intent
|
||||
|
||||
- Verify `test_file` exists and matches `fastvideo/tests/ssim/test_*_similarity.py`.
|
||||
- Grep the file for `*_MODEL_TO_PARAMS` and assert `model_id` is one of its
|
||||
keys. If the file has only a single hardcoded model, accept that model id
|
||||
as the only valid value.
|
||||
- Print the rationale and ask the user to type **`confirm reseed`** (not just
|
||||
`y` — make it deliberate):
|
||||
|
||||
> About to RE-SEED references for model `<model_id>` from test `<test_file>`.
|
||||
> This will OVERWRITE existing refs on
|
||||
> `FastVideo/ssim-reference-videos/reference_videos/default/L40S_reference_videos/<model_id>/`
|
||||
> after backup + Modal regen + eyeball.
|
||||
>
|
||||
> Reason: `<intent_rationale>`
|
||||
> HEAD: `<git rev-parse --short=12 HEAD>`
|
||||
>
|
||||
> Reply `confirm reseed` to proceed, anything else to abort.
|
||||
|
||||
Stop until the user types exactly `confirm reseed`. Anything else aborts
|
||||
with no side effects.
|
||||
|
||||
### 2. Back up existing refs
|
||||
|
||||
Always required. The backup is the only graceful path back if anything goes
|
||||
wrong later.
|
||||
|
||||
```bash
|
||||
SHORT_COMMIT=$(git rev-parse --short=12 HEAD)
|
||||
TIMESTAMP=$(date -u +%Y%m%d_%H%M%S)
|
||||
MODEL_SAFE=$(echo "<model_id>" | tr '/' '_')
|
||||
BACKUP_DIR="ssim_reseed_backup/${TIMESTAMP}_${SHORT_COMMIT}_${MODEL_SAFE}"
|
||||
mkdir -p "$BACKUP_DIR"
|
||||
|
||||
hf download \
|
||||
--repo-type dataset FastVideo/ssim-reference-videos \
|
||||
--include "reference_videos/default/L40S_reference_videos/<model_id>/**" \
|
||||
--local-dir "$BACKUP_DIR"
|
||||
|
||||
mp4_count=$(find "$BACKUP_DIR" -name "*.mp4" | wc -l)
|
||||
echo "Backup mp4 count: $mp4_count"
|
||||
[ "$mp4_count" -gt 0 ] || {
|
||||
echo "ERROR: backup is empty for <model_id>. Either the model id is wrong"
|
||||
echo "or there are no existing refs (use seed-ssim-references instead)."
|
||||
exit 1
|
||||
}
|
||||
|
||||
# Provenance — used in the PR description
|
||||
cat > "$BACKUP_DIR/PROVENANCE.txt" <<EOF
|
||||
test_file: <test_file>
|
||||
model_id: <model_id>
|
||||
head_commit: $(git rev-parse HEAD)
|
||||
timestamp_utc: $(date -u +%FT%TZ)
|
||||
reason: <intent_rationale>
|
||||
EOF
|
||||
```
|
||||
|
||||
If the `hf download` produces zero mp4s, abort — the user has either picked a
|
||||
non-existent `model_id` or there are no refs yet (in which case
|
||||
`seed-ssim-references` is the right tool).
|
||||
|
||||
### 3. Regenerate on Modal L40S
|
||||
|
||||
Mirror CI's exact env recipe so the regenerated refs are byte-comparable to
|
||||
what CI will produce on the same commit. Two differences from CI:
|
||||
|
||||
1. **Pass the same env prefix CI uses** (`IMAGE_VERSION`, `BUILDKITE_*`) — see
|
||||
`.buildkite/pipeline.yml:1-3` and `.buildkite/scripts/pr_test.sh:62-83`.
|
||||
Without this, `ssim_test.py:17-18` resolves a different GHCR image tag
|
||||
(default is `latest`, CI is `py3.12-latest`), and `ssim_test.py:38-46`
|
||||
bakes different values into the image's frozen env block. **Mismatched
|
||||
image or env is the most common source of SSIM drift between reseed and
|
||||
CI runs.**
|
||||
2. **Do not pass `--skip-reference-download`**. Letting the test fetch the
|
||||
existing refs and run the full SSIM compare gives "before" SSIM numbers
|
||||
for the PR description, and the test still produces the new mp4s
|
||||
regardless of whether the comparison passes or fails.
|
||||
|
||||
```bash
|
||||
SUBDIR="${TIMESTAMP}_${SHORT_COMMIT}"
|
||||
|
||||
IMAGE_VERSION="py3.12-latest" \
|
||||
BUILDKITE_REPO="$(git config --get remote.origin.url)" \
|
||||
BUILDKITE_COMMIT="$(git rev-parse HEAD)" \
|
||||
BUILDKITE_PULL_REQUEST="${BUILDKITE_PULL_REQUEST:-false}" \
|
||||
modal run fastvideo/tests/modal/ssim_test.py \
|
||||
--git-repo="$(git config --get remote.origin.url)" \
|
||||
--git-commit="$(git rev-parse HEAD)" \
|
||||
--hf-api-key="$HF_API_KEY" \
|
||||
--test-files="<test_file>" \
|
||||
--sync-generated-to-volume \
|
||||
--generated-volume-subdir="$SUBDIR" \
|
||||
--no-fail-fast
|
||||
```
|
||||
|
||||
Capture the printed `modal volume get ...` hint — its `<SUBDIR>` matches
|
||||
`$SUBDIR` and is needed for step 4. Capture the SSIM numbers from the test
|
||||
output (or from the JSON next to the generated mp4) for the PR description.
|
||||
|
||||
### 4. Download generated videos
|
||||
|
||||
```bash
|
||||
modal volume get --force hf-model-weights \
|
||||
ssim_generated_videos/default/"$SUBDIR"/generated_videos \
|
||||
./generated_videos_modal/default
|
||||
```
|
||||
|
||||
After this, the new mp4s live at:
|
||||
|
||||
```
|
||||
./generated_videos_modal/default/generated_videos/L40S_reference_videos/<model_id>/<backend>/<prompt>.mp4
|
||||
```
|
||||
|
||||
`--force` is required when `./generated_videos_modal/default` already exists
|
||||
from a prior run; safe on the first run too.
|
||||
|
||||
### 5. PAUSE — user reviews quality side-by-side
|
||||
|
||||
Print the diff and the comparison:
|
||||
|
||||
```bash
|
||||
echo "=== File list diff (backup vs new) ==="
|
||||
diff -u \
|
||||
<(find "$BACKUP_DIR/reference_videos/default/L40S_reference_videos/<model_id>" -name "*.mp4" \
|
||||
| sed "s|$BACKUP_DIR/reference_videos/default/L40S_reference_videos/||" | sort) \
|
||||
<(find ./generated_videos_modal/default/generated_videos/L40S_reference_videos/<model_id> -name "*.mp4" \
|
||||
| sed "s|./generated_videos_modal/default/generated_videos/L40S_reference_videos/||" | sort) \
|
||||
|| true
|
||||
|
||||
echo
|
||||
echo "=== SSIM numbers from this run (paste into PR) ==="
|
||||
find ./generated_videos_modal/default/generated_videos/L40S_reference_videos/<model_id> -name "*_ssim.json" -exec cat {} \;
|
||||
```
|
||||
|
||||
Then stop and tell the user:
|
||||
|
||||
> Old refs backed up to `$BACKUP_DIR`.
|
||||
> New videos in `./generated_videos_modal/default/generated_videos/L40S_reference_videos/<model_id>/`.
|
||||
>
|
||||
> Open both in a video player. Confirm the new videos:
|
||||
> 1. Look correct (no obvious artifacts, no black/static frames).
|
||||
> 2. Are *intentionally* different from the backup in the way described
|
||||
> in `<intent_rationale>` (e.g. slight numerical drift only, not a
|
||||
> different scene / different motion / corrupted output).
|
||||
>
|
||||
> Reply **`upload`** to overwrite HF, anything else to abort.
|
||||
> Aborting leaves the backup and new videos on disk for inspection — nothing
|
||||
> on HF changes.
|
||||
|
||||
Do not proceed until the user types exactly `upload`. If they abort, leave
|
||||
everything on disk and stop here.
|
||||
|
||||
### 6. Copy into the local reference layout
|
||||
|
||||
Same as `seed-ssim-references` step 5:
|
||||
|
||||
```bash
|
||||
python fastvideo/tests/ssim/reference_videos_cli.py copy-local \
|
||||
--quality-tier default \
|
||||
--device-folder L40S_reference_videos \
|
||||
--generated-dir ./generated_videos_modal/default/generated_videos/L40S_reference_videos
|
||||
```
|
||||
|
||||
Result: `fastvideo/tests/ssim/reference_videos/default/L40S_reference_videos/<model_id>/<backend>/<prompt>.mp4`.
|
||||
|
||||
### 7. Upload with `--force`, scoped to `--model-id`
|
||||
|
||||
The `--force` flag is what makes this skill different from `seed-ssim-references`.
|
||||
Always pair it with `--model-id` so a typo cannot accidentally overwrite a
|
||||
neighboring model's refs.
|
||||
|
||||
```bash
|
||||
python fastvideo/tests/ssim/reference_videos_cli.py upload \
|
||||
--quality-tier default \
|
||||
--device-folder L40S_reference_videos \
|
||||
--model-id "<model_id>" \
|
||||
--force
|
||||
```
|
||||
|
||||
The CLI's overwrite guard refuses without `--force`; with `--force` it
|
||||
overwrites only files under
|
||||
`reference_videos/default/L40S_reference_videos/<model_id>/`.
|
||||
|
||||
### 8. Report success and retention guidance
|
||||
|
||||
Print:
|
||||
|
||||
- The HF path that was overwritten (`<repo>/reference_videos/default/L40S_reference_videos/<model_id>/`).
|
||||
- The local backup directory path.
|
||||
- The new SSIM numbers from step 5.
|
||||
- This restore command, in case the PR review surfaces a problem after
|
||||
upload:
|
||||
|
||||
```bash
|
||||
python fastvideo/tests/ssim/reference_videos_cli.py upload \
|
||||
--quality-tier default \
|
||||
--device-folder L40S_reference_videos \
|
||||
--model-id "<model_id>" \
|
||||
--reference-dir "$BACKUP_DIR/reference_videos/default/L40S_reference_videos" \
|
||||
--force
|
||||
```
|
||||
|
||||
- This PR-description checklist (see `fastvideo/tests/ssim/AGENTS.md` →
|
||||
*Updating Reference Videos*):
|
||||
1. Source commit that produced the new refs (HEAD at re-seed time).
|
||||
2. Test command and GPU SKU (`L40S`).
|
||||
3. Before/after SSIM numbers.
|
||||
4. The `<intent_rationale>` from step 1.
|
||||
5. A note that the backup lives at `$BACKUP_DIR` and should be retained
|
||||
until CI on the PR is green.
|
||||
|
||||
Do **not** auto-rerun the SSIM test — the user does that as part of the PR.
|
||||
|
||||
## Failure modes and how to handle them
|
||||
|
||||
- **`HF_API_KEY` unset.** Stop before step 2.
|
||||
- **Backup is empty (zero mp4s).** Stop before step 3 — the model id is
|
||||
wrong or the refs don't exist yet (use `seed-ssim-references`).
|
||||
- **Modal run fails before generation.** No mp4s on the volume. Don't
|
||||
upload. Investigate the failure (test crash, OOM, partition exhaustion),
|
||||
fix, then retry from step 3. Backup is still intact.
|
||||
- **Quality regressed (visual or metric).** User aborts at step 5. Backup
|
||||
retained. New videos retained on disk for inspection. Nothing on HF
|
||||
changed. Either fix the underlying code change or abandon the re-seed.
|
||||
- **User confirmed `upload` but later realized the new refs are wrong.**
|
||||
Run the restore command from step 8 with the backup `--reference-dir`.
|
||||
This is exactly why the backup exists.
|
||||
- **Multi-model test, only one model is being re-seeded.** Run the skill
|
||||
once per model id. The `--model-id` scope on upload guarantees the others
|
||||
are untouched.
|
||||
|
||||
## Design notes (for future skill maintainers)
|
||||
|
||||
- Per-`model_id` scope is mandatory. The dataset houses many model subtrees;
|
||||
re-seeding the wrong one is hard to undo without backup.
|
||||
- `default` tier only; `full_quality` is a separate, deliberate operation
|
||||
with different params and ~doubled runtime, and isn't what CI gates on.
|
||||
- The skill deliberately does **not** pass `--skip-reference-download` to
|
||||
Modal so we get pre-reseed SSIM numbers for the PR. The `seed`-skill
|
||||
passes it because no refs exist yet; for re-seed, refs do exist and
|
||||
exposing the comparison is informative.
|
||||
- The two-token confirm (`confirm reseed`, then `upload`) is intentional.
|
||||
Re-seeding is high-blast-radius and should not be one-keystroke.
|
||||
- The backup directory is plain mp4s + `PROVENANCE.txt`. No HF metadata is
|
||||
preserved; the restore path uses `reference_videos_cli.py upload
|
||||
--reference-dir` which doesn't need it.
|
||||
|
||||
## References
|
||||
|
||||
- `.agents/skills/seed-ssim-references/SKILL.md` — the first-time seed
|
||||
skill this one parallels. Read it for the Modal flag rationale shared
|
||||
between the two flows.
|
||||
- `fastvideo/tests/ssim/AGENTS.md` — directory rules, including the PR
|
||||
expectations for any reference-video change (rationale, before/after
|
||||
SSIM, source commit/model/backend).
|
||||
- `fastvideo/tests/ssim/reference_videos_cli.py` — `copy-local`, `upload`
|
||||
(with `--model-id`, `--force`), `download`. The overwrite guard at
|
||||
`upload_reference_videos` is the safety net this skill leans on.
|
||||
- `fastvideo/tests/modal/ssim_test.py` — Modal orchestrator;
|
||||
`--sync-generated-to-volume`, `--generated-volume-subdir`,
|
||||
`--skip-reference-download`, `--no-fail-fast`.
|
||||
|
||||
## Changelog
|
||||
|
||||
| Date | Change |
|
||||
|------|--------|
|
||||
| 2026-05-02 | Initial version. Sister skill to `seed-ssim-references`, scoped to single `(test_file, model_id)` re-seeds, with mandatory backup and two-token confirm. |
|
||||
@@ -0,0 +1,376 @@
|
||||
---
|
||||
name: seed-ssim-references
|
||||
description: Seed HF reference artefacts for a single newly-added SSIM test (pixel `.mp4` for `run_text_to_video_similarity_test`-style tests, or latent `.pt` for `run_text_to_latent_similarity_test`-style tests). Runs the test on Modal L40S, downloads the generated artefacts via `modal volume get`, pauses for the user to verify (visual eyeball for mp4, numerics dump for pt), then uploads only that test's files to `FastVideo/ssim-reference-videos`. Use when a new `fastvideo/tests/ssim/test_*_similarity.py` has just been added and has no references on HF yet.
|
||||
---
|
||||
|
||||
# Seed SSIM Reference Artefacts (mp4 or pt)
|
||||
|
||||
## Purpose
|
||||
|
||||
A brand-new SSIM test in `fastvideo/tests/ssim/` fails forever until its
|
||||
reference artefacts exist on the HF dataset
|
||||
(`FastVideo/ssim-reference-videos`). The dataset hosts two kinds of artefacts
|
||||
side-by-side per `(model_id, backend, prompt)`:
|
||||
|
||||
- **`.mp4`** — pixel ground-truth for tests that call
|
||||
`run_text_to_video_similarity_test` / `run_image_to_video_similarity_test`
|
||||
in `inference_similarity_utils.py`. Compared via SSIM.
|
||||
- **`.pt`** — pre-VAE latent bundle (fp16 full latent + fp32 slice +
|
||||
metadata + `slice_spec` + `format_version`) for tests that call
|
||||
`run_text_to_latent_similarity_test` in `latent_similarity_utils.py`.
|
||||
Compared via cosine distance on the slice and the full tensor.
|
||||
|
||||
This skill:
|
||||
|
||||
1. Detects which artefact type the test produces (pixel vs latent).
|
||||
2. Runs the test on Modal's L40S pool to generate the artefacts.
|
||||
3. Downloads them to the local repo via `modal volume get`.
|
||||
4. Pauses so the user can verify quality:
|
||||
- **mp4**: visual eyeball in a video player.
|
||||
- **pt**: numerics dump (shape, slice stats, NaN/Inf check, metadata).
|
||||
5. Uploads only the new test's files to HF, with a guard that refuses to
|
||||
overwrite anything already present.
|
||||
|
||||
The skill is run **manually**, once per new test. Before invoking it, the user
|
||||
has already sanity-tested the new test locally — it launches `VideoGenerator`
|
||||
and writes an artefact without crashing (the missing-reference assertion at
|
||||
the end is expected). The skill does not re-test locally; it goes straight
|
||||
to Modal L40S (which is what CI uses).
|
||||
|
||||
## When to use
|
||||
|
||||
- A new `test_*_similarity.py` file has been added in `fastvideo/tests/ssim/`
|
||||
and the HF dataset has no `reference_videos/default/L40S_reference_videos/<model_id>/`
|
||||
subtree for it yet.
|
||||
|
||||
## When not to use
|
||||
|
||||
- Regular CI runs — once refs exist, `pytest fastvideo/tests/ssim/` downloads
|
||||
them automatically.
|
||||
- Re-seeding an existing test. That requires `--force` on the upload step, and
|
||||
is out of scope here; treat as a separate, deliberate operation.
|
||||
|
||||
## Inputs
|
||||
|
||||
The skill has **one required input**: the path to the new SSIM test file.
|
||||
Prompt the user for it if they didn't supply it.
|
||||
|
||||
| Parameter | Required | Description |
|
||||
|-----------|----------|-------------|
|
||||
| `test_file` | Yes | e.g. `fastvideo/tests/ssim/test_ltx2_similarity.py`. The skill's first action is to ask for this if missing. |
|
||||
|
||||
Everything else is fixed:
|
||||
|
||||
- Modal runner GPU: **L40S** (hardcoded in `fastvideo/tests/modal/ssim_test.py`).
|
||||
- Device folder: `L40S_reference_videos`.
|
||||
- Quality tier: `default` (the tier CI runs). The `full_quality` tier is not
|
||||
seeded by this skill.
|
||||
- HF repo: `FastVideo/ssim-reference-videos` (dataset).
|
||||
- Multi-model test files: all model ids in `*_MODEL_TO_PARAMS` are seeded
|
||||
together; the Modal run produces one mp4 per (model, prompt, backend) and
|
||||
the upload scopes by `--model-id`, looping if there is more than one.
|
||||
|
||||
## Prerequisites
|
||||
|
||||
The user has confirmed:
|
||||
|
||||
- `modal` CLI authenticated.
|
||||
- `HF_API_KEY` (or `HUGGINGFACE_HUB_TOKEN` / `HF_TOKEN`) exported with write
|
||||
access to `FastVideo/ssim-reference-videos`.
|
||||
- The test file runs locally end-to-end (generates an mp4; SSIM assertion
|
||||
failure due to missing reference is expected and fine).
|
||||
|
||||
Fail fast if the token env var is missing.
|
||||
|
||||
## Steps
|
||||
|
||||
### 1. Ask for the test file, then detect artefact type
|
||||
|
||||
If the user didn't name one, ask: *"Which SSIM test file do you want to seed
|
||||
references for? (e.g. `fastvideo/tests/ssim/test_ltx2_similarity.py`)"*.
|
||||
|
||||
Validate:
|
||||
|
||||
- Path exists and matches `fastvideo/tests/ssim/test_*_similarity.py`.
|
||||
- File defines a `*_MODEL_TO_PARAMS` dict — grep it to extract the set of
|
||||
model ids. Those ids drive step 5.
|
||||
|
||||
Detect artefact type by inspecting the file's imports / helper call:
|
||||
|
||||
- **latent** (`.pt`) — file imports `run_text_to_latent_similarity_test`
|
||||
from `fastvideo.tests.ssim.latent_similarity_utils` (or any other helper
|
||||
that ends with `_latent_similarity_test`).
|
||||
- **pixel** (`.mp4`) — file imports
|
||||
`run_text_to_video_similarity_test` / `run_image_to_video_similarity_test`
|
||||
from `fastvideo.tests.ssim.inference_similarity_utils`, OR uses the
|
||||
legacy custom-inline helper pattern (see `test_gamecraft`,
|
||||
`test_longcat`, etc.). Default to pixel when both heuristics fail.
|
||||
|
||||
Record `ARTEFACT_TYPE ∈ {pixel, latent}` for use in step 4. Steps 2, 3, 5,
|
||||
and 6 are artefact-type-agnostic — `_iter_reference_files`,
|
||||
`copy_generated_to_reference`, and `upload_reference_videos` already walk
|
||||
both `.mp4` and `.pt` (see `reference_videos_cli.py`).
|
||||
|
||||
If either check fails, stop and tell the user what's wrong.
|
||||
|
||||
### 2. Run the test on Modal L40S
|
||||
|
||||
Pick a subdir name so repeated runs don't collide:
|
||||
|
||||
```bash
|
||||
SHORT_COMMIT=$(git rev-parse --short=12 HEAD)
|
||||
TIMESTAMP=$(date -u +%Y%m%d_%H%M%S)
|
||||
SUBDIR="${TIMESTAMP}_${SHORT_COMMIT}"
|
||||
```
|
||||
|
||||
Then launch the Modal run. The `IMAGE_VERSION` and `BUILDKITE_*` env-prefix
|
||||
**must** match what CI exports in `.buildkite/scripts/pr_test.sh`, otherwise
|
||||
`fastvideo/tests/modal/ssim_test.py` resolves a different GHCR image tag
|
||||
(default is `latest`, CI is `py3.12-latest`) and bakes different values into
|
||||
the image's frozen env block (`ssim_test.py:17-18, 38-46`). Mismatched image
|
||||
or env produces SSIM drift that doesn't show up until the same commit runs
|
||||
in CI.
|
||||
|
||||
```bash
|
||||
IMAGE_VERSION="py3.12-latest" \
|
||||
BUILDKITE_REPO="$(git config --get remote.origin.url)" \
|
||||
BUILDKITE_COMMIT="$(git rev-parse HEAD)" \
|
||||
BUILDKITE_PULL_REQUEST="${BUILDKITE_PULL_REQUEST:-false}" \
|
||||
modal run fastvideo/tests/modal/ssim_test.py \
|
||||
--git-repo="$(git config --get remote.origin.url)" \
|
||||
--git-commit="$(git rev-parse HEAD)" \
|
||||
--hf-api-key="$HF_API_KEY" \
|
||||
--test-files="<test_file>" \
|
||||
--sync-generated-to-volume \
|
||||
--generated-volume-subdir="$SUBDIR" \
|
||||
--skip-reference-download \
|
||||
--no-fail-fast
|
||||
```
|
||||
|
||||
Env prefix rationale (parity with CI; see `.buildkite/pipeline.yml:1-3` and
|
||||
`.buildkite/scripts/pr_test.sh:62-83`):
|
||||
- `IMAGE_VERSION=py3.12-latest`: pins the Modal image tag to the same one CI
|
||||
uses. Without this, `ssim_test.py:17` falls back to `latest`, which on
|
||||
GHCR is built from `Dockerfile.python3.10` — different Python, torch, and
|
||||
flash-attn wheel than CI's `py3.12-latest` (`infra-build-image.yml:51-67`,
|
||||
`_template-build-image.yml:65-101`).
|
||||
- `BUILDKITE_REPO`/`BUILDKITE_COMMIT`/`BUILDKITE_PULL_REQUEST`: mirror what
|
||||
Buildkite exports. `ssim_test.py:38-46` bakes these into the image's
|
||||
`.env(...)` block; mismatched values can perturb in-container code paths
|
||||
that branch on PR-vs-non-PR. `false` for `BUILDKITE_PULL_REQUEST` matches
|
||||
Buildkite's "non-PR build" sentinel.
|
||||
|
||||
Flag rationale:
|
||||
- `--skip-reference-download`: no refs exist yet, so conftest must not try to
|
||||
pull them.
|
||||
- `--no-fail-fast`: lets the test finish generation before `_assert_similarity`
|
||||
raises `FileNotFoundError: Reference video folder does not exist`. The
|
||||
expected failure is what we want — the mp4 has already been written.
|
||||
- `--sync-generated-to-volume` + `--generated-volume-subdir`: copies the
|
||||
generated mp4s to the `hf-model-weights` Modal volume under
|
||||
`ssim_generated_videos/default/<SUBDIR>/generated_videos/` so we can pull
|
||||
them locally.
|
||||
|
||||
The Modal run will end with a nonzero exit (expected) and print a
|
||||
`modal volume get hf-model-weights ssim_generated_videos/default/<SUBDIR>/generated_videos ./generated_videos_modal/default`
|
||||
command. Capture that `<SUBDIR>` — you need it for step 3.
|
||||
|
||||
### 3. Download generated videos locally
|
||||
|
||||
```bash
|
||||
modal volume get --force hf-model-weights \
|
||||
ssim_generated_videos/default/"$SUBDIR"/generated_videos \
|
||||
./generated_videos_modal/default
|
||||
```
|
||||
|
||||
`--force` is required when the parent `./generated_videos_modal/default`
|
||||
already exists; without it, `modal volume get` errors with `[Errno 21] Is a
|
||||
directory`. Safe to pass on the first run too.
|
||||
|
||||
After this, the mp4s live at
|
||||
`./generated_videos_modal/default/generated_videos/L40S_reference_videos/<model_id>/<backend>/<prompt>.mp4`.
|
||||
The extra `generated_videos/` level comes from the volume layout in
|
||||
`_sync_generated_videos_to_volume` (`ssim_test.py`) — the command copies
|
||||
`<repo>/fastvideo/tests/ssim/generated_videos/<tier>` to
|
||||
`ssim_generated_videos/<tier>/<SUBDIR>/generated_videos/`, and `modal volume
|
||||
get` preserves that trailing `generated_videos/` segment.
|
||||
|
||||
### 4. PAUSE — user reviews quality
|
||||
|
||||
Type-aware verification.
|
||||
|
||||
**For `ARTEFACT_TYPE = pixel`** — list the downloaded mp4s and ask the user to
|
||||
open them in a video player:
|
||||
|
||||
> "Generated videos downloaded to `./generated_videos_modal/default/generated_videos/L40S_reference_videos/`. Please open them and confirm the quality looks correct. Reply **`upload`** to continue, or anything else to abort."
|
||||
|
||||
**For `ARTEFACT_TYPE = latent`** — `.pt` files are not human-watchable. Print
|
||||
a numerics dump for each `.pt` so the user can sanity-check shape, distribution,
|
||||
and metadata:
|
||||
|
||||
```python
|
||||
import torch
|
||||
from pathlib import Path
|
||||
ROOT = Path("./generated_videos_modal/default/generated_videos/L40S_reference_videos")
|
||||
for p in sorted(ROOT.rglob("*.pt")):
|
||||
d = torch.load(p, map_location="cpu", weights_only=False)
|
||||
s = d["expected_slice"]
|
||||
L = d["latent"].float()
|
||||
print(f"=== {p.relative_to(ROOT)} ===")
|
||||
print(f" format_version: {d['format_version']}")
|
||||
print(f" shape: {d['shape']}")
|
||||
print(f" dtype_original: {d['dtype_original']}")
|
||||
print(f" slice_spec: {d['slice_spec']}")
|
||||
print(f" slice shape={tuple(s.shape)} mean={s.mean():+.4f} std={s.std():.4f} min={s.min():+.4f} max={s.max():+.4f}")
|
||||
print(f" latent shape={tuple(L.shape)} mean={L.mean():+.4f} std={L.std():.4f} min={L.min():+.4f} max={L.max():+.4f}")
|
||||
print(f" finite: latent NaN={torch.isnan(L).any().item()} Inf={torch.isinf(L).any().item()}; "
|
||||
f"slice NaN={torch.isnan(s).any().item()} Inf={torch.isinf(s).any().item()}")
|
||||
print(f" metadata: {d['metadata']}\n")
|
||||
```
|
||||
|
||||
Sanity criteria:
|
||||
- `format_version == 1` (matches `LATENT_REFERENCE_FORMAT_VERSION`).
|
||||
- `shape` matches what the model produces (e.g. LTX-2 distilled =
|
||||
`[1, 128, T_lat, H_lat, W_lat]`; Stable Audio Open 1.0 = `[1, 64, 1024]`).
|
||||
- `slice_spec.kind` matches a registered kind (`corner_3x3_first_frame`
|
||||
for video, `audio_first_8_timesteps` for audio).
|
||||
- No `NaN`/`Inf`. `mean ≈ 0`, `std ≈ 1` (denoised latents stay close to
|
||||
the initial Gaussian distribution; very wide deviations suggest
|
||||
numerical drift).
|
||||
- `metadata.prompt` matches the test's prompt.
|
||||
|
||||
Then ask:
|
||||
|
||||
> "Numerics look right? Reply **`upload`** to continue, or anything else to abort."
|
||||
|
||||
Do not proceed until the user explicitly says `upload`. If they abort, leave
|
||||
everything on disk so they can inspect further — no cleanup.
|
||||
|
||||
### 5. Copy into the local reference layout
|
||||
|
||||
Scoped copy — only the new test's artefacts. Single command works for both
|
||||
artefact types because `_iter_reference_files` walks `.mp4` and `.pt`:
|
||||
|
||||
```bash
|
||||
python fastvideo/tests/ssim/reference_videos_cli.py copy-local \
|
||||
--quality-tier default \
|
||||
--device-folder L40S_reference_videos \
|
||||
--generated-dir ./generated_videos_modal/default/generated_videos/L40S_reference_videos
|
||||
```
|
||||
|
||||
(The `--generated-dir` points at the device-folder root inside the
|
||||
downloaded tree; `copy-local` walks all `<model>/<backend>/*.{mp4,pt}`
|
||||
underneath it. Since the Modal run was scoped to a single test file via
|
||||
`--test-files`, only that test's model(s) are present — so the copy is
|
||||
implicitly per-test.)
|
||||
|
||||
Result for pixel: `fastvideo/tests/ssim/reference_videos/default/L40S_reference_videos/<model_id>/<backend>/<prompt>.mp4`.
|
||||
Result for latent: same path with `.pt` extension.
|
||||
|
||||
### 6. Upload to HF — scoped per model_id, with overwrite guard
|
||||
|
||||
For each `<model_id>`:
|
||||
|
||||
```bash
|
||||
python fastvideo/tests/ssim/reference_videos_cli.py upload \
|
||||
--quality-tier default \
|
||||
--device-folder L40S_reference_videos \
|
||||
--model-id "<model_id>"
|
||||
```
|
||||
|
||||
The upload command:
|
||||
|
||||
- Uploads **only** `reference_videos/default/L40S_reference_videos/<model_id>/`.
|
||||
- **Refuses** if any file already exists at that path on HF (this is the
|
||||
guard — seeding a new test should never clobber existing refs). To override,
|
||||
the user must re-run with `--force`. If the guard fires, stop and report
|
||||
exactly which files exist; do not silently `--force`.
|
||||
|
||||
Reads the HF token from `HF_API_KEY` / `HUGGINGFACE_HUB_TOKEN` / `HF_TOKEN`.
|
||||
|
||||
### 7. Report success
|
||||
|
||||
List what was uploaded (paths in repo) and remind the user to push any
|
||||
related code changes. Do **not** auto-verify by re-running Modal — the user
|
||||
can run `pytest fastvideo/tests/ssim/<test_file>` later to confirm end-to-end;
|
||||
it will auto-download the refs they just uploaded.
|
||||
|
||||
## Failure modes and how to handle them
|
||||
|
||||
- **`HF_API_KEY` unset.** Stop before step 2. The Modal run needs it (passed
|
||||
via `--hf-api-key`), and step 6 needs it for upload. If the user
|
||||
ran `hf auth login` instead of exporting an env var, read the cached
|
||||
token via `huggingface_hub.get_token()` and forward it to Modal as
|
||||
`--hf-api-key="$CACHED_TOKEN"`.
|
||||
- **Modal run fails before generation.** No artefacts on the volume — nothing
|
||||
to download. Fix the test locally (`pytest fastvideo/tests/ssim/<test_file>`)
|
||||
and retry from step 2.
|
||||
- **`./generated_videos_modal/default/L40S_reference_videos/` missing after
|
||||
`modal volume get`.** The run didn't produce artefacts (most likely the
|
||||
test crashed before writing, or `REQUIRED_GPUS` exceeded the partition
|
||||
capacity — see Modal logs).
|
||||
- **Latent test crashed with FSDP / inference_mode error
|
||||
(`RuntimeError: Inference tensors do not track version counter`).** The
|
||||
test must pass `init_kwargs_override={"use_fsdp_inference": False}` when
|
||||
`sp_size == 1` — see `test_stable_audio_similarity.py` for the pattern.
|
||||
Fix in the test, push, retry.
|
||||
- **Upload guard fires (files already exist).** The test name / model id
|
||||
collides with something already on HF. Verify the user actually wants to
|
||||
replace existing refs; if so, re-run the upload with `--force`. If not,
|
||||
rename the model id in `*_MODEL_TO_PARAMS` and re-seed.
|
||||
- **Quality looks wrong in step 4.** Abort. The artefacts stay on disk for
|
||||
inspection. The fix is usually in the test's params (resolution, steps,
|
||||
seed) — edit the test, then re-run the skill.
|
||||
- For latent: also check `slice_spec.kind` matches the latent rank
|
||||
(`corner_3x3_first_frame` requires 5-D, `audio_first_8_timesteps`
|
||||
requires 3-D); a rank/kind mismatch raises in `_extract_expected_slice`.
|
||||
|
||||
## Design notes (for future skill maintainers)
|
||||
|
||||
- The skill deliberately runs on Modal, **not** locally, because the CI
|
||||
runner is L40S. Seeding from a different GPU SKU produces refs that CI's
|
||||
L40S runs can't match (pixel SSIM drifts across SKUs; latent cosine has
|
||||
tighter cross-SKU bf16 drift but the configured tolerances assume
|
||||
same-SKU seed → same-SKU verify).
|
||||
- The skill is default-tier only. `full_quality` refs are seeded by a
|
||||
separate, deliberate operation — they double runtime and aren't what CI
|
||||
gates on.
|
||||
- The overwrite guard in `reference_videos_cli.py upload` is default-on
|
||||
specifically because this skill exists. Re-seeding is a distinct operation
|
||||
that requires explicit `--force`.
|
||||
- Both artefact types share the same Modal flow: the orchestrator sets
|
||||
`--skip-reference-download` + `--no-fail-fast`, runs pytest, the test's
|
||||
helper writes the artefact (`.mp4` via `imageio` for pixel,
|
||||
`save_latent_reference` → `torch.save` for latent) BEFORE the
|
||||
missing-reference assertion raises. `_sync_generated_videos_to_volume` in
|
||||
`ssim_test.py` does a `shutil.copytree` of the whole `generated_videos/`
|
||||
tree, picking up `.mp4`, `.pt`, and the `*_ssim.json` / `*_latent.json`
|
||||
metric files alongside.
|
||||
|
||||
## References
|
||||
|
||||
- `fastvideo/tests/modal/ssim_test.py` — Modal orchestrator; see
|
||||
`--sync-generated-to-volume`, `--generated-volume-subdir`,
|
||||
`--skip-reference-download`, `--no-fail-fast`.
|
||||
- `fastvideo/tests/ssim/reference_videos_cli.py` — `copy-local`, `upload`
|
||||
(with `--model-id`, `--force`), `download`, `ensure` subcommands.
|
||||
Extension allowlist is `REFERENCE_EXTENSIONS = VIDEO_EXTENSIONS +
|
||||
LATENT_EXTENSIONS` (`.pt`).
|
||||
- `fastvideo/tests/ssim/README.md` — reference layout, HF repo conventions.
|
||||
- `fastvideo/tests/ssim/inference_similarity_utils.py` — pixel helpers
|
||||
(`run_text_to_video_similarity_test`,
|
||||
`run_image_to_video_similarity_test`, `build_init_kwargs`).
|
||||
- `fastvideo/tests/ssim/latent_similarity_utils.py` — latent helper
|
||||
(`run_text_to_latent_similarity_test`), slice spec dispatch
|
||||
(`_extract_expected_slice`), reference schema
|
||||
(`save_latent_reference` / `load_latent_reference`),
|
||||
`LATENT_REFERENCE_FORMAT_VERSION`.
|
||||
|
||||
## Changelog
|
||||
|
||||
| Date | Change |
|
||||
|------|--------|
|
||||
| 2026-04-17 | Initial version (Modal sync-to-volume flow). |
|
||||
| 2026-04-21 | Rewrite: single-test scope, explicit user-review pause, per-`model_id` upload, HF overwrite guard. Dropped `scripts/seed_ssim.sh`. |
|
||||
| 2026-04-21 | Post-first-run fixes: `modal volume get` needs `--force` when parent exists; download tree has an extra `generated_videos/` level so `--generated-dir` must reflect it. |
|
||||
| 2026-05-01 | Latent (`*.pt`) artefact support: artefact-type detection in step 1, type-aware verification (visual eyeball for mp4, numerics dump for pt) in step 4, FSDP+inference_mode failure-mode added, design notes for the unified Modal flow. Triggered by PR #1253 (LTX-2 latent migration + Stable Audio latent test). |
|
||||
@@ -0,0 +1,137 @@
|
||||
---
|
||||
name: summarize-run
|
||||
description: Extract a W&B run summary into a structured experiment report
|
||||
---
|
||||
|
||||
# Summarize Run
|
||||
|
||||
## Purpose
|
||||
After a training run completes (or at any checkpoint), extract key metrics from
|
||||
the W&B run summary and produce a structured markdown report. Supports both
|
||||
online (W&B API) and offline (local `wandb-summary.json`) modes.
|
||||
|
||||
## Prerequisites
|
||||
- Run has completed or reached a checkpoint with a saved summary.
|
||||
- For online: `WANDB_API_KEY` set in environment.
|
||||
- For offline: access to `<output_dir>/tracker/wandb/latest-run/files/wandb-summary.json`.
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Required | Description |
|
||||
|-----------|----------|-------------|
|
||||
| `run_id` | Yes* | W&B run ID for online access |
|
||||
| `output_dir` | Yes* | Local output dir for offline access |
|
||||
| `reference_run` | No | Path to reference `wandb-summary.json` for comparison |
|
||||
| `experiment_name` | No | Name for the journal entry (default: from W&B) |
|
||||
|
||||
\* One of `run_id` or `output_dir` is required.
|
||||
|
||||
## Steps
|
||||
|
||||
### 1. Load run summary
|
||||
|
||||
**Online**:
|
||||
|
||||
```python
|
||||
import wandb
|
||||
api = wandb.Api()
|
||||
run = api.run("<run_id>")
|
||||
summary = dict(run.summary)
|
||||
config = dict(run.config)
|
||||
```
|
||||
|
||||
**Offline** (existing codebase pattern from `fastvideo/tests/training/`):
|
||||
|
||||
```python
|
||||
import json
|
||||
summary_path = f"{output_dir}/tracker/wandb/latest-run/files/wandb-summary.json"
|
||||
with open(summary_path) as f:
|
||||
summary = json.load(f)
|
||||
```
|
||||
|
||||
### 2. Extract key fields
|
||||
|
||||
| Field | Source | Description |
|
||||
|-------|--------|-------------|
|
||||
| `train_loss` | `summary["train_loss"]` | Final training loss |
|
||||
| `avg_step_time` | `summary["avg_step_time"]` | Average seconds per step |
|
||||
| `step_time` | `summary["step_time"]` | Last step time |
|
||||
| `grad_norm` | `summary["grad_norm"]` | Final gradient norm |
|
||||
| `learning_rate` | `summary["learning_rate"]` | Final LR |
|
||||
| `_step` | `summary["_step"]` | Total steps completed |
|
||||
| `_runtime` | `summary["_runtime"]` | Total wall-clock seconds |
|
||||
| `validation_videos_*` | `summary[key]` | Validation video artifacts |
|
||||
|
||||
### 3. Compare against reference (optional)
|
||||
|
||||
Follow the pattern in `fastvideo/tests/training/Vanilla/test_training_loss.py`:
|
||||
|
||||
```python
|
||||
# Fields to compare
|
||||
compare_fields = ["train_loss", "grad_norm", "avg_step_time"]
|
||||
tolerance = 0.05 # 5% relative tolerance
|
||||
|
||||
for field in compare_fields:
|
||||
ref_val = reference_summary[field]
|
||||
cur_val = summary[field]
|
||||
diff_pct = abs(cur_val - ref_val) / abs(ref_val) * 100
|
||||
status = "✅" if diff_pct < tolerance * 100 else "⚠️"
|
||||
print(f"{status} {field}: {cur_val:.4f} (ref: {ref_val:.4f}, diff: {diff_pct:.1f}%)")
|
||||
```
|
||||
|
||||
### 4. Generate report
|
||||
|
||||
```markdown
|
||||
# Run Summary: <experiment_name>
|
||||
|
||||
| Metric | Value | Reference | Diff |
|
||||
|--------|-------|-----------|------|
|
||||
| Train Loss | 0.0788 | 0.0800 | -1.5% ✅ |
|
||||
| Avg Step Time | 2.81s | 2.80s | +0.4% ✅ |
|
||||
| Grad Norm | 0.408 | 0.410 | -0.5% ✅ |
|
||||
| Total Steps | 500 | — | — |
|
||||
| Wall Time | 23m 30s | — | — |
|
||||
|
||||
## Configuration
|
||||
- Model: Wan-AI/Wan2.1-T2V-1.3B-Diffusers
|
||||
- Learning Rate: 1e-6
|
||||
- Batch Size: 1
|
||||
- GPUs: 8 × (SP=1, TP=1)
|
||||
- Mixed Precision: bf16
|
||||
|
||||
## Validation Videos
|
||||
<list of validation video paths if available>
|
||||
|
||||
## Notes
|
||||
<any observations or anomalies>
|
||||
```
|
||||
|
||||
### 5. Update experiment journal
|
||||
|
||||
Append or update the experiment's entry in `.agents/memory/experiment-journal/README.md`
|
||||
with the final metrics and status.
|
||||
|
||||
## Outputs
|
||||
- Structured markdown report.
|
||||
- Updated experiment journal entry.
|
||||
|
||||
## Example Usage
|
||||
|
||||
```
|
||||
Summarize the run in output directory "outputs/wan_finetune":
|
||||
|
||||
output_dir: outputs/wan_finetune
|
||||
reference_run: fastvideo/tests/training/Vanilla/a40_reference_wandb_summary.json
|
||||
experiment_name: wan-t2v-finetune-lr1e6
|
||||
```
|
||||
|
||||
## References
|
||||
- `fastvideo/tests/training/Vanilla/test_training_loss.py` — reference comparison pattern
|
||||
- `fastvideo/tests/training/Vanilla/a40_reference_wandb_summary.json` — example summary
|
||||
- `fastvideo/tests/training/lora/test_lora_training.py` — LoRA summary comparison
|
||||
- `fastvideo/training/trackers.py` — tracker summary generation
|
||||
|
||||
## Changelog
|
||||
| Date | Change |
|
||||
|------|--------|
|
||||
| 2026-03-02 | Initial version |
|
||||
@@ -0,0 +1,93 @@
|
||||
---
|
||||
description: How to develop, validate, and register a new evaluation metric
|
||||
---
|
||||
|
||||
# Evaluation Development SOP
|
||||
|
||||
Standard procedure for adding new video quality evaluation metrics to
|
||||
the FastVideo agent toolkit.
|
||||
|
||||
## When to use
|
||||
|
||||
- You need a metric that does not exist in
|
||||
`.agents/memory/evaluation-registry/README.md`.
|
||||
- An existing metric needs significant changes to its methodology.
|
||||
- You are exploring a new evaluation approach.
|
||||
|
||||
## Steps
|
||||
|
||||
### 1. Research
|
||||
|
||||
- Search `.agents/memory/related-work/` for existing evaluation
|
||||
approaches.
|
||||
- Check `.agents/memory/evaluation-registry/README.md` for current
|
||||
metrics and their limitations.
|
||||
- Review literature: FVD, CLIP-Score, human preference, etc.
|
||||
|
||||
### 2. Prototype
|
||||
|
||||
- Write a standalone script in `.agents/exploration/<metric-name>.md`.
|
||||
- Keep it simple: one script, minimal dependencies.
|
||||
- Test on a few known-good and known-bad video samples.
|
||||
|
||||
### 3. Validate
|
||||
|
||||
- **Known-good test**: metric should score high on reference-quality
|
||||
videos.
|
||||
- **Known-bad test**: metric should score low on degraded or unrelated
|
||||
videos.
|
||||
- **Sensitivity test**: small quality differences should produce
|
||||
meaningful score differences.
|
||||
- Document thresholds and their justification.
|
||||
|
||||
### 4. Register
|
||||
|
||||
Update `.agents/memory/evaluation-registry/README.md`:
|
||||
|
||||
- Add the metric with status `Active`.
|
||||
- Document location, thresholds, and trust level.
|
||||
|
||||
### 5. Integrate
|
||||
|
||||
Update `.agents/skills/evaluate-video-quality/SKILL.md`:
|
||||
|
||||
- Add the new metric as a section.
|
||||
- Include code examples and interpretation guide.
|
||||
|
||||
### 6. Document
|
||||
|
||||
- Move the exploration log content into the skill.
|
||||
- Clean up the exploration file or mark it as `promoted`.
|
||||
- If anything went wrong during development, create a lesson.
|
||||
|
||||
## Where the metrics live
|
||||
|
||||
The eval suite is `fastvideo/eval/`. New metrics register themselves
|
||||
via `@register("<group>.<name>")` and are auto-discovered when
|
||||
`fastvideo.eval.metrics` is imported.
|
||||
|
||||
- **Native metrics** (SSIM, PSNR, LPIPS, optical flow, VLM): add a
|
||||
file under the appropriate group dir
|
||||
(`fastvideo/eval/metrics/common/`, `optical_flow/`, `videoscore2/`,
|
||||
`physics_iq/`).
|
||||
- **Metrics that wrap upstream research code**: follow the vbench
|
||||
pattern in `fastvideo/eval/metrics/vbench/`. The contract is:
|
||||
- Upstream lives as a git submodule under
|
||||
`fastvideo/third_party/eval/<bench>/`, pinned to a SHA in repo-root
|
||||
`.gitmodules`.
|
||||
- The metric package's `__init__.py` inserts the submodule path on
|
||||
`sys.path` and installs runtime compat shims (attribute-level
|
||||
monkey-patches) for any modern-dep drift. Do not modify upstream
|
||||
files on disk, and do not ship a `setup.sh`.
|
||||
- See `fastvideo/eval/README.md` for the worked vbench example.
|
||||
- Full porting guide:
|
||||
[`docs/contributing/eval-metrics.md`](../../docs/contributing/eval-metrics.md).
|
||||
|
||||
## Out of scope of the initial eval port
|
||||
|
||||
The following land in follow-up PRs:
|
||||
|
||||
- **MIND** metrics (depends on a separate `vipe` submodule).
|
||||
- **VBench-2.0** sibling package.
|
||||
- Native conversion of **FVD** under `fastvideo/eval/metrics/fvd/`.
|
||||
- The training-time `EvalCallback`.
|
||||
@@ -0,0 +1,47 @@
|
||||
---
|
||||
description: When and how to log experiments in the experiment journal
|
||||
---
|
||||
|
||||
# Experiment Journaling SOP
|
||||
|
||||
Ensures every experiment is properly recorded with context and outcomes.
|
||||
|
||||
## When to Log
|
||||
|
||||
**Always.** Every experiment — even quick tests — should be journaled.
|
||||
|
||||
## Steps
|
||||
|
||||
### 1. Before Launch — Create Draft Entry
|
||||
|
||||
Use the `log-experiment` skill with `status: running`:
|
||||
- Include hypothesis and config.
|
||||
- Leave metrics, duration, and insight blank.
|
||||
|
||||
### 2. After 30-Minute Check — Update with Initial Metrics
|
||||
|
||||
Update the entry with:
|
||||
- Current loss and its trajectory direction.
|
||||
- Step time.
|
||||
- Number of validation videos generated.
|
||||
- Preliminary go/no-go assessment.
|
||||
|
||||
### 3. On Completion — Fill Final Entry
|
||||
|
||||
Update the entry with `status: completed`:
|
||||
- Final loss, grad norm, avg step time.
|
||||
- Total duration and steps.
|
||||
- Checkpoint path.
|
||||
- Key insight.
|
||||
|
||||
### 4. On Failure — Document Failure Mode
|
||||
|
||||
Update the entry with `status: failed`:
|
||||
- What went wrong (OOM, NaN, crash, etc.).
|
||||
- At what step the failure occurred.
|
||||
- Create a lesson in `.agents/lessons/` for non-trivial failures.
|
||||
|
||||
### 5. Cross-Reference
|
||||
|
||||
- Link related lessons: `**Related lessons**: .agents/lessons/<filename>.md`
|
||||
- Link related experiments: if this is a follow-up, reference the prior entry.
|
||||
@@ -0,0 +1,87 @@
|
||||
---
|
||||
description: End-to-end experiment lifecycle from hypothesis to lessons learned
|
||||
---
|
||||
|
||||
# Experiment Lifecycle SOP
|
||||
|
||||
Standard operating procedure for running ML training experiments on
|
||||
FastVideo-WorldModel. Every experiment should follow this flow.
|
||||
|
||||
## Overview
|
||||
|
||||
```
|
||||
Plan → Launch → Monitor → Summarize → Journal → Reflect
|
||||
```
|
||||
|
||||
## Steps
|
||||
|
||||
### 1. Plan the Experiment
|
||||
|
||||
Before launching:
|
||||
- [ ] Define a clear **hypothesis** (what you expect to learn).
|
||||
- [ ] Select the **model** and **pipeline** type (finetune, distill, lora, etc.).
|
||||
- [ ] Prepare the **dataset** (preprocessed into parquet format).
|
||||
- [ ] Review existing experiments in `.agents/memory/experiment-journal/README.md` for related work.
|
||||
- [ ] Check `.agents/lessons/` for known pitfalls with this configuration.
|
||||
- [ ] Document the plan in the experiment journal as a draft entry.
|
||||
|
||||
### 2. Launch the Experiment
|
||||
|
||||
Use the `launch-experiment` skill:
|
||||
- Provide: pipeline, model, data_path, num_gpus, and any hyperparameter overrides.
|
||||
- The skill generates the `torchrun` command and creates a journal entry.
|
||||
- Verify the command looks correct before executing.
|
||||
|
||||
Reference: `.agents/skills/launch-experiment.md`
|
||||
|
||||
### 3. Monitor the Experiment
|
||||
|
||||
Use the `monitor-experiment` skill:
|
||||
- Provide the W&B run ID (or output_dir for offline).
|
||||
- Monitor alerts: loss spikes, NaN gradients, step time regressions.
|
||||
- At the **30-minute mark**: perform the quality check.
|
||||
- Is loss decreasing?
|
||||
- Are validation videos reasonable?
|
||||
- Is step time consistent?
|
||||
- **Decision point**: Continue or abort based on the 30-min check.
|
||||
|
||||
Reference: `.agents/skills/monitor-experiment.md`
|
||||
|
||||
### 4. Summarize the Run
|
||||
|
||||
After completion (or at any checkpoint), use the `summarize-run` skill:
|
||||
- Extract final metrics from W&B summary.
|
||||
- Compare against reference runs if available.
|
||||
- Generate a structured report.
|
||||
|
||||
Reference: `.agents/skills/summarize-run.md`
|
||||
|
||||
### 5. Update the Experiment Journal
|
||||
|
||||
Use the `log-experiment` skill to update the journal entry:
|
||||
- Fill in final metrics, duration, checkpoint paths.
|
||||
- Record the key insight learned.
|
||||
- Set status to `completed`, `failed`, or `abandoned`.
|
||||
|
||||
Reference: `.agents/skills/log-experiment.md`
|
||||
|
||||
### 6. Reflect and Capture Lessons
|
||||
|
||||
After every experiment:
|
||||
- **What went right?** → Note in the journal insight field.
|
||||
- **What went wrong?** → Create a lesson in `.agents/lessons/`:
|
||||
- Use the template in `.agents/lessons/README.md`.
|
||||
- Cross-reference the experiment journal entry.
|
||||
- **What was surprising?** → Consider creating an exploration log if this
|
||||
warrants further investigation.
|
||||
|
||||
Reference: `.agents/workflows/lesson-capture.md`
|
||||
|
||||
## Validation Criteria
|
||||
|
||||
This SOP is validated when an agent can:
|
||||
1. Follow steps 1–6 end-to-end for a minimal training run
|
||||
(e.g., `examples/training/finetune/wan_t2v_1.3B/crush_smol/finetune_t2v.sh`
|
||||
with `--max_train_steps 5`).
|
||||
2. Produce a complete experiment journal entry.
|
||||
3. Generate a run summary report.
|
||||
@@ -0,0 +1,71 @@
|
||||
---
|
||||
description: Post-experiment reflection to capture lessons learned
|
||||
---
|
||||
|
||||
# Lesson Capture SOP
|
||||
|
||||
Systematic procedure for turning experiment outcomes into persistent knowledge.
|
||||
|
||||
## When to Use
|
||||
|
||||
After **every** completed or failed experiment. Even successful experiments
|
||||
can yield lessons (e.g., "LR 5e-5 works better than 1e-5 for LoRA").
|
||||
|
||||
## Steps
|
||||
|
||||
### 1. Review the Experiment
|
||||
|
||||
Read the experiment journal entry. Ask:
|
||||
- Did anything go wrong?
|
||||
- Was anything surprising?
|
||||
- Did anything take longer than expected?
|
||||
- Was a workaround needed?
|
||||
|
||||
### 2. Decide: Lesson or Not?
|
||||
|
||||
| Situation | Action |
|
||||
|-----------|--------|
|
||||
| Something broke | Create a lesson (category: `infrastructure` or `data`) |
|
||||
| Hyperparameter choice mattered | Create a lesson (category: `hyperparameter`) |
|
||||
| Porting issue found | Create a lesson (category: `porting`) |
|
||||
| Evaluation metric was misleading | Create a lesson (category: `evaluation`) |
|
||||
| Everything went smoothly | No lesson needed, but note in the journal insight |
|
||||
|
||||
### 3. Create the Lesson File
|
||||
|
||||
In `.agents/lessons/`, create `<YYYY-MM-DD>_<short-slug>.md`:
|
||||
|
||||
```markdown
|
||||
---
|
||||
date: <ISO-8601>
|
||||
experiment: <journal entry reference>
|
||||
category: hyperparameter | data | infrastructure | evaluation | porting
|
||||
severity: critical | important | minor
|
||||
---
|
||||
|
||||
# <Short Descriptive Title>
|
||||
|
||||
## What Happened
|
||||
<description>
|
||||
|
||||
## Root Cause
|
||||
<analysis>
|
||||
|
||||
## Fix / Workaround
|
||||
<resolution>
|
||||
|
||||
## Prevention
|
||||
<how to avoid in future>
|
||||
```
|
||||
|
||||
### 4. Cross-Reference
|
||||
|
||||
- Update the experiment journal entry with a link to the lesson file.
|
||||
- If a similar lesson already exists, add a reference or update it.
|
||||
|
||||
### 5. Periodic Pattern Review
|
||||
|
||||
Every ~10 lessons, scan for patterns:
|
||||
- Multiple lessons in the same category → consider a new skill or SOP.
|
||||
- Repeated mistakes → strengthen the relevant SOP with a checklist item.
|
||||
- Infrastructure issues → propose a codebase fix.
|
||||
@@ -0,0 +1,46 @@
|
||||
{
|
||||
"benchmark_id": "wan-t2v-1.3b-2gpu",
|
||||
"description": "Wan2.1 T2V 1.3B inference performance",
|
||||
"model": {
|
||||
"model_path": "Wan-AI/Wan2.1-T2V-1.3B-Diffusers",
|
||||
"model_short_name": "Wan2.1-T2V-1.3B"
|
||||
},
|
||||
"init_kwargs": {
|
||||
"num_gpus": 2,
|
||||
"flow_shift": 7.0,
|
||||
"sp_size": 2,
|
||||
"tp_size": 1,
|
||||
"vae_sp": true,
|
||||
"vae_tiling": true,
|
||||
"text_encoder_precisions": ["fp32"]
|
||||
},
|
||||
"generation_kwargs": {
|
||||
"height": 480,
|
||||
"width": 832,
|
||||
"num_frames": 45,
|
||||
"num_inference_steps": 4,
|
||||
"guidance_scale": 3,
|
||||
"embedded_cfg_scale": 6,
|
||||
"seed": 1024,
|
||||
"fps": 24,
|
||||
"neg_prompt": "Bright tones, overexposed, static, blurred details, subtitles, style, works, paintings, images, static, overall gray, worst quality, low quality, JPEG compression residue, ugly, incomplete, extra fingers, poorly drawn hands, poorly drawn faces, deformed, disfigured, misshapen limbs, fused fingers, still picture, messy background, three legs, many people in the background, walking backwards"
|
||||
},
|
||||
"test_prompts": [
|
||||
"Will Smith casually eats noodles, his relaxed demeanor contrasting with the energetic background of a bustling street food market. The scene captures a mix of humor and authenticity. Mid-shot framing, vibrant lighting."
|
||||
],
|
||||
"run_config": {
|
||||
"num_warmup_runs": 2,
|
||||
"num_measurement_runs": 5,
|
||||
"required_gpus": 2
|
||||
},
|
||||
"thresholds": {
|
||||
"L40S": {
|
||||
"max_generation_time_s": 34.0,
|
||||
"max_peak_memory_mb": 11000.0
|
||||
},
|
||||
"default": {
|
||||
"max_generation_time_s": 120.0,
|
||||
"max_peak_memory_mb": 30000.0
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,464 @@
|
||||
env:
|
||||
IMAGE_VERSION: "py3.12-latest"
|
||||
BUILDKITE_CLEAN_CHECKOUT: true
|
||||
|
||||
notify:
|
||||
- github_commit_status:
|
||||
context: "fastcheck-passed"
|
||||
if: build.env("TEST_SCOPE") == "fastcheck" || build.env("TEST_SCOPE") == null
|
||||
- github_commit_status:
|
||||
context: "full-suite-passed"
|
||||
if: build.env("TEST_SCOPE") == "full"
|
||||
- github_commit_status:
|
||||
context: "direct-test-completed"
|
||||
if: build.env("TEST_SCOPE") == "direct"
|
||||
|
||||
steps:
|
||||
# ============================================================
|
||||
# Direct test: triggered by /test <name> slash command.
|
||||
# Labels match fastcheck/full-suite counterparts so the GitHub
|
||||
# check status overwrites the original failed check.
|
||||
# Only ONE step executes per build (gated by TEST_TYPE).
|
||||
# ============================================================
|
||||
|
||||
# --- Fastcheck-scope direct tests ---
|
||||
- label: ":microscope: Encoder Tests"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "encoder"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":microscope: VAE Tests"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "vae"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":microscope: Transformer Tests"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "transformer"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":microscope: Kernel Tests"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "kernel_tests"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":microscope: Unit Tests"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "unit_test"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
|
||||
# --- Full-suite-scope direct tests ---
|
||||
- label: ":bar_chart: SSIM Tests"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "ssim"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
- exit_status: 1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":test_tube: LoRA Inference Tests"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "inference_lora"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":test_tube: Training Tests"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "training"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":test_tube: Distillation DMD Tests"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "distillation_dmd"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":test_tube: Self-Forcing Tests"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "self_forcing"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":test_tube: LoRA Training Tests"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "training_lora"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
- exit_status: 1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":test_tube: Training Tests VSA"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "training_vsa"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
- exit_status: 1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":test_tube: Inference Tests VMoBA"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "inference_vmoba"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":test_tube: Performance Tests"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "performance"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":test_tube: API Server Tests"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "api_server"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":test_tube: Train Framework Tests"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "train_framework"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
|
||||
# ============================================================
|
||||
# Fastcheck: Runs on every PR (~10-15 min parallel)
|
||||
# Core component validation: encoders, VAEs, transformers,
|
||||
# CUDA kernels, and unit tests.
|
||||
# ============================================================
|
||||
- label: "Trigger Fastcheck"
|
||||
if: build.env("TEST_SCOPE") == "fastcheck" || build.env("TEST_SCOPE") == null
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
plugins:
|
||||
- monorepo-diff#v1.4.0:
|
||||
diff: 'git fetch origin "${BUILDKITE_PULL_REQUEST_BASE_BRANCH:-main}" && git diff --name-only "origin/${BUILDKITE_PULL_REQUEST_BASE_BRANCH:-main}...HEAD"'
|
||||
watch:
|
||||
- path:
|
||||
- "fastvideo/models/encoders/**"
|
||||
- "fastvideo/models/loader/**"
|
||||
- "fastvideo/tests/encoders/**"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile.python3.12"
|
||||
config:
|
||||
command: "timeout 20m .buildkite/scripts/pr_test.sh"
|
||||
label: ":microscope: Encoder Tests"
|
||||
env:
|
||||
- TEST_TYPE=encoder
|
||||
agents:
|
||||
queue: "default"
|
||||
- path:
|
||||
- "fastvideo/models/vaes/**"
|
||||
- "fastvideo/models/loader/**"
|
||||
- "fastvideo/tests/vaes/**"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile.python3.12"
|
||||
config:
|
||||
command: "timeout 20m .buildkite/scripts/pr_test.sh"
|
||||
label: ":microscope: VAE Tests"
|
||||
env:
|
||||
- TEST_TYPE=vae
|
||||
agents:
|
||||
queue: "default"
|
||||
- path:
|
||||
- "fastvideo/models/dits/**"
|
||||
- "fastvideo/models/loader/**"
|
||||
- "fastvideo/tests/transformers/**"
|
||||
- "fastvideo/layers/**"
|
||||
- "fastvideo/attention/**"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile.python3.12"
|
||||
config:
|
||||
command: "timeout 15m .buildkite/scripts/pr_test.sh"
|
||||
label: ":microscope: Transformer Tests"
|
||||
env:
|
||||
- TEST_TYPE=transformer
|
||||
agents:
|
||||
queue: "default"
|
||||
- path:
|
||||
- "fastvideo-kernel/**"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile.python3.12"
|
||||
config:
|
||||
command: "timeout 15m .buildkite/scripts/pr_test.sh"
|
||||
label: ":microscope: Kernel Tests"
|
||||
env:
|
||||
- TEST_TYPE=kernel_tests
|
||||
agents:
|
||||
queue: "default"
|
||||
- path:
|
||||
- "fastvideo/**"
|
||||
- ".buildkite/**"
|
||||
- ".github/**"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile.python3.12"
|
||||
config:
|
||||
command: "timeout 15m .buildkite/scripts/pr_test.sh"
|
||||
label: ":microscope: Unit Tests"
|
||||
env:
|
||||
- TEST_TYPE=unit_test
|
||||
agents:
|
||||
queue: "default"
|
||||
|
||||
# ============================================================
|
||||
# Full Suite: Runs when TEST_SCOPE=full
|
||||
# Triggered by adding the 'ready' label (via ci-trigger-full-suite.yml)
|
||||
# or on-demand via /test full slash command.
|
||||
# Includes integration tests, SSIM regression, training pipelines,
|
||||
# and performance benchmarks.
|
||||
# ============================================================
|
||||
- label: "Trigger Full Suite"
|
||||
if: build.env("TEST_SCOPE") == "full"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
plugins:
|
||||
- monorepo-diff#v1.4.0:
|
||||
diff: 'git fetch origin "${BUILDKITE_PULL_REQUEST_BASE_BRANCH:-main}" && git diff --name-only "origin/${BUILDKITE_PULL_REQUEST_BASE_BRANCH:-main}...HEAD"'
|
||||
watch:
|
||||
- path:
|
||||
- "fastvideo/**/*.py"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile.python3.12"
|
||||
config:
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
label: ":bar_chart: SSIM Tests"
|
||||
env:
|
||||
- TEST_TYPE=ssim
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- path:
|
||||
- "fastvideo/tests/lora/**"
|
||||
- "fastvideo/models/loader/**"
|
||||
- "fastvideo/tests/transformers/**"
|
||||
- "fastvideo/pipelines/**"
|
||||
- "fastvideo/layers/lora/**"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile.python3.12"
|
||||
config:
|
||||
command: "timeout 20m .buildkite/scripts/pr_test.sh"
|
||||
label: ":test_tube: LoRA Inference Tests"
|
||||
env:
|
||||
- TEST_TYPE=inference_lora
|
||||
agents:
|
||||
queue: "default"
|
||||
- path:
|
||||
- "fastvideo/**"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile.python3.12"
|
||||
config:
|
||||
command: "timeout 15m .buildkite/scripts/pr_test.sh"
|
||||
label: ":test_tube: Training Tests"
|
||||
env:
|
||||
- TEST_TYPE=training
|
||||
agents:
|
||||
queue: "default"
|
||||
- path:
|
||||
- "fastvideo/training/*distillation_pipeline.py"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile.python3.12"
|
||||
config:
|
||||
command: "timeout 15m .buildkite/scripts/pr_test.sh"
|
||||
label: ":test_tube: Distillation DMD Tests"
|
||||
env:
|
||||
- TEST_TYPE=distillation_dmd
|
||||
agents:
|
||||
queue: "default"
|
||||
- path:
|
||||
- "fastvideo/training/*self_forcing_distillation_pipeline.py"
|
||||
- "fastvideo/tests/training/self-forcing/**"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile.python3.12"
|
||||
config:
|
||||
command: "timeout 15m .buildkite/scripts/pr_test.sh"
|
||||
label: ":test_tube: Self-Forcing Tests"
|
||||
env:
|
||||
- TEST_TYPE=self_forcing
|
||||
agents:
|
||||
queue: "default"
|
||||
- path:
|
||||
- "fastvideo/**"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile.python3.12"
|
||||
config:
|
||||
command: "timeout 15m .buildkite/scripts/pr_test.sh"
|
||||
label: ":test_tube: LoRA Training Tests"
|
||||
env:
|
||||
- TEST_TYPE=training_lora
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- path:
|
||||
- "fastvideo/**"
|
||||
- "fastvideo-kernel/**"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile.python3.12"
|
||||
config:
|
||||
command: "timeout 15m .buildkite/scripts/pr_test.sh"
|
||||
label: ":test_tube: Training Tests VSA"
|
||||
env:
|
||||
- TEST_TYPE=training_vsa
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- path:
|
||||
- "fastvideo-kernel/**"
|
||||
- "fastvideo/attention/backends/vmoba.py"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile.python3.12"
|
||||
config:
|
||||
command: "timeout 15m .buildkite/scripts/pr_test.sh"
|
||||
label: ":test_tube: Inference Tests VMoBA"
|
||||
env:
|
||||
- TEST_TYPE=inference_vmoba
|
||||
agents:
|
||||
queue: "default"
|
||||
- path:
|
||||
- "fastvideo/models/dits/**"
|
||||
- "fastvideo/pipelines/**"
|
||||
- "fastvideo/attention/**"
|
||||
- "fastvideo/layers/**"
|
||||
- "fastvideo/worker/**"
|
||||
- "fastvideo/entrypoints/**"
|
||||
- "fastvideo/tests/performance/**"
|
||||
- ".buildkite/performance-benchmarks/**"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile.python3.12"
|
||||
config:
|
||||
command: "timeout 30m .buildkite/scripts/pr_test.sh"
|
||||
label: ":test_tube: Performance Tests"
|
||||
env:
|
||||
- TEST_TYPE=performance
|
||||
agents:
|
||||
queue: "default"
|
||||
- path:
|
||||
- "fastvideo/entrypoints/openai/**"
|
||||
- "fastvideo/entrypoints/cli/serve.py"
|
||||
- "fastvideo/tests/entrypoints/test_openai_api_integration.py"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile.python3.12"
|
||||
config:
|
||||
command: "timeout 30m .buildkite/scripts/pr_test.sh"
|
||||
label: ":test_tube: API Server Tests"
|
||||
env:
|
||||
- TEST_TYPE=api_server
|
||||
agents:
|
||||
queue: "default"
|
||||
- path:
|
||||
- "fastvideo/train/**"
|
||||
- "fastvideo/tests/train/models/**"
|
||||
- "fastvideo/tests/train/fixtures/**"
|
||||
- "fastvideo/models/dits/**"
|
||||
- "fastvideo/models/loader/**"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile.python3.12"
|
||||
config:
|
||||
command: "timeout 30m .buildkite/scripts/pr_test.sh"
|
||||
label: ":test_tube: Train Framework Tests"
|
||||
env:
|
||||
- TEST_TYPE=train_framework
|
||||
agents:
|
||||
queue: "default"
|
||||
Executable
+239
@@ -0,0 +1,239 @@
|
||||
#!/bin/bash
|
||||
set -uo pipefail
|
||||
|
||||
log() {
|
||||
echo "[$(date '+%Y-%m-%d %H:%M:%S')] $1"
|
||||
}
|
||||
|
||||
log "=== Starting Modal test execution ==="
|
||||
|
||||
# Change to the project directory
|
||||
cd "$(dirname "$0")/../.."
|
||||
PROJECT_ROOT=$(pwd)
|
||||
log "Project root: $PROJECT_ROOT"
|
||||
|
||||
# Install Modal if not available
|
||||
if ! python3 -m modal --version &> /dev/null; then
|
||||
log "Modal not found, installing..."
|
||||
if ! command -v uv &> /dev/null; then
|
||||
log "uv not found, bootstrapping..."
|
||||
if ! curl -LsSf https://astral.sh/uv/install.sh | sh; then
|
||||
log "Error: Failed to bootstrap uv via astral.sh installer."
|
||||
exit 1
|
||||
fi
|
||||
export PATH="$HOME/.local/bin:$PATH"
|
||||
if ! command -v uv &> /dev/null; then
|
||||
log "Error: uv still not on PATH after bootstrap."
|
||||
exit 1
|
||||
fi
|
||||
fi
|
||||
# --break-system-packages preserves prior `pip install --user` semantics on PEP 668 agents.
|
||||
uv pip install --system --break-system-packages modal
|
||||
|
||||
# Verify installation
|
||||
if ! python3 -m modal --version &> /dev/null; then
|
||||
log "Error: Failed to install modal. Please install it manually."
|
||||
exit 1
|
||||
fi
|
||||
fi
|
||||
|
||||
log "modal version: $(python3 -m modal --version)"
|
||||
|
||||
# Set up Modal authentication using Buildkite secrets
|
||||
log "Setting up Modal authentication from Buildkite secrets..."
|
||||
MODAL_TOKEN_ID=$(buildkite-agent secret get modal_token_id)
|
||||
MODAL_TOKEN_SECRET=$(buildkite-agent secret get modal_token_secret)
|
||||
|
||||
# Retrieve other secrets
|
||||
WANDB_API_KEY=$(buildkite-agent secret get wandb_api_key)
|
||||
HF_API_KEY=$(buildkite-agent secret get hf_api_key)
|
||||
|
||||
if [ -n "$MODAL_TOKEN_ID" ] && [ -n "$MODAL_TOKEN_SECRET" ]; then
|
||||
log "Retrieved Modal credentials from Buildkite secrets"
|
||||
python3 -m modal token set --token-id "$MODAL_TOKEN_ID" --token-secret "$MODAL_TOKEN_SECRET" --profile buildkite-ci --activate --verify
|
||||
if [ $? -eq 0 ]; then
|
||||
log "Modal authentication successful"
|
||||
else
|
||||
log "Error: Failed to set Modal credentials"
|
||||
exit 1
|
||||
fi
|
||||
else
|
||||
log "Error: Could not retrieve Modal credentials from Buildkite secrets."
|
||||
log "Please ensure 'modal_token_id' and 'modal_token_secret' secrets are set in Buildkite."
|
||||
exit 1
|
||||
fi
|
||||
|
||||
MODAL_TEST_FILE="fastvideo/tests/modal/pr_test.py"
|
||||
MODAL_SSIM_TEST_FILE="fastvideo/tests/modal/ssim_test.py"
|
||||
|
||||
if [ -z "${TEST_TYPE:-}" ]; then
|
||||
log "Error: TEST_TYPE environment variable is not set"
|
||||
exit 1
|
||||
fi
|
||||
log "Test type: $TEST_TYPE"
|
||||
|
||||
EFFECTIVE_PR=${BUILDKITE_PULL_REQUEST:-false}
|
||||
if [ "$EFFECTIVE_PR" = "false" ] && [ -n "${PR_NUMBER:-}" ]; then
|
||||
EFFECTIVE_PR=$PR_NUMBER
|
||||
fi
|
||||
MODAL_ENV="BUILDKITE_REPO=$BUILDKITE_REPO BUILDKITE_COMMIT=$BUILDKITE_COMMIT BUILDKITE_PULL_REQUEST=$EFFECTIVE_PR BUILDKITE_BRANCH=${BUILDKITE_BRANCH:-} TEST_SCOPE=${TEST_SCOPE:-} IMAGE_VERSION=$IMAGE_VERSION"
|
||||
|
||||
POST_RUN_HOOK=""
|
||||
|
||||
upload_performance_artifacts() {
|
||||
SHORT_SHA=${BUILDKITE_COMMIT:0:7}
|
||||
LOCAL_DIR="downloaded_reports"
|
||||
|
||||
_download_reports() {
|
||||
log "Downloading perf_reports/ from Modal Volume..."
|
||||
mkdir -p "$LOCAL_DIR"
|
||||
if ! modal volume get hf-model-weights "perf_reports/" "$LOCAL_DIR"; then
|
||||
log "Error: Failed to download perf_reports/ from Modal Volume."
|
||||
return 1
|
||||
fi
|
||||
}
|
||||
|
||||
_upload_dashboard() {
|
||||
local target
|
||||
target=$(find "$LOCAL_DIR" -name "dashboard_${SHORT_SHA}_*" | head -n 1)
|
||||
log "TARGET dashboard: '$target'"
|
||||
|
||||
if [ -n "$target" ]; then
|
||||
log "Found dashboard: $target. Uploading to Buildkite..."
|
||||
buildkite-agent artifact upload "$target"
|
||||
buildkite-agent annotate --style info --context "perf-dashboard" < "$target"
|
||||
else
|
||||
log "Warning: Could not find a dashboard file matching $SHORT_SHA"
|
||||
fi
|
||||
}
|
||||
|
||||
_upload_perf_summary() {
|
||||
local target
|
||||
target=$(find "$LOCAL_DIR" -name "perf_${SHORT_SHA}_*" | head -n 1)
|
||||
log "TARGET perf summary: '$target'"
|
||||
|
||||
if [ -n "$target" ]; then
|
||||
log "Found perf summary: $target. Uploading to Buildkite..."
|
||||
buildkite-agent artifact upload "$target"
|
||||
buildkite-agent annotate --style info --context "perf-summary" < "$target"
|
||||
else
|
||||
log "Warning: Could not find a perf summary file matching $SHORT_SHA"
|
||||
fi
|
||||
}
|
||||
|
||||
_cleanup_modal_volume() {
|
||||
log "Cleaning up perf_reports/ from Modal Volume..."
|
||||
if modal volume rm hf-model-weights "perf_reports/" --recursive; then
|
||||
log "Successfully deleted perf_reports/ from Modal Volume."
|
||||
else
|
||||
log "Warning: Failed to delete perf_reports/ from Modal Volume. Manual cleanup may be required."
|
||||
fi
|
||||
}
|
||||
|
||||
_cleanup_local() {
|
||||
log "Cleaning up local download directory..."
|
||||
rm -rf "$LOCAL_DIR"
|
||||
}
|
||||
|
||||
# --- Main flow ---
|
||||
_download_reports || { _cleanup_local; return 1; }
|
||||
_upload_dashboard
|
||||
_upload_perf_summary
|
||||
_cleanup_modal_volume
|
||||
_cleanup_local
|
||||
}
|
||||
|
||||
case "$TEST_TYPE" in
|
||||
"encoder")
|
||||
log "Running encoder tests..."
|
||||
MODAL_COMMAND="$MODAL_ENV HF_API_KEY=$HF_API_KEY python3 -m modal run $MODAL_TEST_FILE::run_encoder_tests"
|
||||
;;
|
||||
"vae")
|
||||
log "Running VAE tests..."
|
||||
MODAL_COMMAND="$MODAL_ENV HF_API_KEY=$HF_API_KEY python3 -m modal run $MODAL_TEST_FILE::run_vae_tests"
|
||||
;;
|
||||
"transformer")
|
||||
log "Running transformer tests..."
|
||||
MODAL_COMMAND="$MODAL_ENV HF_API_KEY=$HF_API_KEY python3 -m modal run $MODAL_TEST_FILE::run_transformer_tests"
|
||||
;;
|
||||
"ssim")
|
||||
log "Running SSIM tests..."
|
||||
MODAL_COMMAND="$MODAL_ENV HF_API_KEY=$HF_API_KEY python3 -m modal run $MODAL_SSIM_TEST_FILE::run_ssim_tests"
|
||||
;;
|
||||
"training")
|
||||
log "Running training tests..."
|
||||
MODAL_COMMAND="$MODAL_ENV WANDB_API_KEY=$WANDB_API_KEY python3 -m modal run $MODAL_TEST_FILE::run_training_tests"
|
||||
;;
|
||||
"training_lora")
|
||||
log "Running LoRA training tests..."
|
||||
MODAL_COMMAND="$MODAL_ENV WANDB_API_KEY=$WANDB_API_KEY python3 -m modal run $MODAL_TEST_FILE::run_training_lora_tests"
|
||||
;;
|
||||
"training_vsa")
|
||||
log "Running training VSA tests..."
|
||||
MODAL_COMMAND="$MODAL_ENV WANDB_API_KEY=$WANDB_API_KEY python3 -m modal run $MODAL_TEST_FILE::run_training_tests_VSA"
|
||||
;;
|
||||
"kernel_tests")
|
||||
log "Running kernel tests..."
|
||||
MODAL_COMMAND="$MODAL_ENV python3 -m modal run $MODAL_TEST_FILE::run_kernel_tests"
|
||||
;;
|
||||
"inference_lora")
|
||||
log "Running LoRA tests..."
|
||||
MODAL_COMMAND="$MODAL_ENV python3 -m modal run $MODAL_TEST_FILE::run_inference_lora_tests"
|
||||
;;
|
||||
"distillation_dmd")
|
||||
log "Running distillation DMD tests..."
|
||||
MODAL_COMMAND="$MODAL_ENV WANDB_API_KEY=$WANDB_API_KEY python3 -m modal run $MODAL_TEST_FILE::run_distill_dmd_tests"
|
||||
;;
|
||||
# run_inference_tests_vmoba
|
||||
"self_forcing")
|
||||
log "Running self-forcing tests..."
|
||||
MODAL_COMMAND="$MODAL_ENV WANDB_API_KEY=$WANDB_API_KEY python3 -m modal run $MODAL_TEST_FILE::run_self_forcing_tests"
|
||||
;;
|
||||
"inference_vmoba")
|
||||
log "Running V-MoBA inference tests..."
|
||||
MODAL_COMMAND="$MODAL_ENV python3 -m modal run $MODAL_TEST_FILE::run_inference_tests_vmoba"
|
||||
;;
|
||||
"unit_test")
|
||||
log "Running unit tests..."
|
||||
MODAL_COMMAND="$MODAL_ENV python3 -m modal run $MODAL_TEST_FILE::run_unit_test"
|
||||
;;
|
||||
"train_framework")
|
||||
log "Running fastvideo.train framework tests..."
|
||||
MODAL_COMMAND="$MODAL_ENV HF_API_KEY=$HF_API_KEY python3 -m modal run $MODAL_TEST_FILE::run_train_framework_tests"
|
||||
;;
|
||||
"lora_extraction")
|
||||
log "Running LoRA extraction tests..."
|
||||
MODAL_COMMAND="$MODAL_ENV HF_API_KEY=$HF_API_KEY python3 -m modal run $MODAL_TEST_FILE::run_lora_extraction_tests"
|
||||
;;
|
||||
"performance")
|
||||
log "Running performance tests on Modal..."
|
||||
MODAL_COMMAND="$MODAL_ENV HF_API_KEY=$HF_API_KEY python3 -m modal run $MODAL_TEST_FILE::run_performance_tests"
|
||||
POST_RUN_HOOK="upload_performance_artifacts"
|
||||
;;
|
||||
"api_server")
|
||||
log "Running API server integration tests..."
|
||||
MODAL_COMMAND="$MODAL_ENV HF_API_KEY=$HF_API_KEY python3 -m modal run $MODAL_TEST_FILE::run_api_server_tests"
|
||||
;;
|
||||
*)
|
||||
log "Error: Unknown test type: $TEST_TYPE"
|
||||
exit 1
|
||||
;;
|
||||
esac
|
||||
|
||||
log "Executing: $MODAL_COMMAND"
|
||||
eval "$MODAL_COMMAND"
|
||||
TEST_EXIT_CODE=$?
|
||||
|
||||
if [ $TEST_EXIT_CODE -eq 0 ]; then
|
||||
log "Modal test completed successfully"
|
||||
else
|
||||
log "Error: Modal test failed with exit code: $TEST_EXIT_CODE"
|
||||
fi
|
||||
|
||||
if [ -n "$POST_RUN_HOOK" ]; then
|
||||
log "Executing post-run hook: $POST_RUN_HOOK"
|
||||
"$POST_RUN_HOOK"
|
||||
fi
|
||||
|
||||
log "=== Test execution completed with exit code: $TEST_EXIT_CODE ==="
|
||||
exit $TEST_EXIT_CODE
|
||||
@@ -0,0 +1,53 @@
|
||||
#!/bin/bash
|
||||
set -uo pipefail
|
||||
|
||||
log() {
|
||||
echo "[$(date '+%Y-%m-%d %H:%M:%S')] $1"
|
||||
}
|
||||
|
||||
log "=== Starting pre-commit checks ==="
|
||||
|
||||
cd "$(dirname "$0")/../.."
|
||||
PROJECT_ROOT=$(pwd)
|
||||
log "Project root: $PROJECT_ROOT"
|
||||
|
||||
if ! python3 -m pre_commit --version &> /dev/null; then
|
||||
log "pre-commit not found, installing..."
|
||||
if ! command -v uv &> /dev/null; then
|
||||
log "uv not found, bootstrapping..."
|
||||
if ! curl -LsSf https://astral.sh/uv/install.sh | sh; then
|
||||
log "Error: Failed to bootstrap uv via astral.sh installer."
|
||||
exit 1
|
||||
fi
|
||||
export PATH="$HOME/.local/bin:$PATH"
|
||||
if ! command -v uv &> /dev/null; then
|
||||
log "Error: uv still not on PATH after bootstrap."
|
||||
exit 1
|
||||
fi
|
||||
fi
|
||||
# --break-system-packages preserves prior `pip install --user` semantics on PEP 668 agents.
|
||||
uv pip install --system --break-system-packages pre-commit==4.0.1
|
||||
|
||||
if ! python3 -m pre_commit --version &> /dev/null; then
|
||||
log "Error: Failed to install pre-commit."
|
||||
exit 1
|
||||
fi
|
||||
fi
|
||||
|
||||
log "Pre-commit version: $(python3 -m pre_commit --version)"
|
||||
|
||||
log "Installing/updating pre-commit hooks..."
|
||||
python3 -m pre_commit install --install-hooks
|
||||
|
||||
log "Running pre-commit checks on all files..."
|
||||
python3 -m pre_commit run --all-files
|
||||
PRE_COMMIT_EXIT_CODE=$?
|
||||
|
||||
if [ $PRE_COMMIT_EXIT_CODE -eq 0 ]; then
|
||||
log "Pre-commit checks completed successfully"
|
||||
else
|
||||
log "Error: Pre-commit checks failed with exit code: $PRE_COMMIT_EXIT_CODE"
|
||||
fi
|
||||
|
||||
log "=== Pre-commit checks completed with exit code: $PRE_COMMIT_EXIT_CODE ==="
|
||||
exit $PRE_COMMIT_EXIT_CODE
|
||||
@@ -4,14 +4,6 @@ title: "[Bug] "
|
||||
labels: ['Bug']
|
||||
|
||||
body:
|
||||
- type: textarea
|
||||
attributes:
|
||||
label: Environment
|
||||
description: |
|
||||
Please share your environment with us. You can run the command **python fastvideo/utils/env_utils.py** and copy-paste its output below.
|
||||
placeholder: FastVideo version, platform, python version, cuda version...
|
||||
validations:
|
||||
required: true
|
||||
- type: textarea
|
||||
attributes:
|
||||
label: Describe the bug
|
||||
@@ -25,5 +17,13 @@ body:
|
||||
What command or script did you run? Which **model** are you using?
|
||||
placeholder: |
|
||||
A placeholder for the command.
|
||||
validations:
|
||||
required: true
|
||||
- type: textarea
|
||||
attributes:
|
||||
label: Environment
|
||||
description: |
|
||||
Please share your environment with us. You can run the command **python collect_env.py** and copy-paste its output below.
|
||||
placeholder: FastVideo version, platform, python version, cuda version...
|
||||
validations:
|
||||
required: true
|
||||
@@ -0,0 +1,56 @@
|
||||
name: 💬 Request for comments (RFC).
|
||||
description: Ask for feedback on major architectural changes or design choices.
|
||||
title: "[RFC]: "
|
||||
labels: ["RFC"]
|
||||
|
||||
body:
|
||||
- type: markdown
|
||||
attributes:
|
||||
value: >
|
||||
#### Please take a look at previous [RFCs](https://github.com/hao-ai-lab/FastVideo/issues?q=label%3ARFC+sort%3Aupdated-desc) for reference.
|
||||
- type: textarea
|
||||
attributes:
|
||||
label: Motivation.
|
||||
description: >
|
||||
The motivation of the RFC.
|
||||
validations:
|
||||
required: true
|
||||
- type: textarea
|
||||
attributes:
|
||||
label: Proposed Change.
|
||||
description: >
|
||||
The proposed change of the RFC.
|
||||
validations:
|
||||
required: true
|
||||
- type: textarea
|
||||
attributes:
|
||||
label: Feedback Period.
|
||||
description: >
|
||||
The feedback period of the RFC. Usually at least one week.
|
||||
validations:
|
||||
required: false
|
||||
- type: textarea
|
||||
attributes:
|
||||
label: CC List.
|
||||
description: >
|
||||
The list of people you want to CC.
|
||||
validations:
|
||||
required: false
|
||||
- type: textarea
|
||||
attributes:
|
||||
label: Any Other Things.
|
||||
description: >
|
||||
Any other things you would like to mention.
|
||||
validations:
|
||||
required: false
|
||||
- type: markdown
|
||||
attributes:
|
||||
value: >
|
||||
Thanks for contributing 🎉!
|
||||
- type: checkboxes
|
||||
id: askllm
|
||||
attributes:
|
||||
label: Before submitting a new issue...
|
||||
options:
|
||||
- label: Make sure you already searched for relevant issues.
|
||||
required: true
|
||||
@@ -0,0 +1 @@
|
||||
blank_issues_enabled: false
|
||||
@@ -0,0 +1,63 @@
|
||||
<!--
|
||||
PR TITLE: Must start with a type tag, e.g.:
|
||||
[feat] Add new model [bugfix] Fix VAE tiling [refactor] Restructure pipeline
|
||||
[perf] Optimize kernel [ci] Update tests [docs] Add guide
|
||||
[misc] Cleanup configs [new-model] Port Flux2 [infra] Add trace hooks
|
||||
[skill] Add agent skill
|
||||
|
||||
MERGE WORKFLOW:
|
||||
1. Ensure pre-commit passes and you have at least 1 approval
|
||||
2. Comment /merge (or add the "ready" label) to enter the Merge Queue
|
||||
3. Full Test Suite runs automatically on a staging branch → auto-merge on success
|
||||
|
||||
ON-DEMAND TESTING (write access required):
|
||||
/test full — Full Test Suite /test ssim — SSIM regression
|
||||
/test training — Training pipeline /test encoder — Encoder tests
|
||||
/test transformer — Transformer tests /test vae — VAE tests
|
||||
/test kernel — CUDA kernel tests /test unit — Unit tests
|
||||
See docs/contributing/pull_requests.md for all 17 test commands
|
||||
-->
|
||||
|
||||
## Purpose
|
||||
|
||||
<!-- What does this PR do? Link the related issue if applicable. -->
|
||||
|
||||
Fixes #
|
||||
|
||||
## Changes
|
||||
|
||||
<!-- Describe your changes concisely. What approach did you take? -->
|
||||
|
||||
-
|
||||
|
||||
## Test Plan
|
||||
|
||||
<!-- How did you verify your changes? Paste exact commands and output. -->
|
||||
|
||||
```bash
|
||||
# Commands you ran
|
||||
```
|
||||
|
||||
## Test Results
|
||||
|
||||
<!-- Paste test output, before/after comparisons, or SSIM scores for model changes. -->
|
||||
|
||||
<details>
|
||||
<summary>Test output</summary>
|
||||
|
||||
```
|
||||
# Paste output here
|
||||
```
|
||||
|
||||
</details>
|
||||
|
||||
## Checklist
|
||||
|
||||
- [ ] I ran `pre-commit run --all-files` and fixed all issues
|
||||
- [ ] I added or updated tests for my changes
|
||||
- [ ] I updated documentation if needed
|
||||
- [ ] I considered GPU memory impact of my changes
|
||||
|
||||
**For model/pipeline changes, also check:**
|
||||
- [ ] I verified SSIM regression tests pass
|
||||
- [ ] I updated the support matrix if adding a new model
|
||||
@@ -0,0 +1,334 @@
|
||||
merge_protections:
|
||||
- name: PR merge requirements
|
||||
if:
|
||||
- base = main
|
||||
success_conditions:
|
||||
- "title~=(?i)^\\[(feat|feature|bugfix|fix|refactor|perf|ci|doc|docs|misc|chore|kernel|new.?model|skill|skills|infra)\\]"
|
||||
- "#approved-reviews-by>=1"
|
||||
- check-success~=pre-commit
|
||||
- check-success=fastcheck-passed
|
||||
- check-success=full-suite-passed
|
||||
|
||||
pull_request_rules:
|
||||
|
||||
# ============================================================
|
||||
# Type labels (from PR title prefix)
|
||||
# ============================================================
|
||||
|
||||
- name: "label type: feat"
|
||||
conditions:
|
||||
- "title~=(?i)^\\[(feat|feature)\\]"
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["type: feat"]
|
||||
|
||||
- name: "label type: bugfix"
|
||||
conditions:
|
||||
- "title~=(?i)^\\[(bug)?fix\\]"
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["type: bugfix"]
|
||||
|
||||
- name: "label type: refactor"
|
||||
conditions:
|
||||
- "title~=(?i)^\\[refactor\\]"
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["type: refactor"]
|
||||
|
||||
- name: "label type: perf"
|
||||
conditions:
|
||||
- "title~=(?i)^\\[perf\\]"
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["type: perf"]
|
||||
|
||||
- name: "label type: ci"
|
||||
conditions:
|
||||
- "title~=(?i)^\\[ci\\]"
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["type: ci"]
|
||||
|
||||
- name: "label type: docs"
|
||||
conditions:
|
||||
- "title~=(?i)^\\[(doc|docs)\\]"
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["type: docs"]
|
||||
|
||||
- name: "label type: misc"
|
||||
conditions:
|
||||
- "title~=(?i)^\\[(misc|chore)\\]"
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["type: misc"]
|
||||
|
||||
- name: "label type: new-model"
|
||||
conditions:
|
||||
- "title~=(?i)^\\[new.?model\\]"
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["type: new-model"]
|
||||
|
||||
- name: "label type: infra"
|
||||
conditions:
|
||||
- "title~=(?i)^\\[infra\\]"
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["type: infra"]
|
||||
|
||||
- name: "label type: skill"
|
||||
conditions:
|
||||
- "title~=(?i)^\\[skills?\\]"
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["type: skill"]
|
||||
|
||||
# ============================================================
|
||||
# Scope labels (from changed files)
|
||||
# ============================================================
|
||||
|
||||
- name: "label scope: training"
|
||||
conditions:
|
||||
- or:
|
||||
- files~=^fastvideo/train/
|
||||
- files~=^fastvideo/training/
|
||||
- files~=^fastvideo/distillation/
|
||||
- files~=^examples/train/
|
||||
- files~=^examples/training/
|
||||
- files~=^examples/distill/
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["scope: training"]
|
||||
|
||||
- name: "label scope: inference"
|
||||
conditions:
|
||||
- or:
|
||||
- files~=^fastvideo/pipelines/basic/
|
||||
- files~=^fastvideo/pipelines/stages/
|
||||
- files~=^fastvideo/pipelines/samplers/
|
||||
- files~=^fastvideo/entrypoints/
|
||||
- files~=^fastvideo/worker/
|
||||
- files~=^fastvideo/api/sampling_param
|
||||
- files~=^fastvideo/configs/pipelines/
|
||||
- files~=^examples/inference/
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["scope: inference"]
|
||||
|
||||
- name: "label scope: attention"
|
||||
conditions:
|
||||
- files~=^fastvideo/attention/
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["scope: attention"]
|
||||
|
||||
- name: "label scope: kernel"
|
||||
conditions:
|
||||
- or:
|
||||
- files~=^fastvideo-kernel/
|
||||
- files~=^csrc/
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["scope: kernel"]
|
||||
|
||||
- name: "label scope: data"
|
||||
conditions:
|
||||
- or:
|
||||
- files~=^fastvideo/dataset/
|
||||
- files~=^fastvideo/pipelines/preprocess/
|
||||
- files~=^examples/preprocessing/
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["scope: data"]
|
||||
|
||||
- name: "label scope: infra"
|
||||
conditions:
|
||||
- or:
|
||||
- files~=^\.github/
|
||||
- files~=^\.buildkite/
|
||||
- files~=^fastvideo/tests/
|
||||
- files~=^docker/
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["scope: infra"]
|
||||
|
||||
- name: "label scope: distributed"
|
||||
conditions:
|
||||
- files~=^fastvideo/distributed/
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["scope: distributed"]
|
||||
|
||||
- name: "label scope: docs"
|
||||
conditions:
|
||||
- files~=^docs/
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["scope: docs"]
|
||||
|
||||
- name: "label scope: ui"
|
||||
conditions:
|
||||
- files~=^ui/
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["scope: ui"]
|
||||
|
||||
- name: "label scope: model"
|
||||
conditions:
|
||||
- or:
|
||||
- files~=^fastvideo/models/
|
||||
- files~=^fastvideo/layers/
|
||||
- files~=^fastvideo/configs/models/
|
||||
- -closed
|
||||
actions:
|
||||
label:
|
||||
add: ["scope: model"]
|
||||
|
||||
# ============================================================
|
||||
# Pre-commit failure help comment
|
||||
# ============================================================
|
||||
|
||||
- name: comment on pre-commit failure
|
||||
conditions:
|
||||
- check-failure~=pre-commit
|
||||
- -closed
|
||||
actions:
|
||||
comment:
|
||||
message: |
|
||||
## Pre-commit checks failed
|
||||
|
||||
Hi @{{author}}, the pre-commit checks have failed. To fix them locally:
|
||||
|
||||
```bash
|
||||
# Install pre-commit if you haven't already
|
||||
uv pip install pre-commit
|
||||
pre-commit install
|
||||
|
||||
# Run all checks and auto-fix what's possible
|
||||
pre-commit run --all-files
|
||||
```
|
||||
|
||||
Common fixes:
|
||||
- **yapf**: `yapf -i <file>` (formatting)
|
||||
- **ruff**: `ruff check --fix <file>` (linting)
|
||||
- **codespell**: `codespell --write-changes <file>` (spelling)
|
||||
|
||||
After fixing, commit and push the changes. The checks will re-run automatically.
|
||||
|
||||
For future commits, `pre-commit` will run automatically on changed files before each commit.
|
||||
|
||||
|
||||
# ============================================================
|
||||
# Merge conflict detection
|
||||
# ============================================================
|
||||
|
||||
- name: label conflicting PRs
|
||||
conditions:
|
||||
- conflict
|
||||
- -closed
|
||||
- label!=stale
|
||||
actions:
|
||||
label:
|
||||
add: [needs-rebase]
|
||||
comment:
|
||||
message: |
|
||||
This PR has merge conflicts with the base branch. Please rebase:
|
||||
|
||||
```bash
|
||||
git fetch origin main
|
||||
git rebase origin/main
|
||||
# Resolve any conflicts, then:
|
||||
git push --force-with-lease
|
||||
```
|
||||
|
||||
- name: remove conflict label when resolved
|
||||
conditions:
|
||||
- -conflict
|
||||
- -closed
|
||||
- label=needs-rebase
|
||||
actions:
|
||||
label:
|
||||
remove: [needs-rebase]
|
||||
|
||||
# ============================================================
|
||||
# Auto-merge and auto-rebase
|
||||
# ============================================================
|
||||
|
||||
- name: auto-merge when ready and all checks pass
|
||||
conditions:
|
||||
- label=ready
|
||||
- "title~=(?i)^\\[(feat|feature|bugfix|fix|refactor|perf|ci|doc|docs|misc|chore|kernel|new.?model|skill|skills|infra)\\]"
|
||||
- "#approved-reviews-by>=1"
|
||||
- check-success~=pre-commit
|
||||
- check-success=fastcheck-passed
|
||||
- check-success=full-suite-passed
|
||||
- -conflict
|
||||
- -closed
|
||||
- -draft
|
||||
actions:
|
||||
merge:
|
||||
method: squash
|
||||
|
||||
- name: auto-update when ready
|
||||
conditions:
|
||||
- label=ready
|
||||
- "#approved-reviews-by>=1"
|
||||
- -conflict
|
||||
- -closed
|
||||
- -draft
|
||||
actions:
|
||||
update: {}
|
||||
|
||||
# ============================================================
|
||||
# PR title format help
|
||||
# ============================================================
|
||||
|
||||
- name: comment on invalid PR title format
|
||||
conditions:
|
||||
- -closed
|
||||
- -draft
|
||||
- "-title~=(?i)^\\[(feat|feature|bugfix|fix|refactor|perf|ci|doc|docs|misc|chore|kernel|new.?model|skill|skills|infra)\\]"
|
||||
actions:
|
||||
comment:
|
||||
message: |
|
||||
## ⚠️ PR title format required
|
||||
|
||||
Your PR title must start with a type tag in brackets. Examples:
|
||||
- `[feat] Add new model support`
|
||||
- `[bugfix] Fix VAE tiling corruption`
|
||||
- `[refactor] Restructure training pipeline`
|
||||
- `[perf] Optimize attention kernel`
|
||||
- `[ci] Update test infrastructure`
|
||||
- `[infra] Add activation trace hooks`
|
||||
- `[docs] Add inference guide`
|
||||
- `[misc] Clean up configs`
|
||||
- `[new-model] Port Flux2 to FastVideo`
|
||||
- `[skill] Add add-model agent skill`
|
||||
|
||||
Valid tags: `feat`, `feature`, `bugfix`, `fix`, `refactor`, `perf`, `ci`, `infra`, `doc`, `docs`, `misc`, `chore`, `kernel`, `new-model`, `skill`, `skills`
|
||||
|
||||
Please update your PR title and the merge protection check will pass automatically.
|
||||
|
||||
merge_protections_settings:
|
||||
reporting_method: check-runs
|
||||
@@ -0,0 +1,106 @@
|
||||
name: Build Image Template
|
||||
|
||||
on:
|
||||
workflow_call:
|
||||
inputs:
|
||||
python_version:
|
||||
required: true
|
||||
type: string
|
||||
dockerfile_path:
|
||||
required: true
|
||||
type: string
|
||||
tag_suffix:
|
||||
required: true
|
||||
type: string
|
||||
|
||||
jobs:
|
||||
build-and-push:
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
contents: read
|
||||
packages: write
|
||||
|
||||
steps:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Free up disk space
|
||||
run: |
|
||||
# Display initial space
|
||||
echo "Initial disk space:"
|
||||
df -h
|
||||
|
||||
# Remove large directories directly
|
||||
sudo rm -rf /usr/share/dotnet
|
||||
sudo rm -rf /usr/local/lib/android
|
||||
sudo rm -rf /opt/ghc
|
||||
sudo rm -rf /usr/local/share/boost
|
||||
sudo rm -rf /usr/share/swift
|
||||
sudo rm -rf /usr/local/lib/node_modules
|
||||
sudo rm -rf /usr/local/share/powershell
|
||||
sudo rm -rf /usr/share/rust
|
||||
sudo rm -rf /usr/local/.ghcup
|
||||
|
||||
# Remove cached files
|
||||
sudo rm -rf /var/lib/apt/lists/*
|
||||
sudo rm -rf /var/cache/apt/archives/*
|
||||
|
||||
# Clean Docker
|
||||
docker system prune -af --volumes
|
||||
|
||||
# Display available space after cleanup
|
||||
echo "Disk space after cleanup:"
|
||||
df -h
|
||||
|
||||
- name: Set up Docker Buildx
|
||||
uses: docker/setup-buildx-action@v3
|
||||
|
||||
- name: Login to GitHub Container Registry
|
||||
uses: docker/login-action@v3
|
||||
with:
|
||||
registry: ghcr.io
|
||||
username: ${{ github.repository_owner }}
|
||||
password: ${{ secrets.GITHUB_TOKEN }}
|
||||
|
||||
- name: Prepare tags
|
||||
id: prepare-tags
|
||||
run: |
|
||||
SHORT_SHA=$(echo ${{ github.sha }} | cut -c1-7)
|
||||
|
||||
TAGS="type=raw,value=${{ inputs.tag_suffix }}-latest"
|
||||
TAGS="${TAGS}\ntype=raw,value=${{ inputs.tag_suffix }}-sha-${SHORT_SHA}"
|
||||
|
||||
# Set Python 3.10 as the default image
|
||||
if [[ "${{ inputs.python_version }}" == "3.10" ]]; then
|
||||
TAGS="${TAGS}\ntype=raw,value=latest"
|
||||
fi
|
||||
|
||||
{
|
||||
echo "tags<<EOF"
|
||||
echo -e "$TAGS"
|
||||
echo "EOF"
|
||||
} >> $GITHUB_OUTPUT
|
||||
|
||||
- name: Extract metadata for Docker
|
||||
id: meta
|
||||
uses: docker/metadata-action@v5
|
||||
with:
|
||||
images: ghcr.io/${{ github.repository }}/fastvideo-dev
|
||||
tags: ${{ steps.prepare-tags.outputs.tags }}
|
||||
|
||||
- name: Build and push Docker image
|
||||
id: build-push
|
||||
uses: docker/build-push-action@v6
|
||||
with:
|
||||
context: .
|
||||
file: ${{ inputs.dockerfile_path }}
|
||||
push: true
|
||||
tags: ${{ steps.meta.outputs.tags }}
|
||||
labels: ${{ steps.meta.outputs.labels }}
|
||||
cache-from: type=gha
|
||||
cache-to: type=gha,mode=max
|
||||
|
||||
- name: Success message
|
||||
run: |
|
||||
echo "✅ Python ${{ inputs.python_version }} image successfully built and pushed to ghcr.io/${{ github.repository }}/fastvideo-dev:${{ inputs.tag_suffix }}-latest"
|
||||
echo "To run tests with this image, manually trigger the 'Run Tests' workflow."
|
||||
@@ -0,0 +1,80 @@
|
||||
name: Aggregate Test Status
|
||||
|
||||
on:
|
||||
status:
|
||||
|
||||
permissions:
|
||||
statuses: write
|
||||
|
||||
jobs:
|
||||
aggregate:
|
||||
if: >-
|
||||
github.event.context == 'direct-test-completed'
|
||||
&& github.event.state == 'success'
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Check and update aggregate status
|
||||
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7.0.1
|
||||
with:
|
||||
script: |
|
||||
const sha = context.payload.sha;
|
||||
|
||||
const { data } = await github.rest.repos.getCombinedStatusForRef({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
ref: sha,
|
||||
per_page: 100,
|
||||
});
|
||||
|
||||
const bkStatuses = data.statuses.filter(
|
||||
s => s.context.startsWith('buildkite/ci/')
|
||||
);
|
||||
|
||||
const FASTCHECK_PREFIX = 'buildkite/ci/microscope-';
|
||||
const FULL_SUITE_PREFIXES = [
|
||||
'buildkite/ci/test-tube-',
|
||||
'buildkite/ci/bar-chart-',
|
||||
];
|
||||
|
||||
const fastcheck = bkStatuses.filter(
|
||||
s => s.context.startsWith(FASTCHECK_PREFIX)
|
||||
);
|
||||
const fullSuite = bkStatuses.filter(
|
||||
s => FULL_SUITE_PREFIXES.some(p => s.context.startsWith(p))
|
||||
);
|
||||
|
||||
if (
|
||||
fastcheck.length > 0
|
||||
&& fastcheck.every(s => s.state === 'success')
|
||||
) {
|
||||
core.info(
|
||||
`All ${fastcheck.length} fastcheck tests passed — updating fastcheck-passed`
|
||||
);
|
||||
await github.rest.repos.createCommitStatus({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
sha,
|
||||
state: 'success',
|
||||
context: 'fastcheck-passed',
|
||||
description:
|
||||
`All ${fastcheck.length} fastcheck tests passed`,
|
||||
});
|
||||
}
|
||||
|
||||
if (
|
||||
fullSuite.length > 0
|
||||
&& fullSuite.every(s => s.state === 'success')
|
||||
) {
|
||||
core.info(
|
||||
`All ${fullSuite.length} full suite tests passed — updating full-suite-passed`
|
||||
);
|
||||
await github.rest.repos.createCommitStatus({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
sha,
|
||||
state: 'success',
|
||||
context: 'full-suite-passed',
|
||||
description:
|
||||
`All ${fullSuite.length} full suite tests passed`,
|
||||
});
|
||||
}
|
||||
@@ -0,0 +1,32 @@
|
||||
name: pre-commit
|
||||
|
||||
on:
|
||||
pull_request:
|
||||
branches: [main]
|
||||
workflow_call:
|
||||
inputs:
|
||||
ref:
|
||||
description: 'Git ref to checkout (defaults to github.ref)'
|
||||
required: false
|
||||
type: string
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
pre-commit:
|
||||
if: github.event_name == 'workflow_call' || github.event.pull_request.draft != true
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
with:
|
||||
ref: ${{ inputs.ref || '' }}
|
||||
- uses: actions/setup-python@v5
|
||||
with:
|
||||
python-version: "3.12"
|
||||
- run: echo "::add-matcher::.github/workflows/matchers/actionlint.json"
|
||||
- run: echo "::add-matcher::.github/workflows/matchers/mypy.json"
|
||||
- run: echo "::add-matcher::.github/workflows/matchers/ruff.json"
|
||||
- uses: pre-commit/action@v3.0.1
|
||||
with:
|
||||
extra_args: --all-files --hook-stage manual
|
||||
@@ -0,0 +1,272 @@
|
||||
name: Slash Commands
|
||||
|
||||
on:
|
||||
issue_comment:
|
||||
types: [created]
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
pull-requests: write
|
||||
statuses: write
|
||||
|
||||
jobs:
|
||||
handle-merge:
|
||||
if: >-
|
||||
github.event.issue.pull_request != null
|
||||
&& startsWith(github.event.comment.body, '/merge')
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Check write permission
|
||||
id: perm
|
||||
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7.0.1
|
||||
with:
|
||||
script: |
|
||||
const { data: perm } = await github.rest.repos.getCollaboratorPermissionLevel({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
username: context.payload.comment.user.login,
|
||||
});
|
||||
const hasWrite = ['admin', 'write'].includes(perm.permission);
|
||||
if (!hasWrite) {
|
||||
core.setFailed(`User ${context.payload.comment.user.login} lacks write permission (has: ${perm.permission}).`);
|
||||
}
|
||||
core.setOutput('has_write', String(hasWrite));
|
||||
|
||||
- name: Add ready label and react
|
||||
id: label
|
||||
if: steps.perm.outputs.has_write == 'true'
|
||||
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7.0.1
|
||||
with:
|
||||
script: |
|
||||
const owner = context.repo.owner;
|
||||
const repo = context.repo.repo;
|
||||
const prNumber = context.payload.issue.number;
|
||||
try { await github.rest.issues.removeLabel({ owner, repo, issue_number: prNumber, name: 'ready' }); } catch {}
|
||||
await github.rest.issues.addLabels({ owner, repo, issue_number: prNumber, labels: ['ready'] });
|
||||
await github.rest.reactions.createForIssueComment({
|
||||
owner, repo,
|
||||
comment_id: context.payload.comment.id,
|
||||
content: 'rocket',
|
||||
});
|
||||
const { data: pr } = await github.rest.pulls.get({ owner, repo, pull_number: prNumber });
|
||||
core.setOutput('pr_sha', pr.head.sha);
|
||||
core.setOutput('pr_branch', pr.head.ref);
|
||||
core.setOutput('pr_number', String(prNumber));
|
||||
|
||||
- name: Trigger Full Suite
|
||||
if: steps.perm.outputs.has_write == 'true'
|
||||
env:
|
||||
BUILDKITE_API_TOKEN: ${{ secrets.BUILDKITE_API_TOKEN }}
|
||||
PR_SHA: ${{ steps.label.outputs.pr_sha }}
|
||||
PR_BRANCH: ${{ steps.label.outputs.pr_branch }}
|
||||
PR_NUMBER: ${{ steps.label.outputs.pr_number }}
|
||||
BK_ORG: ${{ vars.BUILDKITE_ORG_SLUG }}
|
||||
BK_PIPELINE: ${{ vars.BUILDKITE_PIPELINE_SLUG }}
|
||||
run: |
|
||||
curl -sS --fail-with-body -X POST \
|
||||
"https://api.buildkite.com/v2/organizations/${BK_ORG}/pipelines/${BK_PIPELINE}/builds" \
|
||||
-H "Authorization: Bearer $BUILDKITE_API_TOKEN" \
|
||||
-H "Content-Type: application/json" \
|
||||
--data-raw "$(jq -n \
|
||||
--arg commit "$PR_SHA" \
|
||||
--arg branch "$PR_BRANCH" \
|
||||
--arg message "Full Suite for PR #${PR_NUMBER} (via /merge)" \
|
||||
--argjson pr_id "$PR_NUMBER" \
|
||||
'{
|
||||
commit: $commit,
|
||||
branch: $branch,
|
||||
message: $message,
|
||||
ignore_pipeline_branch_filters: true,
|
||||
pull_request_id: $pr_id,
|
||||
pull_request_base_branch: "main",
|
||||
env: {
|
||||
TEST_SCOPE: "full",
|
||||
FULL_SUITE: "true",
|
||||
PR_NUMBER: ($pr_id | tostring)
|
||||
}
|
||||
}')"
|
||||
|
||||
parse-command:
|
||||
if: >-
|
||||
github.event.issue.pull_request != null
|
||||
&& startsWith(github.event.comment.body, '/test')
|
||||
runs-on: ubuntu-latest
|
||||
outputs:
|
||||
test_type: ${{ steps.parse.outputs.test_type }}
|
||||
test_scope: ${{ steps.parse.outputs.test_scope }}
|
||||
full_suite: ${{ steps.parse.outputs.full_suite }}
|
||||
pr_sha: ${{ steps.pr.outputs.sha }}
|
||||
pr_branch: ${{ steps.pr.outputs.branch }}
|
||||
has_write: ${{ steps.perm.outputs.has_write }}
|
||||
steps:
|
||||
- name: Check write permission
|
||||
id: perm
|
||||
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7.0.1
|
||||
with:
|
||||
script: |
|
||||
const { data: perm } = await github.rest.repos.getCollaboratorPermissionLevel({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
username: context.payload.comment.user.login,
|
||||
});
|
||||
const hasWrite = ['admin', 'write'].includes(perm.permission);
|
||||
core.setOutput('has_write', String(hasWrite));
|
||||
if (!hasWrite) {
|
||||
core.info(`User ${context.payload.comment.user.login} lacks write permission — ignoring.`);
|
||||
}
|
||||
|
||||
- name: Parse /test command
|
||||
id: parse
|
||||
if: steps.perm.outputs.has_write == 'true'
|
||||
shell: bash
|
||||
env:
|
||||
COMMENT: ${{ github.event.comment.body }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
TEST_NAME=$(echo "$COMMENT" | grep -oP '(?<=/test\s)\S+' | head -1 || true)
|
||||
|
||||
VALID="encoder vae transformer kernel unit ssim training lora-inference lora-training distillation self-forcing vsa vmoba performance api train-framework full fastcheck pre-commit"
|
||||
if [ -z "$TEST_NAME" ] || ! echo "$VALID" | grep -qw "$TEST_NAME"; then
|
||||
echo "Unknown test: '$TEST_NAME'. Valid: $VALID"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
declare -A MAP=(
|
||||
[encoder]=encoder [vae]=vae [transformer]=transformer
|
||||
[kernel]=kernel_tests [unit]=unit_test
|
||||
[ssim]=ssim [training]=training
|
||||
[lora-inference]=inference_lora [lora-training]=training_lora
|
||||
[distillation]=distillation_dmd [self-forcing]=self_forcing
|
||||
[vsa]=training_vsa [vmoba]=inference_vmoba
|
||||
[performance]=performance [api]=api_server
|
||||
[train-framework]=train_framework
|
||||
)
|
||||
|
||||
if [ "$TEST_NAME" = "full" ]; then
|
||||
{
|
||||
echo "test_type=all"
|
||||
echo "test_scope=full"
|
||||
echo "full_suite=true"
|
||||
} >> "$GITHUB_OUTPUT"
|
||||
elif [ "$TEST_NAME" = "fastcheck" ]; then
|
||||
{
|
||||
echo "test_type=fastcheck"
|
||||
echo "test_scope=fastcheck"
|
||||
echo "full_suite=false"
|
||||
} >> "$GITHUB_OUTPUT"
|
||||
elif [ "$TEST_NAME" = "pre-commit" ]; then
|
||||
{
|
||||
echo "test_type="
|
||||
echo "test_scope=precommit"
|
||||
echo "full_suite=false"
|
||||
} >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
{
|
||||
echo "test_type=${MAP[$TEST_NAME]}"
|
||||
echo "test_scope=direct"
|
||||
echo "full_suite=false"
|
||||
} >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
|
||||
- name: Get PR details
|
||||
id: pr
|
||||
if: steps.perm.outputs.has_write == 'true'
|
||||
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7.0.1
|
||||
with:
|
||||
script: |
|
||||
const { data: pr } = await github.rest.pulls.get({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
pull_number: context.payload.issue.number,
|
||||
});
|
||||
core.setOutput('sha', pr.head.sha);
|
||||
core.setOutput('branch', pr.head.ref);
|
||||
|
||||
- name: React to comment
|
||||
if: steps.perm.outputs.has_write == 'true'
|
||||
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7.0.1
|
||||
with:
|
||||
script: |
|
||||
await github.rest.reactions.createForIssueComment({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
comment_id: context.payload.comment.id,
|
||||
content: 'rocket',
|
||||
});
|
||||
|
||||
pre-commit:
|
||||
needs: parse-command
|
||||
if: >-
|
||||
needs.parse-command.outputs.has_write == 'true'
|
||||
&& needs.parse-command.outputs.test_scope == 'precommit'
|
||||
uses: ./.github/workflows/ci-precommit.yml
|
||||
with:
|
||||
ref: refs/pull/${{ github.event.issue.number }}/merge
|
||||
|
||||
post-precommit-status:
|
||||
needs: [parse-command, pre-commit]
|
||||
if: always() && needs.parse-command.outputs.test_scope == 'precommit'
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7.0.1
|
||||
env:
|
||||
PR_SHA: ${{ needs.parse-command.outputs.pr_sha }}
|
||||
RESULT: ${{ needs.pre-commit.result }}
|
||||
with:
|
||||
script: |
|
||||
const state = process.env.RESULT === 'success' ? 'success' : 'failure';
|
||||
await github.rest.repos.createCommitStatus({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
sha: process.env.PR_SHA,
|
||||
state,
|
||||
context: 'pre-commit',
|
||||
description: `Triggered via /test pre-commit (${state})`,
|
||||
});
|
||||
|
||||
trigger-buildkite:
|
||||
needs: parse-command
|
||||
if: >-
|
||||
needs.parse-command.outputs.has_write == 'true'
|
||||
&& needs.parse-command.outputs.test_type != ''
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Trigger Buildkite
|
||||
env:
|
||||
BUILDKITE_API_TOKEN: ${{ secrets.BUILDKITE_API_TOKEN }}
|
||||
PR_SHA: ${{ needs.parse-command.outputs.pr_sha }}
|
||||
PR_BRANCH: ${{ needs.parse-command.outputs.pr_branch }}
|
||||
PR_NUMBER: ${{ github.event.issue.number }}
|
||||
TEST_SCOPE: ${{ needs.parse-command.outputs.test_scope }}
|
||||
FULL_SUITE: ${{ needs.parse-command.outputs.full_suite }}
|
||||
TEST_TYPE: ${{ needs.parse-command.outputs.test_type }}
|
||||
BK_ORG: ${{ vars.BUILDKITE_ORG_SLUG }}
|
||||
BK_PIPELINE: ${{ vars.BUILDKITE_PIPELINE_SLUG }}
|
||||
run: |
|
||||
curl -sS --fail-with-body -X POST \
|
||||
"https://api.buildkite.com/v2/organizations/${BK_ORG}/pipelines/${BK_PIPELINE}/builds" \
|
||||
-H "Authorization: Bearer $BUILDKITE_API_TOKEN" \
|
||||
-H "Content-Type: application/json" \
|
||||
--data-raw "$(jq -n \
|
||||
--arg commit "$PR_SHA" \
|
||||
--arg branch "$PR_BRANCH" \
|
||||
--arg message "/test ${TEST_TYPE} on PR #${PR_NUMBER}" \
|
||||
--argjson pr_id "$PR_NUMBER" \
|
||||
--arg test_scope "$TEST_SCOPE" \
|
||||
--arg full_suite "$FULL_SUITE" \
|
||||
--arg test_type "$TEST_TYPE" \
|
||||
--arg pr_number "$PR_NUMBER" \
|
||||
'{
|
||||
commit: $commit,
|
||||
branch: $branch,
|
||||
message: $message,
|
||||
ignore_pipeline_branch_filters: true,
|
||||
pull_request_id: $pr_id,
|
||||
pull_request_base_branch: "main",
|
||||
env: {
|
||||
TEST_SCOPE: $test_scope,
|
||||
FULL_SUITE: $full_suite,
|
||||
TEST_TYPE: $test_type,
|
||||
PR_NUMBER: $pr_number
|
||||
}
|
||||
}')"
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user