Compare commits

...
Author SHA1 Message Date
Peiyuan Zhang a25930febb syn with main 2025-02-25 21:51:00 +00:00
Peiyuan Zhang 5ebcb25e1a Merge branch 'main' of github.com:hao-ai-lab/FastVideo into infer_sta_tea 2025-02-21 20:51:31 +00:00
Peiyuan Zhang f683bbf3fa add STA multiple gpus support 2025-02-21 20:51:24 +00:00
Peiyuan Zhang 0fb05601d3 Merge branch 'infer_sta_tea' of github.com:hao-ai-lab/FastVideo into infer_sta_tea 2025-02-20 21:28:28 +00:00
peiyuan zhang c5f40ef1eb Merge branch 'main' of https://github.com/hao-ai-lab/FastVideo into infer_sta_tea 2025-02-20 21:13:45 +00:00
Peiyuan Zhang f994b3ad93 syn 2025-02-20 20:26:08 +00:00
Peiyuan Zhang fefe22c53b syn with main 2025-02-20 19:56:22 +00:00
Peiyuan Zhang 091d58b418 update README 2025-02-20 19:54:24 +00:00
Peiyuan Zhang 5183349eae syn with main 2025-02-20 19:51:31 +00:00
Peiyuan Zhang 7754387cd8 syn 2025-02-20 19:48:26 +00:00
Peiyuan Zhang 79c3f050a7 pass test 2025-02-19 22:13:37 +00:00
Peiyuan Zhang a8a711b136 syn with main 2025-02-19 22:03:24 +00:00
Peiyuan Zhang d1bf08309d fix ori hunyuan issue 2025-02-19 21:57:03 +00:00
Peiyuan Zhang 7e996b8383 fix ori hunyuan inference bug 2025-02-19 21:53:49 +00:00
Peiyuan Zhang 6b34969257 Merge branch 'main' of github.com:hao-ai-lab/FastVideo into infer_sta_tea 2025-02-19 21:48:27 +00:00
Peiyuan Zhang cdf6ab1254 syn 2025-02-18 23:39:50 +00:00
Peiyuan Zhang 8bfe08b826 update README 2025-02-18 19:01:51 +00:00
Peiyuan Zhang 3fe810464a add readme 2025-02-18 18:53:54 +00:00
Peiyuan Zhang 5a55a6d10c add compile 2025-02-18 18:13:55 +00:00
Peiyuan Zhang f70ee71b58 Merge branch 'main' of github.com:hao-ai-lab/FastVideo into infer_sta_tea 2025-02-17 07:12:49 +00:00
peiyuan zhang 29cac0d5c8 update 2025-02-16 00:13:25 +00:00
peiyuan zhang 6fe702e4b7 update 2025-02-16 00:11:59 +00:00
peiyuan zhang 2afbcabd21 update 2025-02-16 00:09:38 +00:00
peiyuan zhang 3084207579 Merge branch 'infer_sta_tea' of https://github.com/hao-ai-lab/FastVideo into infer_sta_tea 2025-02-15 23:58:25 +00:00
peiyuan zhang ad8cd4cde5 pass test 2025-02-15 23:56:21 +00:00
Peiyuan Zhang 34916a6adc Merge branch 'infer_sta_tea' of github.com:hao-ai-lab/FastVideo into infer_sta_tea 2025-02-15 23:52:56 +00:00
Peiyuan Zhang 78a50149b6 remove unuse 2025-02-15 23:52:34 +00:00
peiyuan zhang c76b8614fb Merge branch 'main' of https://github.com/hao-ai-lab/FastVideo into infer_sta_tea 2025-02-15 23:46:00 +00:00
Peiyuan Zhang d0787398bc syn 2025-02-15 09:10:47 +00:00
Peiyuan Zhang 9b9b05fb1a STA done 2025-02-15 09:06:32 +00:00
Peiyuan Zhang 90c1a6710c fix sp issue 2025-02-15 08:30:20 +00:00
Peiyuan Zhang a468e82fd6 add sta teacache with ori sp issue 2025-02-15 08:07:29 +00:00
+4 -4
View File
@@ -83,16 +83,16 @@ def parallel_attention(q, k, v, img_q_len, img_kv_len, text_mask, mask_strategy=
encoder_sequence_length = encoder_query.size(1)
if mask_strategy[0] is not None:
query = torch.cat([tile(query, nccl_info.sp_size), encoder_query], dim=1).transpose(1, 2)
key = torch.cat([tile(key, nccl_info.sp_size), encoder_key], dim=1).transpose(1, 2)
value = torch.cat([tile(value, nccl_info.sp_size), encoder_value], dim=1).transpose(1, 2)
query = torch.cat([tile(query, nccl_info.sp_size), encoder_query], dim=1).transpose(1, 2).contiguous()
key = torch.cat([tile(key, nccl_info.sp_size), encoder_key], dim=1).transpose(1, 2).contiguous()
value = torch.cat([tile(value, nccl_info.sp_size), encoder_value], dim=1).transpose(1, 2).contiguous()
head_num = query.size(1)
current_rank = nccl_info.rank_within_group
start_head = current_rank * head_num
windows = [mask_strategy[head_idx + start_head] for head_idx in range(head_num)]
hidden_states = sliding_tile_attention(query, key, value, windows, text_length).transpose(1, 2)
hidden_states = sliding_tile_attention(query, key, value, windows, text_length).transpose(1, 2).contiguous()
else:
query = torch.cat([query, encoder_query], dim=1)
key = torch.cat([key, encoder_key], dim=1)