[bugfix] Fixed the logic for processing return values after computing attention in flashAttention3

This commit is contained in:
wangxin68
2025-12-10 17:27:37 +08:00
parent dd58511d4e
commit 38a48c670a
+1 -1
View File
@@ -66,7 +66,7 @@ else:
0, dtype=torch.int32).to(q.device, non_blocking=True),
seqused_q=None, seqused_k=None, max_seqlen_q=lq, max_seqlen_k=lk,
softmax_scale=softmax_scale, causal=causal,
deterministic=deterministic)[0].unflatten(0, (b, lq))
deterministic=deterministic).unflatten(0, (b, lq))
else:
assert FLASH_ATTN_2_AVAILABLE
x = flash_attn.flash_attn_varlen_func(q=q, k=k, v=v,