This commit is contained in:
xiaoxiaohehe001
2025-09-21 22:04:59 +08:00
committed by GitHub
parent 5223065d59
commit 9f1882d9a8
4 changed files with 17 additions and 7 deletions

View File

@@ -283,6 +283,7 @@ class FlashAttentionBackend(AttentionBackend):
metadata.kv_token_num_cpu[0].item(),
self.max_seq_len,
getattr(layer, "cache_quant_type_str", "none"),
self.rope_3d,
)
res = self.flash_attn_func(