[AMD] [Fix] Fix --attention-backend triton work for DeepSeek MLA on MI355 (null-K + decode dispatch + RoPE) (#30355)

This commit is contained in:
jacky.cheng
2026-07-15 14:19:23 -07:00
committed by GitHub
parent e76cc75cfa
commit e78051a419
3 changed files with 80 additions and 8 deletions
@@ -31,7 +31,11 @@ class AttentionBackendRegistry:
def _dispatch_mla_subtype(attn, forward_batch):
if _is_hip:
if attn.rocm_fused_decode_mla and forward_batch.forward_mode.is_decode():
if (
attn.rocm_fused_decode_mla
and forward_batch.forward_mode.is_decode()
and attn.current_attention_backend == "aiter"
):
return AttnForwardMethod.MLA_FUSED_ROPE_ROCM
else:
return AttnForwardMethod.MLA
@@ -552,7 +552,12 @@ class DeepseekMLAForwardMixin:
and (not fuse_rope_for_trtllm_mla)
and (not skip_rope_for_dsa_tilelang_fused)
and (not skip_rope_for_aiter_fused_mla)
and (not _use_aiter or not _is_gfx95_supported or self.use_dsa)
and (
not _use_aiter
or not _is_gfx95_supported
or self.use_dsa
or self.current_attention_backend == "triton"
)
):
q_pe, k_pe = self.rotary_emb(positions, q_pe, k_pe)
@@ -758,7 +763,7 @@ class DeepseekMLAForwardMixin:
),
)
else:
if _use_aiter_gfx95:
if _use_aiter_gfx95 and self.current_attention_backend == "aiter":
cos = self.rotary_emb.cos_cache
sin = self.rotary_emb.sin_cache
@@ -1038,11 +1043,7 @@ class DeepseekMLAForwardMixin:
when running aiter-backend MLA on gfx95 (i.e., the `else` branch in forward_absorb_core
that calls fused_qk_rope_cat_and_cache_mla).
"""
return (
_use_aiter_gfx95
and self.current_attention_backend
not in FORWARD_ABSORB_CORE_ATTENTION_BACKENDS
)
return _use_aiter_gfx95 and self.current_attention_backend == "aiter"
# Fuses the absorb BMM (`q_nope @ w_kc`) with `unified_attention_with_output`