[Bugfix] Fix Ascend NPU CP attention for batch size > 1 (#26705)
This commit is contained in:
@@ -820,12 +820,15 @@ class AscendAttnBackend(AttentionBackend):
|
|||||||
"""
|
"""
|
||||||
cp_meta = forward_batch.attn_cp_metadata
|
cp_meta = forward_batch.attn_cp_metadata
|
||||||
|
|
||||||
# Split Q into prev/next halves per zigzag pattern.
|
# Local tokens are laid out [all_seqs_prev, all_seqs_next]; split at
|
||||||
# torch.chunk(q, 2) gives ceil(n/2) and floor(n/2), matching
|
# total_q_prev_tokens rather than the midpoint to support bs > 1.
|
||||||
# actual_seq_q_prev and actual_seq_q_next.
|
split = cp_meta.total_q_prev_tokens
|
||||||
q_prev, q_next = torch.chunk(q, 2, dim=0)
|
q_prev = (
|
||||||
q_prev = q_prev.contiguous().reshape(-1, layer.tp_q_head_num, layer.qk_head_dim)
|
q[:split].contiguous().reshape(-1, layer.tp_q_head_num, layer.qk_head_dim)
|
||||||
q_next = q_next.contiguous().reshape(-1, layer.tp_q_head_num, layer.qk_head_dim)
|
)
|
||||||
|
q_next = (
|
||||||
|
q[split:].contiguous().reshape(-1, layer.tp_q_head_num, layer.qk_head_dim)
|
||||||
|
)
|
||||||
|
|
||||||
k_cache_paged = k_cache.view(
|
k_cache_paged = k_cache.view(
|
||||||
-1, self.page_size, layer.tp_k_head_num * layer.qk_head_dim
|
-1, self.page_size, layer.tp_k_head_num * layer.qk_head_dim
|
||||||
@@ -847,8 +850,8 @@ class AscendAttnBackend(AttentionBackend):
|
|||||||
sparse_mode=3,
|
sparse_mode=3,
|
||||||
next_tokens=0,
|
next_tokens=0,
|
||||||
scale=layer.scaling,
|
scale=layer.scaling,
|
||||||
actual_seq_lengths=[cp_meta.actual_seq_q_prev],
|
actual_seq_lengths=np.cumsum(cp_meta.actual_seq_q_prev_list).tolist(),
|
||||||
actual_seq_lengths_kv=[cp_meta.kv_len_prev],
|
actual_seq_lengths_kv=cp_meta.kv_len_prev_list,
|
||||||
)
|
)
|
||||||
|
|
||||||
attn_out_next, _ = torch.ops.npu.npu_fused_infer_attention_score(
|
attn_out_next, _ = torch.ops.npu.npu_fused_infer_attention_score(
|
||||||
@@ -864,8 +867,8 @@ class AscendAttnBackend(AttentionBackend):
|
|||||||
sparse_mode=3,
|
sparse_mode=3,
|
||||||
next_tokens=0,
|
next_tokens=0,
|
||||||
scale=layer.scaling,
|
scale=layer.scaling,
|
||||||
actual_seq_lengths=[cp_meta.actual_seq_q_next],
|
actual_seq_lengths=np.cumsum(cp_meta.actual_seq_q_next_list).tolist(),
|
||||||
actual_seq_lengths_kv=[cp_meta.kv_len_next],
|
actual_seq_lengths_kv=cp_meta.kv_len_next_list,
|
||||||
)
|
)
|
||||||
|
|
||||||
attn_out = torch.cat([attn_out_prev, attn_out_next], dim=0)
|
attn_out = torch.cat([attn_out_prev, attn_out_next], dim=0)
|
||||||
|
|||||||
Reference in New Issue
Block a user