[BUGFIX]Fix Ascend backend pre-allocated range in NPU Graph Mode. (#22778)
This commit is contained in:
@@ -389,16 +389,19 @@ class AscendAttnBackend(AttentionBackend):
|
||||
self.graph_mode = False
|
||||
|
||||
def init_cuda_graph_state(self, max_bs: int, max_num_tokens: int):
|
||||
total_context_len = self.max_context_len + self.page_size - 1
|
||||
if self.speculative_num_draft_tokens is not None:
|
||||
total_context_len += self.speculative_num_draft_tokens
|
||||
self.graph_metadata = {
|
||||
"block_tables": torch.empty(
|
||||
(max_bs, (self.max_context_len + self.page_size - 1) // self.page_size),
|
||||
(max_bs, total_context_len // self.page_size),
|
||||
dtype=torch.int32,
|
||||
device=self.device,
|
||||
),
|
||||
}
|
||||
if self.is_hybrid_swa:
|
||||
self.graph_metadata["block_tables_swa"] = torch.empty(
|
||||
(max_bs, (self.max_context_len + self.page_size - 1) // self.page_size),
|
||||
(max_bs, total_context_len // self.page_size),
|
||||
dtype=torch.int32,
|
||||
device=self.device,
|
||||
)
|
||||
|
||||
Reference in New Issue
Block a user