fix: copy seq_lens in TRTLLM MHA draft decode cuda graph capture (#26521)
This commit is contained in:
@@ -324,6 +324,9 @@ class TRTLLMHAAttnBackend(FlashInferAttnBackend):
|
|||||||
metadata.cache_seqlens_int32 = self.decode_cuda_graph_metadata[
|
metadata.cache_seqlens_int32 = self.decode_cuda_graph_metadata[
|
||||||
"cache_seqlens"
|
"cache_seqlens"
|
||||||
][:bs]
|
][:bs]
|
||||||
|
metadata.cache_seqlens_int32.copy_(
|
||||||
|
seq_lens + self.speculative_step_id + 1
|
||||||
|
)
|
||||||
metadata.max_seq_len_k = seq_lens.max().item() + (
|
metadata.max_seq_len_k = seq_lens.max().item() + (
|
||||||
self.speculative_step_id + 1
|
self.speculative_step_id + 1
|
||||||
)
|
)
|
||||||
|
|||||||
Reference in New Issue
Block a user