[Fix] Fix Nemotron-H Mamba illegal memory access under DP attention with CUDA graph (#34561)
Co-authored-by: Brayden Zhong <b8zhong@uwaterloo.ca>
This commit is contained in:
co-authored by
Brayden Zhong
parent
ed12d6827d
commit
a6bc0532c9
@@ -563,7 +563,16 @@ class NemotronHMambaDecoderLayer(NemotronHAttnLikeDecoderLayer):
|
|||||||
if get_real_num_tokens(hidden_states, forward_batch) == 0:
|
if get_real_num_tokens(hidden_states, forward_batch) == 0:
|
||||||
return torch.zeros_like(hidden_states), residual
|
return torch.zeros_like(hidden_states), residual
|
||||||
|
|
||||||
output = self._forward_mamba(hidden_states, forward_batch)
|
if is_in_breakable_cuda_graph():
|
||||||
|
output = torch.empty_like(hidden_states)
|
||||||
|
breakable_nemotron_mamba2_with_output(
|
||||||
|
hidden_states, output, self.layer_id, False
|
||||||
|
)
|
||||||
|
elif is_in_tc_piecewise_cuda_graph():
|
||||||
|
output = torch.empty_like(hidden_states)
|
||||||
|
nemotron_mamba2_with_output(hidden_states, output, self.layer_id, False)
|
||||||
|
else:
|
||||||
|
output = self._forward_mamba(hidden_states, forward_batch)
|
||||||
return output, residual
|
return output, residual
|
||||||
|
|
||||||
hidden_states, residual = input_norm_maybe_fuse_allreduce(
|
hidden_states, residual = input_norm_maybe_fuse_allreduce(
|
||||||
|
|||||||
@@ -52,8 +52,6 @@ DP_ATTENTION_EP_ARGS = [
|
|||||||
"0.93",
|
"0.93",
|
||||||
"--max-prefill-tokens",
|
"--max-prefill-tokens",
|
||||||
"8192",
|
"8192",
|
||||||
"--cuda-graph-backend-prefill",
|
|
||||||
"disabled",
|
|
||||||
]
|
]
|
||||||
|
|
||||||
MTP_ARGS = [
|
MTP_ARGS = [
|
||||||
|
|||||||
Reference in New Issue
Block a user