Reenable breakable CUDA graph for NemotronH (#34538)

This commit is contained in:
elvischenv
2026-08-13 13:58:10 +08:00
committed by GitHub
parent ef7208d41d
commit 035c622a14
3 changed files with 3 additions and 16 deletions
@@ -133,14 +133,6 @@ def is_deepseek_v4(config) -> bool:
)
def is_nemotron_h(config) -> bool:
return _hf_arch(config) in (
"NemotronHForCausalLM",
"NemotronHPuzzleForCausalLM",
"NemotronHForCausalLMMTP",
)
def get_dsa_index_head_dim(config: PretrainedConfig) -> int:
assert is_deepseek_dsa(config) or is_deepseek_v4(config)
return config.index_head_dim
+1 -8
View File
@@ -4568,17 +4568,10 @@ class ServerArgs:
memory-saver rejection in its own __init__; config-time rules can be
added here as they're discovered.
"""
from sglang.srt.configs.model_config import is_deepseek_v4, is_nemotron_h
from sglang.srt.configs.model_config import is_deepseek_v4
from sglang.srt.layers.cp.bcg import supports_prefill_cp_bcg
rules = [
# NemotronH's hybrid Mamba2 prefill is not BCG-safe: the mamba
# state-track write is not wired into the captured buffers, so a
# replay can commit a cache slot it never wrote.
(
"NemotronH (hybrid Mamba2 prefill)",
lambda: is_nemotron_h(self.get_model_config().hf_config),
),
# DSV4 is BCG-compatible but introduces heavy memory pressure: the
# c4 indexer scratch is pinned in the capture pool and OOMs. Disable.
(