[BCG][6/N] Allow prefill breakable CUDA graph for the Kimi archs (#34245)

Co-authored-by: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
Yuwei An
2026-08-16 15:57:14 -07:00
committed by GitHub
co-authored by Claude Opus 5
parent b7eccd642f
commit a508d60295
3 changed files with 13 additions and 5 deletions
@@ -1905,11 +1905,17 @@ multimodal_piecewise_cuda_graph_supported_model_archs = [
# Multimodal archs whose LM prefill is validated under breakable CUDA graph;
# embed-carrying batches are rejected at replay (can_run_graph) and run eager.
# The Kimi archs are structurally multimodal -- their configs always carry a
# vision_config, so is_multimodal is True even for text-only serving -- and the
# generic multimodal rule disabled prefill CG for them despite the LM prefill
# capturing cleanly.
multimodal_breakable_cuda_graph_supported_model_archs = [
"InternS2MobiusForConditionalGeneration",
"Qwen3_5ForConditionalGeneration",
"Qwen3_5MoeForConditionalGeneration",
"MuseGlimmerForConditionalGeneration",
"KimiK3ForConditionalGeneration",
"KimiK25ForConditionalGeneration",
]
if external_mm_model_arch := envs.SGLANG_EXTERNAL_MM_MODEL_ARCH.get():
+5 -4
View File
@@ -4520,13 +4520,14 @@ class ServerArgs:
if (Phase.PREFILL, "backend") in self._cuda_graph_config_locked:
return
# Breakable is the general CUDA default, but it is not compatible with
# multimodal prefill. Models on this allowlist have had their decoder
# prefill validated under tc_piecewise; the vision encoder remains
# eager outside the captured LM forward.
# Breakable is the CUDA default but not multimodal-compatible;
# piecewise-allowlisted archs run their validated decoder prefill
# there instead. Archs also on the breakable allowlist keep it --
# this runs first, so piecewise would otherwise silently win.
if (
self.cuda_graph_config.prefill.backend == Backend.BREAKABLE
and self.get_model_config().is_multimodal_piecewise_cuda_graph_supported
and not self.get_model_config().is_multimodal_breakable_cuda_graph_supported
# Keep trtllm_mla on the preferred breakable path, which now serves
# MLA by falling back to the flashinfer MLA impl for extend.
and self._resolved_attention_backends()[0] != "trtllm_mla"