diff --git a/python/sglang/srt/layers/flashinfer_comm_fusion.py b/python/sglang/srt/layers/flashinfer_comm_fusion.py index a4c854bbd..9392521d4 100644 --- a/python/sglang/srt/layers/flashinfer_comm_fusion.py +++ b/python/sglang/srt/layers/flashinfer_comm_fusion.py @@ -204,10 +204,11 @@ if is_flashinfer_available(): # trtllm | Yes | Yes | Yes | Yes | No | # mnnvl | Yes | Yes | Single-node | Yes | Blackwell | # -# FlashInfer allreduce fusion requires SM90 or SM10X. auto resolves to trtllm -# on single-node systems and to mnnvl on Blackwell multi-node systems. -# Non-Blackwell multi-node allreduce fusion is rejected. Explicit mnnvl remains -# available on SM90 single-node systems. +# FlashInfer allreduce fusion requires SM90 or SM10X. auto resolves to mnnvl +# on Blackwell (SM100/SM103) systems (single- and multi-node) and to trtllm on +# SM90 single-node systems. SM90 multi-node and non-SM90/SM10X configurations +# are rejected. Either mnnvl or trtllm can be requested explicitly on +# single-node systems, and mnnvl additionally on Blackwell multi-node. def is_flashinfer_allreduce_unavailable() -> bool: diff --git a/python/sglang/srt/server_args.py b/python/sglang/srt/server_args.py index caa6967e0..01b21de03 100644 --- a/python/sglang/srt/server_args.py +++ b/python/sglang/srt/server_args.py @@ -2122,8 +2122,8 @@ class ServerArgs: "Enable FlashInfer allreduce fusion and choose backend. " "Requires SM90 or SM10X NVIDIA GPUs. " "Defaults to auto. " - "'auto': choose trtllm on single-node systems and mnnvl on " - "SM100/SM103 multi-node systems. " + "'auto': choose mnnvl on Blackwell (SM100/SM103) systems " + "(single- and multi-node) and trtllm on SM90 single-node systems. " "'trtllm': available on single-node systems only. " "'mnnvl': available on SM90 single-node systems and SM100/SM103 " "single-node or multi-node systems via MNNVL fabric. " @@ -4397,8 +4397,8 @@ class ServerArgs: # Auto-enable FlashInfer AllReduce Fusion on SM90/SM100, for models with # explicit support (DeepseekV3, GptOss, Glm4Moe, MistralLarge3, - # Qwen3/Qwen3-VL/Qwen3Next/Qwen3.5 MoE families). auto resolves to trtllm on - # single-node systems and mnnvl on Blackwell multi-node systems. + # Qwen3/Qwen3-VL/Qwen3Next/Qwen3.5 MoE families). auto resolves to mnnvl on + # Blackwell (single- and multi-node) and trtllm on SM90 single-node systems. if ( self.flashinfer_allreduce_fusion_backend is None and model_arch