Sync backend docs with #29063 (#29233)

This commit is contained in:
Mohammad Miadh Angkad
2026-06-24 19:48:34 -07:00
committed by GitHub
parent 8314247d9d
commit 40439acd0a
2 changed files with 9 additions and 8 deletions
@@ -204,10 +204,11 @@ if is_flashinfer_available():
# trtllm | Yes | Yes | Yes | Yes | No |
# mnnvl | Yes | Yes | Single-node | Yes | Blackwell |
#
# FlashInfer allreduce fusion requires SM90 or SM10X. auto resolves to trtllm
# on single-node systems and to mnnvl on Blackwell multi-node systems.
# Non-Blackwell multi-node allreduce fusion is rejected. Explicit mnnvl remains
# available on SM90 single-node systems.
# FlashInfer allreduce fusion requires SM90 or SM10X. auto resolves to mnnvl
# on Blackwell (SM100/SM103) systems (single- and multi-node) and to trtllm on
# SM90 single-node systems. SM90 multi-node and non-SM90/SM10X configurations
# are rejected. Either mnnvl or trtllm can be requested explicitly on
# single-node systems, and mnnvl additionally on Blackwell multi-node.
def is_flashinfer_allreduce_unavailable() -> bool:
+4 -4
View File
@@ -2122,8 +2122,8 @@ class ServerArgs:
"Enable FlashInfer allreduce fusion and choose backend. "
"Requires SM90 or SM10X NVIDIA GPUs. "
"Defaults to auto. "
"'auto': choose trtllm on single-node systems and mnnvl on "
"SM100/SM103 multi-node systems. "
"'auto': choose mnnvl on Blackwell (SM100/SM103) systems "
"(single- and multi-node) and trtllm on SM90 single-node systems. "
"'trtllm': available on single-node systems only. "
"'mnnvl': available on SM90 single-node systems and SM100/SM103 "
"single-node or multi-node systems via MNNVL fabric. "
@@ -4397,8 +4397,8 @@ class ServerArgs:
# Auto-enable FlashInfer AllReduce Fusion on SM90/SM100, for models with
# explicit support (DeepseekV3, GptOss, Glm4Moe, MistralLarge3,
# Qwen3/Qwen3-VL/Qwen3Next/Qwen3.5 MoE families). auto resolves to trtllm on
# single-node systems and mnnvl on Blackwell multi-node systems.
# Qwen3/Qwen3-VL/Qwen3Next/Qwen3.5 MoE families). auto resolves to mnnvl on
# Blackwell (single- and multi-node) and trtllm on SM90 single-node systems.
if (
self.flashinfer_allreduce_fusion_backend is None
and model_arch