Reduce startup log noise and fix Dynamo / CUDA-graph edge cases (#33428)
This commit is contained in:
@@ -169,6 +169,12 @@ def _jit_build_dir_name(module_name: str) -> str:
|
|||||||
return f"{module_name}__arch_{arch}__tvmffi_{_tvm_ffi_version()}"
|
return f"{module_name}__arch_{arch}__tvmffi_{_tvm_ffi_version()}"
|
||||||
|
|
||||||
|
|
||||||
|
# JIT compilation is pure Python/filesystem plumbing (path `.resolve()` calls
|
||||||
|
# `os.lstat`, etc.) that Dynamo cannot trace. When a lazily-loaded kernel is
|
||||||
|
# first reached from inside a `@torch.compile`d region, tracing into it produces
|
||||||
|
# spurious "Dynamo does not know how to trace the builtin `posix.lstat`" graph
|
||||||
|
# breaks. The load happens once and is memoized, so keep it out of the graph.
|
||||||
|
@torch.compiler.disable
|
||||||
def load_jit(
|
def load_jit(
|
||||||
*args: str,
|
*args: str,
|
||||||
cpp_files: List[str] | None = None,
|
cpp_files: List[str] | None = None,
|
||||||
|
|||||||
@@ -2235,6 +2235,12 @@ def _a2a_backend_overrides(view: Any) -> dict:
|
|||||||
@register_post_process
|
@register_post_process
|
||||||
def _a2a_ep_size(view: Any) -> dict:
|
def _a2a_ep_size(view: Any) -> dict:
|
||||||
if view.moe_a2a_backend in _A2A_EP_SPANNING_BACKENDS:
|
if view.moe_a2a_backend in _A2A_EP_SPANNING_BACKENDS:
|
||||||
|
if view.ep_size != view.tp_size:
|
||||||
|
logger.info(
|
||||||
|
f"{view.moe_a2a_backend} MoE is enabled. The expert parallel size "
|
||||||
|
f"is adjusted from {view.ep_size} to the tensor parallel size "
|
||||||
|
f"[{view.tp_size}]."
|
||||||
|
)
|
||||||
return {"ep_size": view.tp_size}
|
return {"ep_size": view.tp_size}
|
||||||
return {}
|
return {}
|
||||||
|
|
||||||
|
|||||||
@@ -1503,20 +1503,6 @@ class ModelConfig:
|
|||||||
f"({self.quantization})."
|
f"({self.quantization})."
|
||||||
)
|
)
|
||||||
|
|
||||||
# Warn if DeepGemm is enabled for a non-ue8m0 checkpoint on Blackwell.
|
|
||||||
# MXFP8 stores E8M0 block scales that DeepGemm consumes losslessly, so skip the warning there.
|
|
||||||
self.use_scale_ue8m0 = quant_cfg.get("scale_fmt", None) == "ue8m0"
|
|
||||||
from sglang.srt.layers import deep_gemm_wrapper
|
|
||||||
|
|
||||||
if (
|
|
||||||
not self.use_scale_ue8m0
|
|
||||||
and deep_gemm_wrapper.DEEPGEMM_SCALE_UE8M0
|
|
||||||
and self.quantization != "mxfp8"
|
|
||||||
):
|
|
||||||
logger.warning(
|
|
||||||
"DeepGemm is enabled but the scale_fmt of checkpoint is not ue8m0. This might cause accuracy degradation on Blackwell."
|
|
||||||
)
|
|
||||||
|
|
||||||
if self.quantization is not None:
|
if self.quantization is not None:
|
||||||
if self.quantization not in supported_quantization:
|
if self.quantization not in supported_quantization:
|
||||||
raise ValueError(
|
raise ValueError(
|
||||||
|
|||||||
@@ -93,6 +93,11 @@ _is_cpu = is_cpu()
|
|||||||
_is_npu = is_npu()
|
_is_npu = is_npu()
|
||||||
_use_aiter = get_bool_env_var("SGLANG_USE_AITER") and _is_hip
|
_use_aiter = get_bool_env_var("SGLANG_USE_AITER") and _is_hip
|
||||||
|
|
||||||
|
# Log the deferred-finalize config at most once per process (rank). Different MoE
|
||||||
|
# layers can resolve to different quant methods, so print_info_once (keyed on the
|
||||||
|
# full message) would otherwise fire once per distinct quant method.
|
||||||
|
_deferred_finalize_info_logged = False
|
||||||
|
|
||||||
|
|
||||||
def _copy_weight_view_before_h2d(loaded_weight: torch.Tensor) -> torch.Tensor:
|
def _copy_weight_view_before_h2d(loaded_weight: torch.Tensor) -> torch.Tensor:
|
||||||
"""Copy a CPU tensor view into independent contiguous storage."""
|
"""Copy a CPU tensor view into independent contiguous storage."""
|
||||||
@@ -379,12 +384,15 @@ class FusedMoE(torch.nn.Module):
|
|||||||
and get_moe_runner_backend().is_flashinfer_trtllm()
|
and get_moe_runner_backend().is_flashinfer_trtllm()
|
||||||
and isinstance(self.quant_method, ModelOptNvFp4FusedMoEMethod)
|
and isinstance(self.quant_method, ModelOptNvFp4FusedMoEMethod)
|
||||||
)
|
)
|
||||||
print_info_once(
|
global _deferred_finalize_info_logged
|
||||||
"FlashInfer TRTLLM MoE deferred finalize is "
|
if not _deferred_finalize_info_logged:
|
||||||
f"{'enabled' if self.supports_deferred_finalize else 'disabled'} "
|
_deferred_finalize_info_logged = True
|
||||||
f"(moe_runner_backend={get_exec().moe.moe_runner_backend}, "
|
logging.getLogger(__name__).info(
|
||||||
f"quant_method={type(self.quant_method).__name__})."
|
"FlashInfer TRTLLM MoE deferred finalize is "
|
||||||
)
|
f"{'enabled' if self.supports_deferred_finalize else 'disabled'} "
|
||||||
|
f"(moe_runner_backend={get_exec().moe.moe_runner_backend}, "
|
||||||
|
f"quant_method={type(self.quant_method).__name__})."
|
||||||
|
)
|
||||||
|
|
||||||
self.quant_method.create_weights(
|
self.quant_method.create_weights(
|
||||||
layer=self,
|
layer=self,
|
||||||
|
|||||||
@@ -119,7 +119,7 @@ class BreakableCudaGraphBackend(DedupedCudaGraphMixin, BaseCudaGraphBackend):
|
|||||||
if post_warmup_hook is not None:
|
if post_warmup_hook is not None:
|
||||||
post_warmup_hook()
|
post_warmup_hook()
|
||||||
|
|
||||||
graph = BreakableCUDAGraph()
|
graph = BreakableCUDAGraph(self.deduped_cuda_graph)
|
||||||
captured_fn = (
|
captured_fn = (
|
||||||
eager_on_graph(True)(forward_fn) if self._debug_eager else forward_fn
|
eager_on_graph(True)(forward_fn) if self._debug_eager else forward_fn
|
||||||
)
|
)
|
||||||
|
|||||||
@@ -4785,9 +4785,15 @@ class ServerArgs:
|
|||||||
)
|
)
|
||||||
|
|
||||||
# Multimodal models need more memory for the image processing,
|
# Multimodal models need more memory for the image processing,
|
||||||
# so we adjust the mem_fraction_static accordingly.
|
# so we adjust the mem_fraction_static accordingly. The VLM encoder
|
||||||
|
# only runs on the prefill stage, so PD decode engines do not need
|
||||||
|
# this headroom; prefill engines and normal (non-PD) engines do.
|
||||||
model_config = self.get_model_config()
|
model_config = self.get_model_config()
|
||||||
if model_config.is_multimodal and not self.language_only:
|
if (
|
||||||
|
model_config.is_multimodal
|
||||||
|
and not self.language_only
|
||||||
|
and self.disaggregation_mode != "decode"
|
||||||
|
):
|
||||||
self.adjust_mem_fraction_for_vlm(model_config)
|
self.adjust_mem_fraction_for_vlm(model_config)
|
||||||
|
|
||||||
# If symm mem is enabled and prealloc size is not set, set it to 4GB
|
# If symm mem is enabled and prealloc size is not set, set it to 4GB
|
||||||
@@ -6611,10 +6617,6 @@ class ServerArgs:
|
|||||||
if a2a_backend == "megamoe":
|
if a2a_backend == "megamoe":
|
||||||
if not envs.SGLANG_OPT_FIX_MEGA_MOE_MEMORY.is_set():
|
if not envs.SGLANG_OPT_FIX_MEGA_MOE_MEMORY.is_set():
|
||||||
envs.SGLANG_OPT_FIX_MEGA_MOE_MEMORY.set(True)
|
envs.SGLANG_OPT_FIX_MEGA_MOE_MEMORY.set(True)
|
||||||
logger.info(
|
|
||||||
f"Mega MoE is enabled. The expert parallel size is adjusted "
|
|
||||||
f"to be the same as the tensor parallel size[{self.tp_size}]."
|
|
||||||
)
|
|
||||||
|
|
||||||
if a2a_backend == "deepep":
|
if a2a_backend == "deepep":
|
||||||
if self.moe_runner_backend == "flashinfer_cutedsl":
|
if self.moe_runner_backend == "flashinfer_cutedsl":
|
||||||
@@ -6637,19 +6639,6 @@ class ServerArgs:
|
|||||||
logger.warning("Cuda graph is disabled because deepep_mode=`normal`")
|
logger.warning("Cuda graph is disabled because deepep_mode=`normal`")
|
||||||
self.cuda_graph_config.decode.backend = Backend.DISABLED
|
self.cuda_graph_config.decode.backend = Backend.DISABLED
|
||||||
self.cuda_graph_config.prefill.backend = Backend.DISABLED
|
self.cuda_graph_config.prefill.backend = Backend.DISABLED
|
||||||
logger.warning(
|
|
||||||
f"DeepEP MoE is enabled. The expert parallel size is adjusted to be the same as the tensor parallel size[{self.tp_size}]."
|
|
||||||
)
|
|
||||||
|
|
||||||
if a2a_backend == "mooncake":
|
|
||||||
logger.warning(
|
|
||||||
f"Mooncake MoE is enabled. The expert parallel size is adjusted to be the same as the tensor parallel size[{self.tp_size}]."
|
|
||||||
)
|
|
||||||
|
|
||||||
if a2a_backend == "nixl":
|
|
||||||
logger.warning(
|
|
||||||
f"Nixl MoE is enabled. The expert parallel size is adjusted to be the same as the tensor parallel size[{self.tp_size}]."
|
|
||||||
)
|
|
||||||
|
|
||||||
if (
|
if (
|
||||||
self.moe_a2a_backend == "none" and is_npu()
|
self.moe_a2a_backend == "none" and is_npu()
|
||||||
@@ -6657,17 +6646,10 @@ class ServerArgs:
|
|||||||
# FIXME (OrangeRedeng): for some reasons if pass "ascend_tp" accuracy drops to zero
|
# FIXME (OrangeRedeng): for some reasons if pass "ascend_tp" accuracy drops to zero
|
||||||
self.moe_a2a_backend = "none"
|
self.moe_a2a_backend = "none"
|
||||||
|
|
||||||
if self.moe_a2a_backend == "ascend_fuseep":
|
|
||||||
logger.warning(
|
|
||||||
f"Ascend fused EP MoE is enabled. The expert parallel size is adjusted to be the same as the tensor parallel size[{self.tp_size}]."
|
|
||||||
)
|
|
||||||
if self.moe_a2a_backend == "flashinfer":
|
if self.moe_a2a_backend == "flashinfer":
|
||||||
assert (
|
assert (
|
||||||
resolved_view(self).enable_dp_attention and self.dp_size == self.tp_size
|
resolved_view(self).enable_dp_attention and self.dp_size == self.tp_size
|
||||||
), "Flashinfer MoE A2A is only supported with dp_size == tp_size and --enable-dp-attention"
|
), "Flashinfer MoE A2A is only supported with dp_size == tp_size and --enable-dp-attention"
|
||||||
logger.warning(
|
|
||||||
f"Flashinfer MoE A2A is enabled. The expert parallel size is adjusted to be the same as the tensor parallel size[{self.tp_size}]."
|
|
||||||
)
|
|
||||||
if self.deepep_mode != "auto":
|
if self.deepep_mode != "auto":
|
||||||
logger.warning("--deepep-mode is ignored for Flashinfer MoE A2A")
|
logger.warning("--deepep-mode is ignored for Flashinfer MoE A2A")
|
||||||
if not envs.SGLANG_MOE_NVFP4_DISPATCH.is_set() and (
|
if not envs.SGLANG_MOE_NVFP4_DISPATCH.is_set() and (
|
||||||
@@ -6688,9 +6670,6 @@ class ServerArgs:
|
|||||||
if self.deepep_mode == "auto":
|
if self.deepep_mode == "auto":
|
||||||
self.deepep_mode = "normal"
|
self.deepep_mode = "normal"
|
||||||
logger.warning("auto set deepep_mode=`normal` for MORI EP")
|
logger.warning("auto set deepep_mode=`normal` for MORI EP")
|
||||||
logger.warning(
|
|
||||||
f"MoRI MoE is enabled. The expert parallel size is adjusted to be the same as the tensor parallel size[{self.tp_size}]."
|
|
||||||
)
|
|
||||||
|
|
||||||
# Check chunked prefill for mori
|
# Check chunked prefill for mori
|
||||||
# Skip validation if chunked prefill is disabled (i.e., size <= 0).
|
# Skip validation if chunked prefill is disabled (i.e., size <= 0).
|
||||||
@@ -6731,9 +6710,6 @@ class ServerArgs:
|
|||||||
if self.moe_runner_backend == "auto":
|
if self.moe_runner_backend == "auto":
|
||||||
self.moe_runner_backend = "deep_gemm"
|
self.moe_runner_backend = "deep_gemm"
|
||||||
logger.warning("auto set moe_runner_backend=`deep_gemm` for PPLX EP")
|
logger.warning("auto set moe_runner_backend=`deep_gemm` for PPLX EP")
|
||||||
logger.warning(
|
|
||||||
f"PPLX MoE is enabled. The expert parallel size is adjusted to be the same as the tensor parallel size[{self.tp_size}]."
|
|
||||||
)
|
|
||||||
|
|
||||||
# Check per-rank dispatch tokens for pplx
|
# Check per-rank dispatch tokens for pplx
|
||||||
# Skip validation if chunked prefill is disabled (i.e., size <= 0)
|
# Skip validation if chunked prefill is disabled (i.e., size <= 0)
|
||||||
|
|||||||
@@ -489,12 +489,13 @@ def get_generation_config(
|
|||||||
return GenerationConfig.from_pretrained(
|
return GenerationConfig.from_pretrained(
|
||||||
model, trust_remote_code=trust_remote_code, revision=revision, **kwargs
|
model, trust_remote_code=trust_remote_code, revision=revision, **kwargs
|
||||||
)
|
)
|
||||||
except FileNotFoundError:
|
except (FileNotFoundError, OSError) as e:
|
||||||
return None
|
# A missing generation_config.json is normal for many checkpoints and
|
||||||
except OSError as e:
|
# is surfaced by HF as a generic OSError (not FileNotFoundError). Treat
|
||||||
logger.warning(
|
# it as benign — proceed without a generation config, at DEBUG level so
|
||||||
"Failed to load generation config for %s: %s. "
|
# normal startup logs stay quiet.
|
||||||
"Proceeding without generation config.",
|
logger.debug(
|
||||||
|
"No generation config for %s: %s. Proceeding without it.",
|
||||||
model,
|
model,
|
||||||
e,
|
e,
|
||||||
)
|
)
|
||||||
|
|||||||
@@ -2010,7 +2010,9 @@ class TestGoldenModelOverrides(_IsolatedPublish):
|
|||||||
|
|
||||||
self.assertEqual(
|
self.assertEqual(
|
||||||
_a2a_ep_size(
|
_a2a_ep_size(
|
||||||
ResolvedView(SimpleNamespace(moe_a2a_backend="deepep", tp_size=8))
|
ResolvedView(
|
||||||
|
SimpleNamespace(moe_a2a_backend="deepep", ep_size=1, tp_size=8)
|
||||||
|
)
|
||||||
),
|
),
|
||||||
{"ep_size": 8},
|
{"ep_size": 8},
|
||||||
)
|
)
|
||||||
|
|||||||
Reference in New Issue
Block a user