use envs in server_args (#22994)
This commit is contained in:
@@ -315,8 +315,10 @@ class Envs:
|
|||||||
|
|
||||||
# AMD & ROCm
|
# AMD & ROCm
|
||||||
SGLANG_USE_AITER = EnvBool(False)
|
SGLANG_USE_AITER = EnvBool(False)
|
||||||
|
SGLANG_USE_AITER_UNIFIED_ATTN = EnvBool(False)
|
||||||
SGLANG_ROCM_FUSED_DECODE_MLA = EnvBool(False)
|
SGLANG_ROCM_FUSED_DECODE_MLA = EnvBool(False)
|
||||||
SGLANG_ROCM_DISABLE_LINEARQUANT = EnvBool(False)
|
SGLANG_ROCM_DISABLE_LINEARQUANT = EnvBool(False)
|
||||||
|
SGLANG_MORI_NUM_MAX_DISPATCH_TOKENS_PER_RANK = EnvInt(4096)
|
||||||
|
|
||||||
# MPS (Apple Silicon)
|
# MPS (Apple Silicon)
|
||||||
SGLANG_USE_MLX = EnvBool(False)
|
SGLANG_USE_MLX = EnvBool(False)
|
||||||
|
|||||||
@@ -37,12 +37,10 @@ from sglang.srt.utils.common import (
|
|||||||
LORA_TARGET_ALL_MODULES,
|
LORA_TARGET_ALL_MODULES,
|
||||||
SUPPORTED_LORA_TARGET_MODULES,
|
SUPPORTED_LORA_TARGET_MODULES,
|
||||||
cpu_has_amx_support,
|
cpu_has_amx_support,
|
||||||
get_bool_env_var,
|
|
||||||
get_device,
|
get_device,
|
||||||
get_device_memory_capacity,
|
get_device_memory_capacity,
|
||||||
get_device_name,
|
get_device_name,
|
||||||
get_device_sm,
|
get_device_sm,
|
||||||
get_int_env_var,
|
|
||||||
get_nvidia_driver_version,
|
get_nvidia_driver_version,
|
||||||
get_quantization_config,
|
get_quantization_config,
|
||||||
human_readable_int,
|
human_readable_int,
|
||||||
@@ -1025,7 +1023,7 @@ class ServerArgs:
|
|||||||
self.mm_process_config = {}
|
self.mm_process_config = {}
|
||||||
|
|
||||||
# Handle ModelScope model downloads
|
# Handle ModelScope model downloads
|
||||||
if get_bool_env_var("SGLANG_USE_MODELSCOPE"):
|
if envs.SGLANG_USE_MODELSCOPE.get():
|
||||||
self._handle_modelscope_paths()
|
self._handle_modelscope_paths()
|
||||||
|
|
||||||
# Mamba scheduler strategy
|
# Mamba scheduler strategy
|
||||||
@@ -1851,13 +1849,13 @@ class ServerArgs:
|
|||||||
"Detected SM120 and MXFP4 quantization format for GPT-OSS model, enabling triton_kernel MOE kernel."
|
"Detected SM120 and MXFP4 quantization format for GPT-OSS model, enabling triton_kernel MOE kernel."
|
||||||
)
|
)
|
||||||
elif (
|
elif (
|
||||||
is_hip() and get_bool_env_var("SGLANG_USE_AITER")
|
is_hip() and envs.SGLANG_USE_AITER.get()
|
||||||
) and is_mxfp4_quant_format:
|
) and is_mxfp4_quant_format:
|
||||||
self.moe_runner_backend = "auto"
|
self.moe_runner_backend = "auto"
|
||||||
logger.warning(
|
logger.warning(
|
||||||
"Detected ROCm and MXFP4 quantization format for GPT-OSS model, enabling aiter MXFP4 MOE kernel."
|
"Detected ROCm and MXFP4 quantization format for GPT-OSS model, enabling aiter MXFP4 MOE kernel."
|
||||||
)
|
)
|
||||||
elif is_hip() and get_bool_env_var("SGLANG_USE_AITER"):
|
elif is_hip() and envs.SGLANG_USE_AITER.get():
|
||||||
# For GPT-OSS bf16 on ROCm with aiter, use triton backend
|
# For GPT-OSS bf16 on ROCm with aiter, use triton backend
|
||||||
# because aiter CK kernel doesn't support all GEMM dimensions
|
# because aiter CK kernel doesn't support all GEMM dimensions
|
||||||
self.moe_runner_backend = "triton"
|
self.moe_runner_backend = "triton"
|
||||||
@@ -2175,7 +2173,7 @@ class ServerArgs:
|
|||||||
if (
|
if (
|
||||||
model_arch in ["Qwen3VLForConditionalGeneration"]
|
model_arch in ["Qwen3VLForConditionalGeneration"]
|
||||||
and is_hip()
|
and is_hip()
|
||||||
and get_bool_env_var("SGLANG_USE_AITER_UNIFIED_ATTN")
|
and envs.SGLANG_USE_AITER_UNIFIED_ATTN.get()
|
||||||
and self.page_size is None
|
and self.page_size is None
|
||||||
):
|
):
|
||||||
self.page_size = 16
|
self.page_size = 16
|
||||||
@@ -2851,7 +2849,7 @@ class ServerArgs:
|
|||||||
"FlashInfer TRTLLM routed MoE is enabled. --disable-shared-experts-fusion is automatically set."
|
"FlashInfer TRTLLM routed MoE is enabled. --disable-shared-experts-fusion is automatically set."
|
||||||
)
|
)
|
||||||
|
|
||||||
if get_bool_env_var("SGLANG_CUTLASS_MOE"):
|
if envs.SGLANG_CUTLASS_MOE.get():
|
||||||
logger.warning(
|
logger.warning(
|
||||||
"SGLANG_CUTLASS_MOE is deprecated, use --moe-runner-backend=cutlass and/or --speculative-moe-runner-backend=cutlass instead"
|
"SGLANG_CUTLASS_MOE is deprecated, use --moe-runner-backend=cutlass and/or --speculative-moe-runner-backend=cutlass instead"
|
||||||
)
|
)
|
||||||
@@ -2911,12 +2909,12 @@ class ServerArgs:
|
|||||||
logger.warning(
|
logger.warning(
|
||||||
f"Ascend fused EP MoE is enabled. The expert parallel size is adjusted to be the same as the tensor parallel size[{self.tp_size}]."
|
f"Ascend fused EP MoE is enabled. The expert parallel size is adjusted to be the same as the tensor parallel size[{self.tp_size}]."
|
||||||
)
|
)
|
||||||
fuse_mode = os.environ.get("SGLANG_NPU_FUSED_MOE_MODE", None)
|
fuse_mode = envs.SGLANG_NPU_FUSED_MOE_MODE.get()
|
||||||
if fuse_mode not in ["1", "2"]:
|
if fuse_mode not in [1, 2]:
|
||||||
raise ValueError(
|
raise ValueError(
|
||||||
f"Wrong value of {fuse_mode=}, the NPU only support 1 or 2."
|
f"Wrong value of {fuse_mode=}, the NPU only support 1 or 2."
|
||||||
)
|
)
|
||||||
elif fuse_mode == "2":
|
elif fuse_mode == 2:
|
||||||
assert (
|
assert (
|
||||||
self.quantization == "modelslim"
|
self.quantization == "modelslim"
|
||||||
), "When fuse_mode is set to 2, the NPU supports only ModelSlim quantization."
|
), "When fuse_mode is set to 2, the NPU supports only ModelSlim quantization."
|
||||||
@@ -2931,7 +2929,7 @@ class ServerArgs:
|
|||||||
)
|
)
|
||||||
if self.deepep_mode != "auto":
|
if self.deepep_mode != "auto":
|
||||||
logger.warning("--deepep-mode is ignored for Flashinfer MoE A2A")
|
logger.warning("--deepep-mode is ignored for Flashinfer MoE A2A")
|
||||||
if os.environ.get("SGLANG_MOE_NVFP4_DISPATCH") is None:
|
if not envs.SGLANG_MOE_NVFP4_DISPATCH.is_set():
|
||||||
envs.SGLANG_MOE_NVFP4_DISPATCH.set(True)
|
envs.SGLANG_MOE_NVFP4_DISPATCH.set(True)
|
||||||
logger.warning(
|
logger.warning(
|
||||||
"SGLANG_MOE_NVFP4_DISPATCH is set to True for Flashinfer MoE A2A"
|
"SGLANG_MOE_NVFP4_DISPATCH is set to True for Flashinfer MoE A2A"
|
||||||
@@ -2953,9 +2951,9 @@ class ServerArgs:
|
|||||||
# Skip validation if chunked prefill is disabled (i.e., size <= 0).
|
# Skip validation if chunked prefill is disabled (i.e., size <= 0).
|
||||||
# Skip validation if disaggregation mode is decode.
|
# Skip validation if disaggregation mode is decode.
|
||||||
if self.chunked_prefill_size > 0 and self.disaggregation_mode != "decode":
|
if self.chunked_prefill_size > 0 and self.disaggregation_mode != "decode":
|
||||||
assert (self.chunked_prefill_size) <= get_int_env_var(
|
assert (
|
||||||
"SGLANG_MORI_NUM_MAX_DISPATCH_TOKENS_PER_RANK", 4096
|
self.chunked_prefill_size
|
||||||
), "SGLANG_MORI_NUM_MAX_DISPATCH_TOKENS_PER_RANK (default 4096) must be larger or equal to chunked_prefill_size"
|
) <= envs.SGLANG_MORI_NUM_MAX_DISPATCH_TOKENS_PER_RANK.get(), "SGLANG_MORI_NUM_MAX_DISPATCH_TOKENS_PER_RANK (default 4096) must be larger or equal to chunked_prefill_size"
|
||||||
|
|
||||||
def _handle_eplb_and_dispatch(self):
|
def _handle_eplb_and_dispatch(self):
|
||||||
if self.enable_eplb and (self.expert_distribution_recorder_mode is None):
|
if self.enable_eplb and (self.expert_distribution_recorder_mode is None):
|
||||||
@@ -3739,9 +3737,9 @@ class ServerArgs:
|
|||||||
self.enable_deterministic_inference = True
|
self.enable_deterministic_inference = True
|
||||||
|
|
||||||
# For VLM
|
# For VLM
|
||||||
os.environ["SGLANG_VLM_CACHE_SIZE_MB"] = "0"
|
envs.SGLANG_VLM_CACHE_SIZE_MB.set(0)
|
||||||
# TODO remove this environment variable as a whole
|
# TODO remove this environment variable as a whole
|
||||||
os.environ["SGLANG_ENABLE_DETERMINISTIC_INFERENCE"] = "1"
|
envs.SGLANG_ENABLE_DETERMINISTIC_INFERENCE.set(True)
|
||||||
|
|
||||||
if self.enable_deterministic_inference:
|
if self.enable_deterministic_inference:
|
||||||
if self.enable_aiter_allreduce_fusion:
|
if self.enable_aiter_allreduce_fusion:
|
||||||
|
|||||||
Reference in New Issue
Block a user