From 3b5909de0ef5e16188e459502745712d9d9e4615 Mon Sep 17 00:00:00 2001 From: Baizhou Zhang Date: Fri, 21 Aug 2026 18:44:18 -0700 Subject: [PATCH] [DeepSeek V4] Add W4A4 MegaMoE server flag (#35918) --- .../templates/config.jsx.tmpl | 9 ++--- .../autoregressive/DeepSeek/DeepSeek-V4.mdx | 9 ++--- docs/scripts/check_cookbook_configs.mjs | 16 ++++++++ docs/src/snippets/_playground.jsx | 23 ++++++----- docs/src/snippets/configs/Qwen/qwen3.8.jsx | 7 +--- .../configs/deepseek-ai/deepseek-v4.jsx | 7 +--- .../snippets/configs/moonshotai/kimi-k3.jsx | 7 +--- python/sglang/srt/arg_groups/mega_moe_hook.py | 38 +++++++++++++++++++ python/sglang/srt/environ.py | 16 +++----- python/sglang/srt/layers/moe/mega_moe.py | 24 +----------- python/sglang/srt/server_args.py | 25 +++++------- .../cp/test_deepseek_v4_flash_fp4_b200_cp.py | 3 +- ...test_deepseek_v4_flash_fp4_megamoe_b200.py | 3 +- .../unit/server_args/test_server_args.py | 32 ++++++++++++++++ test/registered/unit/test_environ.py | 15 ++++++++ 15 files changed, 147 insertions(+), 87 deletions(-) create mode 100644 python/sglang/srt/arg_groups/mega_moe_hook.py diff --git a/.claude/skills/cookbook-add-model/templates/config.jsx.tmpl b/.claude/skills/cookbook-add-model/templates/config.jsx.tmpl index 51d87188a..6cfaf3d28 100644 --- a/.claude/skills/cookbook-add-model/templates/config.jsx.tmpl +++ b/.claude/skills/cookbook-add-model/templates/config.jsx.tmpl @@ -211,7 +211,7 @@ sgl-eval run gsm8k \\ ], }, // MegaMoE quantization sub-select — shown only when backend === "megamoe". - // W4A4 adds the FP4-activations env vars; both strip the DeepEP dispatch env. + // W4A4 adds the FP4-activations server flag; both strip the DeepEP dispatch env. // DELETE this block if there's no MegaMoE backend option above. megamoeQuant: { stripEnv: ["SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK"], @@ -219,11 +219,8 @@ sgl-eval run gsm8k \\ { id: "w4a8", label: "W4A8", env: ["SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK=8320"] }, { id: "w4a4", label: "W4A4", - env: [ - "SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK=8320", - "SGLANG_OPT_DEEPGEMM_MEGA_MOE_USE_FP4_ACTS=1", - "SGLANG_OPT_DEEPGEMM_MEGA_MOE_USE_MXF4_KIND=1", - ] }, + flags: ["--enable-w4a4-mxfp4-megamoe"], + env: ["SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK=8320"] }, ], }, ep: { label: "EP", values: [ diff --git a/docs/cookbook/autoregressive/DeepSeek/DeepSeek-V4.mdx b/docs/cookbook/autoregressive/DeepSeek/DeepSeek-V4.mdx index 720224104..f0ae0012c 100644 --- a/docs/cookbook/autoregressive/DeepSeek/DeepSeek-V4.mdx +++ b/docs/cookbook/autoregressive/DeepSeek/DeepSeek-V4.mdx @@ -332,14 +332,13 @@ HiCache and MegaMoE are **not** supported on RTX PRO 6000. MegaMoE fuses expert dispatch + GEMM into a single kernel for higher throughput on MoE layers. To enable it, use the **MegaMoE** chip in the Playground below — the playground will swap `--moe-a2a-backend deepep` for -`--moe-a2a-backend megamoe` and add the relevant env vars automatically. +`--moe-a2a-backend megamoe` and add the relevant launch settings automatically. Two variants are exposed: - **W4A8** — default MegaMoE kernel (FP4 weights, FP8 activations). -- **W4A4** — adds `SGLANG_OPT_DEEPGEMM_MEGA_MOE_USE_FP4_ACTS=1` and - `SGLANG_OPT_DEEPGEMM_MEGA_MOE_USE_MXF4_KIND=1` to run the custom W4A4 - kernel (FP4 activations). Higher throughput with negligible accuracy drop - (~89.5 GPQA on Pro). +- **W4A4** — adds `--enable-w4a4-mxfp4-megamoe` to run the custom W4A4 kernel + (FP4 activations). The flag configures the required DeepGEMM settings. + Higher throughput with negligible accuracy drop (~89.5 GPQA on Pro). Notes: - The W4A8 / W4A4 variants above are **Blackwell-only** (B200 / B300 / GB200 / GB300). On **Hopper (SM90, H100 / H200)** use the all-FP8 MegaMoE path described below instead. diff --git a/docs/scripts/check_cookbook_configs.mjs b/docs/scripts/check_cookbook_configs.mjs index 54e23eeb7..efe42fdbc 100755 --- a/docs/scripts/check_cookbook_configs.mjs +++ b/docs/scripts/check_cookbook_configs.mjs @@ -27,6 +27,9 @@ import { fileURLToPath } from "node:url"; const SNIPPETS = join(dirname(fileURLToPath(import.meta.url)), "..", "src", "snippets"); const CONFIGS = join(SNIPPETS, "configs"); const DIFFUSION_COOKBOOK = join(SNIPPETS, "..", "..", "cookbook", "diffusion"); +const COOKBOOK_MODEL_TEMPLATE = join( + SNIPPETS, "..", "..", "..", ".claude", "skills", "cookbook-add-model", + "templates", "config.jsx.tmpl"); const LEGACY_DIMS = ["variants", "quantizations", "strategies", "nodesOptions"]; const failures = []; @@ -71,6 +74,19 @@ if (/\bmatchedCell\s*!==\s*baseCell\b/.test(playgroundSource)) { fail("_playground.jsx", "sibling detection compares cloned cells by object identity"); } +const cookbookModelTemplate = readFileSync(COOKBOOK_MODEL_TEMPLATE, "utf8"); +for (const oldName of [ + "SGLANG_OPT_DEEPGEMM_MEGA_MOE_USE_FP4_ACTS", + "SGLANG_OPT_DEEPGEMM_MEGA_MOE_USE_MXF4_KIND", +]) { + if (cookbookModelTemplate.includes(oldName)) { + fail("config.jsx.tmpl", `still emits removed W4A4 setting ${oldName}`); + } +} +if (!cookbookModelTemplate.includes("--enable-w4a4-mxfp4-megamoe")) { + fail("config.jsx.tmpl", "W4A4 MegaMoE option is missing the server flag"); +} + // --------------------------------------------------------------- 3/4. Configs // Configs are .jsx with a single `export const config` literal; import them // through a data: URL so no temp file is needed. diff --git a/docs/src/snippets/_playground.jsx b/docs/src/snippets/_playground.jsx index 2e07cd9f1..ba8ccded9 100644 --- a/docs/src/snippets/_playground.jsx +++ b/docs/src/snippets/_playground.jsx @@ -611,23 +611,21 @@ export const Playground = ({ config }) => { // ---- Axis: MoE Parallelism ---------------------------------------------- // Backend single-select + EP numeric knob; either is optional. Picking the // "megamoe" backend reveals a Quantization sub-select (W4A8 / W4A4) in the same - // row — W4A4 adds the FP4-activations env vars. + // row — W4A4 adds the FP4-activations server flag. moe: { initState: () => ({ backend: null, ep: null, mmQuant: null }), // Prefer --moe-a2a-backend over --moe-runner-backend when both present. - // mmQuant is derived from the base env (FP4 activations present → W4A4). + // mmQuant is derived from the base flag (FP4 activations present → W4A4). deriveFromBase: (cell, fc, h) => { const flags = (cell && cell.flags) || []; - const baseEnv = (cell && cell.env) || []; const a2a = h.findFlagArg(flags, "--moe-a2a-backend"); const runner = h.findFlagArg(flags, "--moe-runner-backend"); - const fp4Acts = baseEnv.some( - (e) => e.startsWith("SGLANG_OPT_DEEPGEMM_MEGA_MOE_USE_FP4_ACTS")); + const w4a4 = h.hasFlag(flags, "--enable-w4a4-mxfp4-megamoe"); return { backend: a2a || runner || null, ep: h.parseIntFlagAny(flags, h.EP_HEADS), - mmQuant: fp4Acts ? "w4a4" : "w4a8", + mmQuant: w4a4 ? "w4a4" : "w4a8", }; }, @@ -651,15 +649,19 @@ export const Playground = ({ config }) => { if (opt?.env?.length) env = [...env, ...opt.env]; } // MegaMoE owns the MoE path: when the effective backend is megamoe, strip the - // DeepEP dispatch + any prior megamoe env, then re-add the selected quant's - // env. When the backend is explicitly switched away from megamoe, only drop - // the megamoe quant env (leave DeepEP dispatch intact). + // DeepEP dispatch + any prior MegaMoE quant settings, then re-add the + // selected quant's flags/env. When the backend is explicitly switched + // away from MegaMoE, only drop the MegaMoE quant settings (leave DeepEP + // dispatch intact). const mq = fc.megamoeQuant; if (mq) { const quantKeys = []; + const quantFlagHeads = []; for (const o of (mq.options || [])) { for (const e of (o.env || [])) quantKeys.push(e.split("=")[0]); + for (const f of (o.flags || [])) quantFlagHeads.push(f.split(/[\s=]/)[0]); } + flags = h.stripFlagsByFirstToken(flags, quantFlagHeads); const effBackend = value.backend !== null ? value.backend : (derived && derived.backend); if (effBackend === "megamoe") { @@ -667,6 +669,9 @@ export const Playground = ({ config }) => { const quant = value.mmQuant != null ? value.mmQuant : ((derived && derived.mmQuant) || "w4a8"); const opt = (mq.options || []).find((o) => o.id === quant); + if (opt?.flags?.length) { + flags = h.insertAfter(flags, h.ANCHOR_NEAR_MOE, opt.flags); + } if (opt?.env?.length) env = [...env, ...opt.env]; } else if (value.backend !== null) { env = h.stripEnvByPrefix(env, quantKeys); diff --git a/docs/src/snippets/configs/Qwen/qwen3.8.jsx b/docs/src/snippets/configs/Qwen/qwen3.8.jsx index bddb104c2..b7b5b4e8d 100644 --- a/docs/src/snippets/configs/Qwen/qwen3.8.jsx +++ b/docs/src/snippets/configs/Qwen/qwen3.8.jsx @@ -178,11 +178,8 @@ export const config = { { id: "w4a8", label: "W4A8", env: ["SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK=8320"] }, { id: "w4a4", label: "W4A4", - env: [ - "SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK=8320", - "SGLANG_OPT_DEEPGEMM_MEGA_MOE_USE_FP4_ACTS=1", - "SGLANG_OPT_DEEPGEMM_MEGA_MOE_USE_MXF4_KIND=1", - ] }, + flags: ["--enable-w4a4-mxfp4-megamoe"], + env: ["SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK=8320"] }, ], }, // WideEP — the launch post's large-scale-EP claim — so the range goes past diff --git a/docs/src/snippets/configs/deepseek-ai/deepseek-v4.jsx b/docs/src/snippets/configs/deepseek-ai/deepseek-v4.jsx index 13622017e..37e9ddd20 100644 --- a/docs/src/snippets/configs/deepseek-ai/deepseek-v4.jsx +++ b/docs/src/snippets/configs/deepseek-ai/deepseek-v4.jsx @@ -258,11 +258,8 @@ sgl-eval run aime25 \\ { id: "w4a8", label: "W4A8", env: ["SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK=8320"] }, { id: "w4a4", label: "W4A4", - env: [ - "SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK=8320", - "SGLANG_OPT_DEEPGEMM_MEGA_MOE_USE_FP4_ACTS=1", - "SGLANG_OPT_DEEPGEMM_MEGA_MOE_USE_MXF4_KIND=1", - ] }, + flags: ["--enable-w4a4-mxfp4-megamoe"], + env: ["SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK=8320"] }, ], }, ep: { label: "EP", values: [ diff --git a/docs/src/snippets/configs/moonshotai/kimi-k3.jsx b/docs/src/snippets/configs/moonshotai/kimi-k3.jsx index 601e1508d..b38f44271 100644 --- a/docs/src/snippets/configs/moonshotai/kimi-k3.jsx +++ b/docs/src/snippets/configs/moonshotai/kimi-k3.jsx @@ -513,11 +513,8 @@ export const config = { { id: "w4a8", label: "W4A8", env: ["SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK=8320"] }, { id: "w4a4", label: "W4A4", - env: [ - "SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK=8320", - "SGLANG_OPT_DEEPGEMM_MEGA_MOE_USE_FP4_ACTS=1", - "SGLANG_OPT_DEEPGEMM_MEGA_MOE_USE_MXF4_KIND=1", - ] }, + flags: ["--enable-w4a4-mxfp4-megamoe"], + env: ["SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK=8320"] }, ], }, ep: { label: "EP", values: [ diff --git a/python/sglang/srt/arg_groups/mega_moe_hook.py b/python/sglang/srt/arg_groups/mega_moe_hook.py new file mode 100644 index 000000000..89382c728 --- /dev/null +++ b/python/sglang/srt/arg_groups/mega_moe_hook.py @@ -0,0 +1,38 @@ +from __future__ import annotations + +import logging +import os +from typing import TYPE_CHECKING + +if TYPE_CHECKING: + from sglang.srt.server_args import ServerArgs + +logger = logging.getLogger(__name__) + + +def handle_mega_moe(server_args: ServerArgs) -> None: + handle_moe_runner_backend_alias(server_args) + handle_w4a4_mxfp4_megamoe_env(server_args) + + +def handle_moe_runner_backend_alias(server_args: ServerArgs) -> None: + if server_args.moe_runner_backend != "megamoe": + return + + if server_args.moe_a2a_backend not in ("none", "megamoe"): + logger.warning( + "--moe-runner-backend megamoe is an alias for " + "--moe-a2a-backend megamoe; overriding " + "--moe-a2a-backend %s.", + server_args.moe_a2a_backend, + ) + server_args.moe_runner_backend = "auto" + server_args.moe_a2a_backend = "megamoe" + + +def handle_w4a4_mxfp4_megamoe_env(server_args: ServerArgs) -> None: + if not server_args.enable_w4a4_mxfp4_megamoe: + return + + os.environ["DG_USE_FP4_ACTS"] = "1" + os.environ["DG_USE_MXF4_KIND"] = "1" diff --git a/python/sglang/srt/environ.py b/python/sglang/srt/environ.py index 86b95e63f..98199ea98 100644 --- a/python/sglang/srt/environ.py +++ b/python/sglang/srt/environ.py @@ -1052,16 +1052,6 @@ class Envs: # DeepGEMM Mega MoE # =================================================================== SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK = EnvInt(8192) - # When set, the mega-MoE x slot is packed E2M1 (FP4) instead of FP8 E4M3. - # Halves symm-buffer footprint and unlocks the MXF4 mainloop downstream. - # Setting this also exports DG_USE_FP4_ACTS=1 so DeepGEMM's symm-buffer - # sizing + fp8_fp4_mega_moe pick up the FP4 layout. - SGLANG_OPT_DEEPGEMM_MEGA_MOE_USE_FP4_ACTS = EnvBool(False) - # Switches the L1+L2 mainloops from kind::mxf8f6f4 (K=32 with-padding) to - # kind::mxf4 (K=64 dense) inside fp8_fp4_mega_moe. No effect unless - # SGLANG_OPT_DEEPGEMM_MEGA_MOE_USE_FP4_ACTS is also set; DeepGEMM asserts - # this combination on the host side. - SGLANG_OPT_DEEPGEMM_MEGA_MOE_USE_MXF4_KIND = EnvBool(False) # =================================================================== # Top-k kernels @@ -1631,6 +1621,12 @@ _DEPRECATED_ENVS: Dict[str, _DeprecatedEnv] = { note="Please use '--moe-runner-backend=cutlass' and/or " "'--speculative-moe-runner-backend=cutlass' instead." ), + "SGLANG_OPT_DEEPGEMM_MEGA_MOE_USE_FP4_ACTS": _DeprecatedEnv( + note="Please use '--enable-w4a4-mxfp4-megamoe' instead." + ), + "SGLANG_OPT_DEEPGEMM_MEGA_MOE_USE_MXF4_KIND": _DeprecatedEnv( + note="Please use '--enable-w4a4-mxfp4-megamoe' instead." + ), "SGLANG_DFLASH_PREFILL_REFILL_TARGET": _DeprecatedEnv( note="DFlash now auto-enables the min-free-slots delay; unset this env. " "To override the threshold, use '--min-free-slots-delay'." diff --git a/python/sglang/srt/layers/moe/mega_moe.py b/python/sglang/srt/layers/moe/mega_moe.py index daec9bd2b..5709b9342 100644 --- a/python/sglang/srt/layers/moe/mega_moe.py +++ b/python/sglang/srt/layers/moe/mega_moe.py @@ -42,26 +42,6 @@ if TYPE_CHECKING: _MEGA_MOE_SYMM_BUFFER: dict = {} -_MEGA_MOE_DG_ENV_APPLIED = False - - -def _apply_mega_moe_dg_env() -> None: - """Forward sglang's FP4/MXF4 opt-in flags to DeepGEMM via env vars. - - DeepGEMM reads `DG_USE_FP4_ACTS` (and `DG_USE_MXF4_KIND`) at host-function - call time — both `get_symm_buffer_for_mega_moe` and `fp8_fp4_mega_moe`. - Forwarding once at first use is sufficient (these are static config - flags, not per-request state) and matches the `setdefault` pattern so - explicit `DG_USE_*` overrides from outside still win. - """ - global _MEGA_MOE_DG_ENV_APPLIED - if _MEGA_MOE_DG_ENV_APPLIED: - return - if envs.SGLANG_OPT_DEEPGEMM_MEGA_MOE_USE_FP4_ACTS.get(): - os.environ.setdefault("DG_USE_FP4_ACTS", "1") - if envs.SGLANG_OPT_DEEPGEMM_MEGA_MOE_USE_MXF4_KIND.get(): - os.environ.setdefault("DG_USE_MXF4_KIND", "1") - _MEGA_MOE_DG_ENV_APPLIED = True def _get_mega_moe_symm_buffer( @@ -74,8 +54,6 @@ def _get_mega_moe_symm_buffer( ) -> SymmBuffer: import deep_gemm - _apply_mega_moe_dg_env() - key = ( id(group), num_max_tokens_per_rank, @@ -232,7 +210,7 @@ def _run_mega_routed( num_tokens, ) - use_fp4_acts = envs.SGLANG_OPT_DEEPGEMM_MEGA_MOE_USE_FP4_ACTS.get() + use_fp4_acts = os.getenv("DG_USE_FP4_ACTS") == "1" if use_fp4_acts: # FP4 path goes through DeepGEMM's mega_moe_pre_dispatch which # handles the E2M1 packing variant. The jit implementation diff --git a/python/sglang/srt/server_args.py b/python/sglang/srt/server_args.py index 912701c51..53316be92 100644 --- a/python/sglang/srt/server_args.py +++ b/python/sglang/srt/server_args.py @@ -2390,6 +2390,13 @@ class ServerArgs: ), NS("exec.moe"), ] = "none" + enable_w4a4_mxfp4_megamoe: A[ + bool, + "Enable the W4A4 MXFP4 MegaMoE path by setting DeepGEMM's " + "DG_USE_FP4_ACTS=1 and DG_USE_MXF4_KIND=1. Use with " + "--moe-a2a-backend megamoe.", + NS("exec.moe"), + ] = False moe_runner_backend: A[ str, Arg( @@ -3651,7 +3658,9 @@ class ServerArgs: # _handle_model_specific_adjustments never runs. self._resolved_overrides = [] - self._handle_moe_runner_backend_alias() + from sglang.srt.arg_groups.mega_moe_hook import handle_mega_moe + + handle_mega_moe(self) self._handle_return_hidden_states_mode() self._handle_media_url_security() self._handle_hicache_ratio_default() @@ -3824,20 +3833,6 @@ class ServerArgs: materialize_declarations(self) - def _handle_moe_runner_backend_alias(self): - if self.moe_runner_backend != "megamoe": - return - - if self.moe_a2a_backend not in ("none", "megamoe"): - logger.warning( - "--moe-runner-backend megamoe is an alias for " - "--moe-a2a-backend megamoe; overriding " - "--moe-a2a-backend %s.", - self.moe_a2a_backend, - ) - self.moe_runner_backend = "auto" - self.moe_a2a_backend = "megamoe" - def _handle_return_hidden_states_mode(self): if self.return_hidden_states_mode not in (None, "last", "full"): raise ValueError( diff --git a/test/registered/cp/test_deepseek_v4_flash_fp4_b200_cp.py b/test/registered/cp/test_deepseek_v4_flash_fp4_b200_cp.py index 2975c9d40..56c1f0748 100644 --- a/test/registered/cp/test_deepseek_v4_flash_fp4_b200_cp.py +++ b/test/registered/cp/test_deepseek_v4_flash_fp4_b200_cp.py @@ -41,8 +41,6 @@ _DEEPEP_ENV = { _MEGAMOE_ENV = { "SGLANG_ENABLE_CP_V2": "1", "SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK": "8320", - "SGLANG_OPT_DEEPGEMM_MEGA_MOE_USE_FP4_ACTS": "1", - "SGLANG_OPT_DEEPGEMM_MEGA_MOE_USE_MXF4_KIND": "1", } @@ -126,6 +124,7 @@ class TestDSV4FlashFP4B200Balanced_CP_Megamoe( "--enable-dp-attention", "--moe-a2a-backend", "megamoe", + "--enable-w4a4-mxfp4-megamoe", "--speculative-algorithm", "EAGLE", "--speculative-num-steps", diff --git a/test/registered/models_e2e/test_deepseek_v4_flash_fp4_megamoe_b200.py b/test/registered/models_e2e/test_deepseek_v4_flash_fp4_megamoe_b200.py index 8ffa029ca..c1a377a20 100644 --- a/test/registered/models_e2e/test_deepseek_v4_flash_fp4_megamoe_b200.py +++ b/test/registered/models_e2e/test_deepseek_v4_flash_fp4_megamoe_b200.py @@ -34,8 +34,6 @@ _W4A8_MEGAMOE_ENV = { _W4A4_MEGAMOE_ENV = { "SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK": "4096", - "SGLANG_OPT_DEEPGEMM_MEGA_MOE_USE_FP4_ACTS": "1", - "SGLANG_OPT_DEEPGEMM_MEGA_MOE_USE_MXF4_KIND": "1", } @@ -115,6 +113,7 @@ class TestDSV4FlashFP4B200W4A4MegaMoE( "--enable-dp-attention", "--moe-a2a-backend", "megamoe", + "--enable-w4a4-mxfp4-megamoe", "--speculative-algorithm", "EAGLE", "--speculative-num-steps", diff --git a/test/registered/unit/server_args/test_server_args.py b/test/registered/unit/server_args/test_server_args.py index 28ea5f7ed..d41c69772 100644 --- a/test/registered/unit/server_args/test_server_args.py +++ b/test/registered/unit/server_args/test_server_args.py @@ -43,6 +43,38 @@ _mock_device.start() class TestPrepareServerArgs(CustomTestCase): + def test_enable_w4a4_mxfp4_megamoe_sets_deepgemm_env(self): + deepgemm_env = { + "DG_USE_FP4_ACTS": "0", + "DG_USE_MXF4_KIND": "0", + } + with patch.dict(os.environ, deepgemm_env, clear=False): + try: + args = prepare_server_args( + ["--model-path", "dummy", "--enable-w4a4-mxfp4-megamoe"] + ) + except SystemExit as exc: + self.fail( + "--enable-w4a4-mxfp4-megamoe must be accepted by the CLI " + f"parser, got SystemExit({exc.code})" + ) + + self.assertTrue(args.enable_w4a4_mxfp4_megamoe) + self.assertEqual(os.environ["DG_USE_FP4_ACTS"], "1") + self.assertEqual(os.environ["DG_USE_MXF4_KIND"], "1") + + def test_w4a4_mxfp4_megamoe_disabled_preserves_deepgemm_env(self): + deepgemm_env = { + "DG_USE_FP4_ACTS": "0", + "DG_USE_MXF4_KIND": "0", + } + with patch.dict(os.environ, deepgemm_env, clear=False): + args = prepare_server_args(["--model-path", "dummy"]) + + self.assertFalse(args.enable_w4a4_mxfp4_megamoe) + self.assertEqual(os.environ["DG_USE_FP4_ACTS"], "0") + self.assertEqual(os.environ["DG_USE_MXF4_KIND"], "0") + def test_prefill_decode_interval(self): args = ServerArgs(model_path="dummy", prefill_decode_interval=16) self.assertEqual(args.prefill_decode_interval, 16) diff --git a/test/registered/unit/test_environ.py b/test/registered/unit/test_environ.py index 42c3db309..54ad9a00a 100644 --- a/test/registered/unit/test_environ.py +++ b/test/registered/unit/test_environ.py @@ -114,6 +114,21 @@ class TestDeprecatedEnvRegistry(unittest.TestCase): caught = self._apply(old_name, _DeprecatedEnv()) self.assertIn(f"{old_name} is deprecated", str(caught[0].message)) + def test_w4a4_mxfp4_megamoe_envs_warn_to_use_cli_flag(self): + old_names = ( + "SGLANG_OPT_DEEPGEMM_MEGA_MOE_USE_FP4_ACTS", + "SGLANG_OPT_DEEPGEMM_MEGA_MOE_USE_MXF4_KIND", + ) + for old_name in old_names: + with self.subTest(old_name=old_name): + os.environ[old_name] = "1" + self.addCleanup(os.environ.pop, old_name, None) + + caught = self._apply(old_name, _DEPRECATED_ENVS[old_name]) + + self.assertIn("--enable-w4a4-mxfp4-megamoe", str(caught[0].message)) + self.assertIsNone(_DEPRECATED_ENVS[old_name].replacement) + def test_renamed_env_forwards_value(self): old_name, new_name = "SGLANG_TEST_OLD_ENV", "SGLANG_TEST_NEW_ENV" os.environ[old_name] = "abc"