diff --git a/python/sglang/srt/server_args.py b/python/sglang/srt/server_args.py index aa4a13580..c30fcdd6a 100644 --- a/python/sglang/srt/server_args.py +++ b/python/sglang/srt/server_args.py @@ -8467,11 +8467,6 @@ class ServerArgs: "(--weight-cache-mode off) for this configuration." ) - if cfg.weight_cache_mode != "off" and cfg.enable_eplb: - raise ValueError( - "--weight-cache-mode is not supported together with --enable-eplb." - ) - def _is_mistral_native_format(self) -> bool: """True iff the checkpoint requires load_format=mistral. diff --git a/python/sglang/srt/weight_cache/daemon.py b/python/sglang/srt/weight_cache/daemon.py index c61dd8394..30240cd00 100644 --- a/python/sglang/srt/weight_cache/daemon.py +++ b/python/sglang/srt/weight_cache/daemon.py @@ -316,6 +316,7 @@ class WeightCacheDaemon: # The initialized groups are the authority for rank identity. This # avoids maintaining a second copy of the model-parallel hierarchy. self._init_distributed(server_args, model_config) + self._initialize_eplb_expert_location_metadata(model_config) moe_dp_rank = get_parallel().moe_dp_rank moe_ep_rank = get_parallel().moe_ep_rank self.config = CacheConfig( @@ -457,6 +458,23 @@ class WeightCacheDaemon: f"metadata size ~{total_bytes / 1024 / 1024:.1f} MB" ) + def _initialize_eplb_expert_location_metadata(self, model_config) -> None: + """Build the same initial physical expert layout as the engine.""" + if not self.server_args.enable_eplb: + return + + from sglang.srt.eplb.expert_location import ( + compute_initial_expert_location_metadata, + set_global_expert_location_metadata, + ) + + set_global_expert_location_metadata( + compute_initial_expert_location_metadata( + model_config=model_config, + moe_ep_rank=get_parallel().moe_ep_rank, + ) + ) + def serve(self): """Block and serve IPC handles over Unix socket.""" # Do NOT unlink an existing socket here: stale-file cleanup is the launch diff --git a/test/registered/model_loading/test_weight_cache_daemon.py b/test/registered/model_loading/test_weight_cache_daemon.py index 00669caab..6be3a01fc 100644 --- a/test/registered/model_loading/test_weight_cache_daemon.py +++ b/test/registered/model_loading/test_weight_cache_daemon.py @@ -10,6 +10,7 @@ import torch from sglang.srt.utils import kill_process_tree from sglang.test.ci.ci_register import register_cuda_ci from sglang.test.test_utils import ( + DEFAULT_TARGET_MODEL_EAGLE_DP_ATTN, DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH, DEFAULT_URL_FOR_TEST, CustomTestCase, @@ -26,7 +27,7 @@ DEFAULT_MODEL = "Qwen/Qwen3-0.6B" # IPC handoff is exercised on every PR. Since the CI runner executes the whole # file per suite, TestWeightCacheDaemonTP2 self-skips when fewer than 2 GPUs are # visible (i.e. on the 1-gpu runner). -register_cuda_ci(est_time=100, stage="extra-a", runner_config="2-gpu-large") +register_cuda_ci(est_time=280, stage="extra-a", runner_config="2-gpu-large") register_cuda_ci(est_time=45, stage="base-b", runner_config="1-gpu-small") # Capture the client server's logs so test_loaded_via_ipc can assert the IPC @@ -50,9 +51,13 @@ PROMPTS = [ class TestWeightCacheDaemonTP2(CustomTestCase): """E2E test: start weight cache daemons, then launch server in client mode with TP2.""" + model_override = None + daemon_args = [] + server_args = [] + @classmethod def setUpClass(cls): - cls.model = DEFAULT_MODEL + cls.model = cls.model_override or DEFAULT_MODEL cls.base_url = DEFAULT_URL_FOR_TEST cls.tp_size = 2 @@ -74,6 +79,7 @@ class TestWeightCacheDaemonTP2(CustomTestCase): cls.model, "--tp-size", str(cls.tp_size), + *cls.daemon_args, ] ) @@ -107,6 +113,7 @@ class TestWeightCacheDaemonTP2(CustomTestCase): str(cls.tp_size), "--weight-cache-mode", "client", + *cls.server_args, ], return_stdout_stderr=(cls.stdout, cls.stderr), ) @@ -334,5 +341,39 @@ class TestWeightCacheDaemonTP1Smoke(CustomTestCase): ) +class TestWeightCacheDaemonQwen3MoeDP(TestWeightCacheDaemonTP2): + """Qwen3 with static attention DP through the existing IPC fixture.""" + + daemon_args = [ + "--dp", + "2", + "--ep-size", + "1", + "--enable-dp-attention", + "--enable-dp-lm-head", + "--random-seed", + "42", + ] + server_args = daemon_args[:] + + +class TestWeightCacheDaemonQwen3MoeEP(TestWeightCacheDaemonTP2): + """Qwen3 MoE with static expert parallelism through the existing IPC fixture.""" + + model_override = DEFAULT_TARGET_MODEL_EAGLE_DP_ATTN + daemon_args = [ + "--dp", + "1", + "--ep-size", + "2", + "--enable-eplb", + "--ep-num-redundant-experts", + "2", + "--random-seed", + "42", + ] + server_args = daemon_args[:] + + if __name__ == "__main__": unittest.main() diff --git a/test/registered/unit/model_loader/test_weight_cache_protocol.py b/test/registered/unit/model_loader/test_weight_cache_protocol.py index d130f98f0..bacbfa415 100644 --- a/test/registered/unit/model_loader/test_weight_cache_protocol.py +++ b/test/registered/unit/model_loader/test_weight_cache_protocol.py @@ -269,8 +269,9 @@ class TestDaemonLaunchConfiguration(CustomTestCase): from sglang.srt.weight_cache import daemon # The spawn helper receives Engine's already-resolved ServerArgs. A - # minimal namespace keeps this projection test CPU-only and - # model-independent; importantly, no EPLB configuration is involved. + # minimal namespace keeps this forwarding test CPU-only and + # model-independent while covering the static DP/EP and EPLB fields + # consumed by the daemon. server_args = SimpleNamespace( model_path="/models/demo", tp_size=8, @@ -284,6 +285,8 @@ class TestDaemonLaunchConfiguration(CustomTestCase): moe_dense_tp_size=1, moe_a2a_backend="mooncake", deepep_mode="low_latency", + enable_eplb=True, + ep_num_redundant_experts=72, load_format="safetensors", dtype="bfloat16", quantization="fp8", diff --git a/test/registered/unit/server_args/test_server_args.py b/test/registered/unit/server_args/test_server_args.py index e8890e2af..a4b937185 100644 --- a/test/registered/unit/server_args/test_server_args.py +++ b/test/registered/unit/server_args/test_server_args.py @@ -45,6 +45,17 @@ _mock_device.start() class TestPrepareServerArgs(CustomTestCase): + def test_weight_cache_daemon_allows_static_eplb(self): + args = ServerArgs( + model_path="dummy", + weight_cache_mode="daemon", + enable_eplb=True, + ) + + # This validation runs before model construction and should allow the + # daemon to build the same static EPLB layout as the engine. + args._handle_load_format() + def test_enable_w4a4_mxfp4_megamoe_sets_deepgemm_env(self): deepgemm_env = { "DG_USE_FP4_ACTS": "0",