Revert "[MoE] Decouple Mega MoE from DeepEP backend" (#25317)
This commit is contained in:
@@ -429,6 +429,7 @@ export const DeepSeekV4Deployment = () => {
|
|||||||
"SGLANG_OPT_USE_JIT_NORM=1",
|
"SGLANG_OPT_USE_JIT_NORM=1",
|
||||||
"SGLANG_OPT_USE_JIT_INDEXER_METADATA=1",
|
"SGLANG_OPT_USE_JIT_INDEXER_METADATA=1",
|
||||||
"SGLANG_OPT_USE_TOPK_V2=1",
|
"SGLANG_OPT_USE_TOPK_V2=1",
|
||||||
|
"SGLANG_OPT_USE_CUSTOM_ALL_REDUCE_V2=1",
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
} else if (recipe === "balanced") {
|
} else if (recipe === "balanced") {
|
||||||
@@ -444,6 +445,14 @@ export const DeepSeekV4Deployment = () => {
|
|||||||
"SGLANG_OPT_USE_JIT_NORM=1",
|
"SGLANG_OPT_USE_JIT_NORM=1",
|
||||||
"SGLANG_OPT_USE_JIT_INDEXER_METADATA=1",
|
"SGLANG_OPT_USE_JIT_INDEXER_METADATA=1",
|
||||||
"SGLANG_OPT_USE_TOPK_V2=1",
|
"SGLANG_OPT_USE_TOPK_V2=1",
|
||||||
|
"SGLANG_OPT_USE_CUSTOM_ALL_REDUCE_V2=1",
|
||||||
|
"SGLANG_OPT_SWA_EVICT_DROP_PAGE_MARGIN=1",
|
||||||
|
"SGLANG_OPT_USE_DEEPGEMM_MEGA_MOE=0",
|
||||||
|
"SGLANG_OPT_USE_FAST_MASK_EP=1",
|
||||||
|
"SGLANG_OPT_FIX_MEGA_MOE_MEMORY=1",
|
||||||
|
"SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK=4096",
|
||||||
|
"SGLANG_OPT_FIX_NEXTN_MEGA_MOE=1",
|
||||||
|
"SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK=0",
|
||||||
);
|
);
|
||||||
} else {
|
} else {
|
||||||
recipeEnv.push(isBig
|
recipeEnv.push(isBig
|
||||||
@@ -466,9 +475,15 @@ export const DeepSeekV4Deployment = () => {
|
|||||||
"SGLANG_OPT_USE_JIT_NORM=1",
|
"SGLANG_OPT_USE_JIT_NORM=1",
|
||||||
"SGLANG_OPT_USE_JIT_INDEXER_METADATA=1",
|
"SGLANG_OPT_USE_JIT_INDEXER_METADATA=1",
|
||||||
"SGLANG_OPT_USE_TOPK_V2=1",
|
"SGLANG_OPT_USE_TOPK_V2=1",
|
||||||
|
"SGLANG_OPT_USE_CUSTOM_ALL_REDUCE_V2=1",
|
||||||
|
"SGLANG_OPT_SWA_EVICT_DROP_PAGE_MARGIN=1",
|
||||||
|
"SGLANG_OPT_USE_FAST_MASK_EP=1",
|
||||||
|
"SGLANG_OPT_FIX_MEGA_MOE_MEMORY=1",
|
||||||
|
"SGLANG_OPT_FIX_NEXTN_MEGA_MOE=1",
|
||||||
"SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK=0",
|
"SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK=0",
|
||||||
"NVSHMEM_DISABLE_IB=1",
|
"NVSHMEM_DISABLE_IB=1",
|
||||||
"SGLANG_OPT_SWA_RELEASE_LEAF_LOCK_AFTER_WINDOW=1",
|
"SGLANG_OPT_SWA_RELEASE_LEAF_LOCK_AFTER_WINDOW=1",
|
||||||
|
"SGLANG_OPT_USE_DEEPGEMM_MEGA_MOE=1",
|
||||||
"SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK=8320",
|
"SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK=8320",
|
||||||
);
|
);
|
||||||
} else {
|
} else {
|
||||||
@@ -595,11 +610,7 @@ export const DeepSeekV4Deployment = () => {
|
|||||||
flags.push(` --dp ${tp}`);
|
flags.push(` --dp ${tp}`);
|
||||||
flags.push(" --enable-dp-attention");
|
flags.push(" --enable-dp-attention");
|
||||||
if (multinode) flags.push(...multiNodeFlags(nnodes));
|
if (multinode) flags.push(...multiNodeFlags(nnodes));
|
||||||
if (isBig && hardware === "b200") {
|
flags.push(" --moe-a2a-backend deepep");
|
||||||
flags.push(" --moe-a2a-backend megamoe");
|
|
||||||
} else {
|
|
||||||
flags.push(" --moe-a2a-backend deepep");
|
|
||||||
}
|
|
||||||
if (hardware === "h200" && isBig) {
|
if (hardware === "h200" && isBig) {
|
||||||
flags.push(" --mem-fraction-static 0.88");
|
flags.push(" --mem-fraction-static 0.88");
|
||||||
} else if (isBig && hardware === "gb300") {
|
} else if (isBig && hardware === "gb300") {
|
||||||
@@ -982,9 +993,12 @@ python3 -m sglang_router.launch_router \\
|
|||||||
</p>
|
</p>
|
||||||
<pre style={{ margin: "8px 0 0 0", padding: "8px 12px", background: isDark ? "#111827" : "#f5f5f5", borderRadius: "4px", fontSize: "12px", lineHeight: "1.5", overflowX: "auto" }}>{
|
<pre style={{ margin: "8px 0 0 0", padding: "8px 12px", background: isDark ? "#111827" : "#f5f5f5", borderRadius: "4px", fontSize: "12px", lineHeight: "1.5", overflowX: "auto" }}>{
|
||||||
`# Add this flag to the sglang serve command:
|
`# Add this flag to the sglang serve command:
|
||||||
--moe-a2a-backend megamoe
|
--moe-a2a-backend deepep
|
||||||
|
|
||||||
# And set these env vars:
|
# And set these env vars:
|
||||||
|
SGLANG_OPT_USE_DEEPGEMM_MEGA_MOE=1
|
||||||
|
SGLANG_OPT_FIX_MEGA_MOE_MEMORY=1
|
||||||
|
SGLANG_OPT_FIX_NEXTN_MEGA_MOE=1
|
||||||
SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK=8320
|
SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK=8320
|
||||||
SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK=0`
|
SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK=0`
|
||||||
}</pre>
|
}</pre>
|
||||||
|
|||||||
@@ -82,7 +82,7 @@ _use_aiter = get_bool_env_var("SGLANG_USE_AITER") and _is_hip
|
|||||||
|
|
||||||
def create_moe_dispatcher(moe_runner_config: MoeRunnerConfig) -> BaseDispatcher:
|
def create_moe_dispatcher(moe_runner_config: MoeRunnerConfig) -> BaseDispatcher:
|
||||||
a2a_backend = get_moe_a2a_backend()
|
a2a_backend = get_moe_a2a_backend()
|
||||||
if a2a_backend.is_none() or a2a_backend.is_megamoe():
|
if a2a_backend.is_none():
|
||||||
return StandardDispatcher(moe_runner_config)
|
return StandardDispatcher(moe_runner_config)
|
||||||
elif (
|
elif (
|
||||||
a2a_backend.is_deepep()
|
a2a_backend.is_deepep()
|
||||||
|
|||||||
@@ -25,7 +25,6 @@ from sglang.jit_kernel.deepseek_v4 import mega_moe_pre_dispatch
|
|||||||
from sglang.srt.environ import envs
|
from sglang.srt.environ import envs
|
||||||
from sglang.srt.eplb.expert_location_dispatch import ExpertLocationDispatchInfo
|
from sglang.srt.eplb.expert_location_dispatch import ExpertLocationDispatchInfo
|
||||||
from sglang.srt.layers.dp_attention import get_dp_global_num_tokens
|
from sglang.srt.layers.dp_attention import get_dp_global_num_tokens
|
||||||
from sglang.srt.layers.moe.utils import get_moe_a2a_backend
|
|
||||||
from sglang.srt.model_executor.cuda_graph_runner import get_is_capture_mode
|
from sglang.srt.model_executor.cuda_graph_runner import get_is_capture_mode
|
||||||
|
|
||||||
if TYPE_CHECKING:
|
if TYPE_CHECKING:
|
||||||
@@ -95,7 +94,7 @@ def _get_mega_moe_symm_buffer(
|
|||||||
|
|
||||||
|
|
||||||
def should_use_mega_moe(moe: "DeepseekV2MoE", hidden_states: torch.Tensor) -> bool:
|
def should_use_mega_moe(moe: "DeepseekV2MoE", hidden_states: torch.Tensor) -> bool:
|
||||||
if not get_moe_a2a_backend().is_megamoe():
|
if not envs.SGLANG_OPT_USE_DEEPGEMM_MEGA_MOE.get():
|
||||||
return False
|
return False
|
||||||
if not getattr(moe.experts, "_mega_moe_weights_built", False):
|
if not getattr(moe.experts, "_mega_moe_weights_built", False):
|
||||||
return False
|
return False
|
||||||
|
|||||||
@@ -131,6 +131,7 @@ class DeepGemmRunnerCore(MoeRunnerCore):
|
|||||||
if envs.SGLANG_OPT_FIX_MEGA_MOE_MEMORY.get():
|
if envs.SGLANG_OPT_FIX_MEGA_MOE_MEMORY.get():
|
||||||
assert envs.SGLANG_OPT_SWIGLU_CLAMP_FUSION.get()
|
assert envs.SGLANG_OPT_SWIGLU_CLAMP_FUSION.get()
|
||||||
assert envs.SGLANG_OPT_USE_JIT_EP_ACTIVATION.get()
|
assert envs.SGLANG_OPT_USE_JIT_EP_ACTIVATION.get()
|
||||||
|
assert envs.SGLANG_OPT_USE_DEEPGEMM_MEGA_MOE.get()
|
||||||
self.use_swizzle = True
|
self.use_swizzle = True
|
||||||
|
|
||||||
def run(
|
def run(
|
||||||
|
|||||||
@@ -29,7 +29,6 @@ class MoeA2ABackend(Enum):
|
|||||||
MORI = "mori"
|
MORI = "mori"
|
||||||
ASCEND_FUSEEP = "ascend_fuseep"
|
ASCEND_FUSEEP = "ascend_fuseep"
|
||||||
FLASHINFER = "flashinfer"
|
FLASHINFER = "flashinfer"
|
||||||
MEGAMOE = "megamoe"
|
|
||||||
CUSTOMIZED = "customized"
|
CUSTOMIZED = "customized"
|
||||||
|
|
||||||
@classmethod
|
@classmethod
|
||||||
@@ -62,9 +61,6 @@ class MoeA2ABackend(Enum):
|
|||||||
def is_mori(self):
|
def is_mori(self):
|
||||||
return self == MoeA2ABackend.MORI
|
return self == MoeA2ABackend.MORI
|
||||||
|
|
||||||
def is_megamoe(self):
|
|
||||||
return self == MoeA2ABackend.MEGAMOE
|
|
||||||
|
|
||||||
def is_customized(self):
|
def is_customized(self):
|
||||||
return self == MoeA2ABackend.CUSTOMIZED
|
return self == MoeA2ABackend.CUSTOMIZED
|
||||||
|
|
||||||
|
|||||||
@@ -1193,7 +1193,7 @@ class Fp8MoEMethod(FusedMoEMethodBase):
|
|||||||
layer.w13_weight.data = layer.w13_weight.data.view(torch.int8)
|
layer.w13_weight.data = layer.w13_weight.data.view(torch.int8)
|
||||||
layer.w2_weight.data = layer.w2_weight.data.view(torch.int8)
|
layer.w2_weight.data = layer.w2_weight.data.view(torch.int8)
|
||||||
|
|
||||||
if get_moe_a2a_backend().is_megamoe():
|
if envs.SGLANG_OPT_USE_DEEPGEMM_MEGA_MOE.get():
|
||||||
from sglang.srt.layers.moe.mega_moe import (
|
from sglang.srt.layers.moe.mega_moe import (
|
||||||
build_mega_moe_experts_weights,
|
build_mega_moe_experts_weights,
|
||||||
)
|
)
|
||||||
|
|||||||
@@ -209,7 +209,6 @@ MOE_A2A_BACKEND_CHOICES = [
|
|||||||
"mori",
|
"mori",
|
||||||
"ascend_fuseep",
|
"ascend_fuseep",
|
||||||
"flashinfer",
|
"flashinfer",
|
||||||
"megamoe",
|
|
||||||
]
|
]
|
||||||
|
|
||||||
FP8_GEMM_RUNNER_BACKEND_CHOICES = [
|
FP8_GEMM_RUNNER_BACKEND_CHOICES = [
|
||||||
@@ -610,14 +609,7 @@ class ServerArgs:
|
|||||||
# Expert parallelism
|
# Expert parallelism
|
||||||
ep_size: int = 1
|
ep_size: int = 1
|
||||||
moe_a2a_backend: Literal[
|
moe_a2a_backend: Literal[
|
||||||
"none",
|
"none", "deepep", "mooncake", "nixl", "mori", "ascend_fuseep", "flashinfer"
|
||||||
"deepep",
|
|
||||||
"mooncake",
|
|
||||||
"nixl",
|
|
||||||
"mori",
|
|
||||||
"ascend_fuseep",
|
|
||||||
"flashinfer",
|
|
||||||
"megamoe",
|
|
||||||
] = "none"
|
] = "none"
|
||||||
moe_runner_backend: str = "auto"
|
moe_runner_backend: str = "auto"
|
||||||
record_nolora_graph: bool = True
|
record_nolora_graph: bool = True
|
||||||
@@ -3192,25 +3184,6 @@ class ServerArgs:
|
|||||||
)
|
)
|
||||||
self.moe_a2a_backend = "deepep"
|
self.moe_a2a_backend = "deepep"
|
||||||
|
|
||||||
if (
|
|
||||||
envs.SGLANG_OPT_USE_DEEPGEMM_MEGA_MOE.get()
|
|
||||||
and self.moe_a2a_backend != "megamoe"
|
|
||||||
):
|
|
||||||
self.moe_a2a_backend = "megamoe"
|
|
||||||
logger.info(
|
|
||||||
"SGLANG_OPT_USE_DEEPGEMM_MEGA_MOE is set, "
|
|
||||||
"auto-configuring --moe-a2a-backend megamoe."
|
|
||||||
)
|
|
||||||
|
|
||||||
if self.moe_a2a_backend == "megamoe":
|
|
||||||
self.ep_size = self.tp_size
|
|
||||||
if not envs.SGLANG_OPT_FIX_MEGA_MOE_MEMORY.is_set():
|
|
||||||
envs.SGLANG_OPT_FIX_MEGA_MOE_MEMORY.set(True)
|
|
||||||
logger.info(
|
|
||||||
f"Mega MoE is enabled. The expert parallel size is adjusted "
|
|
||||||
f"to be the same as the tensor parallel size[{self.tp_size}]."
|
|
||||||
)
|
|
||||||
|
|
||||||
if self.moe_a2a_backend == "deepep":
|
if self.moe_a2a_backend == "deepep":
|
||||||
if self.deepep_mode == "normal":
|
if self.deepep_mode == "normal":
|
||||||
logger.warning("Cuda graph is disabled because deepep_mode=`normal`")
|
logger.warning("Cuda graph is disabled because deepep_mode=`normal`")
|
||||||
|
|||||||
@@ -28,12 +28,18 @@ SERVER_LAUNCH_TIMEOUT = 3600
|
|||||||
|
|
||||||
|
|
||||||
_W4A8_MEGAMOE_ENV = {
|
_W4A8_MEGAMOE_ENV = {
|
||||||
|
"SGLANG_OPT_USE_DEEPGEMM_MEGA_MOE": "1",
|
||||||
|
"SGLANG_OPT_FIX_MEGA_MOE_MEMORY": "1",
|
||||||
|
"SGLANG_OPT_FIX_NEXTN_MEGA_MOE": "1",
|
||||||
"SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK": "4096",
|
"SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK": "4096",
|
||||||
"SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK": "0",
|
"SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK": "0",
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
_W4A4_MEGAMOE_ENV = {
|
_W4A4_MEGAMOE_ENV = {
|
||||||
|
"SGLANG_OPT_USE_DEEPGEMM_MEGA_MOE": "1",
|
||||||
|
"SGLANG_OPT_FIX_MEGA_MOE_MEMORY": "1",
|
||||||
|
"SGLANG_OPT_FIX_NEXTN_MEGA_MOE": "1",
|
||||||
"SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK": "4096",
|
"SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK": "4096",
|
||||||
"SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK": "0",
|
"SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK": "0",
|
||||||
"SGLANG_OPT_DEEPGEMM_MEGA_MOE_USE_FP4_ACTS": "1",
|
"SGLANG_OPT_DEEPGEMM_MEGA_MOE_USE_FP4_ACTS": "1",
|
||||||
@@ -75,7 +81,7 @@ class TestDSV4FlashFP4B200W4A8MegaMoE(ServerSanityMixin, CustomTestCase):
|
|||||||
"4",
|
"4",
|
||||||
"--enable-dp-attention",
|
"--enable-dp-attention",
|
||||||
"--moe-a2a-backend",
|
"--moe-a2a-backend",
|
||||||
"megamoe",
|
"deepep",
|
||||||
"--speculative-algorithm",
|
"--speculative-algorithm",
|
||||||
"EAGLE",
|
"EAGLE",
|
||||||
"--speculative-num-steps",
|
"--speculative-num-steps",
|
||||||
@@ -116,7 +122,7 @@ class TestDSV4FlashFP4B200W4A4MegaMoE(ServerSanityMixin, CustomTestCase):
|
|||||||
"4",
|
"4",
|
||||||
"--enable-dp-attention",
|
"--enable-dp-attention",
|
||||||
"--moe-a2a-backend",
|
"--moe-a2a-backend",
|
||||||
"megamoe",
|
"deepep",
|
||||||
"--speculative-algorithm",
|
"--speculative-algorithm",
|
||||||
"EAGLE",
|
"EAGLE",
|
||||||
"--speculative-num-steps",
|
"--speculative-num-steps",
|
||||||
|
|||||||
Reference in New Issue
Block a user