[Fix][AMD] Qwen3.5 MoE: disable global-slot shared-expert fusion under per-rank EP backends (MoRI + dp-attention init crash) (#31793)
This commit is contained in:
@@ -68,6 +68,7 @@ from sglang.srt.layers.moe.utils import (
|
|||||||
RoutingMethodType,
|
RoutingMethodType,
|
||||||
filter_moe_weight_param_global_expert,
|
filter_moe_weight_param_global_expert,
|
||||||
is_deepep_class_backend,
|
is_deepep_class_backend,
|
||||||
|
uses_per_rank_fused_shared_slots,
|
||||||
)
|
)
|
||||||
from sglang.srt.layers.quantization.base_config import QuantizationConfig
|
from sglang.srt.layers.quantization.base_config import QuantizationConfig
|
||||||
from sglang.srt.layers.radix_attention import RadixAttention
|
from sglang.srt.layers.radix_attention import RadixAttention
|
||||||
@@ -249,6 +250,20 @@ class Qwen2MoeSparseMoeBlock(nn.Module):
|
|||||||
self.num_shared_experts > 0
|
self.num_shared_experts > 0
|
||||||
and can_fuse_shared_expert(config, quant_config)
|
and can_fuse_shared_expert(config, quant_config)
|
||||||
)
|
)
|
||||||
|
if (
|
||||||
|
self.enable_shared_expert_fusion
|
||||||
|
and uses_per_rank_fused_shared_slots()
|
||||||
|
and get_parallel().moe_ep_size > 1
|
||||||
|
):
|
||||||
|
logger.warning_once(
|
||||||
|
"Disabling Qwen shared-expert fusion: it uses a single global "
|
||||||
|
"shared slot with a per-token gate, which is incompatible with "
|
||||||
|
"per-rank EP shared-slot backends (e.g. DeepEP/MoRI) at "
|
||||||
|
"moe_ep_size=%d. Using the separate shared-expert MLP instead.",
|
||||||
|
get_parallel().moe_ep_size,
|
||||||
|
)
|
||||||
|
self.enable_shared_expert_fusion = False
|
||||||
|
|
||||||
if self.enable_shared_expert_fusion:
|
if self.enable_shared_expert_fusion:
|
||||||
self.num_fused_shared_experts = self.num_shared_experts
|
self.num_fused_shared_experts = self.num_shared_experts
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user