Check the topology identities where the layout is written, and build at the published widths (#40340)

This commit is contained in:
Cheng Wan
2026-09-21 12:22:59 -07:00
committed by GitHub
parent d5fdab7022
commit 2d0e94e3a3
43 changed files with 843 additions and 261 deletions
@@ -26,6 +26,7 @@ from sglang.srt.layers.moe.topk import (
select_experts,
)
from sglang.srt.server_args import ServerArgs, set_global_server_args_for_scheduler
from sglang.test.test_utils import publish_build_topology
def fused_moe_triton_api(
@@ -227,10 +228,8 @@ def main():
backend="nccl" if torch.cuda.is_available() else "gloo",
)
initialize_model_parallel(
tensor_model_parallel_size=1,
expert_model_parallel_size=1,
)
publish_build_topology()
initialize_model_parallel()
model_config = get_model_config(args.model, args.tp_size, args.ep_size)
benchmark.run(
@@ -15,6 +15,7 @@ from sglang.srt.distributed.parallel_state import (
from sglang.srt.layers.moe.moe_runner.triton_utils.fused_moe import (
fused_moe as fused_moe_sglang,
)
from sglang.test.test_utils import publish_build_topology
from .common_utils import get_model_config
@@ -243,10 +244,8 @@ def main():
backend="nccl" if torch.cuda.is_available() else "gloo",
)
initialize_model_parallel(
tensor_model_parallel_size=1,
pipeline_model_parallel_size=1,
)
publish_build_topology()
initialize_model_parallel()
shape_configs = get_model_config(args.model, args.tp_size, args.ep_size)
benchmark.run(