# MI355X GLM-5.2 (MXFP4) 2-node 1P1D disaggregation recipe — DP8 + narrow EP8. # # DSA attention is auto-selected (see 1p1d.yaml). DP-attention 8 + within-node # EP8, same topology as the DeepSeek-V4 dp8ep8 leg. # # Consumed by: # * scripts/ci/slurm/process_result.py reads `resources` and # `backend.sglang_config` (TP/EP/DP + worker counts) for the summary table. # * scripts/ci/slurm/launch_mi355x.sh reads `runtime`, `bench`, and `mtp`. resources: prefill_workers: 1 decode_workers: 1 backend: sglang_config: prefill: tensor-parallel-size: 8 expert-parallel-size: 8 data-parallel-size: 8 decode: tensor-parallel-size: 8 expert-parallel-size: 8 data-parallel-size: 8 # Model-specific docker env + sglang server args (generic launcher path; keeps # GLM off the hardcoded DeepSeek-V4 parser branch in launch_mi355x.sh). DSA # attention is auto-selected for GlmMoeDsaForCausalLM, so no --attention-backend # is set. GLM has a shared expert (n_shared_experts=1); shared-experts-fusion is # disabled to mirror the DeepSeek-V4 path. model: env: SGLANG_USE_AITER: 1 server_args: - --reasoning-parser - glm45 - --tool-call-parser - glm45 - --disable-shared-experts-fusion runtime: image: lmsysorg/sglang-rocm:v0.5.15.post1-rocm720-mi35x-20260722 # attention_backend intentionally unset: DSA is auto-selected for # GlmMoeDsaForCausalLM. # RoCE HCAs MORI uses for cross-node KV transfer. ib_devices: rdma0,rdma1,rdma2,rdma3 prefill_port: 30025 decode_port: 30026 prefill_bootstrap_port: 8998 decode_bootstrap_port: 9001 lb_port: 8000 mem_fraction_static: 0.90 page_size: 256 max_running_requests: 256 chunked_prefill_size: 8192 swa_full_tokens_ratio: 0.1 bench: # bench_serving --max-concurrency sweep; one result JSON per concurrency. concurrencies: [1, 8, 16, 32, 64, 128, 256] num_prompts_factor: 4 # num-prompts = concurrency * factor random_range_ratio: 1.0 # Correctness gate run through the PD path before the perf sweep (full GSM8K, # 8-shot, accuracy > 0.91). A regression here fails the nightly even when # throughput looks fine ("fast but wrong"). accuracy: enabled: true num_shots: 8 num_questions: 1319 # full GSM8K test set threshold: 0.91