# MI355X GLM-5.2 (MXFP4) 2-node 1P1D disaggregation recipe. # # GLM-5.2 uses GlmMoeDsaForCausalLM (DeepSeek Sparse Attention). sglang # auto-selects the DSA attention backend for this architecture, so this recipe # leaves `attention_backend` unset (empty) and lets the server pick DSA. MXFP4 # enables SGLANG_DSV4_FP4_EXPERTS in launch_mi355x.sh (driven by PRECISION), # same as the DeepSeek-V4 FP4 path. # # NOTE: the checkpoint on disk is amd/GLM-5.1-MXFP4 (only GLM MXFP4 build # currently mirrored on /it-share). model_path in nightly-configs.yaml points at # it; the recipe naming tracks the model line we are wiring CI for. # # Consumed by: # * scripts/ci/slurm/process_result.py reads `resources` and # `backend.sglang_config` (TP/EP/DP + worker counts) for the summary table. # * scripts/ci/slurm/launch_mi355x.sh reads `runtime` and `bench`. resources: prefill_workers: 1 decode_workers: 1 backend: sglang_config: prefill: tensor-parallel-size: 8 expert-parallel-size: 1 data-parallel-size: 1 decode: tensor-parallel-size: 8 expert-parallel-size: 1 data-parallel-size: 1 # Model-specific docker env + sglang server args (generic launcher path; keeps # GLM off the hardcoded DeepSeek-V4 parser branch in launch_mi355x.sh). DSA # attention is auto-selected for GlmMoeDsaForCausalLM, so no --attention-backend # is set. GLM has a shared expert (n_shared_experts=1); shared-experts-fusion is # disabled to mirror the DeepSeek-V4 path. model: env: SGLANG_USE_AITER: 1 server_args: - --reasoning-parser - glm45 - --tool-call-parser - glm45 - --disable-shared-experts-fusion runtime: image: lmsysorg/sglang-rocm:v0.5.15.post1-rocm720-mi35x-20260722 # attention_backend intentionally unset: DSA is auto-selected for # GlmMoeDsaForCausalLM. # RoCE HCAs MORI uses for cross-node KV transfer. ib_devices: rdma0,rdma1,rdma2,rdma3 prefill_port: 30025 decode_port: 30026 prefill_bootstrap_port: 8998 decode_bootstrap_port: 9001 lb_port: 8000 mem_fraction_static: 0.90 page_size: 256 max_running_requests: 256 chunked_prefill_size: 8192 swa_full_tokens_ratio: 0.1 bench: # bench_serving --max-concurrency sweep; one result JSON per concurrency. concurrencies: [1, 8, 16, 32, 64, 128, 256] num_prompts_factor: 4 # num-prompts = concurrency * factor random_range_ratio: 1.0 # Correctness gate run through the PD path before the perf sweep # (full GSM8K, 8-shot, accuracy > 0.91). A regression here fails the nightly # even when throughput looks fine ("fast but wrong"). accuracy: enabled: true num_shots: 8 num_questions: 1319 # full GSM8K test set threshold: 0.91