[AMD][DI][CI] 8/N Add GLM-5.2 MXFP4 1P1D DI/CI recipes (base + MTP + DP8/EP8) (#32120)
This commit is contained in:
@@ -0,0 +1,78 @@
|
||||
# MI355X GLM-5.2 (MXFP4) 2-node 1P1D disaggregation recipe.
|
||||
#
|
||||
# GLM-5.2 uses GlmMoeDsaForCausalLM (DeepSeek Sparse Attention). sglang
|
||||
# auto-selects the DSA attention backend for this architecture, so this recipe
|
||||
# leaves `attention_backend` unset (empty) and lets the server pick DSA. MXFP4
|
||||
# enables SGLANG_DSV4_FP4_EXPERTS in launch_mi355x.sh (driven by PRECISION),
|
||||
# same as the DeepSeek-V4 FP4 path.
|
||||
#
|
||||
# NOTE: the checkpoint on disk is amd/GLM-5.1-MXFP4 (only GLM MXFP4 build
|
||||
# currently mirrored on /it-share). model_path in nightly-configs.yaml points at
|
||||
# it; the recipe naming tracks the model line we are wiring CI for.
|
||||
#
|
||||
# Consumed by:
|
||||
# * scripts/ci/slurm/process_result.py reads `resources` and
|
||||
# `backend.sglang_config` (TP/EP/DP + worker counts) for the summary table.
|
||||
# * scripts/ci/slurm/launch_mi355x.sh reads `runtime` and `bench`.
|
||||
|
||||
resources:
|
||||
prefill_workers: 1
|
||||
decode_workers: 1
|
||||
|
||||
backend:
|
||||
sglang_config:
|
||||
prefill:
|
||||
tensor-parallel-size: 8
|
||||
expert-parallel-size: 1
|
||||
data-parallel-size: 1
|
||||
decode:
|
||||
tensor-parallel-size: 8
|
||||
expert-parallel-size: 1
|
||||
data-parallel-size: 1
|
||||
|
||||
# Model-specific docker env + sglang server args (generic launcher path; keeps
|
||||
# GLM off the hardcoded DeepSeek-V4 parser branch in launch_mi355x.sh). DSA
|
||||
# attention is auto-selected for GlmMoeDsaForCausalLM, so no --attention-backend
|
||||
# is set. GLM has a shared expert (n_shared_experts=1); shared-experts-fusion is
|
||||
# disabled to mirror the DeepSeek-V4 path.
|
||||
model:
|
||||
env:
|
||||
SGLANG_USE_AITER: 1
|
||||
server_args:
|
||||
- --reasoning-parser
|
||||
- glm45
|
||||
- --tool-call-parser
|
||||
- glm45
|
||||
- --disable-shared-experts-fusion
|
||||
|
||||
runtime:
|
||||
image: lmsysorg/sglang-rocm:v0.5.15.post1-rocm720-mi35x-20260722
|
||||
# attention_backend intentionally unset: DSA is auto-selected for
|
||||
# GlmMoeDsaForCausalLM.
|
||||
# RoCE HCAs MORI uses for cross-node KV transfer.
|
||||
ib_devices: rdma0,rdma1,rdma2,rdma3
|
||||
prefill_port: 30025
|
||||
decode_port: 30026
|
||||
prefill_bootstrap_port: 8998
|
||||
decode_bootstrap_port: 9001
|
||||
lb_port: 8000
|
||||
mem_fraction_static: 0.90
|
||||
page_size: 256
|
||||
max_running_requests: 256
|
||||
chunked_prefill_size: 8192
|
||||
swa_full_tokens_ratio: 0.1
|
||||
|
||||
bench:
|
||||
# bench_serving --max-concurrency sweep; one result JSON per concurrency.
|
||||
concurrencies: [1, 8, 16, 32, 64, 128, 256]
|
||||
num_prompts_factor: 4 # num-prompts = concurrency * factor
|
||||
random_range_ratio: 1.0
|
||||
|
||||
# Correctness gate run through the PD path before the perf sweep
|
||||
# (full GSM8K, 8-shot, accuracy > 0.91). A regression here fails the nightly
|
||||
# even when throughput looks fine ("fast but wrong").
|
||||
accuracy:
|
||||
enabled: true
|
||||
num_shots: 8
|
||||
num_questions: 1319 # full GSM8K test set
|
||||
threshold: 0.91
|
||||
Reference in New Issue
Block a user