82 lines
2.6 KiB
YAML
82 lines
2.6 KiB
YAML
# MI355X GLM-5.2 (MXFP4) 2-node 1P1D disaggregation recipe — TP8 + MTP.
|
|
#
|
|
# GLM-5.2 ships a built-in NextN MTP head (num_nextn_predict_layers=1), same
|
|
# mechanism as DeepSeek-V4 — no external draft checkpoint. DSA attention is
|
|
# auto-selected (see 1p1d.yaml).
|
|
#
|
|
# Consumed by:
|
|
# * scripts/ci/slurm/process_result.py reads `resources` and
|
|
# `backend.sglang_config` (TP/EP/DP + worker counts) for the summary table.
|
|
# * scripts/ci/slurm/launch_mi355x.sh reads `runtime`, `bench`, and `mtp`.
|
|
|
|
resources:
|
|
prefill_workers: 1
|
|
decode_workers: 1
|
|
|
|
backend:
|
|
sglang_config:
|
|
prefill:
|
|
tensor-parallel-size: 8
|
|
expert-parallel-size: 1
|
|
data-parallel-size: 1
|
|
decode:
|
|
tensor-parallel-size: 8
|
|
expert-parallel-size: 1
|
|
data-parallel-size: 1
|
|
|
|
# Model-specific docker env + sglang server args (generic launcher path; keeps
|
|
# GLM off the hardcoded DeepSeek-V4 parser branch in launch_mi355x.sh). DSA
|
|
# attention is auto-selected for GlmMoeDsaForCausalLM, so no --attention-backend
|
|
# is set. GLM has a shared expert (n_shared_experts=1); shared-experts-fusion is
|
|
# disabled to mirror the DeepSeek-V4 path.
|
|
model:
|
|
env:
|
|
SGLANG_USE_AITER: 1
|
|
server_args:
|
|
- --reasoning-parser
|
|
- glm45
|
|
- --tool-call-parser
|
|
- glm45
|
|
- --disable-shared-experts-fusion
|
|
|
|
runtime:
|
|
image: lmsysorg/sglang-rocm:v0.5.15.post1-rocm720-mi35x-20260722
|
|
# attention_backend intentionally unset: DSA is auto-selected for
|
|
# GlmMoeDsaForCausalLM.
|
|
# RoCE HCAs MORI uses for cross-node KV transfer.
|
|
ib_devices: rdma0,rdma1,rdma2,rdma3
|
|
prefill_port: 30025
|
|
decode_port: 30026
|
|
prefill_bootstrap_port: 8998
|
|
decode_bootstrap_port: 9001
|
|
lb_port: 8000
|
|
mem_fraction_static: 0.90
|
|
page_size: 256
|
|
max_running_requests: 256
|
|
chunked_prefill_size: 8192
|
|
swa_full_tokens_ratio: 0.1
|
|
|
|
# MTP / EAGLE speculative decoding (built-in NextN head from the base model).
|
|
# Applied to both prefill and decode. No draft_model_path: the NextN head lives
|
|
# in the base checkpoint.
|
|
mtp:
|
|
enabled: true
|
|
num_steps: 3
|
|
eagle_topk: 1
|
|
num_draft_tokens: 4
|
|
|
|
bench:
|
|
# bench_serving --max-concurrency sweep; one result JSON per concurrency.
|
|
concurrencies: [1, 8, 16, 32, 64, 128, 256]
|
|
num_prompts_factor: 4 # num-prompts = concurrency * factor
|
|
random_range_ratio: 1.0
|
|
|
|
# Correctness gate run through the PD path before the perf sweep (full GSM8K,
|
|
# 8-shot, accuracy > 0.91). A regression here fails the nightly even when
|
|
# throughput looks fine ("fast but wrong").
|
|
accuracy:
|
|
enabled: true
|
|
num_shots: 8
|
|
num_questions: 1319 # full GSM8K test set
|
|
threshold: 0.91
|