[AMD][DI][CI] 2/N Add DSV4 DP8/EP8 and MTP MI355X 1P1D nightly recipes (#29784)

Co-authored-by: bingxche <bingxche@amd.com>
This commit is contained in:
Zhaoyi Li
2026-07-01 23:46:48 +08:00
committed by GitHub
co-authored by bingxche
parent 2a6e5c60fe
commit 9bb7de9258
15 changed files with 970 additions and 3 deletions
@@ -0,0 +1,61 @@
# MI355X DeepSeek-V4-Flash-FP8 2-node 1P1D disaggregation recipe — DP8 + narrow EP8 + MTP.
#
# Consumed by:
# * scripts/ci/slurm/process_result.py reads `resources` and
# `backend.sglang_config` (TP/EP/DP + worker counts) for the summary table.
# * scripts/ci/slurm/launch_mi355x.sh reads `runtime`, `bench`, and `mtp`.
resources:
prefill_workers: 1
decode_workers: 1
backend:
sglang_config:
prefill:
tensor-parallel-size: 8
expert-parallel-size: 8
data-parallel-size: 8
decode:
tensor-parallel-size: 8
expert-parallel-size: 8
data-parallel-size: 8
runtime:
image: lmsysorg/sglang-rocm:v0.5.13.post1-rocm720-mi35x-20260623
attention_backend: dsv4
# RoCE HCAs MORI uses for cross-node KV transfer.
ib_devices: rdma0,rdma1,rdma2,rdma3
prefill_port: 30025
decode_port: 30026
prefill_bootstrap_port: 8998
decode_bootstrap_port: 9001
lb_port: 8000
mem_fraction_static: 0.90
page_size: 256
max_running_requests: 256
chunked_prefill_size: 8192
swa_full_tokens_ratio: 0.1
# MTP / EAGLE speculative decoding (NextN head from the base model). Applied to
# both prefill and decode. Flags mirror
# test/registered/amd/test_deepseek_v4_pro_fp4_mtp.py.
mtp:
enabled: true
num_steps: 3
eagle_topk: 1
num_draft_tokens: 4
bench:
# bench_serving --max-concurrency sweep; one result JSON per concurrency.
concurrencies: [1, 8, 16, 32, 64, 128, 256]
num_prompts_factor: 4 # num-prompts = concurrency * factor
random_range_ratio: 1.0
# Correctness gate run through the PD path before the perf sweep (full GSM8K,
# 8-shot, accuracy > 0.91). A regression here fails the nightly even when
# throughput looks fine ("fast but wrong").
accuracy:
enabled: true
num_shots: 8
num_questions: 1319 # full GSM8K test set
threshold: 0.91
@@ -0,0 +1,52 @@
# MI355X DeepSeek-V4-Flash-FP8 2-node 1P1D disaggregation recipe — DP8 + narrow EP8.
#
# Consumed by:
# * scripts/ci/slurm/process_result.py reads `resources` and
# `backend.sglang_config` (TP/EP/DP + worker counts) for the summary table.
# * scripts/ci/slurm/launch_mi355x.sh reads `runtime`, `bench`, and `mtp`.
resources:
prefill_workers: 1
decode_workers: 1
backend:
sglang_config:
prefill:
tensor-parallel-size: 8
expert-parallel-size: 8
data-parallel-size: 8
decode:
tensor-parallel-size: 8
expert-parallel-size: 8
data-parallel-size: 8
runtime:
image: lmsysorg/sglang-rocm:v0.5.13.post1-rocm720-mi35x-20260623
attention_backend: dsv4
# RoCE HCAs MORI uses for cross-node KV transfer.
ib_devices: rdma0,rdma1,rdma2,rdma3
prefill_port: 30025
decode_port: 30026
prefill_bootstrap_port: 8998
decode_bootstrap_port: 9001
lb_port: 8000
mem_fraction_static: 0.90
page_size: 256
max_running_requests: 256
chunked_prefill_size: 8192
swa_full_tokens_ratio: 0.1
bench:
# bench_serving --max-concurrency sweep; one result JSON per concurrency.
concurrencies: [1, 8, 16, 32, 64, 128, 256]
num_prompts_factor: 4 # num-prompts = concurrency * factor
random_range_ratio: 1.0
# Correctness gate run through the PD path before the perf sweep (full GSM8K,
# 8-shot, accuracy > 0.91). A regression here fails the nightly even when
# throughput looks fine ("fast but wrong").
accuracy:
enabled: true
num_shots: 8
num_questions: 1319 # full GSM8K test set
threshold: 0.91
@@ -0,0 +1,61 @@
# MI355X DeepSeek-V4-Flash-FP8 2-node 1P1D disaggregation recipe — TP8 + MTP.
#
# Consumed by:
# * scripts/ci/slurm/process_result.py reads `resources` and
# `backend.sglang_config` (TP/EP/DP + worker counts) for the summary table.
# * scripts/ci/slurm/launch_mi355x.sh reads `runtime`, `bench`, and `mtp`.
resources:
prefill_workers: 1
decode_workers: 1
backend:
sglang_config:
prefill:
tensor-parallel-size: 8
expert-parallel-size: 1
data-parallel-size: 1
decode:
tensor-parallel-size: 8
expert-parallel-size: 1
data-parallel-size: 1
runtime:
image: lmsysorg/sglang-rocm:v0.5.13.post1-rocm720-mi35x-20260623
attention_backend: dsv4
# RoCE HCAs MORI uses for cross-node KV transfer.
ib_devices: rdma0,rdma1,rdma2,rdma3
prefill_port: 30025
decode_port: 30026
prefill_bootstrap_port: 8998
decode_bootstrap_port: 9001
lb_port: 8000
mem_fraction_static: 0.90
page_size: 256
max_running_requests: 256
chunked_prefill_size: 8192
swa_full_tokens_ratio: 0.1
# MTP / EAGLE speculative decoding (NextN head from the base model). Applied to
# both prefill and decode. Flags mirror
# test/registered/amd/test_deepseek_v4_pro_fp4_mtp.py.
mtp:
enabled: true
num_steps: 3
eagle_topk: 1
num_draft_tokens: 4
bench:
# bench_serving --max-concurrency sweep; one result JSON per concurrency.
concurrencies: [1, 8, 16, 32, 64, 128, 256]
num_prompts_factor: 4 # num-prompts = concurrency * factor
random_range_ratio: 1.0
# Correctness gate run through the PD path before the perf sweep (full GSM8K,
# 8-shot, accuracy > 0.91). A regression here fails the nightly even when
# throughput looks fine ("fast but wrong").
accuracy:
enabled: true
num_shots: 8
num_questions: 1319 # full GSM8K test set
threshold: 0.91
@@ -0,0 +1,61 @@
# MI355X DeepSeek-V4-Pro-FP8 2-node 1P1D disaggregation recipe — DP8 + narrow EP8 + MTP.
#
# Consumed by:
# * scripts/ci/slurm/process_result.py reads `resources` and
# `backend.sglang_config` (TP/EP/DP + worker counts) for the summary table.
# * scripts/ci/slurm/launch_mi355x.sh reads `runtime`, `bench`, and `mtp`.
resources:
prefill_workers: 1
decode_workers: 1
backend:
sglang_config:
prefill:
tensor-parallel-size: 8
expert-parallel-size: 8
data-parallel-size: 8
decode:
tensor-parallel-size: 8
expert-parallel-size: 8
data-parallel-size: 8
runtime:
image: lmsysorg/sglang-rocm:v0.5.13.post1-rocm720-mi35x-20260623
attention_backend: dsv4
# RoCE HCAs MORI uses for cross-node KV transfer.
ib_devices: rdma0,rdma1,rdma2,rdma3
prefill_port: 30025
decode_port: 30026
prefill_bootstrap_port: 8998
decode_bootstrap_port: 9001
lb_port: 8000
mem_fraction_static: 0.90
page_size: 256
max_running_requests: 256
chunked_prefill_size: 8192
swa_full_tokens_ratio: 0.1
# MTP / EAGLE speculative decoding (NextN head from the base model). Applied to
# both prefill and decode. Flags mirror
# test/registered/amd/test_deepseek_v4_pro_fp4_mtp.py.
mtp:
enabled: true
num_steps: 3
eagle_topk: 1
num_draft_tokens: 4
bench:
# bench_serving --max-concurrency sweep; one result JSON per concurrency.
concurrencies: [1, 8, 16, 32, 64, 128, 256]
num_prompts_factor: 4 # num-prompts = concurrency * factor
random_range_ratio: 1.0
# Correctness gate run through the PD path before the perf sweep (full GSM8K,
# 8-shot, accuracy > 0.91). A regression here fails the nightly even when
# throughput looks fine ("fast but wrong").
accuracy:
enabled: true
num_shots: 8
num_questions: 1319 # full GSM8K test set
threshold: 0.91
@@ -0,0 +1,52 @@
# MI355X DeepSeek-V4-Pro-FP8 2-node 1P1D disaggregation recipe — DP8 + narrow EP8.
#
# Consumed by:
# * scripts/ci/slurm/process_result.py reads `resources` and
# `backend.sglang_config` (TP/EP/DP + worker counts) for the summary table.
# * scripts/ci/slurm/launch_mi355x.sh reads `runtime`, `bench`, and `mtp`.
resources:
prefill_workers: 1
decode_workers: 1
backend:
sglang_config:
prefill:
tensor-parallel-size: 8
expert-parallel-size: 8
data-parallel-size: 8
decode:
tensor-parallel-size: 8
expert-parallel-size: 8
data-parallel-size: 8
runtime:
image: lmsysorg/sglang-rocm:v0.5.13.post1-rocm720-mi35x-20260623
attention_backend: dsv4
# RoCE HCAs MORI uses for cross-node KV transfer.
ib_devices: rdma0,rdma1,rdma2,rdma3
prefill_port: 30025
decode_port: 30026
prefill_bootstrap_port: 8998
decode_bootstrap_port: 9001
lb_port: 8000
mem_fraction_static: 0.90
page_size: 256
max_running_requests: 256
chunked_prefill_size: 8192
swa_full_tokens_ratio: 0.1
bench:
# bench_serving --max-concurrency sweep; one result JSON per concurrency.
concurrencies: [1, 8, 16, 32, 64, 128, 256]
num_prompts_factor: 4 # num-prompts = concurrency * factor
random_range_ratio: 1.0
# Correctness gate run through the PD path before the perf sweep (full GSM8K,
# 8-shot, accuracy > 0.91). A regression here fails the nightly even when
# throughput looks fine ("fast but wrong").
accuracy:
enabled: true
num_shots: 8
num_questions: 1319 # full GSM8K test set
threshold: 0.91
@@ -0,0 +1,68 @@
# MI355X DeepSeek-V4-Pro-FP8 2-node 1P1D disaggregation recipe — TP8 + MTP.
#
# Consumed by:
# * scripts/ci/slurm/process_result.py reads `resources` and
# `backend.sglang_config` (TP/EP/DP + worker counts) for the summary table.
# * scripts/ci/slurm/launch_mi355x.sh reads `runtime`, `bench`, and `mtp`.
resources:
prefill_workers: 1
decode_workers: 1
backend:
sglang_config:
prefill:
tensor-parallel-size: 8
expert-parallel-size: 1
data-parallel-size: 1
decode:
tensor-parallel-size: 8
expert-parallel-size: 1
data-parallel-size: 1
runtime:
image: lmsysorg/sglang-rocm:v0.5.13.post1-rocm720-mi35x-20260623
attention_backend: dsv4
# RoCE HCAs MORI uses for cross-node KV transfer.
ib_devices: rdma0,rdma1,rdma2,rdma3
prefill_port: 30025
decode_port: 30026
prefill_bootstrap_port: 8998
decode_bootstrap_port: 9001
lb_port: 8000
mem_fraction_static: 0.90
page_size: 256
max_running_requests: 256
chunked_prefill_size: 8192
swa_full_tokens_ratio: 0.1
# MTP / EAGLE speculative decoding (NextN head from the base model). Applied to
# both prefill and decode. Flags mirror
# test/registered/amd/test_deepseek_v4_pro_fp4_mtp.py.
mtp:
enabled: true
num_steps: 3
eagle_topk: 1
num_draft_tokens: 4
bench:
# bench_serving --max-concurrency sweep; one result JSON per concurrency.
# conc256 is excluded for this leg only: at conc256 the disagg-decode SWA
# hybrid KV pool fills, the scheduler retracts running requests, and the
# retract->offload_kv_cache path calls get_cpu_copy() which is unimplemented
# for the SWA hybrid pool (raises NotImplementedError, crashes the decode
# scheduler). Raising swa_full_tokens_ratio (tried up to 0.3) does not help
# -- usage climbs to fill the larger budget and still retracts. Until the
# upstream get_cpu_copy stub is implemented, cap this leg at 128.
concurrencies: [1, 8, 16, 32, 64, 128]
num_prompts_factor: 4 # num-prompts = concurrency * factor
random_range_ratio: 1.0
# Correctness gate run through the PD path before the perf sweep (full GSM8K,
# 8-shot, accuracy > 0.91). A regression here fails the nightly even when
# throughput looks fine ("fast but wrong").
accuracy:
enabled: true
num_shots: 8
num_questions: 1319 # full GSM8K test set
threshold: 0.91