# MI355X DeepSeek-V4-Pro FP4 2-node 1P1D disaggregation recipe — DP8 + narrow EP8 + MTP. # # Consumed by: # * scripts/ci/slurm/process_result.py reads `resources` and # `backend.sglang_config` (TP/EP/DP + worker counts) for the summary table. # * scripts/ci/slurm/launch_mi355x.sh reads `runtime`, `bench`, and `mtp`. resources: prefill_workers: 1 decode_workers: 1 backend: sglang_config: prefill: tensor-parallel-size: 8 expert-parallel-size: 8 data-parallel-size: 8 decode: tensor-parallel-size: 8 expert-parallel-size: 8 data-parallel-size: 8 runtime: image: lmsysorg/sglang-rocm:v0.5.13.post1-rocm720-mi35x-20260623 attention_backend: dsv4 # RoCE HCAs MORI uses for cross-node KV transfer. ib_devices: rdma0,rdma1,rdma2,rdma3 prefill_port: 30025 decode_port: 30026 prefill_bootstrap_port: 8998 decode_bootstrap_port: 9001 lb_port: 8000 mem_fraction_static: 0.90 page_size: 256 max_running_requests: 256 chunked_prefill_size: 8192 swa_full_tokens_ratio: 0.1 # MTP / EAGLE speculative decoding (NextN head from the base model). Applied to # both prefill and decode. Flags mirror # test/registered/amd/test_deepseek_v4_pro_fp4_mtp.py. mtp: enabled: true num_steps: 3 eagle_topk: 1 num_draft_tokens: 4 bench: # bench_serving --max-concurrency sweep; one result JSON per concurrency. concurrencies: [1, 8, 16, 32, 64, 128, 256] num_prompts_factor: 4 # num-prompts = concurrency * factor random_range_ratio: 1.0 # Correctness gate run through the PD path before the perf sweep (full GSM8K, # 8-shot, accuracy > 0.91). A regression here fails the nightly even when # throughput looks fine ("fast but wrong"). accuracy: enabled: true num_shots: 8 num_questions: 1319 # full GSM8K test set threshold: 0.91