# MI355X DeepSeek-V4-Pro-FP8 2-node 1P1D disaggregation recipe — TP8 + MTP. # # Consumed by: # * scripts/ci/slurm/process_result.py reads `resources` and # `backend.sglang_config` (TP/EP/DP + worker counts) for the summary table. # * scripts/ci/slurm/launch_mi355x.sh reads `runtime`, `bench`, and `mtp`. resources: prefill_workers: 1 decode_workers: 1 backend: sglang_config: prefill: tensor-parallel-size: 8 expert-parallel-size: 1 data-parallel-size: 1 decode: tensor-parallel-size: 8 expert-parallel-size: 1 data-parallel-size: 1 runtime: image: lmsysorg/sglang-rocm:v0.5.13.post1-rocm720-mi35x-20260623 attention_backend: dsv4 # RoCE HCAs MORI uses for cross-node KV transfer. ib_devices: rdma0,rdma1,rdma2,rdma3 prefill_port: 30025 decode_port: 30026 prefill_bootstrap_port: 8998 decode_bootstrap_port: 9001 lb_port: 8000 mem_fraction_static: 0.90 page_size: 256 max_running_requests: 256 chunked_prefill_size: 8192 swa_full_tokens_ratio: 0.1 # MTP / EAGLE speculative decoding (NextN head from the base model). Applied to # both prefill and decode. Flags mirror # test/registered/amd/test_deepseek_v4_pro_fp4_mtp.py. mtp: enabled: true num_steps: 3 eagle_topk: 1 num_draft_tokens: 4 bench: # bench_serving --max-concurrency sweep; one result JSON per concurrency. # conc256 is excluded for this leg only: at conc256 the disagg-decode SWA # hybrid KV pool fills, the scheduler retracts running requests, and the # retract->offload_kv_cache path calls get_cpu_copy() which is unimplemented # for the SWA hybrid pool (raises NotImplementedError, crashes the decode # scheduler). Raising swa_full_tokens_ratio (tried up to 0.3) does not help # -- usage climbs to fill the larger budget and still retracts. Until the # upstream get_cpu_copy stub is implemented, cap this leg at 128. concurrencies: [1, 8, 16, 32, 64, 128] num_prompts_factor: 4 # num-prompts = concurrency * factor random_range_ratio: 1.0 # Correctness gate run through the PD path before the perf sweep (full GSM8K, # 8-shot, accuracy > 0.91). A regression here fails the nightly even when # throughput looks fine ("fast but wrong"). accuracy: enabled: true num_shots: 8 num_questions: 1319 # full GSM8K test set threshold: 0.91