[AMD][DI][CI] 2/N Add DSV4 DP8/EP8 and MTP MI355X 1P1D nightly recipes (#29784)
Co-authored-by: bingxche <bingxche@amd.com>
This commit is contained in:
@@ -142,8 +142,40 @@ jobs:
|
|||||||
# Manual dispatch input wins; otherwise use the latest image resolved
|
# Manual dispatch input wins; otherwise use the latest image resolved
|
||||||
# by the setup job; otherwise the launcher falls back to the recipe default.
|
# by the setup job; otherwise the launcher falls back to the recipe default.
|
||||||
IMAGE_OVERRIDE: ${{ inputs.image != '' && inputs.image || needs.setup.outputs.image }}
|
IMAGE_OVERRIDE: ${{ inputs.image != '' && inputs.image || needs.setup.outputs.image }}
|
||||||
|
# Keep the scheduler off mia1-p01-g20: its ionic RDMA driver ABI
|
||||||
|
# mismatches the container, so MORI reports "no active RDMA device"
|
||||||
|
# and the disagg server dies on init. Remove once the node is fixed.
|
||||||
|
SLURM_EXCLUDE: mia1-p01-g20
|
||||||
run: bash scripts/ci/slurm/launch_mi355x.sh
|
run: bash scripts/ci/slurm/launch_mi355x.sh
|
||||||
|
|
||||||
|
- name: Pack logs
|
||||||
|
if: always()
|
||||||
|
run: |
|
||||||
|
LOG_DIR="$HOME/.mi355x_ci/${MATRIX_CONFIG_NAME}"
|
||||||
|
LOG_BUNDLE="${GITHUB_WORKSPACE}/${RESULT_FILENAME}_logs.tar.gz"
|
||||||
|
if [ ! -d "$LOG_DIR" ]; then
|
||||||
|
echo "WARN: log directory not found: $LOG_DIR"
|
||||||
|
exit 0
|
||||||
|
fi
|
||||||
|
|
||||||
|
mapfile -d '' LOG_FILES < <(
|
||||||
|
find "$LOG_DIR" -maxdepth 1 -type f \
|
||||||
|
\( -name '*.log' -o -name 'server_exit_*' -o -name 'bench_exit' \) \
|
||||||
|
-printf '%P\0' | sort -z
|
||||||
|
)
|
||||||
|
if [ "${#LOG_FILES[@]}" -eq 0 ]; then
|
||||||
|
echo "WARN: no log files found in $LOG_DIR"
|
||||||
|
exit 0
|
||||||
|
fi
|
||||||
|
|
||||||
|
STAGING_DIR="$(mktemp -d)"
|
||||||
|
trap 'rm -rf "$STAGING_DIR"' EXIT
|
||||||
|
for log_file in "${LOG_FILES[@]}"; do
|
||||||
|
cp -a "$LOG_DIR/$log_file" "$STAGING_DIR/$log_file" || true
|
||||||
|
done
|
||||||
|
tar czf "$LOG_BUNDLE" -C "$STAGING_DIR" .
|
||||||
|
echo "Packed ${#LOG_FILES[@]} log file(s) -> $LOG_BUNDLE"
|
||||||
|
|
||||||
- name: Process results
|
- name: Process results
|
||||||
if: always()
|
if: always()
|
||||||
run: |
|
run: |
|
||||||
@@ -164,7 +196,9 @@ jobs:
|
|||||||
uses: actions/upload-artifact@v4
|
uses: actions/upload-artifact@v4
|
||||||
with:
|
with:
|
||||||
name: mi355x-${{ matrix.config.name }}-${{ github.run_id }}
|
name: mi355x-${{ matrix.config.name }}-${{ github.run_id }}
|
||||||
path: ${{ github.workspace }}/*.json
|
path: |
|
||||||
|
${{ github.workspace }}/*.json
|
||||||
|
${{ github.workspace }}/*_logs.tar.gz
|
||||||
retention-days: 30
|
retention-days: 30
|
||||||
if-no-files-found: warn
|
if-no-files-found: warn
|
||||||
|
|
||||||
|
|||||||
@@ -22,6 +22,8 @@
|
|||||||
# MODEL_PATH - local snapshot dir (preferred over downloading MODEL)
|
# MODEL_PATH - local snapshot dir (preferred over downloading MODEL)
|
||||||
# SLURM_PARTITION - default: amd-sglang
|
# SLURM_PARTITION - default: amd-sglang
|
||||||
# SLURM_NODELIST - optional explicit node pin (else scheduler chooses)
|
# SLURM_NODELIST - optional explicit node pin (else scheduler chooses)
|
||||||
|
# SLURM_EXCLUDE - optional comma-separated nodes to keep the scheduler
|
||||||
|
# off (e.g. hosts with a broken RDMA driver)
|
||||||
# RUNNER_NAME - GitHub runner name (a built-in default env var)
|
# RUNNER_NAME - GitHub runner name (a built-in default env var)
|
||||||
# GITHUB_RUN_ID - GitHub Actions run id (a built-in default env var)
|
# GITHUB_RUN_ID - GitHub Actions run id (a built-in default env var)
|
||||||
# The allocation is named
|
# The allocation is named
|
||||||
@@ -101,6 +103,13 @@ emit("CHUNK", rt["chunked_prefill_size"])
|
|||||||
emit("SWA", rt["swa_full_tokens_ratio"])
|
emit("SWA", rt["swa_full_tokens_ratio"])
|
||||||
emit("PTP", b["prefill"]["tensor-parallel-size"])
|
emit("PTP", b["prefill"]["tensor-parallel-size"])
|
||||||
emit("DTP", b["decode"]["tensor-parallel-size"])
|
emit("DTP", b["decode"]["tensor-parallel-size"])
|
||||||
|
emit("PEP", b["prefill"].get("expert-parallel-size", 1))
|
||||||
|
emit("PDP", b["prefill"].get("data-parallel-size", 1))
|
||||||
|
m = r.get("mtp", {}) or {}
|
||||||
|
emit("MTP_ENABLED", 1 if m.get("enabled") else 0)
|
||||||
|
emit("MTP_STEPS", m.get("num_steps", 3))
|
||||||
|
emit("MTP_TOPK", m.get("eagle_topk", 1))
|
||||||
|
emit("MTP_DRAFT", m.get("num_draft_tokens", 4))
|
||||||
# Worker counts double as node counts here: one server per node (TP == GPUs/node).
|
# Worker counts double as node counts here: one server per node (TP == GPUs/node).
|
||||||
# 1P1D today; bumping these reserves 2P2D / 1P3D / 3P1D. Multi-node-per-worker
|
# 1P1D today; bumping these reserves 2P2D / 1P3D / 3P1D. Multi-node-per-worker
|
||||||
# (TP > GPUs/node, needs --dist-init-addr/--nnodes/--node-rank) is out of scope.
|
# (TP > GPUs/node, needs --dist-init-addr/--nnodes/--node-rank) is out of scope.
|
||||||
@@ -175,12 +184,24 @@ DSV4_ENV=(
|
|||||||
DSV4_ENV_STR="${DSV4_ENV[*]}"
|
DSV4_ENV_STR="${DSV4_ENV[*]}"
|
||||||
MORI_ENV="-e MORI_DISABLE_AUTO_XGMI=1 -e NCCL_IB_HCA=ionic -e NCCL_IB_GID_INDEX=1 -e NCCL_CROSS_NIC=1"
|
MORI_ENV="-e MORI_DISABLE_AUTO_XGMI=1 -e NCCL_IB_HCA=ionic -e NCCL_IB_GID_INDEX=1 -e NCCL_CROSS_NIC=1"
|
||||||
|
|
||||||
|
# Optional topology / speculative-decode flags driven by the recipe. Base recipes
|
||||||
|
# (EP1/DP1, no mtp) leave EXTRA_FLAGS empty, preserving prior behavior exactly.
|
||||||
|
EXTRA_FLAGS=""
|
||||||
|
(( PDP > 1 )) && EXTRA_FLAGS="$EXTRA_FLAGS --enable-dp-attention --dp-size $PDP"
|
||||||
|
(( PEP > 1 )) && EXTRA_FLAGS="$EXTRA_FLAGS --ep-size $PEP"
|
||||||
|
if [[ "$MTP_ENABLED" == "1" ]]; then
|
||||||
|
EXTRA_FLAGS="$EXTRA_FLAGS --speculative-algorithm EAGLE \
|
||||||
|
--speculative-num-steps $MTP_STEPS --speculative-eagle-topk $MTP_TOPK \
|
||||||
|
--speculative-num-draft-tokens $MTP_DRAFT"
|
||||||
|
fi
|
||||||
|
echo "extra flags: ${EXTRA_FLAGS:-<none>} (pep=$PEP pdp=$PDP mtp=$MTP_ENABLED)"
|
||||||
|
|
||||||
COMMON_FLAGS="--trust-remote-code --tp $PTP --disable-radix-cache \
|
COMMON_FLAGS="--trust-remote-code --tp $PTP --disable-radix-cache \
|
||||||
--attention-backend $ATTN --max-running-requests $MAXREQ --page-size $PAGE \
|
--attention-backend $ATTN --max-running-requests $MAXREQ --page-size $PAGE \
|
||||||
--mem-fraction-static $MEMFRAC --swa-full-tokens-ratio $SWA \
|
--mem-fraction-static $MEMFRAC --swa-full-tokens-ratio $SWA \
|
||||||
--chunked-prefill-size $CHUNK --disable-shared-experts-fusion \
|
--chunked-prefill-size $CHUNK --disable-shared-experts-fusion \
|
||||||
--tool-call-parser deepseekv4 --reasoning-parser deepseek-v4 \
|
--tool-call-parser deepseekv4 --reasoning-parser deepseek-v4 \
|
||||||
--disaggregation-transfer-backend mori --disaggregation-ib-device $IB"
|
--disaggregation-transfer-backend mori --disaggregation-ib-device $IB$EXTRA_FLAGS"
|
||||||
|
|
||||||
DOCKER_COMMON="--rm --network host --ipc host --shm-size 32g --privileged \
|
DOCKER_COMMON="--rm --network host --ipc host --shm-size 32g --privileged \
|
||||||
--security-opt seccomp=unconfined \
|
--security-opt seccomp=unconfined \
|
||||||
@@ -367,6 +388,12 @@ NODELIST_ARG=()
|
|||||||
EXCLUSIVE_ARG=()
|
EXCLUSIVE_ARG=()
|
||||||
[[ "${SLURM_EXCLUSIVE:-1}" == "1" ]] && EXCLUSIVE_ARG=(--exclusive)
|
[[ "${SLURM_EXCLUSIVE:-1}" == "1" ]] && EXCLUSIVE_ARG=(--exclusive)
|
||||||
|
|
||||||
|
# Keep the scheduler off known-bad nodes (e.g. a host whose ionic RDMA driver
|
||||||
|
# ABI mismatches the container, where MORI reports "no active RDMA device" and
|
||||||
|
# the disagg server dies on init). Comma-separated node list.
|
||||||
|
EXCLUDE_ARG=()
|
||||||
|
[[ -n "${SLURM_EXCLUDE:-}" ]] && EXCLUDE_ARG=(--exclude="$SLURM_EXCLUDE")
|
||||||
|
|
||||||
# One node per prefill/decode worker (TP == GPUs/node). 1P1D -> 2 nodes.
|
# One node per prefill/decode worker (TP == GPUs/node). 1P1D -> 2 nodes.
|
||||||
TOTAL_NODES=$((PW + DW))
|
TOTAL_NODES=$((PW + DW))
|
||||||
|
|
||||||
@@ -377,7 +404,7 @@ TOTAL_NODES=$((PW + DW))
|
|||||||
JOB_NAME="mi355x-ci-${RUNNER_NAME:-norunner}-${GITHUB_RUN_ID:-0}-${MATRIX_CONFIG_NAME}"
|
JOB_NAME="mi355x-ci-${RUNNER_NAME:-norunner}-${GITHUB_RUN_ID:-0}-${MATRIX_CONFIG_NAME}"
|
||||||
|
|
||||||
set +e
|
set +e
|
||||||
salloc -p "$SLURM_PARTITION" -N"$TOTAL_NODES" "${NODELIST_ARG[@]}" "${EXCLUSIVE_ARG[@]}" \
|
salloc -p "$SLURM_PARTITION" -N"$TOTAL_NODES" "${NODELIST_ARG[@]}" "${EXCLUDE_ARG[@]}" "${EXCLUSIVE_ARG[@]}" \
|
||||||
--job-name "$JOB_NAME" -t "$TIME_LIMIT" \
|
--job-name "$JOB_NAME" -t "$TIME_LIMIT" \
|
||||||
bash "$WORKDIR/drive.sh" "$WORKDIR" "$PW" "$DW"
|
bash "$WORKDIR/drive.sh" "$WORKDIR" "$PW" "$DW"
|
||||||
SALLOC_RC=$?
|
SALLOC_RC=$?
|
||||||
|
|||||||
@@ -117,3 +117,206 @@ dsv4pro-fp4-mi355x-sglang:
|
|||||||
search-space:
|
search-space:
|
||||||
- conc-list: [1, 8, 16, 32, 64, 128, 256]
|
- conc-list: [1, 8, 16, 32, 64, 128, 256]
|
||||||
config_file: scripts/ci/slurm/recipes/mi355x-fp4/dsv4pro/1k1k/1p1d.yaml
|
config_file: scripts/ci/slurm/recipes/mi355x-fp4/dsv4pro/1k1k/1p1d.yaml
|
||||||
|
|
||||||
|
# AMD MI355X 2-node 1P1D disaggregation over MORI with extra topology / MTP
|
||||||
|
# coverage on the SAME four DeepSeek-V4 model x precision combos as the base
|
||||||
|
# MORI blocks above (which stay TP8, no MTP). Three variants per model:
|
||||||
|
# * -mtp : TP8 + EAGLE MTP (recipe `mtp.enabled`)
|
||||||
|
# * -dp8ep8 : DP-attention 8 + narrow within-node EP8
|
||||||
|
# * -dp8ep8-mtp : DP8 + narrow EP8 + EAGLE MTP
|
||||||
|
# launch_mi355x.sh reads expert-/data-parallel-size and the `mtp:` section to
|
||||||
|
# append --ep-size / --enable-dp-attention --dp-size / --speculative-* flags.
|
||||||
|
|
||||||
|
dsv4flash-fp8-mi355x-mtp-sglang:
|
||||||
|
model: sgl-project/DeepSeek-V4-Flash-FP8
|
||||||
|
model-prefix: dsv4flash
|
||||||
|
model_path: /it-share/model_coverage/models--sgl-project--DeepSeek-V4-Flash-FP8
|
||||||
|
runner: mi355x
|
||||||
|
precision: fp8
|
||||||
|
framework: sglang
|
||||||
|
multinode: true
|
||||||
|
disagg: true
|
||||||
|
seq-len-configs:
|
||||||
|
- isl: 1024
|
||||||
|
osl: 1024
|
||||||
|
search-space:
|
||||||
|
- conc-list: [1, 8, 16, 32, 64, 128, 256]
|
||||||
|
config_file: scripts/ci/slurm/recipes/mi355x-fp8/dsv4flash/1k1k/1p1d-mtp.yaml
|
||||||
|
|
||||||
|
dsv4flash-fp8-mi355x-dp8ep8-sglang:
|
||||||
|
model: sgl-project/DeepSeek-V4-Flash-FP8
|
||||||
|
model-prefix: dsv4flash
|
||||||
|
model_path: /it-share/model_coverage/models--sgl-project--DeepSeek-V4-Flash-FP8
|
||||||
|
runner: mi355x
|
||||||
|
precision: fp8
|
||||||
|
framework: sglang
|
||||||
|
multinode: true
|
||||||
|
disagg: true
|
||||||
|
seq-len-configs:
|
||||||
|
- isl: 1024
|
||||||
|
osl: 1024
|
||||||
|
search-space:
|
||||||
|
- conc-list: [1, 8, 16, 32, 64, 128, 256]
|
||||||
|
config_file: scripts/ci/slurm/recipes/mi355x-fp8/dsv4flash/1k1k/1p1d-dp8ep8.yaml
|
||||||
|
|
||||||
|
dsv4flash-fp8-mi355x-dp8ep8-mtp-sglang:
|
||||||
|
model: sgl-project/DeepSeek-V4-Flash-FP8
|
||||||
|
model-prefix: dsv4flash
|
||||||
|
model_path: /it-share/model_coverage/models--sgl-project--DeepSeek-V4-Flash-FP8
|
||||||
|
runner: mi355x
|
||||||
|
precision: fp8
|
||||||
|
framework: sglang
|
||||||
|
multinode: true
|
||||||
|
disagg: true
|
||||||
|
seq-len-configs:
|
||||||
|
- isl: 1024
|
||||||
|
osl: 1024
|
||||||
|
search-space:
|
||||||
|
- conc-list: [1, 8, 16, 32, 64, 128, 256]
|
||||||
|
config_file: scripts/ci/slurm/recipes/mi355x-fp8/dsv4flash/1k1k/1p1d-dp8ep8-mtp.yaml
|
||||||
|
|
||||||
|
dsv4pro-fp8-mi355x-mtp-sglang:
|
||||||
|
model: sgl-project/DeepSeek-V4-Pro-FP8
|
||||||
|
model-prefix: dsv4pro
|
||||||
|
model_path: /it-share/model_coverage/models--sgl-project--DeepSeek-V4-Pro-FP8
|
||||||
|
runner: mi355x
|
||||||
|
precision: fp8
|
||||||
|
framework: sglang
|
||||||
|
multinode: true
|
||||||
|
disagg: true
|
||||||
|
seq-len-configs:
|
||||||
|
- isl: 1024
|
||||||
|
osl: 1024
|
||||||
|
search-space:
|
||||||
|
# conc256 excluded: disagg-decode SWA hybrid pool retract->get_cpu_copy
|
||||||
|
# is an upstream NotImplementedError (crashes decode). See recipe.
|
||||||
|
- conc-list: [1, 8, 16, 32, 64, 128]
|
||||||
|
config_file: scripts/ci/slurm/recipes/mi355x-fp8/dsv4pro/1k1k/1p1d-mtp.yaml
|
||||||
|
|
||||||
|
dsv4pro-fp8-mi355x-dp8ep8-sglang:
|
||||||
|
model: sgl-project/DeepSeek-V4-Pro-FP8
|
||||||
|
model-prefix: dsv4pro
|
||||||
|
model_path: /it-share/model_coverage/models--sgl-project--DeepSeek-V4-Pro-FP8
|
||||||
|
runner: mi355x
|
||||||
|
precision: fp8
|
||||||
|
framework: sglang
|
||||||
|
multinode: true
|
||||||
|
disagg: true
|
||||||
|
seq-len-configs:
|
||||||
|
- isl: 1024
|
||||||
|
osl: 1024
|
||||||
|
search-space:
|
||||||
|
- conc-list: [1, 8, 16, 32, 64, 128, 256]
|
||||||
|
config_file: scripts/ci/slurm/recipes/mi355x-fp8/dsv4pro/1k1k/1p1d-dp8ep8.yaml
|
||||||
|
|
||||||
|
dsv4pro-fp8-mi355x-dp8ep8-mtp-sglang:
|
||||||
|
model: sgl-project/DeepSeek-V4-Pro-FP8
|
||||||
|
model-prefix: dsv4pro
|
||||||
|
model_path: /it-share/model_coverage/models--sgl-project--DeepSeek-V4-Pro-FP8
|
||||||
|
runner: mi355x
|
||||||
|
precision: fp8
|
||||||
|
framework: sglang
|
||||||
|
multinode: true
|
||||||
|
disagg: true
|
||||||
|
seq-len-configs:
|
||||||
|
- isl: 1024
|
||||||
|
osl: 1024
|
||||||
|
search-space:
|
||||||
|
- conc-list: [1, 8, 16, 32, 64, 128, 256]
|
||||||
|
config_file: scripts/ci/slurm/recipes/mi355x-fp8/dsv4pro/1k1k/1p1d-dp8ep8-mtp.yaml
|
||||||
|
|
||||||
|
dsv4flash-fp4-mi355x-mtp-sglang:
|
||||||
|
model: deepseek-ai/DeepSeek-V4-Flash
|
||||||
|
model-prefix: dsv4flash
|
||||||
|
model_path: /it-share/model_coverage/models--deepseek-ai--DeepSeek-V4-Flash
|
||||||
|
runner: mi355x
|
||||||
|
precision: fp4
|
||||||
|
framework: sglang
|
||||||
|
multinode: true
|
||||||
|
disagg: true
|
||||||
|
seq-len-configs:
|
||||||
|
- isl: 1024
|
||||||
|
osl: 1024
|
||||||
|
search-space:
|
||||||
|
- conc-list: [1, 8, 16, 32, 64, 128, 256]
|
||||||
|
config_file: scripts/ci/slurm/recipes/mi355x-fp4/dsv4flash/1k1k/1p1d-mtp.yaml
|
||||||
|
|
||||||
|
dsv4flash-fp4-mi355x-dp8ep8-sglang:
|
||||||
|
model: deepseek-ai/DeepSeek-V4-Flash
|
||||||
|
model-prefix: dsv4flash
|
||||||
|
model_path: /it-share/model_coverage/models--deepseek-ai--DeepSeek-V4-Flash
|
||||||
|
runner: mi355x
|
||||||
|
precision: fp4
|
||||||
|
framework: sglang
|
||||||
|
multinode: true
|
||||||
|
disagg: true
|
||||||
|
seq-len-configs:
|
||||||
|
- isl: 1024
|
||||||
|
osl: 1024
|
||||||
|
search-space:
|
||||||
|
- conc-list: [1, 8, 16, 32, 64, 128, 256]
|
||||||
|
config_file: scripts/ci/slurm/recipes/mi355x-fp4/dsv4flash/1k1k/1p1d-dp8ep8.yaml
|
||||||
|
|
||||||
|
dsv4flash-fp4-mi355x-dp8ep8-mtp-sglang:
|
||||||
|
model: deepseek-ai/DeepSeek-V4-Flash
|
||||||
|
model-prefix: dsv4flash
|
||||||
|
model_path: /it-share/model_coverage/models--deepseek-ai--DeepSeek-V4-Flash
|
||||||
|
runner: mi355x
|
||||||
|
precision: fp4
|
||||||
|
framework: sglang
|
||||||
|
multinode: true
|
||||||
|
disagg: true
|
||||||
|
seq-len-configs:
|
||||||
|
- isl: 1024
|
||||||
|
osl: 1024
|
||||||
|
search-space:
|
||||||
|
- conc-list: [1, 8, 16, 32, 64, 128, 256]
|
||||||
|
config_file: scripts/ci/slurm/recipes/mi355x-fp4/dsv4flash/1k1k/1p1d-dp8ep8-mtp.yaml
|
||||||
|
|
||||||
|
dsv4pro-fp4-mi355x-mtp-sglang:
|
||||||
|
model: deepseek-ai/DeepSeek-V4-Pro
|
||||||
|
model-prefix: dsv4pro
|
||||||
|
model_path: /it-share/model_coverage/models--deepseek-ai--DeepSeek-V4-Pro
|
||||||
|
runner: mi355x
|
||||||
|
precision: fp4
|
||||||
|
framework: sglang
|
||||||
|
multinode: true
|
||||||
|
disagg: true
|
||||||
|
seq-len-configs:
|
||||||
|
- isl: 1024
|
||||||
|
osl: 1024
|
||||||
|
search-space:
|
||||||
|
- conc-list: [1, 8, 16, 32, 64, 128, 256]
|
||||||
|
config_file: scripts/ci/slurm/recipes/mi355x-fp4/dsv4pro/1k1k/1p1d-mtp.yaml
|
||||||
|
|
||||||
|
dsv4pro-fp4-mi355x-dp8ep8-sglang:
|
||||||
|
model: deepseek-ai/DeepSeek-V4-Pro
|
||||||
|
model-prefix: dsv4pro
|
||||||
|
model_path: /it-share/model_coverage/models--deepseek-ai--DeepSeek-V4-Pro
|
||||||
|
runner: mi355x
|
||||||
|
precision: fp4
|
||||||
|
framework: sglang
|
||||||
|
multinode: true
|
||||||
|
disagg: true
|
||||||
|
seq-len-configs:
|
||||||
|
- isl: 1024
|
||||||
|
osl: 1024
|
||||||
|
search-space:
|
||||||
|
- conc-list: [1, 8, 16, 32, 64, 128, 256]
|
||||||
|
config_file: scripts/ci/slurm/recipes/mi355x-fp4/dsv4pro/1k1k/1p1d-dp8ep8.yaml
|
||||||
|
|
||||||
|
dsv4pro-fp4-mi355x-dp8ep8-mtp-sglang:
|
||||||
|
model: deepseek-ai/DeepSeek-V4-Pro
|
||||||
|
model-prefix: dsv4pro
|
||||||
|
model_path: /it-share/model_coverage/models--deepseek-ai--DeepSeek-V4-Pro
|
||||||
|
runner: mi355x
|
||||||
|
precision: fp4
|
||||||
|
framework: sglang
|
||||||
|
multinode: true
|
||||||
|
disagg: true
|
||||||
|
seq-len-configs:
|
||||||
|
- isl: 1024
|
||||||
|
osl: 1024
|
||||||
|
search-space:
|
||||||
|
- conc-list: [1, 8, 16, 32, 64, 128, 256]
|
||||||
|
config_file: scripts/ci/slurm/recipes/mi355x-fp4/dsv4pro/1k1k/1p1d-dp8ep8-mtp.yaml
|
||||||
|
|||||||
@@ -0,0 +1,61 @@
|
|||||||
|
# MI355X DeepSeek-V4-Flash FP4 2-node 1P1D disaggregation recipe — DP8 + narrow EP8 + MTP.
|
||||||
|
#
|
||||||
|
# Consumed by:
|
||||||
|
# * scripts/ci/slurm/process_result.py reads `resources` and
|
||||||
|
# `backend.sglang_config` (TP/EP/DP + worker counts) for the summary table.
|
||||||
|
# * scripts/ci/slurm/launch_mi355x.sh reads `runtime`, `bench`, and `mtp`.
|
||||||
|
|
||||||
|
resources:
|
||||||
|
prefill_workers: 1
|
||||||
|
decode_workers: 1
|
||||||
|
|
||||||
|
backend:
|
||||||
|
sglang_config:
|
||||||
|
prefill:
|
||||||
|
tensor-parallel-size: 8
|
||||||
|
expert-parallel-size: 8
|
||||||
|
data-parallel-size: 8
|
||||||
|
decode:
|
||||||
|
tensor-parallel-size: 8
|
||||||
|
expert-parallel-size: 8
|
||||||
|
data-parallel-size: 8
|
||||||
|
|
||||||
|
runtime:
|
||||||
|
image: lmsysorg/sglang-rocm:v0.5.13.post1-rocm720-mi35x-20260623
|
||||||
|
attention_backend: dsv4
|
||||||
|
# RoCE HCAs MORI uses for cross-node KV transfer.
|
||||||
|
ib_devices: rdma0,rdma1,rdma2,rdma3
|
||||||
|
prefill_port: 30025
|
||||||
|
decode_port: 30026
|
||||||
|
prefill_bootstrap_port: 8998
|
||||||
|
decode_bootstrap_port: 9001
|
||||||
|
lb_port: 8000
|
||||||
|
mem_fraction_static: 0.90
|
||||||
|
page_size: 256
|
||||||
|
max_running_requests: 256
|
||||||
|
chunked_prefill_size: 8192
|
||||||
|
swa_full_tokens_ratio: 0.1
|
||||||
|
|
||||||
|
# MTP / EAGLE speculative decoding (NextN head from the base model). Applied to
|
||||||
|
# both prefill and decode. Flags mirror
|
||||||
|
# test/registered/amd/test_deepseek_v4_pro_fp4_mtp.py.
|
||||||
|
mtp:
|
||||||
|
enabled: true
|
||||||
|
num_steps: 3
|
||||||
|
eagle_topk: 1
|
||||||
|
num_draft_tokens: 4
|
||||||
|
|
||||||
|
bench:
|
||||||
|
# bench_serving --max-concurrency sweep; one result JSON per concurrency.
|
||||||
|
concurrencies: [1, 8, 16, 32, 64, 128, 256]
|
||||||
|
num_prompts_factor: 4 # num-prompts = concurrency * factor
|
||||||
|
random_range_ratio: 1.0
|
||||||
|
|
||||||
|
# Correctness gate run through the PD path before the perf sweep (full GSM8K,
|
||||||
|
# 8-shot, accuracy > 0.91). A regression here fails the nightly even when
|
||||||
|
# throughput looks fine ("fast but wrong").
|
||||||
|
accuracy:
|
||||||
|
enabled: true
|
||||||
|
num_shots: 8
|
||||||
|
num_questions: 1319 # full GSM8K test set
|
||||||
|
threshold: 0.91
|
||||||
@@ -0,0 +1,52 @@
|
|||||||
|
# MI355X DeepSeek-V4-Flash FP4 2-node 1P1D disaggregation recipe — DP8 + narrow EP8.
|
||||||
|
#
|
||||||
|
# Consumed by:
|
||||||
|
# * scripts/ci/slurm/process_result.py reads `resources` and
|
||||||
|
# `backend.sglang_config` (TP/EP/DP + worker counts) for the summary table.
|
||||||
|
# * scripts/ci/slurm/launch_mi355x.sh reads `runtime`, `bench`, and `mtp`.
|
||||||
|
|
||||||
|
resources:
|
||||||
|
prefill_workers: 1
|
||||||
|
decode_workers: 1
|
||||||
|
|
||||||
|
backend:
|
||||||
|
sglang_config:
|
||||||
|
prefill:
|
||||||
|
tensor-parallel-size: 8
|
||||||
|
expert-parallel-size: 8
|
||||||
|
data-parallel-size: 8
|
||||||
|
decode:
|
||||||
|
tensor-parallel-size: 8
|
||||||
|
expert-parallel-size: 8
|
||||||
|
data-parallel-size: 8
|
||||||
|
|
||||||
|
runtime:
|
||||||
|
image: lmsysorg/sglang-rocm:v0.5.13.post1-rocm720-mi35x-20260623
|
||||||
|
attention_backend: dsv4
|
||||||
|
# RoCE HCAs MORI uses for cross-node KV transfer.
|
||||||
|
ib_devices: rdma0,rdma1,rdma2,rdma3
|
||||||
|
prefill_port: 30025
|
||||||
|
decode_port: 30026
|
||||||
|
prefill_bootstrap_port: 8998
|
||||||
|
decode_bootstrap_port: 9001
|
||||||
|
lb_port: 8000
|
||||||
|
mem_fraction_static: 0.90
|
||||||
|
page_size: 256
|
||||||
|
max_running_requests: 256
|
||||||
|
chunked_prefill_size: 8192
|
||||||
|
swa_full_tokens_ratio: 0.1
|
||||||
|
|
||||||
|
bench:
|
||||||
|
# bench_serving --max-concurrency sweep; one result JSON per concurrency.
|
||||||
|
concurrencies: [1, 8, 16, 32, 64, 128, 256]
|
||||||
|
num_prompts_factor: 4 # num-prompts = concurrency * factor
|
||||||
|
random_range_ratio: 1.0
|
||||||
|
|
||||||
|
# Correctness gate run through the PD path before the perf sweep (full GSM8K,
|
||||||
|
# 8-shot, accuracy > 0.91). A regression here fails the nightly even when
|
||||||
|
# throughput looks fine ("fast but wrong").
|
||||||
|
accuracy:
|
||||||
|
enabled: true
|
||||||
|
num_shots: 8
|
||||||
|
num_questions: 1319 # full GSM8K test set
|
||||||
|
threshold: 0.91
|
||||||
@@ -0,0 +1,61 @@
|
|||||||
|
# MI355X DeepSeek-V4-Flash FP4 2-node 1P1D disaggregation recipe — TP8 + MTP.
|
||||||
|
#
|
||||||
|
# Consumed by:
|
||||||
|
# * scripts/ci/slurm/process_result.py reads `resources` and
|
||||||
|
# `backend.sglang_config` (TP/EP/DP + worker counts) for the summary table.
|
||||||
|
# * scripts/ci/slurm/launch_mi355x.sh reads `runtime`, `bench`, and `mtp`.
|
||||||
|
|
||||||
|
resources:
|
||||||
|
prefill_workers: 1
|
||||||
|
decode_workers: 1
|
||||||
|
|
||||||
|
backend:
|
||||||
|
sglang_config:
|
||||||
|
prefill:
|
||||||
|
tensor-parallel-size: 8
|
||||||
|
expert-parallel-size: 1
|
||||||
|
data-parallel-size: 1
|
||||||
|
decode:
|
||||||
|
tensor-parallel-size: 8
|
||||||
|
expert-parallel-size: 1
|
||||||
|
data-parallel-size: 1
|
||||||
|
|
||||||
|
runtime:
|
||||||
|
image: lmsysorg/sglang-rocm:v0.5.13.post1-rocm720-mi35x-20260623
|
||||||
|
attention_backend: dsv4
|
||||||
|
# RoCE HCAs MORI uses for cross-node KV transfer.
|
||||||
|
ib_devices: rdma0,rdma1,rdma2,rdma3
|
||||||
|
prefill_port: 30025
|
||||||
|
decode_port: 30026
|
||||||
|
prefill_bootstrap_port: 8998
|
||||||
|
decode_bootstrap_port: 9001
|
||||||
|
lb_port: 8000
|
||||||
|
mem_fraction_static: 0.90
|
||||||
|
page_size: 256
|
||||||
|
max_running_requests: 256
|
||||||
|
chunked_prefill_size: 8192
|
||||||
|
swa_full_tokens_ratio: 0.1
|
||||||
|
|
||||||
|
# MTP / EAGLE speculative decoding (NextN head from the base model). Applied to
|
||||||
|
# both prefill and decode. Flags mirror
|
||||||
|
# test/registered/amd/test_deepseek_v4_pro_fp4_mtp.py.
|
||||||
|
mtp:
|
||||||
|
enabled: true
|
||||||
|
num_steps: 3
|
||||||
|
eagle_topk: 1
|
||||||
|
num_draft_tokens: 4
|
||||||
|
|
||||||
|
bench:
|
||||||
|
# bench_serving --max-concurrency sweep; one result JSON per concurrency.
|
||||||
|
concurrencies: [1, 8, 16, 32, 64, 128, 256]
|
||||||
|
num_prompts_factor: 4 # num-prompts = concurrency * factor
|
||||||
|
random_range_ratio: 1.0
|
||||||
|
|
||||||
|
# Correctness gate run through the PD path before the perf sweep (full GSM8K,
|
||||||
|
# 8-shot, accuracy > 0.91). A regression here fails the nightly even when
|
||||||
|
# throughput looks fine ("fast but wrong").
|
||||||
|
accuracy:
|
||||||
|
enabled: true
|
||||||
|
num_shots: 8
|
||||||
|
num_questions: 1319 # full GSM8K test set
|
||||||
|
threshold: 0.91
|
||||||
@@ -0,0 +1,61 @@
|
|||||||
|
# MI355X DeepSeek-V4-Pro FP4 2-node 1P1D disaggregation recipe — DP8 + narrow EP8 + MTP.
|
||||||
|
#
|
||||||
|
# Consumed by:
|
||||||
|
# * scripts/ci/slurm/process_result.py reads `resources` and
|
||||||
|
# `backend.sglang_config` (TP/EP/DP + worker counts) for the summary table.
|
||||||
|
# * scripts/ci/slurm/launch_mi355x.sh reads `runtime`, `bench`, and `mtp`.
|
||||||
|
|
||||||
|
resources:
|
||||||
|
prefill_workers: 1
|
||||||
|
decode_workers: 1
|
||||||
|
|
||||||
|
backend:
|
||||||
|
sglang_config:
|
||||||
|
prefill:
|
||||||
|
tensor-parallel-size: 8
|
||||||
|
expert-parallel-size: 8
|
||||||
|
data-parallel-size: 8
|
||||||
|
decode:
|
||||||
|
tensor-parallel-size: 8
|
||||||
|
expert-parallel-size: 8
|
||||||
|
data-parallel-size: 8
|
||||||
|
|
||||||
|
runtime:
|
||||||
|
image: lmsysorg/sglang-rocm:v0.5.13.post1-rocm720-mi35x-20260623
|
||||||
|
attention_backend: dsv4
|
||||||
|
# RoCE HCAs MORI uses for cross-node KV transfer.
|
||||||
|
ib_devices: rdma0,rdma1,rdma2,rdma3
|
||||||
|
prefill_port: 30025
|
||||||
|
decode_port: 30026
|
||||||
|
prefill_bootstrap_port: 8998
|
||||||
|
decode_bootstrap_port: 9001
|
||||||
|
lb_port: 8000
|
||||||
|
mem_fraction_static: 0.90
|
||||||
|
page_size: 256
|
||||||
|
max_running_requests: 256
|
||||||
|
chunked_prefill_size: 8192
|
||||||
|
swa_full_tokens_ratio: 0.1
|
||||||
|
|
||||||
|
# MTP / EAGLE speculative decoding (NextN head from the base model). Applied to
|
||||||
|
# both prefill and decode. Flags mirror
|
||||||
|
# test/registered/amd/test_deepseek_v4_pro_fp4_mtp.py.
|
||||||
|
mtp:
|
||||||
|
enabled: true
|
||||||
|
num_steps: 3
|
||||||
|
eagle_topk: 1
|
||||||
|
num_draft_tokens: 4
|
||||||
|
|
||||||
|
bench:
|
||||||
|
# bench_serving --max-concurrency sweep; one result JSON per concurrency.
|
||||||
|
concurrencies: [1, 8, 16, 32, 64, 128, 256]
|
||||||
|
num_prompts_factor: 4 # num-prompts = concurrency * factor
|
||||||
|
random_range_ratio: 1.0
|
||||||
|
|
||||||
|
# Correctness gate run through the PD path before the perf sweep (full GSM8K,
|
||||||
|
# 8-shot, accuracy > 0.91). A regression here fails the nightly even when
|
||||||
|
# throughput looks fine ("fast but wrong").
|
||||||
|
accuracy:
|
||||||
|
enabled: true
|
||||||
|
num_shots: 8
|
||||||
|
num_questions: 1319 # full GSM8K test set
|
||||||
|
threshold: 0.91
|
||||||
@@ -0,0 +1,52 @@
|
|||||||
|
# MI355X DeepSeek-V4-Pro FP4 2-node 1P1D disaggregation recipe — DP8 + narrow EP8.
|
||||||
|
#
|
||||||
|
# Consumed by:
|
||||||
|
# * scripts/ci/slurm/process_result.py reads `resources` and
|
||||||
|
# `backend.sglang_config` (TP/EP/DP + worker counts) for the summary table.
|
||||||
|
# * scripts/ci/slurm/launch_mi355x.sh reads `runtime`, `bench`, and `mtp`.
|
||||||
|
|
||||||
|
resources:
|
||||||
|
prefill_workers: 1
|
||||||
|
decode_workers: 1
|
||||||
|
|
||||||
|
backend:
|
||||||
|
sglang_config:
|
||||||
|
prefill:
|
||||||
|
tensor-parallel-size: 8
|
||||||
|
expert-parallel-size: 8
|
||||||
|
data-parallel-size: 8
|
||||||
|
decode:
|
||||||
|
tensor-parallel-size: 8
|
||||||
|
expert-parallel-size: 8
|
||||||
|
data-parallel-size: 8
|
||||||
|
|
||||||
|
runtime:
|
||||||
|
image: lmsysorg/sglang-rocm:v0.5.13.post1-rocm720-mi35x-20260623
|
||||||
|
attention_backend: dsv4
|
||||||
|
# RoCE HCAs MORI uses for cross-node KV transfer.
|
||||||
|
ib_devices: rdma0,rdma1,rdma2,rdma3
|
||||||
|
prefill_port: 30025
|
||||||
|
decode_port: 30026
|
||||||
|
prefill_bootstrap_port: 8998
|
||||||
|
decode_bootstrap_port: 9001
|
||||||
|
lb_port: 8000
|
||||||
|
mem_fraction_static: 0.90
|
||||||
|
page_size: 256
|
||||||
|
max_running_requests: 256
|
||||||
|
chunked_prefill_size: 8192
|
||||||
|
swa_full_tokens_ratio: 0.1
|
||||||
|
|
||||||
|
bench:
|
||||||
|
# bench_serving --max-concurrency sweep; one result JSON per concurrency.
|
||||||
|
concurrencies: [1, 8, 16, 32, 64, 128, 256]
|
||||||
|
num_prompts_factor: 4 # num-prompts = concurrency * factor
|
||||||
|
random_range_ratio: 1.0
|
||||||
|
|
||||||
|
# Correctness gate run through the PD path before the perf sweep (full GSM8K,
|
||||||
|
# 8-shot, accuracy > 0.91). A regression here fails the nightly even when
|
||||||
|
# throughput looks fine ("fast but wrong").
|
||||||
|
accuracy:
|
||||||
|
enabled: true
|
||||||
|
num_shots: 8
|
||||||
|
num_questions: 1319 # full GSM8K test set
|
||||||
|
threshold: 0.91
|
||||||
@@ -0,0 +1,61 @@
|
|||||||
|
# MI355X DeepSeek-V4-Pro FP4 2-node 1P1D disaggregation recipe — TP8 + MTP.
|
||||||
|
#
|
||||||
|
# Consumed by:
|
||||||
|
# * scripts/ci/slurm/process_result.py reads `resources` and
|
||||||
|
# `backend.sglang_config` (TP/EP/DP + worker counts) for the summary table.
|
||||||
|
# * scripts/ci/slurm/launch_mi355x.sh reads `runtime`, `bench`, and `mtp`.
|
||||||
|
|
||||||
|
resources:
|
||||||
|
prefill_workers: 1
|
||||||
|
decode_workers: 1
|
||||||
|
|
||||||
|
backend:
|
||||||
|
sglang_config:
|
||||||
|
prefill:
|
||||||
|
tensor-parallel-size: 8
|
||||||
|
expert-parallel-size: 1
|
||||||
|
data-parallel-size: 1
|
||||||
|
decode:
|
||||||
|
tensor-parallel-size: 8
|
||||||
|
expert-parallel-size: 1
|
||||||
|
data-parallel-size: 1
|
||||||
|
|
||||||
|
runtime:
|
||||||
|
image: lmsysorg/sglang-rocm:v0.5.13.post1-rocm720-mi35x-20260623
|
||||||
|
attention_backend: dsv4
|
||||||
|
# RoCE HCAs MORI uses for cross-node KV transfer.
|
||||||
|
ib_devices: rdma0,rdma1,rdma2,rdma3
|
||||||
|
prefill_port: 30025
|
||||||
|
decode_port: 30026
|
||||||
|
prefill_bootstrap_port: 8998
|
||||||
|
decode_bootstrap_port: 9001
|
||||||
|
lb_port: 8000
|
||||||
|
mem_fraction_static: 0.90
|
||||||
|
page_size: 256
|
||||||
|
max_running_requests: 256
|
||||||
|
chunked_prefill_size: 8192
|
||||||
|
swa_full_tokens_ratio: 0.1
|
||||||
|
|
||||||
|
# MTP / EAGLE speculative decoding (NextN head from the base model). Applied to
|
||||||
|
# both prefill and decode. Flags mirror
|
||||||
|
# test/registered/amd/test_deepseek_v4_pro_fp4_mtp.py.
|
||||||
|
mtp:
|
||||||
|
enabled: true
|
||||||
|
num_steps: 3
|
||||||
|
eagle_topk: 1
|
||||||
|
num_draft_tokens: 4
|
||||||
|
|
||||||
|
bench:
|
||||||
|
# bench_serving --max-concurrency sweep; one result JSON per concurrency.
|
||||||
|
concurrencies: [1, 8, 16, 32, 64, 128, 256]
|
||||||
|
num_prompts_factor: 4 # num-prompts = concurrency * factor
|
||||||
|
random_range_ratio: 1.0
|
||||||
|
|
||||||
|
# Correctness gate run through the PD path before the perf sweep (full GSM8K,
|
||||||
|
# 8-shot, accuracy > 0.91). A regression here fails the nightly even when
|
||||||
|
# throughput looks fine ("fast but wrong").
|
||||||
|
accuracy:
|
||||||
|
enabled: true
|
||||||
|
num_shots: 8
|
||||||
|
num_questions: 1319 # full GSM8K test set
|
||||||
|
threshold: 0.91
|
||||||
@@ -0,0 +1,61 @@
|
|||||||
|
# MI355X DeepSeek-V4-Flash-FP8 2-node 1P1D disaggregation recipe — DP8 + narrow EP8 + MTP.
|
||||||
|
#
|
||||||
|
# Consumed by:
|
||||||
|
# * scripts/ci/slurm/process_result.py reads `resources` and
|
||||||
|
# `backend.sglang_config` (TP/EP/DP + worker counts) for the summary table.
|
||||||
|
# * scripts/ci/slurm/launch_mi355x.sh reads `runtime`, `bench`, and `mtp`.
|
||||||
|
|
||||||
|
resources:
|
||||||
|
prefill_workers: 1
|
||||||
|
decode_workers: 1
|
||||||
|
|
||||||
|
backend:
|
||||||
|
sglang_config:
|
||||||
|
prefill:
|
||||||
|
tensor-parallel-size: 8
|
||||||
|
expert-parallel-size: 8
|
||||||
|
data-parallel-size: 8
|
||||||
|
decode:
|
||||||
|
tensor-parallel-size: 8
|
||||||
|
expert-parallel-size: 8
|
||||||
|
data-parallel-size: 8
|
||||||
|
|
||||||
|
runtime:
|
||||||
|
image: lmsysorg/sglang-rocm:v0.5.13.post1-rocm720-mi35x-20260623
|
||||||
|
attention_backend: dsv4
|
||||||
|
# RoCE HCAs MORI uses for cross-node KV transfer.
|
||||||
|
ib_devices: rdma0,rdma1,rdma2,rdma3
|
||||||
|
prefill_port: 30025
|
||||||
|
decode_port: 30026
|
||||||
|
prefill_bootstrap_port: 8998
|
||||||
|
decode_bootstrap_port: 9001
|
||||||
|
lb_port: 8000
|
||||||
|
mem_fraction_static: 0.90
|
||||||
|
page_size: 256
|
||||||
|
max_running_requests: 256
|
||||||
|
chunked_prefill_size: 8192
|
||||||
|
swa_full_tokens_ratio: 0.1
|
||||||
|
|
||||||
|
# MTP / EAGLE speculative decoding (NextN head from the base model). Applied to
|
||||||
|
# both prefill and decode. Flags mirror
|
||||||
|
# test/registered/amd/test_deepseek_v4_pro_fp4_mtp.py.
|
||||||
|
mtp:
|
||||||
|
enabled: true
|
||||||
|
num_steps: 3
|
||||||
|
eagle_topk: 1
|
||||||
|
num_draft_tokens: 4
|
||||||
|
|
||||||
|
bench:
|
||||||
|
# bench_serving --max-concurrency sweep; one result JSON per concurrency.
|
||||||
|
concurrencies: [1, 8, 16, 32, 64, 128, 256]
|
||||||
|
num_prompts_factor: 4 # num-prompts = concurrency * factor
|
||||||
|
random_range_ratio: 1.0
|
||||||
|
|
||||||
|
# Correctness gate run through the PD path before the perf sweep (full GSM8K,
|
||||||
|
# 8-shot, accuracy > 0.91). A regression here fails the nightly even when
|
||||||
|
# throughput looks fine ("fast but wrong").
|
||||||
|
accuracy:
|
||||||
|
enabled: true
|
||||||
|
num_shots: 8
|
||||||
|
num_questions: 1319 # full GSM8K test set
|
||||||
|
threshold: 0.91
|
||||||
@@ -0,0 +1,52 @@
|
|||||||
|
# MI355X DeepSeek-V4-Flash-FP8 2-node 1P1D disaggregation recipe — DP8 + narrow EP8.
|
||||||
|
#
|
||||||
|
# Consumed by:
|
||||||
|
# * scripts/ci/slurm/process_result.py reads `resources` and
|
||||||
|
# `backend.sglang_config` (TP/EP/DP + worker counts) for the summary table.
|
||||||
|
# * scripts/ci/slurm/launch_mi355x.sh reads `runtime`, `bench`, and `mtp`.
|
||||||
|
|
||||||
|
resources:
|
||||||
|
prefill_workers: 1
|
||||||
|
decode_workers: 1
|
||||||
|
|
||||||
|
backend:
|
||||||
|
sglang_config:
|
||||||
|
prefill:
|
||||||
|
tensor-parallel-size: 8
|
||||||
|
expert-parallel-size: 8
|
||||||
|
data-parallel-size: 8
|
||||||
|
decode:
|
||||||
|
tensor-parallel-size: 8
|
||||||
|
expert-parallel-size: 8
|
||||||
|
data-parallel-size: 8
|
||||||
|
|
||||||
|
runtime:
|
||||||
|
image: lmsysorg/sglang-rocm:v0.5.13.post1-rocm720-mi35x-20260623
|
||||||
|
attention_backend: dsv4
|
||||||
|
# RoCE HCAs MORI uses for cross-node KV transfer.
|
||||||
|
ib_devices: rdma0,rdma1,rdma2,rdma3
|
||||||
|
prefill_port: 30025
|
||||||
|
decode_port: 30026
|
||||||
|
prefill_bootstrap_port: 8998
|
||||||
|
decode_bootstrap_port: 9001
|
||||||
|
lb_port: 8000
|
||||||
|
mem_fraction_static: 0.90
|
||||||
|
page_size: 256
|
||||||
|
max_running_requests: 256
|
||||||
|
chunked_prefill_size: 8192
|
||||||
|
swa_full_tokens_ratio: 0.1
|
||||||
|
|
||||||
|
bench:
|
||||||
|
# bench_serving --max-concurrency sweep; one result JSON per concurrency.
|
||||||
|
concurrencies: [1, 8, 16, 32, 64, 128, 256]
|
||||||
|
num_prompts_factor: 4 # num-prompts = concurrency * factor
|
||||||
|
random_range_ratio: 1.0
|
||||||
|
|
||||||
|
# Correctness gate run through the PD path before the perf sweep (full GSM8K,
|
||||||
|
# 8-shot, accuracy > 0.91). A regression here fails the nightly even when
|
||||||
|
# throughput looks fine ("fast but wrong").
|
||||||
|
accuracy:
|
||||||
|
enabled: true
|
||||||
|
num_shots: 8
|
||||||
|
num_questions: 1319 # full GSM8K test set
|
||||||
|
threshold: 0.91
|
||||||
@@ -0,0 +1,61 @@
|
|||||||
|
# MI355X DeepSeek-V4-Flash-FP8 2-node 1P1D disaggregation recipe — TP8 + MTP.
|
||||||
|
#
|
||||||
|
# Consumed by:
|
||||||
|
# * scripts/ci/slurm/process_result.py reads `resources` and
|
||||||
|
# `backend.sglang_config` (TP/EP/DP + worker counts) for the summary table.
|
||||||
|
# * scripts/ci/slurm/launch_mi355x.sh reads `runtime`, `bench`, and `mtp`.
|
||||||
|
|
||||||
|
resources:
|
||||||
|
prefill_workers: 1
|
||||||
|
decode_workers: 1
|
||||||
|
|
||||||
|
backend:
|
||||||
|
sglang_config:
|
||||||
|
prefill:
|
||||||
|
tensor-parallel-size: 8
|
||||||
|
expert-parallel-size: 1
|
||||||
|
data-parallel-size: 1
|
||||||
|
decode:
|
||||||
|
tensor-parallel-size: 8
|
||||||
|
expert-parallel-size: 1
|
||||||
|
data-parallel-size: 1
|
||||||
|
|
||||||
|
runtime:
|
||||||
|
image: lmsysorg/sglang-rocm:v0.5.13.post1-rocm720-mi35x-20260623
|
||||||
|
attention_backend: dsv4
|
||||||
|
# RoCE HCAs MORI uses for cross-node KV transfer.
|
||||||
|
ib_devices: rdma0,rdma1,rdma2,rdma3
|
||||||
|
prefill_port: 30025
|
||||||
|
decode_port: 30026
|
||||||
|
prefill_bootstrap_port: 8998
|
||||||
|
decode_bootstrap_port: 9001
|
||||||
|
lb_port: 8000
|
||||||
|
mem_fraction_static: 0.90
|
||||||
|
page_size: 256
|
||||||
|
max_running_requests: 256
|
||||||
|
chunked_prefill_size: 8192
|
||||||
|
swa_full_tokens_ratio: 0.1
|
||||||
|
|
||||||
|
# MTP / EAGLE speculative decoding (NextN head from the base model). Applied to
|
||||||
|
# both prefill and decode. Flags mirror
|
||||||
|
# test/registered/amd/test_deepseek_v4_pro_fp4_mtp.py.
|
||||||
|
mtp:
|
||||||
|
enabled: true
|
||||||
|
num_steps: 3
|
||||||
|
eagle_topk: 1
|
||||||
|
num_draft_tokens: 4
|
||||||
|
|
||||||
|
bench:
|
||||||
|
# bench_serving --max-concurrency sweep; one result JSON per concurrency.
|
||||||
|
concurrencies: [1, 8, 16, 32, 64, 128, 256]
|
||||||
|
num_prompts_factor: 4 # num-prompts = concurrency * factor
|
||||||
|
random_range_ratio: 1.0
|
||||||
|
|
||||||
|
# Correctness gate run through the PD path before the perf sweep (full GSM8K,
|
||||||
|
# 8-shot, accuracy > 0.91). A regression here fails the nightly even when
|
||||||
|
# throughput looks fine ("fast but wrong").
|
||||||
|
accuracy:
|
||||||
|
enabled: true
|
||||||
|
num_shots: 8
|
||||||
|
num_questions: 1319 # full GSM8K test set
|
||||||
|
threshold: 0.91
|
||||||
@@ -0,0 +1,61 @@
|
|||||||
|
# MI355X DeepSeek-V4-Pro-FP8 2-node 1P1D disaggregation recipe — DP8 + narrow EP8 + MTP.
|
||||||
|
#
|
||||||
|
# Consumed by:
|
||||||
|
# * scripts/ci/slurm/process_result.py reads `resources` and
|
||||||
|
# `backend.sglang_config` (TP/EP/DP + worker counts) for the summary table.
|
||||||
|
# * scripts/ci/slurm/launch_mi355x.sh reads `runtime`, `bench`, and `mtp`.
|
||||||
|
|
||||||
|
resources:
|
||||||
|
prefill_workers: 1
|
||||||
|
decode_workers: 1
|
||||||
|
|
||||||
|
backend:
|
||||||
|
sglang_config:
|
||||||
|
prefill:
|
||||||
|
tensor-parallel-size: 8
|
||||||
|
expert-parallel-size: 8
|
||||||
|
data-parallel-size: 8
|
||||||
|
decode:
|
||||||
|
tensor-parallel-size: 8
|
||||||
|
expert-parallel-size: 8
|
||||||
|
data-parallel-size: 8
|
||||||
|
|
||||||
|
runtime:
|
||||||
|
image: lmsysorg/sglang-rocm:v0.5.13.post1-rocm720-mi35x-20260623
|
||||||
|
attention_backend: dsv4
|
||||||
|
# RoCE HCAs MORI uses for cross-node KV transfer.
|
||||||
|
ib_devices: rdma0,rdma1,rdma2,rdma3
|
||||||
|
prefill_port: 30025
|
||||||
|
decode_port: 30026
|
||||||
|
prefill_bootstrap_port: 8998
|
||||||
|
decode_bootstrap_port: 9001
|
||||||
|
lb_port: 8000
|
||||||
|
mem_fraction_static: 0.90
|
||||||
|
page_size: 256
|
||||||
|
max_running_requests: 256
|
||||||
|
chunked_prefill_size: 8192
|
||||||
|
swa_full_tokens_ratio: 0.1
|
||||||
|
|
||||||
|
# MTP / EAGLE speculative decoding (NextN head from the base model). Applied to
|
||||||
|
# both prefill and decode. Flags mirror
|
||||||
|
# test/registered/amd/test_deepseek_v4_pro_fp4_mtp.py.
|
||||||
|
mtp:
|
||||||
|
enabled: true
|
||||||
|
num_steps: 3
|
||||||
|
eagle_topk: 1
|
||||||
|
num_draft_tokens: 4
|
||||||
|
|
||||||
|
bench:
|
||||||
|
# bench_serving --max-concurrency sweep; one result JSON per concurrency.
|
||||||
|
concurrencies: [1, 8, 16, 32, 64, 128, 256]
|
||||||
|
num_prompts_factor: 4 # num-prompts = concurrency * factor
|
||||||
|
random_range_ratio: 1.0
|
||||||
|
|
||||||
|
# Correctness gate run through the PD path before the perf sweep (full GSM8K,
|
||||||
|
# 8-shot, accuracy > 0.91). A regression here fails the nightly even when
|
||||||
|
# throughput looks fine ("fast but wrong").
|
||||||
|
accuracy:
|
||||||
|
enabled: true
|
||||||
|
num_shots: 8
|
||||||
|
num_questions: 1319 # full GSM8K test set
|
||||||
|
threshold: 0.91
|
||||||
@@ -0,0 +1,52 @@
|
|||||||
|
# MI355X DeepSeek-V4-Pro-FP8 2-node 1P1D disaggregation recipe — DP8 + narrow EP8.
|
||||||
|
#
|
||||||
|
# Consumed by:
|
||||||
|
# * scripts/ci/slurm/process_result.py reads `resources` and
|
||||||
|
# `backend.sglang_config` (TP/EP/DP + worker counts) for the summary table.
|
||||||
|
# * scripts/ci/slurm/launch_mi355x.sh reads `runtime`, `bench`, and `mtp`.
|
||||||
|
|
||||||
|
resources:
|
||||||
|
prefill_workers: 1
|
||||||
|
decode_workers: 1
|
||||||
|
|
||||||
|
backend:
|
||||||
|
sglang_config:
|
||||||
|
prefill:
|
||||||
|
tensor-parallel-size: 8
|
||||||
|
expert-parallel-size: 8
|
||||||
|
data-parallel-size: 8
|
||||||
|
decode:
|
||||||
|
tensor-parallel-size: 8
|
||||||
|
expert-parallel-size: 8
|
||||||
|
data-parallel-size: 8
|
||||||
|
|
||||||
|
runtime:
|
||||||
|
image: lmsysorg/sglang-rocm:v0.5.13.post1-rocm720-mi35x-20260623
|
||||||
|
attention_backend: dsv4
|
||||||
|
# RoCE HCAs MORI uses for cross-node KV transfer.
|
||||||
|
ib_devices: rdma0,rdma1,rdma2,rdma3
|
||||||
|
prefill_port: 30025
|
||||||
|
decode_port: 30026
|
||||||
|
prefill_bootstrap_port: 8998
|
||||||
|
decode_bootstrap_port: 9001
|
||||||
|
lb_port: 8000
|
||||||
|
mem_fraction_static: 0.90
|
||||||
|
page_size: 256
|
||||||
|
max_running_requests: 256
|
||||||
|
chunked_prefill_size: 8192
|
||||||
|
swa_full_tokens_ratio: 0.1
|
||||||
|
|
||||||
|
bench:
|
||||||
|
# bench_serving --max-concurrency sweep; one result JSON per concurrency.
|
||||||
|
concurrencies: [1, 8, 16, 32, 64, 128, 256]
|
||||||
|
num_prompts_factor: 4 # num-prompts = concurrency * factor
|
||||||
|
random_range_ratio: 1.0
|
||||||
|
|
||||||
|
# Correctness gate run through the PD path before the perf sweep (full GSM8K,
|
||||||
|
# 8-shot, accuracy > 0.91). A regression here fails the nightly even when
|
||||||
|
# throughput looks fine ("fast but wrong").
|
||||||
|
accuracy:
|
||||||
|
enabled: true
|
||||||
|
num_shots: 8
|
||||||
|
num_questions: 1319 # full GSM8K test set
|
||||||
|
threshold: 0.91
|
||||||
@@ -0,0 +1,68 @@
|
|||||||
|
# MI355X DeepSeek-V4-Pro-FP8 2-node 1P1D disaggregation recipe — TP8 + MTP.
|
||||||
|
#
|
||||||
|
# Consumed by:
|
||||||
|
# * scripts/ci/slurm/process_result.py reads `resources` and
|
||||||
|
# `backend.sglang_config` (TP/EP/DP + worker counts) for the summary table.
|
||||||
|
# * scripts/ci/slurm/launch_mi355x.sh reads `runtime`, `bench`, and `mtp`.
|
||||||
|
|
||||||
|
resources:
|
||||||
|
prefill_workers: 1
|
||||||
|
decode_workers: 1
|
||||||
|
|
||||||
|
backend:
|
||||||
|
sglang_config:
|
||||||
|
prefill:
|
||||||
|
tensor-parallel-size: 8
|
||||||
|
expert-parallel-size: 1
|
||||||
|
data-parallel-size: 1
|
||||||
|
decode:
|
||||||
|
tensor-parallel-size: 8
|
||||||
|
expert-parallel-size: 1
|
||||||
|
data-parallel-size: 1
|
||||||
|
|
||||||
|
runtime:
|
||||||
|
image: lmsysorg/sglang-rocm:v0.5.13.post1-rocm720-mi35x-20260623
|
||||||
|
attention_backend: dsv4
|
||||||
|
# RoCE HCAs MORI uses for cross-node KV transfer.
|
||||||
|
ib_devices: rdma0,rdma1,rdma2,rdma3
|
||||||
|
prefill_port: 30025
|
||||||
|
decode_port: 30026
|
||||||
|
prefill_bootstrap_port: 8998
|
||||||
|
decode_bootstrap_port: 9001
|
||||||
|
lb_port: 8000
|
||||||
|
mem_fraction_static: 0.90
|
||||||
|
page_size: 256
|
||||||
|
max_running_requests: 256
|
||||||
|
chunked_prefill_size: 8192
|
||||||
|
swa_full_tokens_ratio: 0.1
|
||||||
|
|
||||||
|
# MTP / EAGLE speculative decoding (NextN head from the base model). Applied to
|
||||||
|
# both prefill and decode. Flags mirror
|
||||||
|
# test/registered/amd/test_deepseek_v4_pro_fp4_mtp.py.
|
||||||
|
mtp:
|
||||||
|
enabled: true
|
||||||
|
num_steps: 3
|
||||||
|
eagle_topk: 1
|
||||||
|
num_draft_tokens: 4
|
||||||
|
|
||||||
|
bench:
|
||||||
|
# bench_serving --max-concurrency sweep; one result JSON per concurrency.
|
||||||
|
# conc256 is excluded for this leg only: at conc256 the disagg-decode SWA
|
||||||
|
# hybrid KV pool fills, the scheduler retracts running requests, and the
|
||||||
|
# retract->offload_kv_cache path calls get_cpu_copy() which is unimplemented
|
||||||
|
# for the SWA hybrid pool (raises NotImplementedError, crashes the decode
|
||||||
|
# scheduler). Raising swa_full_tokens_ratio (tried up to 0.3) does not help
|
||||||
|
# -- usage climbs to fill the larger budget and still retracts. Until the
|
||||||
|
# upstream get_cpu_copy stub is implemented, cap this leg at 128.
|
||||||
|
concurrencies: [1, 8, 16, 32, 64, 128]
|
||||||
|
num_prompts_factor: 4 # num-prompts = concurrency * factor
|
||||||
|
random_range_ratio: 1.0
|
||||||
|
|
||||||
|
# Correctness gate run through the PD path before the perf sweep (full GSM8K,
|
||||||
|
# 8-shot, accuracy > 0.91). A regression here fails the nightly even when
|
||||||
|
# throughput looks fine ("fast but wrong").
|
||||||
|
accuracy:
|
||||||
|
enabled: true
|
||||||
|
num_shots: 8
|
||||||
|
num_questions: 1319 # full GSM8K test set
|
||||||
|
threshold: 0.91
|
||||||
Reference in New Issue
Block a user