[AMD][DI][CI] 3/N Add Kimi K2.6 FP8 MI355X 1P1D nightly recipes (#29855)
Co-authored-by: bingxche <bingxche@amd.com>
This commit is contained in:
@@ -59,19 +59,26 @@ if [[ -z "$MODEL_PATH" ]]; then
|
|||||||
fi
|
fi
|
||||||
|
|
||||||
# Resolve a HuggingFace cache dir (models--org--name) to its live snapshot dir.
|
# Resolve a HuggingFace cache dir (models--org--name) to its live snapshot dir.
|
||||||
# Lets nightly-configs point at the shared cache without hardcoding a snapshot
|
# Lets nightly-configs / recipes point at the shared cache without hardcoding a
|
||||||
# hash; if MODEL_PATH is already a concrete snapshot (or plain dir), use as-is.
|
# snapshot hash; a concrete snapshot dir (or plain dir) is returned unchanged.
|
||||||
if [[ -f "$MODEL_PATH/refs/main" && -d "$MODEL_PATH/snapshots" ]]; then
|
# Used for both MODEL_PATH and an optional speculative draft model path.
|
||||||
SNAP_HASH="$(cat "$MODEL_PATH/refs/main")"
|
resolve_snapshot() {
|
||||||
RESOLVED="$MODEL_PATH/snapshots/$SNAP_HASH"
|
local p="$1"
|
||||||
if [[ -d "$RESOLVED" ]]; then
|
if [[ -f "$p/refs/main" && -d "$p/snapshots" ]]; then
|
||||||
echo "resolved snapshot: $MODEL_PATH -> $RESOLVED"
|
local hash resolved
|
||||||
MODEL_PATH="$RESOLVED"
|
hash="$(cat "$p/refs/main")"
|
||||||
else
|
resolved="$p/snapshots/$hash"
|
||||||
echo "ERROR: refs/main=$SNAP_HASH but $RESOLVED missing" >&2
|
if [[ -n "$hash" && -d "$resolved" ]]; then
|
||||||
exit 1
|
echo "resolved snapshot: $p -> $resolved" >&2
|
||||||
|
echo "$resolved"
|
||||||
|
return 0
|
||||||
fi
|
fi
|
||||||
fi
|
echo "ERROR: refs/main=$hash but $resolved missing" >&2
|
||||||
|
return 1
|
||||||
|
fi
|
||||||
|
echo "$p"
|
||||||
|
}
|
||||||
|
MODEL_PATH="$(resolve_snapshot "$MODEL_PATH")" || exit 1
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
# Parse the recipe (runtime + bench + topology) into shell vars.
|
# Parse the recipe (runtime + bench + topology) into shell vars.
|
||||||
@@ -89,7 +96,12 @@ rt = r["runtime"]; b = r["backend"]["sglang_config"]; bn = r["bench"]
|
|||||||
res = r.get("resources", {})
|
res = r.get("resources", {})
|
||||||
def emit(k, v): print(f"{k}={v}")
|
def emit(k, v): print(f"{k}={v}")
|
||||||
emit("IMAGE", rt["image"])
|
emit("IMAGE", rt["image"])
|
||||||
emit("ATTN", rt["attention_backend"])
|
# Attention backend: single (`attention_backend`) or split
|
||||||
|
# (`prefill_attention_backend`/`decode_attention_backend`). Empty when absent so
|
||||||
|
# the flag is dropped for a model that omits it.
|
||||||
|
emit("ATTN", rt.get("attention_backend", ""))
|
||||||
|
emit("PATTN", rt.get("prefill_attention_backend", ""))
|
||||||
|
emit("DATTN", rt.get("decode_attention_backend", ""))
|
||||||
emit("IB", rt["ib_devices"])
|
emit("IB", rt["ib_devices"])
|
||||||
emit("PPORT", rt["prefill_port"])
|
emit("PPORT", rt["prefill_port"])
|
||||||
emit("DPORT", rt["decode_port"])
|
emit("DPORT", rt["decode_port"])
|
||||||
@@ -100,16 +112,23 @@ emit("MEMFRAC", rt["mem_fraction_static"])
|
|||||||
emit("PAGE", rt["page_size"])
|
emit("PAGE", rt["page_size"])
|
||||||
emit("MAXREQ", rt["max_running_requests"])
|
emit("MAXREQ", rt["max_running_requests"])
|
||||||
emit("CHUNK", rt["chunked_prefill_size"])
|
emit("CHUNK", rt["chunked_prefill_size"])
|
||||||
emit("SWA", rt["swa_full_tokens_ratio"])
|
# swa is DSV4-specific; emit empty when a model omits it so the flag is dropped.
|
||||||
|
emit("SWA", rt.get("swa_full_tokens_ratio", ""))
|
||||||
|
# 1 when the recipe carries a `model:` block (env + server_args written to
|
||||||
|
# model_flags.sh); 0 for the DSV4 recipes, which keep the hardcoded DSV4 path.
|
||||||
|
emit("HAS_MODEL", 1 if r.get("model") else 0)
|
||||||
emit("PTP", b["prefill"]["tensor-parallel-size"])
|
emit("PTP", b["prefill"]["tensor-parallel-size"])
|
||||||
emit("DTP", b["decode"]["tensor-parallel-size"])
|
emit("DTP", b["decode"]["tensor-parallel-size"])
|
||||||
emit("PEP", b["prefill"].get("expert-parallel-size", 1))
|
emit("PEP", b["prefill"].get("expert-parallel-size", 1))
|
||||||
emit("PDP", b["prefill"].get("data-parallel-size", 1))
|
emit("PDP", b["prefill"].get("data-parallel-size", 1))
|
||||||
m = r.get("mtp", {}) or {}
|
m = r.get("mtp", {}) or {}
|
||||||
emit("MTP_ENABLED", 1 if m.get("enabled") else 0)
|
emit("MTP_ENABLED", 1 if m.get("enabled") else 0)
|
||||||
|
emit("MTP_ALGO", m.get("algorithm", "EAGLE"))
|
||||||
emit("MTP_STEPS", m.get("num_steps", 3))
|
emit("MTP_STEPS", m.get("num_steps", 3))
|
||||||
emit("MTP_TOPK", m.get("eagle_topk", 1))
|
emit("MTP_TOPK", m.get("eagle_topk", 1))
|
||||||
emit("MTP_DRAFT", m.get("num_draft_tokens", 4))
|
emit("MTP_DRAFT", m.get("num_draft_tokens", 4))
|
||||||
|
# External draft checkpoint (EAGLE3 etc.); empty for DSV4's built-in EAGLE head.
|
||||||
|
emit("MTP_DRAFT_PATH", m.get("draft_model_path", ""))
|
||||||
# Worker counts double as node counts here: one server per node (TP == GPUs/node).
|
# Worker counts double as node counts here: one server per node (TP == GPUs/node).
|
||||||
# 1P1D today; bumping these reserves 2P2D / 1P3D / 3P1D. Multi-node-per-worker
|
# 1P1D today; bumping these reserves 2P2D / 1P3D / 3P1D. Multi-node-per-worker
|
||||||
# (TP > GPUs/node, needs --dist-init-addr/--nnodes/--node-rank) is out of scope.
|
# (TP > GPUs/node, needs --dist-init-addr/--nnodes/--node-rank) is out of scope.
|
||||||
@@ -134,7 +153,7 @@ eval "$RECIPE_VARS"
|
|||||||
if [[ -n "${IMAGE_OVERRIDE:-}" ]]; then
|
if [[ -n "${IMAGE_OVERRIDE:-}" ]]; then
|
||||||
IMAGE="$IMAGE_OVERRIDE"
|
IMAGE="$IMAGE_OVERRIDE"
|
||||||
fi
|
fi
|
||||||
echo "recipe: image=$IMAGE attn=$ATTN ib=$IB ptp=$PTP dtp=$DTP concs=$CONCS isl=$ISL osl=$OSL"
|
echo "recipe: image=$IMAGE attn=${ATTN:-$PATTN/$DATTN} ib=$IB ptp=$PTP dtp=$DTP concs=$CONCS isl=$ISL osl=$OSL"
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
# Shared NFS scratch (visible to login node + compute nodes). Raw bench output
|
# Shared NFS scratch (visible to login node + compute nodes). Raw bench output
|
||||||
@@ -182,26 +201,82 @@ DSV4_ENV=(
|
|||||||
-e AITER_BF16_FP8_MOE_BOUND=0 -e SGLANG_DSV4_FP4_EXPERTS=$FP4_EXPERTS
|
-e AITER_BF16_FP8_MOE_BOUND=0 -e SGLANG_DSV4_FP4_EXPERTS=$FP4_EXPERTS
|
||||||
)
|
)
|
||||||
DSV4_ENV_STR="${DSV4_ENV[*]}"
|
DSV4_ENV_STR="${DSV4_ENV[*]}"
|
||||||
|
# A recipe carrying a `model:` block supplies its OWN docker env (below), so the
|
||||||
|
# DSV4 env must not leak into it; the DSV4 recipes keep the string above.
|
||||||
|
[[ "$HAS_MODEL" == "1" ]] && DSV4_ENV_STR=""
|
||||||
MORI_ENV="-e MORI_DISABLE_AUTO_XGMI=1 -e NCCL_IB_HCA=ionic -e NCCL_IB_GID_INDEX=1 -e NCCL_CROSS_NIC=1"
|
MORI_ENV="-e MORI_DISABLE_AUTO_XGMI=1 -e NCCL_IB_HCA=ionic -e NCCL_IB_GID_INDEX=1 -e NCCL_CROSS_NIC=1"
|
||||||
|
|
||||||
|
# Model-specific docker `-e` env + sglang server args from the recipe's optional
|
||||||
|
# `model:` block, written as bash arrays to model_flags.sh (sourced by
|
||||||
|
# prefill.sh/decode.sh). DSV4 recipes have no `model:` block -> empty arrays, so
|
||||||
|
# their generated docker argv is unchanged. Each server arg + its value MUST be a
|
||||||
|
# separate YAML list item so shlex.quote keeps "--foo" and "bar" as two tokens.
|
||||||
|
python3 - "$CONFIG_FILE" "$WORKDIR/model_flags.sh" <<'PY'
|
||||||
|
import shlex, sys, yaml
|
||||||
|
r = yaml.safe_load(open(sys.argv[1]))
|
||||||
|
model = r.get("model", {}) or {}
|
||||||
|
env = model.get("env", {}) or {}
|
||||||
|
server_args = model.get("server_args", []) or []
|
||||||
|
# YAML true/false parse to Python bool; render lowercase so env values stay
|
||||||
|
# byte-identical to shell (`=false`, not `=False`) -- SGLang parsing is
|
||||||
|
# case-sensitive for some of these.
|
||||||
|
def fmt(v):
|
||||||
|
if isinstance(v, bool):
|
||||||
|
return "true" if v else "false"
|
||||||
|
return str(v)
|
||||||
|
env_args = []
|
||||||
|
for k, v in env.items():
|
||||||
|
env_args += ["-e", f"{k}={fmt(v)}"]
|
||||||
|
def q(items):
|
||||||
|
return " ".join(shlex.quote(fmt(x)) for x in items)
|
||||||
|
with open(sys.argv[2], "w") as f:
|
||||||
|
f.write(f"MODEL_ENV_ARGS=({q(env_args)})\n")
|
||||||
|
f.write(f"MODEL_SERVER_ARGS=({q(server_args)})\n")
|
||||||
|
PY
|
||||||
|
|
||||||
# Optional topology / speculative-decode flags driven by the recipe. Base recipes
|
# Optional topology / speculative-decode flags driven by the recipe. Base recipes
|
||||||
# (EP1/DP1, no mtp) leave EXTRA_FLAGS empty, preserving prior behavior exactly.
|
# (EP1/DP1, no mtp) leave EXTRA_FLAGS empty, preserving prior behavior exactly.
|
||||||
EXTRA_FLAGS=""
|
EXTRA_FLAGS=""
|
||||||
(( PDP > 1 )) && EXTRA_FLAGS="$EXTRA_FLAGS --enable-dp-attention --dp-size $PDP"
|
(( PDP > 1 )) && EXTRA_FLAGS="$EXTRA_FLAGS --enable-dp-attention --dp-size $PDP"
|
||||||
(( PEP > 1 )) && EXTRA_FLAGS="$EXTRA_FLAGS --ep-size $PEP"
|
(( PEP > 1 )) && EXTRA_FLAGS="$EXTRA_FLAGS --ep-size $PEP"
|
||||||
if [[ "$MTP_ENABLED" == "1" ]]; then
|
if [[ "$MTP_ENABLED" == "1" ]]; then
|
||||||
EXTRA_FLAGS="$EXTRA_FLAGS --speculative-algorithm EAGLE \
|
EXTRA_FLAGS="$EXTRA_FLAGS --speculative-algorithm $MTP_ALGO \
|
||||||
--speculative-num-steps $MTP_STEPS --speculative-eagle-topk $MTP_TOPK \
|
--speculative-num-steps $MTP_STEPS --speculative-eagle-topk $MTP_TOPK \
|
||||||
--speculative-num-draft-tokens $MTP_DRAFT"
|
--speculative-num-draft-tokens $MTP_DRAFT"
|
||||||
|
# EAGLE3 (and other draft-model algos) need an external draft checkpoint;
|
||||||
|
# built-in EAGLE (DSV4) omits draft_model_path and this stays unset.
|
||||||
|
if [[ -n "$MTP_DRAFT_PATH" ]]; then
|
||||||
|
DRAFT_RESOLVED="$(resolve_snapshot "$MTP_DRAFT_PATH")" || exit 1
|
||||||
|
EXTRA_FLAGS="$EXTRA_FLAGS --speculative-draft-model-path $DRAFT_RESOLVED"
|
||||||
|
fi
|
||||||
fi
|
fi
|
||||||
echo "extra flags: ${EXTRA_FLAGS:-<none>} (pep=$PEP pdp=$PDP mtp=$MTP_ENABLED)"
|
echo "extra flags: ${EXTRA_FLAGS:-<none>} (pep=$PEP pdp=$PDP mtp=$MTP_ENABLED algo=$MTP_ALGO)"
|
||||||
|
|
||||||
COMMON_FLAGS="--trust-remote-code --tp $PTP --disable-radix-cache \
|
if [[ "$HAS_MODEL" == "1" ]]; then
|
||||||
|
# Generic path (e.g. Kimi): attention + swa from the recipe, model parsers /
|
||||||
|
# quirks ride MODEL_SERVER_ARGS. Single `--attention-backend` when the recipe
|
||||||
|
# sets `attention_backend`; split `--prefill-/--decode-attention-backend` when
|
||||||
|
# it sets the per-role keys. swa dropped when the recipe omits it.
|
||||||
|
ATTN_FLAGS=""
|
||||||
|
[[ -n "$ATTN" ]] && ATTN_FLAGS="$ATTN_FLAGS --attention-backend $ATTN"
|
||||||
|
[[ -n "$PATTN" ]] && ATTN_FLAGS="$ATTN_FLAGS --prefill-attention-backend $PATTN"
|
||||||
|
[[ -n "$DATTN" ]] && ATTN_FLAGS="$ATTN_FLAGS --decode-attention-backend $DATTN"
|
||||||
|
SWA_FLAG=""
|
||||||
|
[[ -n "$SWA" ]] && SWA_FLAG=" --swa-full-tokens-ratio $SWA"
|
||||||
|
COMMON_FLAGS="--trust-remote-code --tp $PTP --disable-radix-cache \
|
||||||
|
$ATTN_FLAGS --max-running-requests $MAXREQ --page-size $PAGE \
|
||||||
|
--mem-fraction-static $MEMFRAC$SWA_FLAG \
|
||||||
|
--chunked-prefill-size $CHUNK \
|
||||||
|
--disaggregation-transfer-backend mori --disaggregation-ib-device $IB$EXTRA_FLAGS"
|
||||||
|
else
|
||||||
|
# DSV4 path: byte-identical to the pre-Kimi launcher.
|
||||||
|
COMMON_FLAGS="--trust-remote-code --tp $PTP --disable-radix-cache \
|
||||||
--attention-backend $ATTN --max-running-requests $MAXREQ --page-size $PAGE \
|
--attention-backend $ATTN --max-running-requests $MAXREQ --page-size $PAGE \
|
||||||
--mem-fraction-static $MEMFRAC --swa-full-tokens-ratio $SWA \
|
--mem-fraction-static $MEMFRAC --swa-full-tokens-ratio $SWA \
|
||||||
--chunked-prefill-size $CHUNK --disable-shared-experts-fusion \
|
--chunked-prefill-size $CHUNK --disable-shared-experts-fusion \
|
||||||
--tool-call-parser deepseekv4 --reasoning-parser deepseek-v4 \
|
--tool-call-parser deepseekv4 --reasoning-parser deepseek-v4 \
|
||||||
--disaggregation-transfer-backend mori --disaggregation-ib-device $IB$EXTRA_FLAGS"
|
--disaggregation-transfer-backend mori --disaggregation-ib-device $IB$EXTRA_FLAGS"
|
||||||
|
fi
|
||||||
|
|
||||||
DOCKER_COMMON="--rm --network host --ipc host --shm-size 32g --privileged \
|
DOCKER_COMMON="--rm --network host --ipc host --shm-size 32g --privileged \
|
||||||
--security-opt seccomp=unconfined \
|
--security-opt seccomp=unconfined \
|
||||||
@@ -211,24 +286,34 @@ DOCKER_COMMON="--rm --network host --ipc host --shm-size 32g --privileged \
|
|||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
# Write per-role scripts that srun dispatches to each compute node.
|
# Write per-role scripts that srun dispatches to each compute node.
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
|
# These are UNQUOTED `<<EOF` heredocs, so $MORI_ENV/$DSV4_ENV_STR/$COMMON_FLAGS/
|
||||||
|
# $IMAGE/$MODEL_PATH expand now (at generation). model_flags.sh is sourced at
|
||||||
|
# runtime for the model's env/server arrays, so the `${MODEL_ENV_ARGS[@]}` /
|
||||||
|
# `${MODEL_SERVER_ARGS[@]}` refs are backslash-escaped to survive into the script
|
||||||
|
# and expand after `source`. For DSV4 those arrays are empty and $DSV4_ENV_STR is
|
||||||
|
# set, so the resulting docker argv is byte-identical to the pre-Kimi launcher.
|
||||||
cat > "$WORKDIR/prefill.sh" <<EOF
|
cat > "$WORKDIR/prefill.sh" <<EOF
|
||||||
#!/bin/bash
|
#!/bin/bash
|
||||||
|
source "$WORKDIR/model_flags.sh"
|
||||||
docker rm -f mi355x_prefill 2>/dev/null || true
|
docker rm -f mi355x_prefill 2>/dev/null || true
|
||||||
docker run $DOCKER_COMMON --name mi355x_prefill \
|
docker run $DOCKER_COMMON --name mi355x_prefill \
|
||||||
-e HIP_VISIBLE_DEVICES=0,1,2,3,4,5,6,7 $MORI_ENV $DSV4_ENV_STR \
|
-e HIP_VISIBLE_DEVICES=0,1,2,3,4,5,6,7 $MORI_ENV $DSV4_ENV_STR "\${MODEL_ENV_ARGS[@]}" \
|
||||||
$IMAGE python3 -m sglang.launch_server \
|
$IMAGE python3 -m sglang.launch_server \
|
||||||
--model-path $MODEL_PATH --host 0.0.0.0 --port $PPORT \
|
--model-path $MODEL_PATH --host 0.0.0.0 --port $PPORT \
|
||||||
$COMMON_FLAGS --disaggregation-mode prefill --disaggregation-bootstrap-port $PBOOT
|
$COMMON_FLAGS "\${MODEL_SERVER_ARGS[@]}" \
|
||||||
|
--disaggregation-mode prefill --disaggregation-bootstrap-port $PBOOT
|
||||||
EOF
|
EOF
|
||||||
|
|
||||||
cat > "$WORKDIR/decode.sh" <<EOF
|
cat > "$WORKDIR/decode.sh" <<EOF
|
||||||
#!/bin/bash
|
#!/bin/bash
|
||||||
|
source "$WORKDIR/model_flags.sh"
|
||||||
docker rm -f mi355x_decode 2>/dev/null || true
|
docker rm -f mi355x_decode 2>/dev/null || true
|
||||||
docker run $DOCKER_COMMON --name mi355x_decode \
|
docker run $DOCKER_COMMON --name mi355x_decode \
|
||||||
-e HIP_VISIBLE_DEVICES=0,1,2,3,4,5,6,7 $MORI_ENV $DSV4_ENV_STR \
|
-e HIP_VISIBLE_DEVICES=0,1,2,3,4,5,6,7 $MORI_ENV $DSV4_ENV_STR "\${MODEL_ENV_ARGS[@]}" \
|
||||||
$IMAGE python3 -m sglang.launch_server \
|
$IMAGE python3 -m sglang.launch_server \
|
||||||
--model-path $MODEL_PATH --host 0.0.0.0 --port $DPORT \
|
--model-path $MODEL_PATH --host 0.0.0.0 --port $DPORT \
|
||||||
$COMMON_FLAGS --disaggregation-mode decode --disaggregation-bootstrap-port $DBOOT
|
$COMMON_FLAGS "\${MODEL_SERVER_ARGS[@]}" \
|
||||||
|
--disaggregation-mode decode --disaggregation-bootstrap-port $DBOOT
|
||||||
EOF
|
EOF
|
||||||
|
|
||||||
# Probe payload + validator (separate files to avoid quoting inside the
|
# Probe payload + validator (separate files to avoid quoting inside the
|
||||||
|
|||||||
@@ -320,3 +320,39 @@ dsv4pro-fp4-mi355x-dp8ep8-mtp-sglang:
|
|||||||
search-space:
|
search-space:
|
||||||
- conc-list: [1, 8, 16, 32, 64, 128, 256]
|
- conc-list: [1, 8, 16, 32, 64, 128, 256]
|
||||||
config_file: scripts/ci/slurm/recipes/mi355x-fp4/dsv4pro/1k1k/1p1d-dp8ep8-mtp.yaml
|
config_file: scripts/ci/slurm/recipes/mi355x-fp4/dsv4pro/1k1k/1p1d-dp8ep8-mtp.yaml
|
||||||
|
|
||||||
|
# Kimi-K2.6 (FP8) 2-node 1P1D. Demonstrates the launcher is model-agnostic: all
|
||||||
|
# Kimi-specific config lives in the recipe's `model:` block + split attention
|
||||||
|
# runtime, with nothing hardcoded in launch_mi355x.sh. Base + EAGLE3 MTP (the
|
||||||
|
# MTP leg uses an external draft checkpoint via mtp.draft_model_path).
|
||||||
|
kimik26-fp8-mi355x-sglang:
|
||||||
|
model: moonshotai/Kimi-K2.6
|
||||||
|
model-prefix: kimik26
|
||||||
|
model_path: /it-share/model_coverage/models--moonshotai--Kimi-K2.6
|
||||||
|
runner: mi355x
|
||||||
|
precision: fp8
|
||||||
|
framework: sglang
|
||||||
|
multinode: true
|
||||||
|
disagg: true
|
||||||
|
seq-len-configs:
|
||||||
|
- isl: 1024
|
||||||
|
osl: 1024
|
||||||
|
search-space:
|
||||||
|
- conc-list: [1, 8, 16, 32, 64, 128, 256]
|
||||||
|
config_file: scripts/ci/slurm/recipes/mi355x-fp8/kimik26/1k1k/1p1d.yaml
|
||||||
|
|
||||||
|
kimik26-fp8-mi355x-mtp-sglang:
|
||||||
|
model: moonshotai/Kimi-K2.6
|
||||||
|
model-prefix: kimik26
|
||||||
|
model_path: /it-share/model_coverage/models--moonshotai--Kimi-K2.6
|
||||||
|
runner: mi355x
|
||||||
|
precision: fp8
|
||||||
|
framework: sglang
|
||||||
|
multinode: true
|
||||||
|
disagg: true
|
||||||
|
seq-len-configs:
|
||||||
|
- isl: 1024
|
||||||
|
osl: 1024
|
||||||
|
search-space:
|
||||||
|
- conc-list: [1, 8, 16, 32, 64, 128, 256]
|
||||||
|
config_file: scripts/ci/slurm/recipes/mi355x-fp8/kimik26/1k1k/1p1d-mtp.yaml
|
||||||
|
|||||||
@@ -0,0 +1,85 @@
|
|||||||
|
# MI355X Kimi-K2.6 (FP8) 2-node 1P1D disaggregation recipe (base + EAGLE3 MTP).
|
||||||
|
#
|
||||||
|
# Self-contained (no inheritance): same as 1p1d.yaml plus the `mtp:` block. Uses
|
||||||
|
# EAGLE3 speculative decoding with an EXTERNAL draft checkpoint (unlike DSV4's
|
||||||
|
# built-in EAGLE NextN head): the launcher resolves mtp.draft_model_path through
|
||||||
|
# the HF-cache snapshot logic and appends --speculative-draft-model-path. The
|
||||||
|
# draft dir must live under /it-share (the container's :ro mount). MTP is applied
|
||||||
|
# to both prefill and decode. Mirrors the Kimi-K2.6 serving cookbook.
|
||||||
|
#
|
||||||
|
# Consumed by:
|
||||||
|
# * scripts/ci/slurm/process_result.py reads `resources` and
|
||||||
|
# `backend.sglang_config` (TP/EP/DP + worker counts) for the summary table.
|
||||||
|
# * scripts/ci/slurm/launch_mi355x.sh reads `runtime`, `bench`, `model`, `mtp`.
|
||||||
|
|
||||||
|
resources:
|
||||||
|
prefill_workers: 1
|
||||||
|
decode_workers: 1
|
||||||
|
|
||||||
|
backend:
|
||||||
|
sglang_config:
|
||||||
|
prefill:
|
||||||
|
tensor-parallel-size: 8
|
||||||
|
expert-parallel-size: 1
|
||||||
|
data-parallel-size: 1
|
||||||
|
decode:
|
||||||
|
tensor-parallel-size: 8
|
||||||
|
expert-parallel-size: 1
|
||||||
|
data-parallel-size: 1
|
||||||
|
|
||||||
|
# Model-specific docker env + sglang server args (written verbatim via
|
||||||
|
# model_flags.sh). Each server arg + value is a SEPARATE list item.
|
||||||
|
model:
|
||||||
|
env:
|
||||||
|
SGLANG_USE_AITER: 1
|
||||||
|
SGLANG_ROCM_FUSED_DECODE_MLA: 0
|
||||||
|
server_args:
|
||||||
|
- --model-loader-extra-config
|
||||||
|
- '{"enable_multithread_load": true}'
|
||||||
|
- --watchdog-timeout
|
||||||
|
- 1200
|
||||||
|
- --reasoning-parser
|
||||||
|
- kimi_k2
|
||||||
|
- --tool-call-parser
|
||||||
|
- kimi_k2
|
||||||
|
|
||||||
|
# EAGLE3 speculative decoding with an external draft checkpoint.
|
||||||
|
mtp:
|
||||||
|
enabled: true
|
||||||
|
algorithm: EAGLE3
|
||||||
|
num_steps: 3
|
||||||
|
eagle_topk: 1
|
||||||
|
num_draft_tokens: 4
|
||||||
|
draft_model_path: /it-share/model_coverage/models--lightseekorg--kimi-k2.6-eagle3.1-mla
|
||||||
|
|
||||||
|
runtime:
|
||||||
|
image: lmsysorg/sglang-rocm:v0.5.13.post1-rocm720-mi35x-20260623
|
||||||
|
# Kimi uses split attention backends (aiter prefill / triton decode), not a
|
||||||
|
# single --attention-backend.
|
||||||
|
prefill_attention_backend: aiter
|
||||||
|
decode_attention_backend: triton
|
||||||
|
# RoCE HCAs MORI uses for cross-node KV transfer.
|
||||||
|
ib_devices: rdma0,rdma1,rdma2,rdma3
|
||||||
|
prefill_port: 30025
|
||||||
|
decode_port: 30026
|
||||||
|
prefill_bootstrap_port: 8998
|
||||||
|
decode_bootstrap_port: 9001
|
||||||
|
lb_port: 8000
|
||||||
|
mem_fraction_static: 0.90
|
||||||
|
page_size: 256
|
||||||
|
max_running_requests: 256
|
||||||
|
chunked_prefill_size: 8192
|
||||||
|
|
||||||
|
bench:
|
||||||
|
# bench_serving --max-concurrency sweep; one result JSON per concurrency.
|
||||||
|
concurrencies: [1, 8, 16, 32, 64, 128, 256]
|
||||||
|
num_prompts_factor: 4 # num-prompts = concurrency * factor
|
||||||
|
random_range_ratio: 1.0
|
||||||
|
|
||||||
|
# Correctness gate run through the PD path before the perf sweep. Mirrors the
|
||||||
|
# registered single-node Kimi-K2.6 eval (full GSM8K, 8-shot, accuracy > 0.92).
|
||||||
|
accuracy:
|
||||||
|
enabled: true
|
||||||
|
num_shots: 8
|
||||||
|
num_questions: 1319 # full GSM8K test set
|
||||||
|
threshold: 0.92
|
||||||
@@ -0,0 +1,75 @@
|
|||||||
|
# MI355X Kimi-K2.6 (FP8) 2-node 1P1D disaggregation recipe (base).
|
||||||
|
#
|
||||||
|
# All Kimi-specific config lives in this recipe's `model:` block (docker env +
|
||||||
|
# sglang server args) and `runtime` (split attention backends); nothing about
|
||||||
|
# Kimi is hardcoded in launch_mi355x.sh. Mirrors the single-node registered test
|
||||||
|
# test/registered/amd/accuracy/mi35x/test_kimi_k26_eval_mi35x.py (TP8, split
|
||||||
|
# attention backends, multithread loader, GSM8K > 0.92).
|
||||||
|
#
|
||||||
|
# Consumed by:
|
||||||
|
# * scripts/ci/slurm/process_result.py reads `resources` and
|
||||||
|
# `backend.sglang_config` (TP/EP/DP + worker counts) for the summary table.
|
||||||
|
# * scripts/ci/slurm/launch_mi355x.sh reads `runtime`, `bench`, `model`, `mtp`.
|
||||||
|
|
||||||
|
resources:
|
||||||
|
prefill_workers: 1
|
||||||
|
decode_workers: 1
|
||||||
|
|
||||||
|
backend:
|
||||||
|
sglang_config:
|
||||||
|
prefill:
|
||||||
|
tensor-parallel-size: 8
|
||||||
|
expert-parallel-size: 1
|
||||||
|
data-parallel-size: 1
|
||||||
|
decode:
|
||||||
|
tensor-parallel-size: 8
|
||||||
|
expert-parallel-size: 1
|
||||||
|
data-parallel-size: 1
|
||||||
|
|
||||||
|
# Model-specific docker env + sglang server args (written verbatim via
|
||||||
|
# model_flags.sh). Each server arg + value is a SEPARATE list item.
|
||||||
|
model:
|
||||||
|
env:
|
||||||
|
SGLANG_USE_AITER: 1
|
||||||
|
SGLANG_ROCM_FUSED_DECODE_MLA: 0
|
||||||
|
server_args:
|
||||||
|
- --model-loader-extra-config
|
||||||
|
- '{"enable_multithread_load": true}'
|
||||||
|
- --watchdog-timeout
|
||||||
|
- 1200
|
||||||
|
- --reasoning-parser
|
||||||
|
- kimi_k2
|
||||||
|
- --tool-call-parser
|
||||||
|
- kimi_k2
|
||||||
|
|
||||||
|
runtime:
|
||||||
|
image: lmsysorg/sglang-rocm:v0.5.13.post1-rocm720-mi35x-20260623
|
||||||
|
# Kimi uses split attention backends (aiter prefill / triton decode), not a
|
||||||
|
# single --attention-backend.
|
||||||
|
prefill_attention_backend: aiter
|
||||||
|
decode_attention_backend: triton
|
||||||
|
# RoCE HCAs MORI uses for cross-node KV transfer.
|
||||||
|
ib_devices: rdma0,rdma1,rdma2,rdma3
|
||||||
|
prefill_port: 30025
|
||||||
|
decode_port: 30026
|
||||||
|
prefill_bootstrap_port: 8998
|
||||||
|
decode_bootstrap_port: 9001
|
||||||
|
lb_port: 8000
|
||||||
|
mem_fraction_static: 0.90
|
||||||
|
page_size: 256
|
||||||
|
max_running_requests: 256
|
||||||
|
chunked_prefill_size: 8192
|
||||||
|
|
||||||
|
bench:
|
||||||
|
# bench_serving --max-concurrency sweep; one result JSON per concurrency.
|
||||||
|
concurrencies: [1, 8, 16, 32, 64, 128, 256]
|
||||||
|
num_prompts_factor: 4 # num-prompts = concurrency * factor
|
||||||
|
random_range_ratio: 1.0
|
||||||
|
|
||||||
|
# Correctness gate run through the PD path before the perf sweep. Mirrors the
|
||||||
|
# registered single-node Kimi-K2.6 eval (full GSM8K, 8-shot, accuracy > 0.92).
|
||||||
|
accuracy:
|
||||||
|
enabled: true
|
||||||
|
num_shots: 8
|
||||||
|
num_questions: 1319 # full GSM8K test set
|
||||||
|
threshold: 0.92
|
||||||
Reference in New Issue
Block a user