[AMD] Add GLM-5.3-Flash recipes for MI300X, MI325X, and MI355X (#36608)
This commit is contained in:
@@ -144,4 +144,19 @@ export const benchmarks = [
|
||||
{ match: { hw: "b300", strategy: "high-throughput" } },
|
||||
{ match: { hw: "gb200", strategy: "low-latency" } },
|
||||
{ match: { hw: "gb200", strategy: "high-throughput" } },
|
||||
{
|
||||
match: { hw: "mi300x", strategy: "high-throughput" },
|
||||
sglang_version: "9e692c9216",
|
||||
accuracy: { gsm8k_pct: 97.35 },
|
||||
notes:
|
||||
"Accuracy-only validation on 8x MI300X (gfx942, TP8) with zai-org/GLM-5.3-Flash revision 3f1971b7b5f7a528c9c4ef6212c8785298a8c24a, SGLang PR #36607 head 9e692c9216c3b5e5c443fecf6b995700eb68d2e4 (validated source manifest 2c240e0e01d5fdf04acc485ebfa25f8a1793ba45fb07f165eecedfba7ec1db80), and lmsysorg/sglang:v0.5.18-rocm720-mi30x with the PR source mounted over the image tree. Full GSM8K scored 1,284/1,319 with a 100% stop rate and zero request errors, empty generations, or truncations. No throughput or latency benchmark was run.",
|
||||
},
|
||||
{ match: { hw: "mi325x", strategy: "high-throughput" } },
|
||||
{
|
||||
match: { hw: "mi355x", strategy: "high-throughput" },
|
||||
sglang_version: "9e692c9216",
|
||||
accuracy: { gsm8k_pct: 97.65 },
|
||||
notes:
|
||||
"Accuracy-only validation on 8x MI355X (gfx950, TP8) with zai-org/GLM-5.3-Flash revision 3f1971b7b5f7a528c9c4ef6212c8785298a8c24a, SGLang PR #36607 head 9e692c9216c3b5e5c443fecf6b995700eb68d2e4 (validated source manifest 2c240e0e01d5fdf04acc485ebfa25f8a1793ba45fb07f165eecedfba7ec1db80), and lmsysorg/sglang:v0.5.18-rocm720-mi35x with the PR source mounted over the image tree. Full GSM8K scored 1,288/1,319 with a 100% stop rate and zero request errors, empty generations, or truncations. No throughput or latency benchmark was run.",
|
||||
},
|
||||
];
|
||||
|
||||
@@ -1,21 +1,32 @@
|
||||
export const config = {
|
||||
modelName: "GLM-5.3-Flash",
|
||||
|
||||
supportedHardware: ["gb300", "h100", "h200", "b200", "b300", "gb200"],
|
||||
supportedHardware: [
|
||||
"gb300", "h100", "h200", "b200", "b300", "gb200",
|
||||
"mi300x", "mi325x", "mi355x",
|
||||
],
|
||||
|
||||
matchDims: [
|
||||
{
|
||||
id: "strategy",
|
||||
title: "Strategy",
|
||||
options: [
|
||||
{ id: "low-latency", label: "Low Latency", subtitle: "Adaptive MTP 5/1/6" },
|
||||
{
|
||||
id: "low-latency",
|
||||
label: "Low Latency",
|
||||
subtitle: "Adaptive MTP 5/1/6",
|
||||
disabled: (s) => ["mi300x", "mi325x", "mi355x"].includes(s.hw),
|
||||
disableReason: "MTP speculative decoding has not been validated for GLM-5.3-Flash on AMD ROCm; use the non-speculative High Throughput recipe.",
|
||||
},
|
||||
{ id: "high-throughput", label: "High Throughput", subtitle: "Spec decode off" },
|
||||
],
|
||||
},
|
||||
],
|
||||
|
||||
isRecommendedSelection(s) {
|
||||
const pairing = ["h100", "h200"].includes(s.hw) ? "bf16-tilelang" : "fp8-trtllm";
|
||||
const pairing = ["h100", "h200", "mi300x", "mi325x", "mi355x"].includes(s.hw)
|
||||
? "bf16-tilelang"
|
||||
: "fp8-trtllm";
|
||||
return (
|
||||
s.kvDsaPair === pairing &&
|
||||
s.mmTransport === "auto" &&
|
||||
@@ -32,8 +43,8 @@ export const config = {
|
||||
{
|
||||
id: "fp8-trtllm",
|
||||
label: "FP8 + TRT-LLM",
|
||||
disabled: (s) => ["h100", "h200"].includes(s.hw),
|
||||
disableReason: "FP8 KV cache with TRT-LLM DSA is not supported on Hopper GPUs.",
|
||||
disabled: (s) => ["h100", "h200", "mi300x", "mi325x", "mi355x"].includes(s.hw),
|
||||
disableReason: "This recipe uses BF16 KV cache with TileLang DSA on Hopper and AMD ROCm GPUs.",
|
||||
stripPrefixes: ["--kv-cache-dtype", "--dsa-prefill-backend", "--dsa-decode-backend"],
|
||||
flags: [
|
||||
"--kv-cache-dtype fp8_e4m3",
|
||||
@@ -147,8 +158,9 @@ sgl-eval run gsm8k \\
|
||||
["gsm8k_pct", "GSM8K", "%"],
|
||||
],
|
||||
|
||||
// Support is not in a public sglang release yet, so the nightly images do
|
||||
// not work; every NVIDIA lane uses the purpose-built CUDA 13 image.
|
||||
// Support is not in a public sglang release yet. NVIDIA uses the
|
||||
// purpose-built CUDA 13 image. AMD validation used these ROCm 7.2 images
|
||||
// with the GLM-5.3 ROCm engine branch mounted over the image source tree.
|
||||
dockerImages: {
|
||||
gb300: "lmsysorg/sglang:glm-5.3-flash",
|
||||
h100: "lmsysorg/sglang:glm-5.3-flash",
|
||||
@@ -156,6 +168,9 @@ sgl-eval run gsm8k \\
|
||||
b200: "lmsysorg/sglang:glm-5.3-flash",
|
||||
b300: "lmsysorg/sglang:glm-5.3-flash",
|
||||
gb200: "lmsysorg/sglang:glm-5.3-flash",
|
||||
mi300x: "lmsysorg/sglang:v0.5.18-rocm720-mi30x",
|
||||
mi325x: "lmsysorg/sglang:v0.5.18-rocm720-mi30x",
|
||||
mi355x: "lmsysorg/sglang:v0.5.18-rocm720-mi35x",
|
||||
},
|
||||
|
||||
github: {
|
||||
@@ -213,6 +228,8 @@ sgl-eval run gsm8k \\
|
||||
id: "deep_gemm",
|
||||
label: "DeepGemm",
|
||||
flags: ["--moe-runner-backend deep_gemm"],
|
||||
disabled: (s) => ["mi300x", "mi325x", "mi355x"].includes(s.hw),
|
||||
disableReason: "The validated AMD ROCm recipe uses the Triton MoE runner.",
|
||||
},
|
||||
],
|
||||
},
|
||||
@@ -525,5 +542,71 @@ sgl-eval run gsm8k \\
|
||||
"--port {{PORT}}",
|
||||
],
|
||||
},
|
||||
// AMD ROCm — one non-speculative TP8 operating point. The explicit BF16
|
||||
// KV and TileLang DSA flags match the resolved defaults observed in the
|
||||
// validation server logs. AITER remains enabled for the ROCm kernel paths,
|
||||
// while Triton owns the MoE runner. CUDA graphs stay disabled because that
|
||||
// is the architecture-gated configuration used for correctness validation.
|
||||
{
|
||||
match: { hw: "mi300x", strategy: "high-throughput" },
|
||||
nnodes: 1,
|
||||
verified: true,
|
||||
env: ["SGLANG_USE_AITER=1"],
|
||||
flags: [
|
||||
"--model-path {{MODEL_NAME}}",
|
||||
"--tp-size 8",
|
||||
"--trust-remote-code",
|
||||
"--disable-cuda-graph",
|
||||
"--dsa-prefill-backend tilelang",
|
||||
"--dsa-decode-backend tilelang",
|
||||
"--kv-cache-dtype bfloat16",
|
||||
"--moe-runner-backend triton",
|
||||
"--reasoning-parser glm45",
|
||||
"--tool-call-parser glm47",
|
||||
"--host {{HOST_IP}}",
|
||||
"--port {{PORT}}",
|
||||
],
|
||||
},
|
||||
{
|
||||
match: { hw: "mi325x", strategy: "high-throughput" },
|
||||
nnodes: 1,
|
||||
verified: false,
|
||||
warn: "This MI325X recipe is inferred from the validated MI300X gfx942 path. It has not been measured directly on MI325X.",
|
||||
env: ["SGLANG_USE_AITER=1"],
|
||||
flags: [
|
||||
"--model-path {{MODEL_NAME}}",
|
||||
"--tp-size 8",
|
||||
"--trust-remote-code",
|
||||
"--disable-cuda-graph",
|
||||
"--dsa-prefill-backend tilelang",
|
||||
"--dsa-decode-backend tilelang",
|
||||
"--kv-cache-dtype bfloat16",
|
||||
"--moe-runner-backend triton",
|
||||
"--reasoning-parser glm45",
|
||||
"--tool-call-parser glm47",
|
||||
"--host {{HOST_IP}}",
|
||||
"--port {{PORT}}",
|
||||
],
|
||||
},
|
||||
{
|
||||
match: { hw: "mi355x", strategy: "high-throughput" },
|
||||
nnodes: 1,
|
||||
verified: true,
|
||||
env: ["SGLANG_USE_AITER=1"],
|
||||
flags: [
|
||||
"--model-path {{MODEL_NAME}}",
|
||||
"--tp-size 8",
|
||||
"--trust-remote-code",
|
||||
"--disable-cuda-graph",
|
||||
"--dsa-prefill-backend tilelang",
|
||||
"--dsa-decode-backend tilelang",
|
||||
"--kv-cache-dtype bfloat16",
|
||||
"--moe-runner-backend triton",
|
||||
"--reasoning-parser glm45",
|
||||
"--tool-call-parser glm47",
|
||||
"--host {{HOST_IP}}",
|
||||
"--port {{PORT}}",
|
||||
],
|
||||
},
|
||||
],
|
||||
};
|
||||
|
||||
Reference in New Issue
Block a user