[AMD] Add GLM-5.3-Flash recipes for MI300X, MI325X, and MI355X (#36608)

This commit is contained in:
andyluo7
2026-08-27 05:19:57 +00:00
committed by GitHub
parent f775db03aa
commit 0f7b5b8b2a
3 changed files with 112 additions and 12 deletions
@@ -144,4 +144,19 @@ export const benchmarks = [
{ match: { hw: "b300", strategy: "high-throughput" } },
{ match: { hw: "gb200", strategy: "low-latency" } },
{ match: { hw: "gb200", strategy: "high-throughput" } },
{
match: { hw: "mi300x", strategy: "high-throughput" },
sglang_version: "9e692c9216",
accuracy: { gsm8k_pct: 97.35 },
notes:
"Accuracy-only validation on 8x MI300X (gfx942, TP8) with zai-org/GLM-5.3-Flash revision 3f1971b7b5f7a528c9c4ef6212c8785298a8c24a, SGLang PR #36607 head 9e692c9216c3b5e5c443fecf6b995700eb68d2e4 (validated source manifest 2c240e0e01d5fdf04acc485ebfa25f8a1793ba45fb07f165eecedfba7ec1db80), and lmsysorg/sglang:v0.5.18-rocm720-mi30x with the PR source mounted over the image tree. Full GSM8K scored 1,284/1,319 with a 100% stop rate and zero request errors, empty generations, or truncations. No throughput or latency benchmark was run.",
},
{ match: { hw: "mi325x", strategy: "high-throughput" } },
{
match: { hw: "mi355x", strategy: "high-throughput" },
sglang_version: "9e692c9216",
accuracy: { gsm8k_pct: 97.65 },
notes:
"Accuracy-only validation on 8x MI355X (gfx950, TP8) with zai-org/GLM-5.3-Flash revision 3f1971b7b5f7a528c9c4ef6212c8785298a8c24a, SGLang PR #36607 head 9e692c9216c3b5e5c443fecf6b995700eb68d2e4 (validated source manifest 2c240e0e01d5fdf04acc485ebfa25f8a1793ba45fb07f165eecedfba7ec1db80), and lmsysorg/sglang:v0.5.18-rocm720-mi35x with the PR source mounted over the image tree. Full GSM8K scored 1,288/1,319 with a 100% stop rate and zero request errors, empty generations, or truncations. No throughput or latency benchmark was run.",
},
];
@@ -1,21 +1,32 @@
export const config = {
modelName: "GLM-5.3-Flash",
supportedHardware: ["gb300", "h100", "h200", "b200", "b300", "gb200"],
supportedHardware: [
"gb300", "h100", "h200", "b200", "b300", "gb200",
"mi300x", "mi325x", "mi355x",
],
matchDims: [
{
id: "strategy",
title: "Strategy",
options: [
{ id: "low-latency", label: "Low Latency", subtitle: "Adaptive MTP 5/1/6" },
{
id: "low-latency",
label: "Low Latency",
subtitle: "Adaptive MTP 5/1/6",
disabled: (s) => ["mi300x", "mi325x", "mi355x"].includes(s.hw),
disableReason: "MTP speculative decoding has not been validated for GLM-5.3-Flash on AMD ROCm; use the non-speculative High Throughput recipe.",
},
{ id: "high-throughput", label: "High Throughput", subtitle: "Spec decode off" },
],
},
],
isRecommendedSelection(s) {
const pairing = ["h100", "h200"].includes(s.hw) ? "bf16-tilelang" : "fp8-trtllm";
const pairing = ["h100", "h200", "mi300x", "mi325x", "mi355x"].includes(s.hw)
? "bf16-tilelang"
: "fp8-trtllm";
return (
s.kvDsaPair === pairing &&
s.mmTransport === "auto" &&
@@ -32,8 +43,8 @@ export const config = {
{
id: "fp8-trtllm",
label: "FP8 + TRT-LLM",
disabled: (s) => ["h100", "h200"].includes(s.hw),
disableReason: "FP8 KV cache with TRT-LLM DSA is not supported on Hopper GPUs.",
disabled: (s) => ["h100", "h200", "mi300x", "mi325x", "mi355x"].includes(s.hw),
disableReason: "This recipe uses BF16 KV cache with TileLang DSA on Hopper and AMD ROCm GPUs.",
stripPrefixes: ["--kv-cache-dtype", "--dsa-prefill-backend", "--dsa-decode-backend"],
flags: [
"--kv-cache-dtype fp8_e4m3",
@@ -147,8 +158,9 @@ sgl-eval run gsm8k \\
["gsm8k_pct", "GSM8K", "%"],
],
// Support is not in a public sglang release yet, so the nightly images do
// not work; every NVIDIA lane uses the purpose-built CUDA 13 image.
// Support is not in a public sglang release yet. NVIDIA uses the
// purpose-built CUDA 13 image. AMD validation used these ROCm 7.2 images
// with the GLM-5.3 ROCm engine branch mounted over the image source tree.
dockerImages: {
gb300: "lmsysorg/sglang:glm-5.3-flash",
h100: "lmsysorg/sglang:glm-5.3-flash",
@@ -156,6 +168,9 @@ sgl-eval run gsm8k \\
b200: "lmsysorg/sglang:glm-5.3-flash",
b300: "lmsysorg/sglang:glm-5.3-flash",
gb200: "lmsysorg/sglang:glm-5.3-flash",
mi300x: "lmsysorg/sglang:v0.5.18-rocm720-mi30x",
mi325x: "lmsysorg/sglang:v0.5.18-rocm720-mi30x",
mi355x: "lmsysorg/sglang:v0.5.18-rocm720-mi35x",
},
github: {
@@ -213,6 +228,8 @@ sgl-eval run gsm8k \\
id: "deep_gemm",
label: "DeepGemm",
flags: ["--moe-runner-backend deep_gemm"],
disabled: (s) => ["mi300x", "mi325x", "mi355x"].includes(s.hw),
disableReason: "The validated AMD ROCm recipe uses the Triton MoE runner.",
},
],
},
@@ -525,5 +542,71 @@ sgl-eval run gsm8k \\
"--port {{PORT}}",
],
},
// AMD ROCm — one non-speculative TP8 operating point. The explicit BF16
// KV and TileLang DSA flags match the resolved defaults observed in the
// validation server logs. AITER remains enabled for the ROCm kernel paths,
// while Triton owns the MoE runner. CUDA graphs stay disabled because that
// is the architecture-gated configuration used for correctness validation.
{
match: { hw: "mi300x", strategy: "high-throughput" },
nnodes: 1,
verified: true,
env: ["SGLANG_USE_AITER=1"],
flags: [
"--model-path {{MODEL_NAME}}",
"--tp-size 8",
"--trust-remote-code",
"--disable-cuda-graph",
"--dsa-prefill-backend tilelang",
"--dsa-decode-backend tilelang",
"--kv-cache-dtype bfloat16",
"--moe-runner-backend triton",
"--reasoning-parser glm45",
"--tool-call-parser glm47",
"--host {{HOST_IP}}",
"--port {{PORT}}",
],
},
{
match: { hw: "mi325x", strategy: "high-throughput" },
nnodes: 1,
verified: false,
warn: "This MI325X recipe is inferred from the validated MI300X gfx942 path. It has not been measured directly on MI325X.",
env: ["SGLANG_USE_AITER=1"],
flags: [
"--model-path {{MODEL_NAME}}",
"--tp-size 8",
"--trust-remote-code",
"--disable-cuda-graph",
"--dsa-prefill-backend tilelang",
"--dsa-decode-backend tilelang",
"--kv-cache-dtype bfloat16",
"--moe-runner-backend triton",
"--reasoning-parser glm45",
"--tool-call-parser glm47",
"--host {{HOST_IP}}",
"--port {{PORT}}",
],
},
{
match: { hw: "mi355x", strategy: "high-throughput" },
nnodes: 1,
verified: true,
env: ["SGLANG_USE_AITER=1"],
flags: [
"--model-path {{MODEL_NAME}}",
"--tp-size 8",
"--trust-remote-code",
"--disable-cuda-graph",
"--dsa-prefill-backend tilelang",
"--dsa-decode-backend tilelang",
"--kv-cache-dtype bfloat16",
"--moe-runner-backend triton",
"--reasoning-parser glm45",
"--tool-call-parser glm47",
"--host {{HOST_IP}}",
"--port {{PORT}}",
],
},
],
};