[AMD] [GLM5] GLM-5.1 MXFP4 (MI355X) + enable EAGLE for gfx950 in cookbook (#29194)

Co-authored-by: Raiden-Makoto <Raiden-Makoto@users.noreply.github.com>
This commit is contained in:
Raiden Makoto
2026-06-25 03:54:48 -07:00
committed by GitHub
co-authored by Raiden-Makoto
parent ddda4f9028
commit 0075c8f02b
2 changed files with 53 additions and 17 deletions
@@ -43,6 +43,7 @@ import { GLM51Deployment } from '/src/snippets/autoregressive/glm-51-deployment.
<th style={{textAlign: "left", padding: "10px 12px", fontWeight: 700, whiteSpace: "nowrap", backgroundColor: "rgba(255,255,255,0.05)"}}>NVFP4</th> <th style={{textAlign: "left", padding: "10px 12px", fontWeight: 700, whiteSpace: "nowrap", backgroundColor: "rgba(255,255,255,0.05)"}}>NVFP4</th>
<th style={{textAlign: "left", padding: "10px 12px", fontWeight: 700, whiteSpace: "nowrap", backgroundColor: "rgba(255,255,255,0.02)"}}>FP8</th> <th style={{textAlign: "left", padding: "10px 12px", fontWeight: 700, whiteSpace: "nowrap", backgroundColor: "rgba(255,255,255,0.02)"}}>FP8</th>
<th style={{textAlign: "left", padding: "10px 12px", fontWeight: 700, whiteSpace: "nowrap", backgroundColor: "rgba(255,255,255,0.05)"}}>BF16</th> <th style={{textAlign: "left", padding: "10px 12px", fontWeight: 700, whiteSpace: "nowrap", backgroundColor: "rgba(255,255,255,0.05)"}}>BF16</th>
<th style={{textAlign: "left", padding: "10px 12px", fontWeight: 700, whiteSpace: "nowrap", backgroundColor: "rgba(255,255,255,0.02)"}}>MXFP4</th>
</tr> </tr>
</thead> </thead>
<tbody> <tbody>
@@ -51,43 +52,49 @@ import { GLM51Deployment } from '/src/snippets/autoregressive/glm-51-deployment.
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>—</td> <td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>—</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.02)"}}>tp=16</td> <td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.02)"}}>tp=16</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>—</td> <td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>—</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.02)"}}>—</td>
</tr> </tr>
<tr> <tr>
<td style={{padding: "9px 12px", fontWeight: 500, backgroundColor: "rgba(255,255,255,0.02)"}}>H200</td> <td style={{padding: "9px 12px", fontWeight: 500, backgroundColor: "rgba(255,255,255,0.02)"}}>H200</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>—</td> <td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>—</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.02)"}}>tp=8</td> <td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.02)"}}>tp=8</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>—</td> <td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>—</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.02)"}}>—</td>
</tr> </tr>
<tr> <tr>
<td style={{padding: "9px 12px", fontWeight: 500, backgroundColor: "rgba(255,255,255,0.02)"}}>B300</td> <td style={{padding: "9px 12px", fontWeight: 500, backgroundColor: "rgba(255,255,255,0.02)"}}>B300</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>tp=8</td> <td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>tp=8</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.02)"}}>—</td> <td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.02)"}}>—</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>—</td> <td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>—</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.02)"}}>—</td>
</tr> </tr>
<tr> <tr>
<td style={{padding: "9px 12px", fontWeight: 500, backgroundColor: "rgba(255,255,255,0.02)"}}>GB300</td> <td style={{padding: "9px 12px", fontWeight: 500, backgroundColor: "rgba(255,255,255,0.02)"}}>GB300</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>tp=4</td> <td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>tp=4</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.02)"}}>—</td> <td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.02)"}}>—</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>—</td> <td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>—</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.02)"}}>—</td>
</tr> </tr>
<tr> <tr>
<td style={{padding: "9px 12px", fontWeight: 500, backgroundColor: "rgba(255,255,255,0.02)"}}>MI300X/MI325X</td> <td style={{padding: "9px 12px", fontWeight: 500, backgroundColor: "rgba(255,255,255,0.02)"}}>MI300X/MI325X</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>—</td> <td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>—</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.02)"}}>tp=8</td> <td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.02)"}}>tp=8</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>tp=8</td> <td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>tp=8</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.02)"}}>—</td>
</tr> </tr>
<tr> <tr>
<td style={{padding: "9px 12px", fontWeight: 500, backgroundColor: "rgba(255,255,255,0.02)"}}>MI355X</td> <td style={{padding: "9px 12px", fontWeight: 500, backgroundColor: "rgba(255,255,255,0.02)"}}>MI355X</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>—</td> <td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>—</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.02)"}}>tp=8</td> <td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.02)"}}>tp=8</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>tp=8</td> <td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>tp=8</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.02)"}}>tp=4</td>
</tr> </tr>
</tbody> </tbody>
</table> </table>
- **H100 and H200**: FP8 is the recommended deployment path. - **H100 and H200**: FP8 is the recommended deployment path.
- **B300 and GB300**: NVFP4 is the recommended deployment path. Use `nvidia/GLM-5.1-NVFP4` with `--quantization modelopt_fp4`. Use `tp=8` on B300 and `tp=4` on GB300. The CUDA 13 image variant is required for B300 and GB300. - **B300 and GB300**: NVFP4 is the recommended deployment path. Use `nvidia/GLM-5.1-NVFP4` with `--quantization modelopt_fp4`. Use `tp=8` on B300 and `tp=4` on GB300. The CUDA 13 image variant is required for B300 and GB300.
- **AMD GPUs**: Both BF16 and FP8 checkpoints are supported on MI300X/MI325X/MI355X at tp=8. Use `--dsa-prefill-backend tilelang --dsa-decode-backend tilelang` for the DSA attention backend. Add `--chunked-prefill-size 131072` and `--watchdog-timeout 1200` (20 minutes for weight loading). FP8 uses approximately half the memory of BF16 (~89 GB/GPU vs ~175 GB/GPU). EAGLE speculative decoding is not currently supported on AMD for GLM-5.1. - **AMD GPUs**: BF16 and FP8 checkpoints run on MI300X/MI325X/MI355X at tp=8. On MI355X (gfx950), the MXFP4 checkpoint `amd/GLM-5.1-MXFP4` is also supported at tp=4 with `--kv-cache-dtype fp8_e4m3`. All AMD paths pass `--dsa-prefill-backend tilelang --dsa-decode-backend tilelang`, `--chunked-prefill-size 131072`, and `--watchdog-timeout 1200` (20 minutes for weight loading). FP8 uses approximately half the memory of BF16 (~89 GB/GPU vs ~175 GB/GPU). EAGLE speculative decoding is supported on MI355X (gfx950) and unverified on MI300X/MI325X (gfx942).
- For other configuration tips (MTP, DSA kernel, Context Parallel, HiSparse, NVFP4, Index Cache), see the [DeepSeek-V3.2 cookbook page](../DeepSeek/DeepSeek-V3_2). GLM-5.1 and DeepSeek-V3.2 share the same model structure, so the optimization techniques are common. - For other configuration tips (MTP, DSA kernel, Context Parallel, HiSparse, NVFP4, Index Cache), see the [DeepSeek-V3.2 cookbook page](../DeepSeek/DeepSeek-V3_2). GLM-5.1 and DeepSeek-V3.2 share the same model structure, so the optimization techniques are common.
- Use `--json-model-override-args '{"index_topk_pattern": "FFSFSSSFSSFFFSSSFFFSFSSSSSSFFSFFSFFSSFFFFFFSFFFFFSFFSSSSSSFSFFFSFSSSFSFFSFFSSS"}'` to enable the [IndexCache](https://github.com/THUDM/IndexCache) method for GLM-5.1. This can improve serving efficiency with only a small accuracy loss. If you are running rigorous accuracy evaluations, do not enable this feature. - Use `--json-model-override-args '{"index_topk_pattern": "FFSFSSSFSSFFFSSSFFFSFSSSSSSFFSFFSFFSSFFFFFFSFFFFFSFFSSSSSSFSFFFSFSSSFSFFSFFSSS"}'` to enable the [IndexCache](https://github.com/THUDM/IndexCache) method for GLM-5.1. This can improve serving efficiency with only a small accuracy loss. If you are running rigorous accuracy evaluations, do not enable this feature.
@@ -154,6 +161,32 @@ sglang serve \
The following ROCm commands are additional options for AMD GPUs and do not replace the NVIDIA instructions above. The following ROCm commands are additional options for AMD GPUs and do not replace the NVIDIA instructions above.
#### MXFP4 (MI355X / gfx950)
On MI355X (gfx950), set `SGLANG_DSA_TRITON_PREFILL=1` to enable a faster Triton attention kernel for the prefill phase (opt-in, off by default). Keep `--dsa-prefill-backend tilelang` as shown. The EAGLE speculative-decoding flags below are optional but recommended on gfx950.
```shell Command
# SGLANG_DSA_TRITON_PREFILL=1 is optional; it enables a faster Triton prefill kernel on gfx950
SGLANG_DSA_TRITON_PREFILL=1 sglang serve \
--model-path amd/GLM-5.1-MXFP4 \
--tp 4 \
--trust-remote-code \
--kv-cache-dtype fp8_e4m3 \
--tool-call-parser glm47 \
--reasoning-parser glm45 \
--dsa-prefill-backend tilelang \
--dsa-decode-backend tilelang \
--chunked-prefill-size 131072 \
--mem-fraction-static 0.85 \
--watchdog-timeout 1200 \
--speculative-algorithm EAGLE \
--speculative-num-steps 3 \
--speculative-eagle-topk 1 \
--speculative-num-draft-tokens 4 \
--host 0.0.0.0 \
--port 30000
```
#### FP8 (Recommended) #### FP8 (Recommended)
```shell Command ```shell Command
@@ -4,7 +4,9 @@ export const GLM51Deployment = () => {
// Recommended quantization per hardware: // Recommended quantization per hardware:
// H100 / H200 → FP8 // H100 / H200 → FP8
// B300 / GB300 → NVFP4 // B300 / GB300 → NVFP4
// MI300X/MI325X/MI355X → BF16 (FP8 not verified on AMD) // MI300X / MI325X → BF16 (FP8 not verified on AMD)
// MI355X (gfx950) → MXFP4 (amd/GLM-5.1-MXFP4); BF16 also supported.
// MI350X is identical to MI355X (cooling only) and is omitted here.
const options = { const options = {
hardware: { hardware: {
name: 'hardware', name: 'hardware',
@@ -25,11 +27,13 @@ export const GLM51Deployment = () => {
getDynamicItems: (values) => { getDynamicItems: (values) => {
const hw = values.hardware; const hw = values.hardware;
const isAMD = ['mi300x', 'mi325x', 'mi355x'].includes(hw); const isAMD = ['mi300x', 'mi325x', 'mi355x'].includes(hw);
const isGfx950 = hw === 'mi355x'; // MI350X identical (cooling only)
const supportsNVFP4 = ['b300', 'gb300'].includes(hw); const supportsNVFP4 = ['b300', 'gb300'].includes(hw);
const isB300 = hw === 'b300'; const isB300 = hw === 'b300';
const isGB300 = hw === 'gb300'; const isGB300 = hw === 'gb300';
return [ return [
{ id: 'bf16', label: 'BF16', subtitle: 'Full Weights', default: isAMD, disabled: !isAMD, disabledReason: supportsNVFP4 ? 'NVFP4 is recommended for this hardware' : 'FP8 is recommended for this hardware' }, { id: 'mxfp4', label: 'MXFP4', subtitle: 'gfx950', default: isGfx950, disabled: !isGfx950, disabledReason: !isGfx950 ? 'MXFP4 verified on MI355X (gfx950)' : '' },
{ id: 'bf16', label: 'BF16', subtitle: 'Full Weights', default: isAMD && !isGfx950, disabled: !isAMD, disabledReason: supportsNVFP4 ? 'NVFP4 is recommended for this hardware' : 'FP8 is recommended for this hardware' },
{ id: 'fp8', label: 'FP8', subtitle: 'High Throughput', default: !isAMD && !supportsNVFP4, disabled: isAMD || supportsNVFP4, disabledReason: isAMD ? 'FP8 not verified on AMD' : (supportsNVFP4 ? 'NVFP4 is recommended for this hardware' : '') }, { id: 'fp8', label: 'FP8', subtitle: 'High Throughput', default: !isAMD && !supportsNVFP4, disabled: isAMD || supportsNVFP4, disabledReason: isAMD ? 'FP8 not verified on AMD' : (supportsNVFP4 ? 'NVFP4 is recommended for this hardware' : '') },
{ id: 'nvfp4', label: 'NVFP4', subtitle: 'Blackwell FP4', default: isB300 || isGB300, disabled: !supportsNVFP4, disabledReason: !supportsNVFP4 ? 'NVFP4 only on B300/GB300' : '' } { id: 'nvfp4', label: 'NVFP4', subtitle: 'Blackwell FP4', default: isB300 || isGB300, disabled: !supportsNVFP4, disabledReason: !supportsNVFP4 ? 'NVFP4 only on B300/GB300' : '' }
]; ];
@@ -58,15 +62,6 @@ export const GLM51Deployment = () => {
{ id: 'disabled', label: 'Disabled', subtitle: 'Low Latency', default: true }, { id: 'disabled', label: 'Disabled', subtitle: 'Low Latency', default: true },
{ id: 'enabled', label: 'Enabled', subtitle: 'High Throughput', default: false } { id: 'enabled', label: 'Enabled', subtitle: 'High Throughput', default: false }
] ]
},
speculative: {
name: 'speculative',
title: 'Speculative Decoding',
condition: (values) => !['mi300x', 'mi325x', 'mi355x'].includes(values.hardware),
items: [
{ id: 'disabled', label: 'Disabled', default: false },
{ id: 'enabled', label: 'Enabled', default: true }
]
} }
}; };
@@ -77,7 +72,7 @@ export const GLM51Deployment = () => {
gb300: { nvfp4: { tp: 4, mem: 0.80 }, fp8: { tp: 4, mem: 0.9 } }, gb300: { nvfp4: { tp: 4, mem: 0.80 }, fp8: { tp: 4, mem: 0.9 } },
mi300x: { bf16: { tp: 8, mem: 0.80 } }, mi300x: { bf16: { tp: 8, mem: 0.80 } },
mi325x: { bf16: { tp: 8, mem: 0.80 } }, mi325x: { bf16: { tp: 8, mem: 0.80 } },
mi355x: { bf16: { tp: 8, mem: 0.80 } } mi355x: { bf16: { tp: 8, mem: 0.80 }, mxfp4: { tp: 4, mem: 0.85 } }
}; };
const resolveItems = (option, values) => { const resolveItems = (option, values) => {
@@ -135,17 +130,22 @@ export const GLM51Deployment = () => {
const generateCommand = () => { const generateCommand = () => {
const { hardware, quantization } = values; const { hardware, quantization } = values;
const isAMD = ['mi300x', 'mi325x', 'mi355x'].includes(hardware); const isAMD = ['mi300x', 'mi325x', 'mi355x'].includes(hardware);
const isGfx950 = hardware === 'mi355x'; // MI350X identical (cooling only)
const recommendsNVFP4 = ['b300', 'gb300'].includes(hardware); const recommendsNVFP4 = ['b300', 'gb300'].includes(hardware);
const effectiveQuant = isAMD ? 'bf16' : (recommendsNVFP4 ? 'nvfp4' : 'fp8'); const effectiveQuant = isAMD
? (isGfx950 && quantization === 'mxfp4' ? 'mxfp4' : 'bf16')
: (recommendsNVFP4 ? 'nvfp4' : 'fp8');
const suffix = effectiveQuant === 'fp8' ? '-FP8' : ''; const suffix = effectiveQuant === 'fp8' ? '-FP8' : '';
const modelName = effectiveQuant === 'nvfp4' ? 'nvidia/GLM-5.1-NVFP4' : `zai-org/GLM-5.1${suffix}`; const modelName =
effectiveQuant === 'nvfp4' ? 'nvidia/GLM-5.1-NVFP4'
: effectiveQuant === 'mxfp4' ? 'amd/GLM-5.1-MXFP4'
: `zai-org/GLM-5.1${suffix}`;
const hwConfig = modelConfigs[hardware][effectiveQuant]; const hwConfig = modelConfigs[hardware][effectiveQuant];
if (!hwConfig) return '# Configuration not available for the selected hardware and quantization.'; if (!hwConfig) return '# Configuration not available for the selected hardware and quantization.';
const tpValue = hwConfig.tp; const tpValue = hwConfig.tp;
const memFraction = hwConfig.mem; const memFraction = hwConfig.mem;
const enableSpec = values.speculative === 'enabled';
let cmd = 'sglang serve \\\n'; let cmd = 'sglang serve \\\n';
cmd += ` --model-path ${modelName}`; cmd += ` --model-path ${modelName}`;
@@ -158,6 +158,7 @@ export const GLM51Deployment = () => {
if (isAMD) { if (isAMD) {
cmd += ' \\\n --trust-remote-code'; cmd += ' \\\n --trust-remote-code';
if (effectiveQuant === 'mxfp4') cmd += ' \\\n --kv-cache-dtype fp8_e4m3';
cmd += ' \\\n --dsa-prefill-backend tilelang'; cmd += ' \\\n --dsa-prefill-backend tilelang';
cmd += ' \\\n --dsa-decode-backend tilelang'; cmd += ' \\\n --dsa-decode-backend tilelang';
cmd += ' \\\n --chunked-prefill-size 131072'; cmd += ' \\\n --chunked-prefill-size 131072';
@@ -169,7 +170,9 @@ export const GLM51Deployment = () => {
} }
if (values.reasoning === 'enabled') cmd += ' \\\n --reasoning-parser glm45'; if (values.reasoning === 'enabled') cmd += ' \\\n --reasoning-parser glm45';
if (values.toolcall === 'enabled') cmd += ' \\\n --tool-call-parser glm47'; if (values.toolcall === 'enabled') cmd += ' \\\n --tool-call-parser glm47';
if (enableSpec) { // EAGLE MTP speculative decoding: emitted by default (recommended). Excluded
// only on MI300X/MI325X (gfx942), where it is not yet verified.
if (!['mi300x', 'mi325x'].includes(hardware)) {
cmd += ' \\\n --speculative-algorithm EAGLE'; cmd += ' \\\n --speculative-algorithm EAGLE';
cmd += ' \\\n --speculative-num-steps 3'; cmd += ' \\\n --speculative-num-steps 3';
cmd += ' \\\n --speculative-eagle-topk 1'; cmd += ' \\\n --speculative-eagle-topk 1';