[Docs] Add GLM-5.1 NVFP4 to cookbook (#27708)

Signed-off-by: Mohammad Miadh Angkad <176301910+mmangkad@users.noreply.github.com>
This commit is contained in:
Mohammad Miadh Angkad
2026-06-10 10:23:15 -07:00
committed by GitHub
parent 8c6bbe0658
commit 91ff7baa28
2 changed files with 93 additions and 39 deletions
@@ -1,7 +1,7 @@
--- ---
title: GLM-5.1 title: GLM-5.1
metatags: metatags:
description: "Deploy GLM-5.1 with SGLang on NVIDIA H100/H200/B200/GB300 and AMD MI300X/MI325X/MI355X." description: "Deploy GLM-5.1 with SGLang on NVIDIA H100/H200/B300/GB300 and AMD MI300X/MI325X/MI355X."
tag: NEW tag: NEW
--- ---
@@ -11,6 +11,7 @@ tag: NEW
- **BF16 (Full precision)**: [zai-org/GLM-5.1](https://huggingface.co/zai-org/GLM-5.1) - **BF16 (Full precision)**: [zai-org/GLM-5.1](https://huggingface.co/zai-org/GLM-5.1)
- **FP8 (8-bit quantized)**: [zai-org/GLM-5.1-FP8](https://huggingface.co/zai-org/GLM-5.1-FP8) - **FP8 (8-bit quantized)**: [zai-org/GLM-5.1-FP8](https://huggingface.co/zai-org/GLM-5.1-FP8)
- **NVFP4 (4-bit quantized)**: [nvidia/GLM-5.1-NVFP4](https://huggingface.co/nvidia/GLM-5.1-NVFP4)
**License:** MIT **License:** MIT
@@ -24,7 +25,7 @@ This section provides deployment configurations optimized for different hardware
### 3.1 Basic Configuration ### 3.1 Basic Configuration
**Interactive Command Generator**: Use the configuration selector below to automatically generate the appropriate deployment command for your hardware platform, quantization method, and capabilities. SGLang supports serving GLM-5.1 on NVIDIA H100, H200, B200, GB300, and AMD MI300X/MI325X/MI355X GPUs. **Interactive Command Generator**: Use the configuration selector below to automatically generate the appropriate deployment command for your hardware platform, quantization method, and capabilities. SGLang supports serving GLM-5.1 on NVIDIA H100, H200, B300, GB300, and AMD MI300X/MI325X/MI355X GPUs.
import { GLM51Deployment } from '/src/snippets/autoregressive/glm-51-deployment.jsx' import { GLM51Deployment } from '/src/snippets/autoregressive/glm-51-deployment.jsx'
@@ -35,61 +36,68 @@ import { GLM51Deployment } from '/src/snippets/autoregressive/glm-51-deployment.
- Speculative decoding (MTP) can significantly reduce latency for interactive use cases. - Speculative decoding (MTP) can significantly reduce latency for interactive use cases.
- **DP Attention**: Enables data parallel attention for higher throughput under high concurrency. Note that DP attention trades off low-concurrency latency for high-concurrency throughput — disable it if your workload is latency-sensitive with few concurrent requests. - **DP Attention**: Enables data parallel attention for higher throughput under high concurrency. Note that DP attention trades off low-concurrency latency for high-concurrency throughput — disable it if your workload is latency-sensitive with few concurrent requests.
- The `--mem-fraction-static` flag is recommended for optimal memory utilization, adjust it based on your hardware and workload. - The `--mem-fraction-static` flag is recommended for optimal memory utilization, adjust it based on your hardware and workload.
- BF16 model always requires **2x GPUs** compared to FP8 on NVIDIA hardware.
<table style={{width: "100%", borderCollapse: "collapse", tableLayout: "fixed"}}> <table style={{width: "100%", borderCollapse: "collapse", tableLayout: "fixed"}}>
<thead> <thead>
<tr style={{borderBottom: "2px solid #d55816"}}> <tr style={{borderBottom: "2px solid #d55816"}}>
<th style={{textAlign: "left", padding: "10px 12px", fontWeight: 700, whiteSpace: "nowrap", backgroundColor: "rgba(255,255,255,0.02)"}}>Hardware</th> <th style={{textAlign: "left", padding: "10px 12px", fontWeight: 700, whiteSpace: "nowrap", backgroundColor: "rgba(255,255,255,0.02)"}}>Hardware</th>
<th style={{textAlign: "left", padding: "10px 12px", fontWeight: 700, whiteSpace: "nowrap", backgroundColor: "rgba(255,255,255,0.05)"}}>FP8</th> <th style={{textAlign: "left", padding: "10px 12px", fontWeight: 700, whiteSpace: "nowrap", backgroundColor: "rgba(255,255,255,0.05)"}}>NVFP4</th>
<th style={{textAlign: "left", padding: "10px 12px", fontWeight: 700, whiteSpace: "nowrap", backgroundColor: "rgba(255,255,255,0.02)"}}>BF16</th> <th style={{textAlign: "left", padding: "10px 12px", fontWeight: 700, whiteSpace: "nowrap", backgroundColor: "rgba(255,255,255,0.02)"}}>FP8</th>
<th style={{textAlign: "left", padding: "10px 12px", fontWeight: 700, whiteSpace: "nowrap", backgroundColor: "rgba(255,255,255,0.05)"}}>BF16</th>
</tr> </tr>
</thead> </thead>
<tbody> <tbody>
<tr> <tr>
<td style={{padding: "9px 12px", fontWeight: 500, backgroundColor: "rgba(255,255,255,0.02)"}}>H100</td> <td style={{padding: "9px 12px", fontWeight: 500, backgroundColor: "rgba(255,255,255,0.02)"}}>H100</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>tp=16</td> <td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>—</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.02)"}}>tp=32</td> <td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.02)"}}>tp=16</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>—</td>
</tr> </tr>
<tr> <tr>
<td style={{padding: "9px 12px", fontWeight: 500, backgroundColor: "rgba(255,255,255,0.02)"}}>H200</td> <td style={{padding: "9px 12px", fontWeight: 500, backgroundColor: "rgba(255,255,255,0.02)"}}>H200</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>tp=8</td> <td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>—</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.02)"}}>tp=16</td> <td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.02)"}}>tp=8</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>—</td>
</tr> </tr>
<tr> <tr>
<td style={{padding: "9px 12px", fontWeight: 500, backgroundColor: "rgba(255,255,255,0.02)"}}>B200</td> <td style={{padding: "9px 12px", fontWeight: 500, backgroundColor: "rgba(255,255,255,0.02)"}}>B300</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>tp=8</td> <td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>tp=8</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.02)"}}>tp=16</td> <td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.02)"}}>—</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>—</td>
</tr> </tr>
<tr> <tr>
<td style={{padding: "9px 12px", fontWeight: 500, backgroundColor: "rgba(255,255,255,0.02)"}}>GB300</td> <td style={{padding: "9px 12px", fontWeight: 500, backgroundColor: "rgba(255,255,255,0.02)"}}>GB300</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>tp=4</td> <td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>tp=4</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.02)"}}>—</td> <td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.02)"}}>—</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>—</td>
</tr> </tr>
<tr> <tr>
<td style={{padding: "9px 12px", fontWeight: 500, backgroundColor: "rgba(255,255,255,0.02)"}}>MI300X/MI325X</td> <td style={{padding: "9px 12px", fontWeight: 500, backgroundColor: "rgba(255,255,255,0.02)"}}>MI300X/MI325X</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>tp=8</td> <td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>—</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.02)"}}>tp=8</td> <td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.02)"}}>tp=8</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>tp=8</td>
</tr> </tr>
<tr> <tr>
<td style={{padding: "9px 12px", fontWeight: 500, backgroundColor: "rgba(255,255,255,0.02)"}}>MI355X</td> <td style={{padding: "9px 12px", fontWeight: 500, backgroundColor: "rgba(255,255,255,0.02)"}}>MI355X</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>tp=8</td> <td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>—</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.02)"}}>tp=8</td> <td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.02)"}}>tp=8</td>
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>tp=8</td>
</tr> </tr>
</tbody> </tbody>
</table> </table>
- **H100 and H200**: FP8 is the recommended deployment path.
- **B300 and GB300**: NVFP4 is the recommended deployment path. Use `nvidia/GLM-5.1-NVFP4` with `--quantization modelopt_fp4`. Use `tp=8` on B300 and `tp=4` on GB300. The CUDA 13 image variant is required for B300 and GB300.
- **AMD GPUs**: Both BF16 and FP8 checkpoints are supported on MI300X/MI325X/MI355X at tp=8. Use `--dsa-prefill-backend tilelang --dsa-decode-backend tilelang` for the DSA attention backend. Add `--chunked-prefill-size 131072` and `--watchdog-timeout 1200` (20 minutes for weight loading). FP8 uses approximately half the memory of BF16 (~89 GB/GPU vs ~175 GB/GPU). EAGLE speculative decoding is not currently supported on AMD for GLM-5.1. - **AMD GPUs**: Both BF16 and FP8 checkpoints are supported on MI300X/MI325X/MI355X at tp=8. Use `--dsa-prefill-backend tilelang --dsa-decode-backend tilelang` for the DSA attention backend. Add `--chunked-prefill-size 131072` and `--watchdog-timeout 1200` (20 minutes for weight loading). FP8 uses approximately half the memory of BF16 (~89 GB/GPU vs ~175 GB/GPU). EAGLE speculative decoding is not currently supported on AMD for GLM-5.1.
- **GB300**: Only the FP8 checkpoint is recommended on GB300, with `tp=4`. For high-throughput DP attention on GB300, use `--dp 4`.
- For other configuration tips (MTP, DSA kernel, Context Parallel, HiSparse, NVFP4, Index Cache), see the [DeepSeek-V3.2 cookbook page](../DeepSeek/DeepSeek-V3_2). GLM-5.1 and DeepSeek-V3.2 share the same model structure, so the optimization techniques are common. - For other configuration tips (MTP, DSA kernel, Context Parallel, HiSparse, NVFP4, Index Cache), see the [DeepSeek-V3.2 cookbook page](../DeepSeek/DeepSeek-V3_2). GLM-5.1 and DeepSeek-V3.2 share the same model structure, so the optimization techniques are common.
- Use `--json-model-override-args '{"index_topk_pattern": "FFSFSSSFSSFFFSSSFFFSFSSSSSSFFSFFSFFSSFFFFFFSFFFFFSFFSSSSSSFSFFFSFSSSFSFFSFFSSS"}'` for GLM-5.1-FP8 if you want to enable the [IndexCache](https://github.com/THUDM/IndexCache) method. This feature is supported through [this PR](https://github.com/sgl-project/sglang/pull/21405) and introduces only a small accuracy loss. However, if you are running rigorous accuracy evaluations, it is not recommended to enable this feature. - Use `--json-model-override-args '{"index_topk_pattern": "FFSFSSSFSSFFFSSSFFFSFSSSSSSFFSFFSFFSSFFFFFFSFFFFFSFFSSSSSSFSFFFSFSSSFSFFSFFSSS"}'` to enable the [IndexCache](https://github.com/THUDM/IndexCache) method for GLM-5.1. This can improve serving efficiency with only a small accuracy loss. If you are running rigorous accuracy evaluations, do not enable this feature.
## 4. Model Invocation ## 4. Model Invocation
Deploy GLM-5.1 with the following command (FP8 on H200, all features enabled): Deploy GLM-5.1 with the following command (FP8 on H200, all features enabled):
```shell Command ```shell Command
SGLANG_ENABLE_SPEC_V2=1 sglang serve \ sglang serve \
--model-path zai-org/GLM-5.1-FP8 \ --model-path zai-org/GLM-5.1-FP8 \
--tp 8 \ --tp 8 \
--tool-call-parser glm47 \ --tool-call-parser glm47 \
@@ -103,7 +111,47 @@ SGLANG_ENABLE_SPEC_V2=1 sglang serve \
--port 30000 --port 30000
``` ```
### 4.1 MI300X/MI325X/MI355X (ROCm) Server Command ### 4.1 B300/GB300 (NVFP4) Server Command
#### B300
```shell Command
sglang serve \
--model-path nvidia/GLM-5.1-NVFP4 \
--tp 8 \
--quantization modelopt_fp4 \
--tool-call-parser glm47 \
--reasoning-parser glm45 \
--speculative-algorithm EAGLE \
--speculative-num-steps 3 \
--speculative-eagle-topk 1 \
--speculative-num-draft-tokens 4 \
--trust-remote-code \
--mem-fraction-static 0.80 \
--host 0.0.0.0 \
--port 30000
```
#### GB300
```shell Command
sglang serve \
--model-path nvidia/GLM-5.1-NVFP4 \
--tp 4 \
--quantization modelopt_fp4 \
--tool-call-parser glm47 \
--reasoning-parser glm45 \
--speculative-algorithm EAGLE \
--speculative-num-steps 3 \
--speculative-eagle-topk 1 \
--speculative-num-draft-tokens 4 \
--trust-remote-code \
--mem-fraction-static 0.80 \
--host 0.0.0.0 \
--port 30000
```
### 4.2 MI300X/MI325X/MI355X (ROCm) Server Command
The following ROCm commands are additional options for AMD GPUs and do not replace the NVIDIA instructions above. The following ROCm commands are additional options for AMD GPUs and do not replace the NVIDIA instructions above.
@@ -141,15 +189,15 @@ sglang serve \
--port 30000 --port 30000
``` ```
### 4.2 Basic Usage ### 4.3 Basic Usage
For basic API usage and request examples, please refer to: For basic API usage and request examples, please refer to:
- [SGLang Basic Usage Guide](../../../docs/basic_usage/send_request) - [SGLang Basic Usage Guide](../../../docs/basic_usage/send_request)
### 4.3 Advanced Usage ### 4.4 Advanced Usage
#### 4.3.1 Reasoning Parser #### 4.4.1 Reasoning Parser
GLM-5.1 supports Thinking mode **by default**. Enable the reasoning parser during deployment to separate the thinking and content sections. The thinking process is returned via `reasoning_content` in the streaming response. GLM-5.1 supports Thinking mode **by default**. Enable the reasoning parser during deployment to separate the thinking and content sections. The thinking process is returned via `reasoning_content` in the streaming response.
@@ -312,7 +360,7 @@ Here is how to calculate it:
2. Multiply the decimal by the number: 0.15 × 240 = 36 2. Multiply the decimal by the number: 0.15 × 240 = 36
``` ```
#### 4.3.2 Tool Calling #### 4.4.2 Tool Calling
GLM-5.1 supports tool calling capabilities. Enable the tool call parser during deployment. Thinking mode is on by default; to disable it for tool calling requests, pass `extra_body={"chat_template_kwargs": {"enable_thinking": False}}`. GLM-5.1 supports tool calling capabilities. Enable the tool call parser during deployment. Thinking mode is on by default; to disable it for tool calling requests, pass `extra_body={"chat_template_kwargs": {"enable_thinking": False}}`.
@@ -1,19 +1,19 @@
export const GLM51Deployment = () => { export const GLM51Deployment = () => {
// Config mirrors sgl-cookbook src/components/autoregressive/GLM51ConfigGenerator/index.js. // Config mirrors sgl-cookbook src/components/autoregressive/GLM51ConfigGenerator/index.js.
// //
// Supported quantization per hardware: // Recommended quantization per hardware:
// H100 / H200 / B200 → BF16 + FP8 // H100 / H200 → FP8
// GB300 → FP8 only // B300 / GB300 → NVFP4
// MI300X/MI325X/MI355X → BF16 (FP8 not verified on AMD) // MI300X/MI325X/MI355X → BF16 (FP8 not verified on AMD)
const options = { const options = {
hardware: { hardware: {
name: 'hardware', name: 'hardware',
title: 'Hardware Platform', title: 'Hardware Platform',
items: [ items: [
{ id: 'h200', label: 'H200', default: true },
{ id: 'b200', label: 'B200', default: false },
{ id: 'gb300', label: 'GB300', default: false },
{ id: 'h100', label: 'H100', default: false }, { id: 'h100', label: 'H100', default: false },
{ id: 'h200', label: 'H200', default: true },
{ id: 'b300', label: 'B300', default: false },
{ id: 'gb300', label: 'GB300', default: false },
{ id: 'mi300x', label: 'MI300X', default: false }, { id: 'mi300x', label: 'MI300X', default: false },
{ id: 'mi325x', label: 'MI325X', default: false }, { id: 'mi325x', label: 'MI325X', default: false },
{ id: 'mi355x', label: 'MI355X', default: false } { id: 'mi355x', label: 'MI355X', default: false }
@@ -25,10 +25,13 @@ export const GLM51Deployment = () => {
getDynamicItems: (values) => { getDynamicItems: (values) => {
const hw = values.hardware; const hw = values.hardware;
const isAMD = ['mi300x', 'mi325x', 'mi355x'].includes(hw); const isAMD = ['mi300x', 'mi325x', 'mi355x'].includes(hw);
const supportsNVFP4 = ['b300', 'gb300'].includes(hw);
const isB300 = hw === 'b300';
const isGB300 = hw === 'gb300'; const isGB300 = hw === 'gb300';
return [ return [
{ id: 'bf16', label: 'BF16', subtitle: 'Full Weights', default: isAMD, disabled: isGB300, disabledReason: isGB300 ? 'BF16 is not recommended on GB300 for GLM-5.1' : '' }, { id: 'bf16', label: 'BF16', subtitle: 'Full Weights', default: isAMD, disabled: !isAMD, disabledReason: supportsNVFP4 ? 'NVFP4 is recommended for this hardware' : 'FP8 is recommended for this hardware' },
{ id: 'fp8', label: 'FP8', subtitle: 'High Throughput', default: !isAMD, disabled: isAMD, disabledReason: isAMD ? 'FP8 not verified on AMD' : '' } { id: 'fp8', label: 'FP8', subtitle: 'High Throughput', default: !isAMD && !supportsNVFP4, disabled: isAMD || supportsNVFP4, disabledReason: isAMD ? 'FP8 not verified on AMD' : (supportsNVFP4 ? 'NVFP4 is recommended for this hardware' : '') },
{ id: 'nvfp4', label: 'NVFP4', subtitle: 'Blackwell FP4', default: isB300 || isGB300, disabled: !supportsNVFP4, disabledReason: !supportsNVFP4 ? 'NVFP4 only on B300/GB300' : '' }
]; ];
} }
}, },
@@ -68,10 +71,10 @@ export const GLM51Deployment = () => {
}; };
const modelConfigs = { const modelConfigs = {
h100: { fp8: { tp: 16, mem: 0.85 }, bf16: { tp: 32, mem: 0.85 } }, h100: { fp8: { tp: 16, mem: 0.85 } },
h200: { fp8: { tp: 8, mem: 0.85 }, bf16: { tp: 16, mem: 0.85 } }, h200: { fp8: { tp: 8, mem: 0.85 } },
b200: { fp8: { tp: 8, mem: 0.9 }, bf16: { tp: 16, mem: 0.9 } }, b300: { nvfp4: { tp: 8, mem: 0.80 }, fp8: { tp: 8, mem: 0.9 }, bf16: { tp: 16, mem: 0.9 } },
gb300: { fp8: { tp: 4, mem: 0.9 } }, gb300: { nvfp4: { tp: 4, mem: 0.80 }, fp8: { tp: 4, mem: 0.9 } },
mi300x: { bf16: { tp: 8, mem: 0.80 } }, mi300x: { bf16: { tp: 8, mem: 0.80 } },
mi325x: { bf16: { tp: 8, mem: 0.80 } }, mi325x: { bf16: { tp: 8, mem: 0.80 } },
mi355x: { bf16: { tp: 8, mem: 0.80 } } mi355x: { bf16: { tp: 8, mem: 0.80 } }
@@ -132,10 +135,10 @@ export const GLM51Deployment = () => {
const generateCommand = () => { const generateCommand = () => {
const { hardware, quantization } = values; const { hardware, quantization } = values;
const isAMD = ['mi300x', 'mi325x', 'mi355x'].includes(hardware); const isAMD = ['mi300x', 'mi325x', 'mi355x'].includes(hardware);
const isGB300 = hardware === 'gb300'; const recommendsNVFP4 = ['b300', 'gb300'].includes(hardware);
const effectiveQuant = isAMD ? 'bf16' : (isGB300 && quantization === 'bf16' ? 'fp8' : quantization); const effectiveQuant = isAMD ? 'bf16' : (recommendsNVFP4 ? 'nvfp4' : 'fp8');
const suffix = effectiveQuant === 'fp8' ? '-FP8' : ''; const suffix = effectiveQuant === 'fp8' ? '-FP8' : '';
const modelName = `zai-org/GLM-5.1${suffix}`; const modelName = effectiveQuant === 'nvfp4' ? 'nvidia/GLM-5.1-NVFP4' : `zai-org/GLM-5.1${suffix}`;
const hwConfig = modelConfigs[hardware][effectiveQuant]; const hwConfig = modelConfigs[hardware][effectiveQuant];
if (!hwConfig) return '# Configuration not available for the selected hardware and quantization.'; if (!hwConfig) return '# Configuration not available for the selected hardware and quantization.';
@@ -144,12 +147,15 @@ export const GLM51Deployment = () => {
const memFraction = hwConfig.mem; const memFraction = hwConfig.mem;
const enableSpec = values.speculative === 'enabled'; const enableSpec = values.speculative === 'enabled';
let cmd = ''; let cmd = 'sglang serve \\\n';
if (enableSpec) cmd += 'SGLANG_ENABLE_SPEC_V2=1 ';
cmd += 'sglang serve \\\n';
cmd += ` --model-path ${modelName}`; cmd += ` --model-path ${modelName}`;
cmd += ` \\\n --tp ${tpValue}`; cmd += ` \\\n --tp ${tpValue}`;
if (effectiveQuant === 'nvfp4') {
cmd += ' \\\n --quantization modelopt_fp4';
cmd += ' \\\n --trust-remote-code';
}
if (isAMD) { if (isAMD) {
cmd += ' \\\n --trust-remote-code'; cmd += ' \\\n --trust-remote-code';
cmd += ' \\\n --dsa-prefill-backend tilelang'; cmd += ' \\\n --dsa-prefill-backend tilelang';