[AMD] [GLM5] GLM-5.1 MXFP4 (MI355X) + enable EAGLE for gfx950 in cookbook (#29194)
Co-authored-by: Raiden-Makoto <Raiden-Makoto@users.noreply.github.com>
This commit is contained in:
co-authored by
Raiden-Makoto
parent
ddda4f9028
commit
0075c8f02b
@@ -43,6 +43,7 @@ import { GLM51Deployment } from '/src/snippets/autoregressive/glm-51-deployment.
|
|||||||
<th style={{textAlign: "left", padding: "10px 12px", fontWeight: 700, whiteSpace: "nowrap", backgroundColor: "rgba(255,255,255,0.05)"}}>NVFP4</th>
|
<th style={{textAlign: "left", padding: "10px 12px", fontWeight: 700, whiteSpace: "nowrap", backgroundColor: "rgba(255,255,255,0.05)"}}>NVFP4</th>
|
||||||
<th style={{textAlign: "left", padding: "10px 12px", fontWeight: 700, whiteSpace: "nowrap", backgroundColor: "rgba(255,255,255,0.02)"}}>FP8</th>
|
<th style={{textAlign: "left", padding: "10px 12px", fontWeight: 700, whiteSpace: "nowrap", backgroundColor: "rgba(255,255,255,0.02)"}}>FP8</th>
|
||||||
<th style={{textAlign: "left", padding: "10px 12px", fontWeight: 700, whiteSpace: "nowrap", backgroundColor: "rgba(255,255,255,0.05)"}}>BF16</th>
|
<th style={{textAlign: "left", padding: "10px 12px", fontWeight: 700, whiteSpace: "nowrap", backgroundColor: "rgba(255,255,255,0.05)"}}>BF16</th>
|
||||||
|
<th style={{textAlign: "left", padding: "10px 12px", fontWeight: 700, whiteSpace: "nowrap", backgroundColor: "rgba(255,255,255,0.02)"}}>MXFP4</th>
|
||||||
</tr>
|
</tr>
|
||||||
</thead>
|
</thead>
|
||||||
<tbody>
|
<tbody>
|
||||||
@@ -51,43 +52,49 @@ import { GLM51Deployment } from '/src/snippets/autoregressive/glm-51-deployment.
|
|||||||
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>—</td>
|
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>—</td>
|
||||||
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.02)"}}>tp=16</td>
|
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.02)"}}>tp=16</td>
|
||||||
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>—</td>
|
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>—</td>
|
||||||
|
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.02)"}}>—</td>
|
||||||
</tr>
|
</tr>
|
||||||
<tr>
|
<tr>
|
||||||
<td style={{padding: "9px 12px", fontWeight: 500, backgroundColor: "rgba(255,255,255,0.02)"}}>H200</td>
|
<td style={{padding: "9px 12px", fontWeight: 500, backgroundColor: "rgba(255,255,255,0.02)"}}>H200</td>
|
||||||
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>—</td>
|
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>—</td>
|
||||||
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.02)"}}>tp=8</td>
|
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.02)"}}>tp=8</td>
|
||||||
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>—</td>
|
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>—</td>
|
||||||
|
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.02)"}}>—</td>
|
||||||
</tr>
|
</tr>
|
||||||
<tr>
|
<tr>
|
||||||
<td style={{padding: "9px 12px", fontWeight: 500, backgroundColor: "rgba(255,255,255,0.02)"}}>B300</td>
|
<td style={{padding: "9px 12px", fontWeight: 500, backgroundColor: "rgba(255,255,255,0.02)"}}>B300</td>
|
||||||
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>tp=8</td>
|
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>tp=8</td>
|
||||||
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.02)"}}>—</td>
|
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.02)"}}>—</td>
|
||||||
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>—</td>
|
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>—</td>
|
||||||
|
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.02)"}}>—</td>
|
||||||
</tr>
|
</tr>
|
||||||
<tr>
|
<tr>
|
||||||
<td style={{padding: "9px 12px", fontWeight: 500, backgroundColor: "rgba(255,255,255,0.02)"}}>GB300</td>
|
<td style={{padding: "9px 12px", fontWeight: 500, backgroundColor: "rgba(255,255,255,0.02)"}}>GB300</td>
|
||||||
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>tp=4</td>
|
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>tp=4</td>
|
||||||
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.02)"}}>—</td>
|
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.02)"}}>—</td>
|
||||||
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>—</td>
|
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>—</td>
|
||||||
|
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.02)"}}>—</td>
|
||||||
</tr>
|
</tr>
|
||||||
<tr>
|
<tr>
|
||||||
<td style={{padding: "9px 12px", fontWeight: 500, backgroundColor: "rgba(255,255,255,0.02)"}}>MI300X/MI325X</td>
|
<td style={{padding: "9px 12px", fontWeight: 500, backgroundColor: "rgba(255,255,255,0.02)"}}>MI300X/MI325X</td>
|
||||||
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>—</td>
|
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>—</td>
|
||||||
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.02)"}}>tp=8</td>
|
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.02)"}}>tp=8</td>
|
||||||
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>tp=8</td>
|
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>tp=8</td>
|
||||||
|
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.02)"}}>—</td>
|
||||||
</tr>
|
</tr>
|
||||||
<tr>
|
<tr>
|
||||||
<td style={{padding: "9px 12px", fontWeight: 500, backgroundColor: "rgba(255,255,255,0.02)"}}>MI355X</td>
|
<td style={{padding: "9px 12px", fontWeight: 500, backgroundColor: "rgba(255,255,255,0.02)"}}>MI355X</td>
|
||||||
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>—</td>
|
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>—</td>
|
||||||
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.02)"}}>tp=8</td>
|
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.02)"}}>tp=8</td>
|
||||||
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>tp=8</td>
|
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.05)"}}>tp=8</td>
|
||||||
|
<td style={{padding: "9px 12px", backgroundColor: "rgba(255,255,255,0.02)"}}>tp=4</td>
|
||||||
</tr>
|
</tr>
|
||||||
</tbody>
|
</tbody>
|
||||||
</table>
|
</table>
|
||||||
|
|
||||||
- **H100 and H200**: FP8 is the recommended deployment path.
|
- **H100 and H200**: FP8 is the recommended deployment path.
|
||||||
- **B300 and GB300**: NVFP4 is the recommended deployment path. Use `nvidia/GLM-5.1-NVFP4` with `--quantization modelopt_fp4`. Use `tp=8` on B300 and `tp=4` on GB300. The CUDA 13 image variant is required for B300 and GB300.
|
- **B300 and GB300**: NVFP4 is the recommended deployment path. Use `nvidia/GLM-5.1-NVFP4` with `--quantization modelopt_fp4`. Use `tp=8` on B300 and `tp=4` on GB300. The CUDA 13 image variant is required for B300 and GB300.
|
||||||
- **AMD GPUs**: Both BF16 and FP8 checkpoints are supported on MI300X/MI325X/MI355X at tp=8. Use `--dsa-prefill-backend tilelang --dsa-decode-backend tilelang` for the DSA attention backend. Add `--chunked-prefill-size 131072` and `--watchdog-timeout 1200` (20 minutes for weight loading). FP8 uses approximately half the memory of BF16 (~89 GB/GPU vs ~175 GB/GPU). EAGLE speculative decoding is not currently supported on AMD for GLM-5.1.
|
- **AMD GPUs**: BF16 and FP8 checkpoints run on MI300X/MI325X/MI355X at tp=8. On MI355X (gfx950), the MXFP4 checkpoint `amd/GLM-5.1-MXFP4` is also supported at tp=4 with `--kv-cache-dtype fp8_e4m3`. All AMD paths pass `--dsa-prefill-backend tilelang --dsa-decode-backend tilelang`, `--chunked-prefill-size 131072`, and `--watchdog-timeout 1200` (20 minutes for weight loading). FP8 uses approximately half the memory of BF16 (~89 GB/GPU vs ~175 GB/GPU). EAGLE speculative decoding is supported on MI355X (gfx950) and unverified on MI300X/MI325X (gfx942).
|
||||||
- For other configuration tips (MTP, DSA kernel, Context Parallel, HiSparse, NVFP4, Index Cache), see the [DeepSeek-V3.2 cookbook page](../DeepSeek/DeepSeek-V3_2). GLM-5.1 and DeepSeek-V3.2 share the same model structure, so the optimization techniques are common.
|
- For other configuration tips (MTP, DSA kernel, Context Parallel, HiSparse, NVFP4, Index Cache), see the [DeepSeek-V3.2 cookbook page](../DeepSeek/DeepSeek-V3_2). GLM-5.1 and DeepSeek-V3.2 share the same model structure, so the optimization techniques are common.
|
||||||
- Use `--json-model-override-args '{"index_topk_pattern": "FFSFSSSFSSFFFSSSFFFSFSSSSSSFFSFFSFFSSFFFFFFSFFFFFSFFSSSSSSFSFFFSFSSSFSFFSFFSSS"}'` to enable the [IndexCache](https://github.com/THUDM/IndexCache) method for GLM-5.1. This can improve serving efficiency with only a small accuracy loss. If you are running rigorous accuracy evaluations, do not enable this feature.
|
- Use `--json-model-override-args '{"index_topk_pattern": "FFSFSSSFSSFFFSSSFFFSFSSSSSSFFSFFSFFSSFFFFFFSFFFFFSFFSSSSSSFSFFFSFSSSFSFFSFFSSS"}'` to enable the [IndexCache](https://github.com/THUDM/IndexCache) method for GLM-5.1. This can improve serving efficiency with only a small accuracy loss. If you are running rigorous accuracy evaluations, do not enable this feature.
|
||||||
|
|
||||||
@@ -154,6 +161,32 @@ sglang serve \
|
|||||||
|
|
||||||
The following ROCm commands are additional options for AMD GPUs and do not replace the NVIDIA instructions above.
|
The following ROCm commands are additional options for AMD GPUs and do not replace the NVIDIA instructions above.
|
||||||
|
|
||||||
|
#### MXFP4 (MI355X / gfx950)
|
||||||
|
|
||||||
|
On MI355X (gfx950), set `SGLANG_DSA_TRITON_PREFILL=1` to enable a faster Triton attention kernel for the prefill phase (opt-in, off by default). Keep `--dsa-prefill-backend tilelang` as shown. The EAGLE speculative-decoding flags below are optional but recommended on gfx950.
|
||||||
|
|
||||||
|
```shell Command
|
||||||
|
# SGLANG_DSA_TRITON_PREFILL=1 is optional; it enables a faster Triton prefill kernel on gfx950
|
||||||
|
SGLANG_DSA_TRITON_PREFILL=1 sglang serve \
|
||||||
|
--model-path amd/GLM-5.1-MXFP4 \
|
||||||
|
--tp 4 \
|
||||||
|
--trust-remote-code \
|
||||||
|
--kv-cache-dtype fp8_e4m3 \
|
||||||
|
--tool-call-parser glm47 \
|
||||||
|
--reasoning-parser glm45 \
|
||||||
|
--dsa-prefill-backend tilelang \
|
||||||
|
--dsa-decode-backend tilelang \
|
||||||
|
--chunked-prefill-size 131072 \
|
||||||
|
--mem-fraction-static 0.85 \
|
||||||
|
--watchdog-timeout 1200 \
|
||||||
|
--speculative-algorithm EAGLE \
|
||||||
|
--speculative-num-steps 3 \
|
||||||
|
--speculative-eagle-topk 1 \
|
||||||
|
--speculative-num-draft-tokens 4 \
|
||||||
|
--host 0.0.0.0 \
|
||||||
|
--port 30000
|
||||||
|
```
|
||||||
|
|
||||||
#### FP8 (Recommended)
|
#### FP8 (Recommended)
|
||||||
|
|
||||||
```shell Command
|
```shell Command
|
||||||
|
|||||||
@@ -4,7 +4,9 @@ export const GLM51Deployment = () => {
|
|||||||
// Recommended quantization per hardware:
|
// Recommended quantization per hardware:
|
||||||
// H100 / H200 → FP8
|
// H100 / H200 → FP8
|
||||||
// B300 / GB300 → NVFP4
|
// B300 / GB300 → NVFP4
|
||||||
// MI300X/MI325X/MI355X → BF16 (FP8 not verified on AMD)
|
// MI300X / MI325X → BF16 (FP8 not verified on AMD)
|
||||||
|
// MI355X (gfx950) → MXFP4 (amd/GLM-5.1-MXFP4); BF16 also supported.
|
||||||
|
// MI350X is identical to MI355X (cooling only) and is omitted here.
|
||||||
const options = {
|
const options = {
|
||||||
hardware: {
|
hardware: {
|
||||||
name: 'hardware',
|
name: 'hardware',
|
||||||
@@ -25,11 +27,13 @@ export const GLM51Deployment = () => {
|
|||||||
getDynamicItems: (values) => {
|
getDynamicItems: (values) => {
|
||||||
const hw = values.hardware;
|
const hw = values.hardware;
|
||||||
const isAMD = ['mi300x', 'mi325x', 'mi355x'].includes(hw);
|
const isAMD = ['mi300x', 'mi325x', 'mi355x'].includes(hw);
|
||||||
|
const isGfx950 = hw === 'mi355x'; // MI350X identical (cooling only)
|
||||||
const supportsNVFP4 = ['b300', 'gb300'].includes(hw);
|
const supportsNVFP4 = ['b300', 'gb300'].includes(hw);
|
||||||
const isB300 = hw === 'b300';
|
const isB300 = hw === 'b300';
|
||||||
const isGB300 = hw === 'gb300';
|
const isGB300 = hw === 'gb300';
|
||||||
return [
|
return [
|
||||||
{ id: 'bf16', label: 'BF16', subtitle: 'Full Weights', default: isAMD, disabled: !isAMD, disabledReason: supportsNVFP4 ? 'NVFP4 is recommended for this hardware' : 'FP8 is recommended for this hardware' },
|
{ id: 'mxfp4', label: 'MXFP4', subtitle: 'gfx950', default: isGfx950, disabled: !isGfx950, disabledReason: !isGfx950 ? 'MXFP4 verified on MI355X (gfx950)' : '' },
|
||||||
|
{ id: 'bf16', label: 'BF16', subtitle: 'Full Weights', default: isAMD && !isGfx950, disabled: !isAMD, disabledReason: supportsNVFP4 ? 'NVFP4 is recommended for this hardware' : 'FP8 is recommended for this hardware' },
|
||||||
{ id: 'fp8', label: 'FP8', subtitle: 'High Throughput', default: !isAMD && !supportsNVFP4, disabled: isAMD || supportsNVFP4, disabledReason: isAMD ? 'FP8 not verified on AMD' : (supportsNVFP4 ? 'NVFP4 is recommended for this hardware' : '') },
|
{ id: 'fp8', label: 'FP8', subtitle: 'High Throughput', default: !isAMD && !supportsNVFP4, disabled: isAMD || supportsNVFP4, disabledReason: isAMD ? 'FP8 not verified on AMD' : (supportsNVFP4 ? 'NVFP4 is recommended for this hardware' : '') },
|
||||||
{ id: 'nvfp4', label: 'NVFP4', subtitle: 'Blackwell FP4', default: isB300 || isGB300, disabled: !supportsNVFP4, disabledReason: !supportsNVFP4 ? 'NVFP4 only on B300/GB300' : '' }
|
{ id: 'nvfp4', label: 'NVFP4', subtitle: 'Blackwell FP4', default: isB300 || isGB300, disabled: !supportsNVFP4, disabledReason: !supportsNVFP4 ? 'NVFP4 only on B300/GB300' : '' }
|
||||||
];
|
];
|
||||||
@@ -58,15 +62,6 @@ export const GLM51Deployment = () => {
|
|||||||
{ id: 'disabled', label: 'Disabled', subtitle: 'Low Latency', default: true },
|
{ id: 'disabled', label: 'Disabled', subtitle: 'Low Latency', default: true },
|
||||||
{ id: 'enabled', label: 'Enabled', subtitle: 'High Throughput', default: false }
|
{ id: 'enabled', label: 'Enabled', subtitle: 'High Throughput', default: false }
|
||||||
]
|
]
|
||||||
},
|
|
||||||
speculative: {
|
|
||||||
name: 'speculative',
|
|
||||||
title: 'Speculative Decoding',
|
|
||||||
condition: (values) => !['mi300x', 'mi325x', 'mi355x'].includes(values.hardware),
|
|
||||||
items: [
|
|
||||||
{ id: 'disabled', label: 'Disabled', default: false },
|
|
||||||
{ id: 'enabled', label: 'Enabled', default: true }
|
|
||||||
]
|
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
@@ -77,7 +72,7 @@ export const GLM51Deployment = () => {
|
|||||||
gb300: { nvfp4: { tp: 4, mem: 0.80 }, fp8: { tp: 4, mem: 0.9 } },
|
gb300: { nvfp4: { tp: 4, mem: 0.80 }, fp8: { tp: 4, mem: 0.9 } },
|
||||||
mi300x: { bf16: { tp: 8, mem: 0.80 } },
|
mi300x: { bf16: { tp: 8, mem: 0.80 } },
|
||||||
mi325x: { bf16: { tp: 8, mem: 0.80 } },
|
mi325x: { bf16: { tp: 8, mem: 0.80 } },
|
||||||
mi355x: { bf16: { tp: 8, mem: 0.80 } }
|
mi355x: { bf16: { tp: 8, mem: 0.80 }, mxfp4: { tp: 4, mem: 0.85 } }
|
||||||
};
|
};
|
||||||
|
|
||||||
const resolveItems = (option, values) => {
|
const resolveItems = (option, values) => {
|
||||||
@@ -135,17 +130,22 @@ export const GLM51Deployment = () => {
|
|||||||
const generateCommand = () => {
|
const generateCommand = () => {
|
||||||
const { hardware, quantization } = values;
|
const { hardware, quantization } = values;
|
||||||
const isAMD = ['mi300x', 'mi325x', 'mi355x'].includes(hardware);
|
const isAMD = ['mi300x', 'mi325x', 'mi355x'].includes(hardware);
|
||||||
|
const isGfx950 = hardware === 'mi355x'; // MI350X identical (cooling only)
|
||||||
const recommendsNVFP4 = ['b300', 'gb300'].includes(hardware);
|
const recommendsNVFP4 = ['b300', 'gb300'].includes(hardware);
|
||||||
const effectiveQuant = isAMD ? 'bf16' : (recommendsNVFP4 ? 'nvfp4' : 'fp8');
|
const effectiveQuant = isAMD
|
||||||
|
? (isGfx950 && quantization === 'mxfp4' ? 'mxfp4' : 'bf16')
|
||||||
|
: (recommendsNVFP4 ? 'nvfp4' : 'fp8');
|
||||||
const suffix = effectiveQuant === 'fp8' ? '-FP8' : '';
|
const suffix = effectiveQuant === 'fp8' ? '-FP8' : '';
|
||||||
const modelName = effectiveQuant === 'nvfp4' ? 'nvidia/GLM-5.1-NVFP4' : `zai-org/GLM-5.1${suffix}`;
|
const modelName =
|
||||||
|
effectiveQuant === 'nvfp4' ? 'nvidia/GLM-5.1-NVFP4'
|
||||||
|
: effectiveQuant === 'mxfp4' ? 'amd/GLM-5.1-MXFP4'
|
||||||
|
: `zai-org/GLM-5.1${suffix}`;
|
||||||
|
|
||||||
const hwConfig = modelConfigs[hardware][effectiveQuant];
|
const hwConfig = modelConfigs[hardware][effectiveQuant];
|
||||||
if (!hwConfig) return '# Configuration not available for the selected hardware and quantization.';
|
if (!hwConfig) return '# Configuration not available for the selected hardware and quantization.';
|
||||||
|
|
||||||
const tpValue = hwConfig.tp;
|
const tpValue = hwConfig.tp;
|
||||||
const memFraction = hwConfig.mem;
|
const memFraction = hwConfig.mem;
|
||||||
const enableSpec = values.speculative === 'enabled';
|
|
||||||
|
|
||||||
let cmd = 'sglang serve \\\n';
|
let cmd = 'sglang serve \\\n';
|
||||||
cmd += ` --model-path ${modelName}`;
|
cmd += ` --model-path ${modelName}`;
|
||||||
@@ -158,6 +158,7 @@ export const GLM51Deployment = () => {
|
|||||||
|
|
||||||
if (isAMD) {
|
if (isAMD) {
|
||||||
cmd += ' \\\n --trust-remote-code';
|
cmd += ' \\\n --trust-remote-code';
|
||||||
|
if (effectiveQuant === 'mxfp4') cmd += ' \\\n --kv-cache-dtype fp8_e4m3';
|
||||||
cmd += ' \\\n --dsa-prefill-backend tilelang';
|
cmd += ' \\\n --dsa-prefill-backend tilelang';
|
||||||
cmd += ' \\\n --dsa-decode-backend tilelang';
|
cmd += ' \\\n --dsa-decode-backend tilelang';
|
||||||
cmd += ' \\\n --chunked-prefill-size 131072';
|
cmd += ' \\\n --chunked-prefill-size 131072';
|
||||||
@@ -169,7 +170,9 @@ export const GLM51Deployment = () => {
|
|||||||
}
|
}
|
||||||
if (values.reasoning === 'enabled') cmd += ' \\\n --reasoning-parser glm45';
|
if (values.reasoning === 'enabled') cmd += ' \\\n --reasoning-parser glm45';
|
||||||
if (values.toolcall === 'enabled') cmd += ' \\\n --tool-call-parser glm47';
|
if (values.toolcall === 'enabled') cmd += ' \\\n --tool-call-parser glm47';
|
||||||
if (enableSpec) {
|
// EAGLE MTP speculative decoding: emitted by default (recommended). Excluded
|
||||||
|
// only on MI300X/MI325X (gfx942), where it is not yet verified.
|
||||||
|
if (!['mi300x', 'mi325x'].includes(hardware)) {
|
||||||
cmd += ' \\\n --speculative-algorithm EAGLE';
|
cmd += ' \\\n --speculative-algorithm EAGLE';
|
||||||
cmd += ' \\\n --speculative-num-steps 3';
|
cmd += ' \\\n --speculative-num-steps 3';
|
||||||
cmd += ' \\\n --speculative-eagle-topk 1';
|
cmd += ' \\\n --speculative-eagle-topk 1';
|
||||||
|
|||||||
Reference in New Issue
Block a user