From 00081a00d5b3b8111df81151db7cbd518612c549 Mon Sep 17 00:00:00 2001 From: Xinyuan Tong <115166877+JustinTong0323@users.noreply.github.com> Date: Tue, 16 Jun 2026 18:18:34 +0100 Subject: [PATCH] docs(cookbook): tune GLM-5.2 MTP to 5-1-6 and simplify launch flags (#28448) --- .../cookbook/autoregressive/GLM/GLM-5.2.mdx | 4 +- docs_new/src/snippets/_deployment.jsx | 4 +- .../configs/zai-org/glm-5.2-benchmarks.jsx | 17 +++-- .../src/snippets/configs/zai-org/glm-5.2.jsx | 65 ++++++------------- 4 files changed, 36 insertions(+), 54 deletions(-) diff --git a/docs_new/cookbook/autoregressive/GLM/GLM-5.2.mdx b/docs_new/cookbook/autoregressive/GLM/GLM-5.2.mdx index 0ef366ce8..31501df1b 100644 --- a/docs_new/cookbook/autoregressive/GLM/GLM-5.2.mdx +++ b/docs_new/cookbook/autoregressive/GLM/GLM-5.2.mdx @@ -92,8 +92,8 @@ import { Playground } from "/src/snippets/_playground.jsx"; ## 2. Configuration Tips -- **DeepSeek Sparse Attention (DSA).** GLM-5.2 uses the `glm_moe_dsa` architecture; SGLang auto-selects the DSA attention backends (`flashmla_sparse` prefill, `fa3` decode, `sgl-kernel` indexer topk). No attention-backend flag is needed on the supported hardware. -- **MTP / speculative decoding.** The checkpoint ships one nextn layer. Enable EAGLE MTP for lower latency (`--speculative-algorithm EAGLE --speculative-num-steps 3 --speculative-eagle-topk 1 --speculative-num-draft-tokens 4` for low-latency; `1-1-2` for balanced). The config's `index_share_for_mtp_iteration` reuses the DSA indexer's topk across draft steps (effective only at `--speculative-eagle-topk 1`). +- **DeepSeek Sparse Attention (DSA).** GLM-5.2 uses the `glm_moe_dsa` architecture; SGLang auto-selects the DSA attention backends (`flashmla_sparse` prefill, `fa3` decode, `sgl-kernel` indexer topk). No attention-backend flag is needed on the supported hardware. SGLang also auto-selects the KV-cache dtype for DSA models — `fp8_e4m3` on Blackwell (B200/GB300/B300, which then routes DSA through the TensorRT-LLM backend) and `bf16` on Hopper (H200) — so no `--kv-cache-dtype` flag is required. +- **MTP / speculative decoding.** The checkpoint ships one nextn layer. Enable EAGLE MTP for lower latency (`--speculative-algorithm EAGLE --speculative-num-steps 5 --speculative-eagle-topk 1 --speculative-num-draft-tokens 6` for low-latency; `1-1-2` for balanced). The config's `index_share_for_mtp_iteration` reuses the DSA indexer's topk across draft steps (effective only at `--speculative-eagle-topk 1`). **Tune the draft length to the accept length.** GLM-5.2's MTP head is strong — accept length runs high (4+ in many workloads, near-saturating at 5–6 in low-latency runs). Watch the server's reported **accept length** and adjust `--speculative-num-steps` / `--speculative-num-draft-tokens` accordingly: while accept length stays close to the draft-token count there is headroom to push them higher (more accepted tokens per step); if it falls well below, lower them — every rejected draft token is wasted verification compute. - **Context Parallelism (CP) for long prefill.** DSA prefill CP splits the long-prefill attention across `--attn-cp-size` ranks. On **Hopper (H200)** this gives a large prefill-latency win at long context — e.g. round-robin CP (`--tp 8 --attn-cp-size 8 --enable-dsa-prefill-context-parallel --dsa-prefill-cp-mode round-robin-split`) cut 64K-token prefill TTFT roughly **2.5–2.8×** vs. plain TP8 in our testing. Trade-offs: CP partitions the KV pool (lower max context at the same `--mem-fraction-static`) and adds some decode-side overhead, so it pays off only for long sequences. **CP is currently verified on Hopper only** — the Blackwell (sm100) DSA-CP FP8 rope kernel is not yet adapted, so leave CP off on B200/GB300. - **Memory.** The FP8 weights are large (MoE total, not active params). Start around `--mem-fraction-static 0.8` on H200 (TP8) and tune up; raise it for the 4-GPU GB300 single-node layout (TP4). - **DP-Attention + DeepEP** for the balanced/high-throughput strategies spreads attention across data-parallel ranks and routes MoE through DeepEP. diff --git a/docs_new/src/snippets/_deployment.jsx b/docs_new/src/snippets/_deployment.jsx index 27378aeb3..dc67a9b40 100644 --- a/docs_new/src/snippets/_deployment.jsx +++ b/docs_new/src/snippets/_deployment.jsx @@ -512,7 +512,9 @@ export const Deployment = ({ config, benchmarks }) => { // just maps the serve port. multinode ? " --network host" : ` -p ${servePort}:${servePort}`, " -v ~/.cache/huggingface:/root/.cache/huggingface", - ` --env "HF_TOKEN={{HF_TOKEN}}"`, + // HF token only for gated checkpoints — configs that declare an HF_TOKEN placeholder. + ...(config.placeholders && config.placeholders.HF_TOKEN + ? [` --env "HF_TOKEN={{HF_TOKEN}}"`] : []), ...cellEnv.map((e) => ` --env ${e}`), " --ipc=host", ` ${image}`, diff --git a/docs_new/src/snippets/configs/zai-org/glm-5.2-benchmarks.jsx b/docs_new/src/snippets/configs/zai-org/glm-5.2-benchmarks.jsx index ac5902665..b1839b6dc 100644 --- a/docs_new/src/snippets/configs/zai-org/glm-5.2-benchmarks.jsx +++ b/docs_new/src/snippets/configs/zai-org/glm-5.2-benchmarks.jsx @@ -5,13 +5,16 @@ export const benchmarks = [ // ---- H200 + FP8 ---- (measured on the v0.5.13.post1 release image, flush-cache on every run) { + // EAGLE MTP 5-1-6 (was 3-1-4): accept ~5.96/6 → +31%/+15% throughput, -25%/-11% TPOT vs 3-1-4. + // KV stays bf16 (Hopper auto-default). fp8 KV measured worse on H200 (slower flashmla_kv prefill + // + lower decode throughput): conc=1 31 gpu / TTFT 838, conc=16 96 gpu / TTFT 6650. match: { hw: "h200", variant: "default", quant: "fp8", strategy: "low-latency", nodes: "single" }, sglang_version: "0.5.13.post1", speed: [ { workload: { dataset: "random", isl: 8192, osl: 1024, max_concurrency: 1 }, - ttft_ms: 740, tpot_ms: 4.06, tokens_per_sec_per_gpu: 26 }, + ttft_ms: 662, tpot_ms: 3.03, tokens_per_sec_per_gpu: 34 }, { workload: { dataset: "random", isl: 8192, osl: 1024, max_concurrency: 16 }, - ttft_ms: 5980, tpot_ms: 13.97, tokens_per_sec_per_gpu: 98 }, + ttft_ms: 5080, tpot_ms: 12.44, tokens_per_sec_per_gpu: 113 }, ], }, { @@ -36,13 +39,14 @@ export const benchmarks = [ }, // ---- B200 + FP8 ---- (measured on the v0.5.13.post1 release image, flush-cache on every run) { + // EAGLE MTP 5-1-6 (was 3-1-4): accept length ~5.98/6 → +33%/+22% throughput, -26%/-15% TPOT vs 3-1-4. match: { hw: "b200", variant: "default", quant: "fp8", strategy: "low-latency", nodes: "single" }, sglang_version: "0.5.13.post1", speed: [ { workload: { dataset: "random", isl: 8192, osl: 1024, max_concurrency: 1 }, - ttft_ms: 335, tpot_ms: 3.12, tokens_per_sec_per_gpu: 36 }, + ttft_ms: 334, tpot_ms: 2.30, tokens_per_sec_per_gpu: 48 }, { workload: { dataset: "random", isl: 8192, osl: 1024, max_concurrency: 16 }, - ttft_ms: 2710, tpot_ms: 8.05, tokens_per_sec_per_gpu: 172 }, + ttft_ms: 2777, tpot_ms: 6.84, tokens_per_sec_per_gpu: 209 }, ], }, { @@ -67,13 +71,14 @@ export const benchmarks = [ }, // ---- GB300 + FP8 ---- (4-GPU single node, TP4; measured on the v0.5.13.post1 release image, flush-cache on every run) { + // EAGLE MTP 5-1-6 (was 3-1-4): accept length ~5.98/6 → +34%/+24% throughput, -28%/-18% TPOT vs 3-1-4. match: { hw: "gb300", variant: "default", quant: "fp8", strategy: "low-latency", nodes: "single" }, sglang_version: "0.5.13.post1", speed: [ { workload: { dataset: "random", isl: 8192, osl: 1024, max_concurrency: 1 }, - ttft_ms: 398, tpot_ms: 3.86, tokens_per_sec_per_gpu: 59 }, + ttft_ms: 393, tpot_ms: 2.78, tokens_per_sec_per_gpu: 79 }, { workload: { dataset: "random", isl: 8192, osl: 1024, max_concurrency: 16 }, - ttft_ms: 3637, tpot_ms: 10.36, tokens_per_sec_per_gpu: 275 }, + ttft_ms: 3201, tpot_ms: 8.53, tokens_per_sec_per_gpu: 341 }, ], }, { diff --git a/docs_new/src/snippets/configs/zai-org/glm-5.2.jsx b/docs_new/src/snippets/configs/zai-org/glm-5.2.jsx index b6f908ff9..173257163 100644 --- a/docs_new/src/snippets/configs/zai-org/glm-5.2.jsx +++ b/docs_new/src/snippets/configs/zai-org/glm-5.2.jsx @@ -36,7 +36,6 @@ export const config = { PORT: { target: "command", label: "Bind port", default: "30000" }, NODE0_IP: { target: "command", label: "Head node IP", default: "" }, NODE_RANK: { target: "command", label: "This node rank", default: "" }, - HF_TOKEN: { target: "command", label: "HF token (Docker)", default: "" }, CURL_HOST: { target: "curl", label: "Server host", default: "localhost" }, CURL_PORT: { target: "curl", label: "Server port", default: "30000" }, }, @@ -141,10 +140,10 @@ sgl-eval run aime25 \\ options: [ { id: "current", label: "Inherited from base" }, { id: "off", label: "Off (greedy)" }, - { id: "mtp-314", label: "EAGLE / MTP 3-1-4", - flags: ["--speculative-algorithm EAGLE", "--speculative-num-steps 3", - "--speculative-eagle-topk 1", "--speculative-num-draft-tokens 4"] }, - { id: "mtp-112", label: "EAGLE / MTP 1-1-2", + { id: "mtp-516", label: "EAGLE / MTP 5-1-6 (low-latency)", + flags: ["--speculative-algorithm EAGLE", "--speculative-num-steps 5", + "--speculative-eagle-topk 1", "--speculative-num-draft-tokens 6"] }, + { id: "mtp-112", label: "EAGLE / MTP 1-1-2 (balanced)", flags: ["--speculative-algorithm EAGLE", "--speculative-num-steps 1", "--speculative-eagle-topk 1", "--speculative-num-draft-tokens 2"] }, ], @@ -174,13 +173,12 @@ sgl-eval run aime25 \\ verified: true, env: [], flags: [ - "--trust-remote-code", "--model-path {{MODEL_NAME}}", "--tp 8", "--speculative-algorithm EAGLE", - "--speculative-num-steps 3", + "--speculative-num-steps 5", "--speculative-eagle-topk 1", - "--speculative-num-draft-tokens 4", + "--speculative-num-draft-tokens 6", "--mem-fraction-static 0.8", "--cuda-graph-max-bs 32", "--host {{HOST_IP}}", @@ -192,7 +190,6 @@ sgl-eval run aime25 \\ verified: true, env: [], flags: [ - "--trust-remote-code", "--model-path {{MODEL_NAME}}", "--tp 8", "--dp 8", @@ -217,7 +214,6 @@ sgl-eval run aime25 \\ verified: true, env: [], flags: [ - "--trust-remote-code", "--model-path {{MODEL_NAME}}", "--tp 8", "--dp 8", @@ -239,13 +235,12 @@ sgl-eval run aime25 \\ verified: true, env: [], flags: [ - "--trust-remote-code", "--model-path {{MODEL_NAME}}", "--tp 8", "--speculative-algorithm EAGLE", - "--speculative-num-steps 3", + "--speculative-num-steps 5", "--speculative-eagle-topk 1", - "--speculative-num-draft-tokens 4", + "--speculative-num-draft-tokens 6", "--mem-fraction-static 0.8", "--cuda-graph-max-bs 32", "--host {{HOST_IP}}", @@ -257,7 +252,6 @@ sgl-eval run aime25 \\ verified: true, env: [], flags: [ - "--trust-remote-code", "--model-path {{MODEL_NAME}}", "--tp 8", "--dp 8", @@ -282,7 +276,6 @@ sgl-eval run aime25 \\ verified: true, env: [], flags: [ - "--trust-remote-code", "--model-path {{MODEL_NAME}}", "--tp 8", "--dp 8", @@ -307,13 +300,12 @@ sgl-eval run aime25 \\ verified: true, env: [], flags: [ - "--trust-remote-code", "--model-path {{MODEL_NAME}}", "--tp 4", "--speculative-algorithm EAGLE", - "--speculative-num-steps 3", + "--speculative-num-steps 5", "--speculative-eagle-topk 1", - "--speculative-num-draft-tokens 4", + "--speculative-num-draft-tokens 6", "--mem-fraction-static 0.85", "--cuda-graph-max-bs 32", "--host {{HOST_IP}}", @@ -325,7 +317,6 @@ sgl-eval run aime25 \\ verified: true, env: [], flags: [ - "--trust-remote-code", "--model-path {{MODEL_NAME}}", "--tp 4", "--dp 4", @@ -349,7 +340,6 @@ sgl-eval run aime25 \\ verified: true, env: [], flags: [ - "--trust-remote-code", "--model-path {{MODEL_NAME}}", "--tp 4", "--dp 4", @@ -373,13 +363,12 @@ sgl-eval run aime25 \\ verified: false, env: [], flags: [ - "--trust-remote-code", "--model-path {{MODEL_NAME}}", "--tp 8", "--speculative-algorithm EAGLE", - "--speculative-num-steps 3", + "--speculative-num-steps 5", "--speculative-eagle-topk 1", - "--speculative-num-draft-tokens 4", + "--speculative-num-draft-tokens 6", "--mem-fraction-static 0.8", "--cuda-graph-max-bs 32", "--host {{HOST_IP}}", @@ -391,7 +380,6 @@ sgl-eval run aime25 \\ verified: false, env: [], flags: [ - "--trust-remote-code", "--model-path {{MODEL_NAME}}", "--tp 8", "--dp 8", @@ -414,7 +402,6 @@ sgl-eval run aime25 \\ verified: false, env: [], flags: [ - "--trust-remote-code", "--model-path {{MODEL_NAME}}", "--tp 8", "--dp 8", @@ -439,13 +426,12 @@ sgl-eval run aime25 \\ verified: false, env: [], flags: [ - "--trust-remote-code", "--model-path {{MODEL_NAME}}", "--tp 8", "--speculative-algorithm EAGLE", - "--speculative-num-steps 3", + "--speculative-num-steps 5", "--speculative-eagle-topk 1", - "--speculative-num-draft-tokens 4", + "--speculative-num-draft-tokens 6", "--mem-fraction-static 0.9", "--cuda-graph-max-bs 32", "--host {{HOST_IP}}", @@ -457,7 +443,6 @@ sgl-eval run aime25 \\ verified: false, env: [], flags: [ - "--trust-remote-code", "--model-path {{MODEL_NAME}}", "--tp 8", "--speculative-algorithm EAGLE", @@ -477,7 +462,6 @@ sgl-eval run aime25 \\ verified: false, env: [], flags: [ - "--trust-remote-code", "--model-path {{MODEL_NAME}}", "--tp 8", "--mem-fraction-static 0.9", @@ -499,13 +483,12 @@ sgl-eval run aime25 \\ verified: false, env: [], flags: [ - "--trust-remote-code", "--model-path {{MODEL_NAME}}", "--tp 16", "--speculative-algorithm EAGLE", - "--speculative-num-steps 3", + "--speculative-num-steps 5", "--speculative-eagle-topk 1", - "--speculative-num-draft-tokens 4", + "--speculative-num-draft-tokens 6", "--mem-fraction-static 0.85", "--cuda-graph-max-bs 32", "--host {{HOST_IP}}", @@ -517,7 +500,6 @@ sgl-eval run aime25 \\ verified: false, env: [], flags: [ - "--trust-remote-code", "--model-path {{MODEL_NAME}}", "--tp 16", "--speculative-algorithm EAGLE", @@ -537,7 +519,6 @@ sgl-eval run aime25 \\ verified: false, env: [], flags: [ - "--trust-remote-code", "--model-path {{MODEL_NAME}}", "--tp 16", "--mem-fraction-static 0.85", @@ -552,13 +533,12 @@ sgl-eval run aime25 \\ verified: false, env: [], flags: [ - "--trust-remote-code", "--model-path {{MODEL_NAME}}", "--tp 16", "--speculative-algorithm EAGLE", - "--speculative-num-steps 3", + "--speculative-num-steps 5", "--speculative-eagle-topk 1", - "--speculative-num-draft-tokens 4", + "--speculative-num-draft-tokens 6", "--mem-fraction-static 0.85", "--cuda-graph-max-bs 32", "--host {{HOST_IP}}", @@ -570,7 +550,6 @@ sgl-eval run aime25 \\ verified: false, env: [], flags: [ - "--trust-remote-code", "--model-path {{MODEL_NAME}}", "--tp 16", "--speculative-algorithm EAGLE", @@ -590,7 +569,6 @@ sgl-eval run aime25 \\ verified: false, env: [], flags: [ - "--trust-remote-code", "--model-path {{MODEL_NAME}}", "--tp 16", "--mem-fraction-static 0.85", @@ -605,13 +583,12 @@ sgl-eval run aime25 \\ verified: false, env: [], flags: [ - "--trust-remote-code", "--model-path {{MODEL_NAME}}", "--tp 8", "--speculative-algorithm EAGLE", - "--speculative-num-steps 3", + "--speculative-num-steps 5", "--speculative-eagle-topk 1", - "--speculative-num-draft-tokens 4", + "--speculative-num-draft-tokens 6", "--mem-fraction-static 0.85", "--cuda-graph-max-bs 32", "--host {{HOST_IP}}", @@ -623,7 +600,6 @@ sgl-eval run aime25 \\ verified: false, env: [], flags: [ - "--trust-remote-code", "--model-path {{MODEL_NAME}}", "--tp 8", "--speculative-algorithm EAGLE", @@ -643,7 +619,6 @@ sgl-eval run aime25 \\ verified: false, env: [], flags: [ - "--trust-remote-code", "--model-path {{MODEL_NAME}}", "--tp 8", "--mem-fraction-static 0.85",