diff --git a/docs/cookbook/autoregressive/Moonshotai/Kimi-K3.mdx b/docs/cookbook/autoregressive/Moonshotai/Kimi-K3.mdx index e316b9b01..c9a88c033 100644 --- a/docs/cookbook/autoregressive/Moonshotai/Kimi-K3.mdx +++ b/docs/cookbook/autoregressive/Moonshotai/Kimi-K3.mdx @@ -40,14 +40,16 @@ For how to launch the image, see [Install → Method 3: Using Docker](../../../d ```bash Command -docker pull quay.io/ascend/sglang:main-cann9.0.0-a3 +docker pull quay.io/ascend/sglang:main-cann9.0.0-a3 # Ascend A3 Series +docker pull swr.cn-southwest-2.myhuaweicloud.com/base_image/dockerhub/lmsysorg/sglang:cann9.1.0-950-B070 # Ascend 950PR/DT Series ``` For host and platform setup, see the [NPU installation guide](../../../docs/hardware-platforms/ascend-npus/getting-started/installation) and the [quick start guide](../../../docs/hardware-platforms/ascend-npus/getting-started/quick_start). -**Weights (NPU):** [Kimi-K3-W4A8](https://www.modelscope.cn/models/sgl-npu/Kimi-K3-W4A8) (W4A8, 1.49 TB) · +**Weights (NPU):** [Kimi-K3](https://modelscope.cn/models/moonshotai/Kimi-K3) (950PR/DT Series, MXFP4) · +[Kimi-K3-W4A8](https://www.modelscope.cn/models/sgl-npu/Kimi-K3-W4A8) (A3 Series, W4A8, 1.49 TB) · [Kimi-K3-DSpark](https://www.modelscope.cn/models/RadixArk/Kimi-K3-DSpark) (DSPARK draft, 4.5 GB) @@ -56,7 +58,7 @@ For host and platform setup, see the -Pick your hardware, then the deployment shape and operating point. Node count follows the hardware recipe (B200 2×8, GB200 4×4, H100 4×8, B300 1×8, H200 2×8 — 4×8 on Unified High-Throughput, GB300 2×4, MI350X/MI355X 1×8, Ascend A3 Series 4×8 — 32 cards / 64 dies), so it is not a separate choice. If you serve the NVFP4 checkpoint (`nvidia/Kimi-K3-NVFP4`, the **Quantization** row in the panel below), use the `lmsysorg/sglang:dev-dev-kimi-k3-nvfp4` image. +Pick your hardware, then the deployment shape and operating point. Node count follows the hardware recipe (B200 2×8, GB200 4×4, H100 4×8, B300 1×8, H200 2×8 — 4×8 on Unified High-Throughput, GB300 2×4, MI350X/MI355X 1×8, Ascend A3 Series 4×8 — 32 cards / 64 dies, Ascend 950PR/DT Series 4×8), so it is not a separate choice. If you serve the NVFP4 checkpoint (`nvidia/Kimi-K3-NVFP4`, the **Quantization** row in the panel below), use the `lmsysorg/sglang:dev-dev-kimi-k3-nvfp4` image. **PD Mode** — `Unified` serves prefill and decode together. `Prefill` / `Decode` split them into dedicated pools (see [PD disaggregation](#3-4-pd-disaggregation)); `Prefill` ships two strategies, both chunked at 16k. On the 8-GPU platforms (B300 1×8, GB300 2×4), `Default` is TP8 and `Long-Context` is `--pp-size 8 --tp-size 1`. On the 16-GPU platforms (B200 2×8, GB200 4×4), both are `--pp-size 16 --tp-size 1` and differ only in `--mem-fraction-static` (0.85 vs 0.90) — deep PP is the throughput shape there, not just the long-context one (see [Deep PP](#deep-pp-for-prefill)). @@ -71,7 +73,7 @@ Pick your hardware, then the deployment shape and operating point. Node count fo **Spec Decode** — layers onto the strategy without changing it, on every platform except B200. DSPARK proposes 7 draft tokens per step (tune in the Playground) and requires `pp_size == 1`, so on B200 it also drops the pipeline and re-lays the same 16 GPUs flat: PP2 × TP8 → TP16, PP2 × DCPEP8 → DCPEP16. DFLASH has no published draft checkpoint. The win is largest on short interactive traffic and fades as the prompt grows. -`--mamba-full-memory-ratio` is the one sizing flag, computed live: set your average request length in the [Mamba ratio calculator](#mamba-ratio-calculator); everything else follows the panels, and the result is pinned into the command. (The Ascend A3 Series uses `--max-mamba-cache-size` instead.) +`--mamba-full-memory-ratio` is the one sizing flag, computed live: set your average request length in the [Mamba ratio calculator](#mamba-ratio-calculator); everything else follows the panels, and the result is pinned into the command. (The Ascend NPU recipes don't use the calculator and set no ratio at all — the Ascend path sizes both pools itself: the A3 Series pins `--max-mamba-cache-size` instead, and the A5 recipe serves with the radix cache off.) import { Deployment } from "/src/snippets/_deployment.jsx"; @@ -146,7 +148,7 @@ not been re-measured on any cell — re-measure before you rely on one. ## 2. Configuration Tips -**Memory: two pools, one flag.** K3 splits static memory into a worst-case-reserved **KDA state pool** (it sets the concurrency ceiling) and a paged **MLA KV pool**, divided by `--mamba-full-memory-ratio`. The command panel pins that flag to the [calculator](#mamba-ratio-calculator)'s output — set your average request length there; every other calculator input follows the panels. (On the Ascend A3 Series: `--max-mamba-cache-size`, no calculator.) After boot, read back `max_total_num_tokens` (the KV side) and the admitted-request cap (the state side). +**Memory: two pools, one flag.** K3 splits static memory into a worst-case-reserved **KDA state pool** (it sets the concurrency ceiling) and a paged **MLA KV pool**, divided by `--mamba-full-memory-ratio`. The command panel pins that flag to the [calculator](#mamba-ratio-calculator)'s output — set your average request length there; every other calculator input follows the panels. (On the Ascend NPU recipes there is no calculator and no ratio flag: the Ascend path sizes both pools itself — the A3 Series pins `--max-mamba-cache-size`, and the A5 recipe runs the radix cache off.) After boot, read back `max_total_num_tokens` (the KV side) and the admitted-request cap (the state side). Capacity levers, all in the Playground. Each trades precision or cache behavior for capacity — re-verify accuracy on your workload: @@ -180,6 +182,7 @@ Speculation: DSPARK holds block size + 1 (= 8) intermediate states per request | H100 4×8 | TP32/EP32, Marlin + FlashMLA | SM90a build of the K3 image; pin NCCL/Gloo to the same NIC on all nodes; least post-weight headroom (80 GB) | | MI350X/MI355X 1×8 | TP8 ROCm/AITER | AITER A8W4 FlyDSL MoE, Triton attention (`SGLANG_MLA_DECODE_TUNE=1` for gfx950 MLA decode geometry), graph bs up to 256, fp8 kvcache; DSPARK supported. Activation-quant and fused-KDA-decode knobs: [AMD ROCm/AITER environment](#amd-env) | | Ascend A3 Series 4×8 (32 cards / 64 dies) | TP64/DP4 + DeepEP | PD-mixed `Unified` only; DSPARK baked in; pin `GLOO`/`HCCL_SOCKET_IFNAME` on every node | +| Ascend 950PR/DT Series 4×8 | TP32/dp1 + DeepEP | PD-mixed `Unified` only; DSPARK baked in; shared experts / dense MLP shard over attention-TP (`--shared-experts-tp-size 4`); radix cache off; pin `GLOO`/`HCCL_SOCKET_IFNAME` on every node | **DCP notes** — the DCP cells are Balanced and High-Throughput on every Blackwell platform, in both the `Unified` and `Decode` roles: @@ -239,7 +242,7 @@ Pending update... ### 3.2 Tool Calling -Enable the `kimi_k3` tool-call parser (toggle **Tool Call Parser** in the **Parsers** card of the [Playground above](#playground)) to surface structured tool calls via `message.tool_calls`. Because K3 is a thinking model, the follow-up turn may put text in `reasoning_content` as well as `content` — print both. (Not yet supported on the Ascend A3 Series.) +Enable the `kimi_k3` tool-call parser (toggle **Tool Call Parser** in the **Parsers** card of the [Playground above](#playground)) to surface structured tool calls via `message.tool_calls`. Because K3 is a thinking model, the follow-up turn may put text in `reasoning_content` as well as `content` — print both. (Not yet supported on the Ascend NPU recipes — A3 and A5.) diff --git a/docs/src/snippets/_deployment.jsx b/docs/src/snippets/_deployment.jsx index 42c127978..b86f83774 100644 --- a/docs/src/snippets/_deployment.jsx +++ b/docs/src/snippets/_deployment.jsx @@ -147,10 +147,14 @@ export const Deployment = ({ config, benchmarks }) => { { id: "mi355x", label: "MI355X", vram: "288GB", multiNodeDockerFlags: [...AMD_RDMA_DOCKER_FLAGS] }, ], - // Ascend A3 Series: 1 card = 2 dies, so --tp-size is 2× the card - // count (32 cards -> --tp-size 64). + // Ascend device layout: one /dev/davinciN per core. An A3 Series card is + // the exception — 2 dies per card, so an 8-card node exposes 16 devices + // and --tp-size is twice the card count. A 950PR/DT Series card is a + // single core, so the device count and --tp-size follow the cards. Both + // counts feed the docker `--device` list (`npuDevices`). npu: [ - { id: "a3", label: "Ascend A3 Series", vram: "64GB/die" }, + { id: "a3", label: "A3 Series", vram: "64GB/die", npuDevices: 16 }, + { id: "a5", label: "950PR/DT Series", vram: "128GB", npuDevices: 8 }, ], }; @@ -819,14 +823,31 @@ export const Deployment = ({ config, benchmarks }) => { return (extra && extra.vendor) || "nvidia"; }; // `config.hardware` overrides by id, as in buildHardwareGroups. - const fabricFlagsOf = (hwId) => { + const catalogEntryOf = (hwId) => { const extra = (config.hardware || []).find((h) => h.id === hwId); - if (extra) return extra.multiNodeDockerFlags || []; + if (extra) return extra; for (const list of Object.values(HARDWARE_CATALOG)) { const hit = list.find((h) => h.id === hwId); - if (hit) return hit.multiNodeDockerFlags || []; + if (hit) return hit; } - return []; + return null; + }; + const fabricFlagsOf = (hwId) => + (catalogEntryOf(hwId) || {}).multiNodeDockerFlags || []; + // NPU cards are reached with --device, one per /dev/davinciN core; + // `npuDevices` carries the per-product-line count (16 on an A3 Series + // node, 8 on a 950PR/DT Series node), four devices per line as the host + // docs show. + const davinciLines = (devices) => { + const lines = []; + for (let i = 0; i < devices; i += 4) { + const group = []; + for (let k = i; k < Math.min(i + 4, devices); k++) { + group.push(`--device=/dev/davinci${k}`); + } + lines.push(" " + group.join(" ")); + } + return lines; }; const gpuAccessLines = vendorOf(sel.hw) === "amd" ? [ @@ -838,14 +859,10 @@ export const Deployment = ({ config, benchmarks }) => { ] : vendorOf(sel.hw) === "npu" ? [ - // NPU: --privileged grants the davinci devices (16 dies on an - // 8-card Ascend A3 Series node); the host CANN driver/firmware/state - // must be mounted in. + // NPU: --privileged grants the davinci devices; the host CANN + // driver/firmware/state must be mounted in. "docker run --privileged --shm-size=16g", - " --device=/dev/davinci0 --device=/dev/davinci1 --device=/dev/davinci2 --device=/dev/davinci3", - " --device=/dev/davinci4 --device=/dev/davinci5 --device=/dev/davinci6 --device=/dev/davinci7", - " --device=/dev/davinci8 --device=/dev/davinci9 --device=/dev/davinci10 --device=/dev/davinci11", - " --device=/dev/davinci12 --device=/dev/davinci13 --device=/dev/davinci14 --device=/dev/davinci15", + ...davinciLines((catalogEntryOf(sel.hw) || {}).npuDevices || 16), " --device=/dev/davinci_manager", " --device=/dev/hisi_hdc", " -v /usr/local/sbin:/usr/local/sbin", diff --git a/docs/src/snippets/_kimi_k3_mamba_ratio_calculator.jsx b/docs/src/snippets/_kimi_k3_mamba_ratio_calculator.jsx index 3a996b2ae..97a0b2baf 100644 --- a/docs/src/snippets/_kimi_k3_mamba_ratio_calculator.jsx +++ b/docs/src/snippets/_kimi_k3_mamba_ratio_calculator.jsx @@ -135,10 +135,18 @@ export const KimiK3MambaRatioCalculator = () => { const eff = derive(cfg.flags, cfg.env); const bs = derive(cfg.baseFlags.length ? cfg.baseFlags : cfg.flags, cfg.baseFlags.length ? cfg.baseEnv : cfg.env); - // A --max-mamba-cache-size cell sizes the pool explicitly — no ratio to - // compute or broadcast. - const explicitSizing = (cfg.baseFlags.length ? cfg.baseFlags : cfg.flags) - .some((f) => f.startsWith("--max-mamba-cache-size")); + // Recipes that size the dual pool without the ratio neither render one nor + // broadcast one: + // - a --max-mamba-cache-size cell pins the KDA slot count explicitly; + // - the Ascend NPU recipes size both pools internally and never set + // --mamba-full-memory-ratio (every NPU recipe carries --device npu). + const baseFlagList = cfg.baseFlags.length ? cfg.baseFlags : cfg.flags; + const explicitSizing = baseFlagList.some((f) => f.startsWith("--max-mamba-cache-size")); + const npuRecipe = baseFlagList.some((f) => { + const [head, ...rest] = f.trim().split(/[\s=]+/); + return head === "--device" && rest[0] === "npu"; + }); + const ratioNotApplicable = explicitSizing || npuRecipe; const { ratio, tp, dp, attnTp, dcp, kvDtype, ssmDtype, radixOff, strategy, skipLock, slots, specOn, replaySpec, block, pdRole } = eff; const valid = Number.isFinite(ratio) && ratio > 0 && length > 0 && 96 % attnTp === 0; const baseValid = Number.isFinite(bs.ratio) && bs.ratio > 0 && length > 0; @@ -155,18 +163,22 @@ export const KimiK3MambaRatioCalculator = () => { const cliFlag = valid ? `--mamba-full-memory-ratio ${result}` : ""; // Broadcast both results: the Deploy command takes the base-config value, - // the Playground's composed command takes the effective one. + // the Playground's composed command takes the effective one. On a recipe + // that does not use the ratio, broadcast nulls instead of skipping the + // dispatch — the panels must drop a ratio pinned for a recipe the reader + // has since left rather than keep injecting it. useEffect(() => { - if (explicitSizing) return; window.dispatchEvent( new CustomEvent("sglang-k3-mamba-ratio", { - detail: { - ratio: valid ? result : null, - baseRatio: baseValid ? baseResult : null, - }, + detail: ratioNotApplicable + ? { ratio: null, baseRatio: null } + : { + ratio: valid ? result : null, + baseRatio: baseValid ? baseResult : null, + }, }) ); - }, [result, valid, baseResult, baseValid, explicitSizing]); + }, [result, valid, baseResult, baseValid, ratioNotApplicable]); const copyFlag = () => { if (!cliFlag || typeof navigator === "undefined" || !navigator.clipboard) return; @@ -235,10 +247,12 @@ export const KimiK3MambaRatioCalculator = () => { specLabel, ].filter(Boolean); - if (explicitSizing) { + if (ratioNotApplicable) { return (
- This recipe sizes the KDA state pool explicitly with --max-mamba-cache-size, so the ratio calculator does not apply. + {explicitSizing + ? <>This recipe sizes the KDA state pool explicitly with --max-mamba-cache-size, so the ratio calculator does not apply. + : <>This recipe runs on Ascend NPUs, which size the KDA state pool and the MLA KV pool internally and never set --mamba-full-memory-ratio, so the ratio calculator does not apply.}
); } diff --git a/docs/src/snippets/_playground.jsx b/docs/src/snippets/_playground.jsx index 2fcf21377..11a35d14e 100644 --- a/docs/src/snippets/_playground.jsx +++ b/docs/src/snippets/_playground.jsx @@ -1723,9 +1723,25 @@ export const Playground = ({ config }) => { mi355x: AMD_RDMA_DOCKER_FLAGS, }; const fabricFlags = HW_MULTINODE_DOCKER_FLAGS[sel.hw] || []; - // Mirrors the vendor branch in _deployment.jsx: ROCm reaches its GPUs - // through /dev/kfd + /dev/dri and the video group, not --gpus all. + // Mirrors the vendor branches in _deployment.jsx: ROCm reaches its GPUs + // through /dev/kfd + /dev/dri and the video group (not --gpus all), and + // Ascend NPUs are reached with --device, one per /dev/davinciN core (16 + // on an A3 Series node, 8 on a 950PR/DT Series node — the catalog's + // `npuDevices`). const isAmdHw = /^mi\d/.test(sel.hw || ""); + const HW_NPU_DEVICES = { a3: 16, a5: 8 }; + const npuDevices = HW_NPU_DEVICES[sel.hw]; + const davinciLines = (devices) => { + const lines = []; + for (let i = 0; i < devices; i += 4) { + const group = []; + for (let k = i; k < Math.min(i + 4, devices); k++) { + group.push(`--device=/dev/davinci${k}`); + } + lines.push(" " + group.join(" ")); + } + return lines; + }; const dockerLines = [ ...(isAmdHw ? [ @@ -1735,6 +1751,21 @@ export const Playground = ({ config }) => { " --cap-add=SYS_PTRACE --security-opt seccomp=unconfined", " --shm-size 32g", ] + : npuDevices + ? [ + // NPU: --privileged grants the davinci devices; the host CANN + // driver/firmware/state must be mounted in. + "docker run --privileged --shm-size=16g", + ...davinciLines(npuDevices), + " --device=/dev/davinci_manager", + " --device=/dev/hisi_hdc", + " -v /usr/local/sbin:/usr/local/sbin", + " -v /usr/local/Ascend/driver:/usr/local/Ascend/driver", + " -v /usr/local/Ascend/firmware:/usr/local/Ascend/firmware", + " -v /etc/ascend_install.info:/etc/ascend_install.info", + " -v /var/queue_schedule:/var/queue_schedule", + " -v ~/.cache/:/root/.cache/", + ] : [ "docker run --gpus all", " --shm-size 32g", @@ -1743,7 +1774,8 @@ export const Playground = ({ config }) => { // A PD pair is cross-host even when each role is a single-node cell, so // the RDMA fabric flags are needed for `pdMode` too, not just multinode. ...((multinode || pdMode) ? fabricFlags.map((x) => " " + x) : []), - " -v ~/.cache/huggingface:/root/.cache/huggingface", + // The NPU device block already mounts ~/.cache/. + ...(npuDevices ? [] : [" -v ~/.cache/huggingface:/root/.cache/huggingface"]), ...(config.dockerMounts || []).map((mount) => ` -v ${mount}`), ` --env "HF_TOKEN={{HF_TOKEN}}"`, ...cellEnv.map((e) => ` --env ${e}`), diff --git a/docs/src/snippets/configs/moonshotai/kimi-k3.jsx b/docs/src/snippets/configs/moonshotai/kimi-k3.jsx index 561382eb2..87f803d7c 100644 --- a/docs/src/snippets/configs/moonshotai/kimi-k3.jsx +++ b/docs/src/snippets/configs/moonshotai/kimi-k3.jsx @@ -14,8 +14,10 @@ export const config = { // Low-Latency, PP2 × DCPEP8 on Balanced and High-Throughput, while DSPARK // re-lays the same 16 as flat TP16 / DCPEP16), GB200 (4×4 TP16 MNNVL), // H200 (2×8 TP16/EP16, or 4×8 TP32/EP32 for High-Throughput), H100 - // (4×8 TP32/EP32), and MI350X/MI355X (1×8 TP8) have serving recipes. - supportedHardware: ["b300", "gb300", "b200", "gb200", "h200", "h100", "mi350x", "mi355x", "a3"], + // (4×8 TP32/EP32), MI350X/MI355X (1×8 TP8), Ascend A3 Series (4×8, TP64 + // over 2-die cards), and Ascend 950PR/DT Series (4×8, TP32, one rank per + // card) have serving recipes. + supportedHardware: ["b300", "gb300", "b200", "gb200", "h200", "h100", "mi350x", "mi355x", "a3", "a5"], // ---- Cell introspection (config-internal; the engines ignore these keys) ---- // @@ -47,6 +49,15 @@ export const config = { const f = config.flagOf(cell, name); return f ? Number(f.split(/[\s=]/)[1]) || 1 : 1; }, + // The two NPU recipes (A3 and A5) each ship exactly one operating point — + // Unified PD, the Balanced strategy, DSPARK, no HiCache, no tool calling — + // so every panel gate below keys off this helper instead of enumerating + // platforms. Shape differences between the two (TP64/DP4 vs TP32/dp1, + // Modelslim W4A8 vs the MXFP4 checkpoint, mamba-cache sizing) are handled by + // the cells, the TP/DP knob rules and the Quantization axis, not here. + isNpuHw(s) { + return s.hw === "a3" || s.hw === "a5"; + }, // Does this recipe shard the TP-replicated MLA KV? Replaces the per-platform // "which cells carry DCP" tables the HiCache tiers used to hardcode. hasDcp(s) { @@ -139,14 +150,14 @@ export const config = { { id: "prefill", label: "Prefill", - disabled: (s) => s.hw === "a3", - disableReason: (s) => (s.hw === "a3" ? "Only Unified PD is supported on this recipe." : ""), + disabled: (s) => config.isNpuHw(s), + disableReason: (s) => (config.isNpuHw(s) ? "Only Unified PD is supported on this recipe." : ""), }, { id: "decode", label: "Decode", - disabled: (s) => s.hw === "a3", - disableReason: (s) => (s.hw === "a3" ? "Only Unified PD is supported on this recipe." : ""), + disabled: (s) => config.isNpuHw(s), + disableReason: (s) => (config.isNpuHw(s) ? "Only Unified PD is supported on this recipe." : ""), }, ], }, @@ -160,16 +171,16 @@ export const config = { id: "low-latency", label: "Low-Latency", showWhen: (s) => s.pdMode !== "prefill", - disabled: (s) => s.hw === "a3", - disableReason: (s) => (s.hw === "a3" ? "Only the Balanced operating point is supported on this recipe." : ""), + disabled: (s) => config.isNpuHw(s), + disableReason: (s) => (config.isNpuHw(s) ? "Only the Balanced operating point is supported on this recipe." : ""), }, { id: "balanced", label: "Balanced", showWhen: (s) => s.pdMode !== "prefill" }, { id: "high-throughput", label: "High-Throughput", showWhen: (s) => s.pdMode !== "prefill", - disabled: (s) => s.hw === "a3", - disableReason: (s) => (s.hw === "a3" ? "Only the Balanced operating point is supported on this recipe." : ""), + disabled: (s) => config.isNpuHw(s), + disableReason: (s) => (config.isNpuHw(s) ? "Only the Balanced operating point is supported on this recipe." : ""), }, { id: "default", label: "Default", showWhen: (s) => s.pdMode === "prefill" }, { id: "long-context", label: "Long-Context", showWhen: (s) => s.pdMode === "prefill" }, @@ -194,11 +205,15 @@ export const config = { default: "mxfp4", options: [ { id: "mxfp4", label: "MXFP4", subtitle: "Moonshot AI checkpoint", + // The 950PR/DT recipe serves this checkpoint; the A3 Series recipe + // serves the W4A8 build under it instead. disabled: (s) => s.hw === "a3", - disableReason: (s) => (s.hw === "a3" ? "Only Modelslim (W4A8) is supported on this recipe." : ""), + disableReason: (s) => (s.hw === "a3" ? "The A3 Series recipe serves the sgl-npu Modelslim (W4A8) checkpoint." : ""), }, { - // A3 Series only (NPU W4A8 checkpoint); hidden on the GPU recipes. + // A3 Series only (ModelSlim W4A8 checkpoint); hidden elsewhere — the + // 950PR/DT recipe serves the MXFP4 checkpoint above with no + // --quantization flag. id: "modelslim", label: "Modelslim (W4A8)", subtitle: "ModelScope NPU checkpoint", @@ -227,7 +242,7 @@ export const config = { id: "mmTransport", title: "VLM Transport", default: "auto", - showWhen: (s) => s.pdMode !== "decode" && s.hw !== "a3", + showWhen: (s) => s.pdMode !== "decode" && !config.isNpuHw(s), options: [ { id: "auto", @@ -266,8 +281,8 @@ export const config = { options: [ { id: "none", label: "Non-Spec", env: (s) => (["mi350x", "mi355x"].includes(s.hw) ? ["SGLANG_MLA_DECODE_TUNE=1"] : []), - disabled: (s) => s.hw === "a3", - disableReason: (s) => (s.hw === "a3" ? "Only DSPARK is supported on this recipe." : ""), + disabled: (s) => config.isNpuHw(s), + disableReason: (s) => (config.isNpuHw(s) ? "Only DSPARK is supported on this recipe." : ""), }, { id: "dspark", @@ -296,10 +311,10 @@ export const config = { "--speculative-algorithm DSPARK", "--speculative-draft-model-path RadixArk/Kimi-K3-DSpark", "--speculative-dspark-block-size 7", - // The NPU recipe adds the NPU draft path's own knobs (draft - // attention backend, topk 1, unquantized draft weights) on top of - // the common trio. - ...(s.hw === "a3" + // The NPU recipes (A3 and A5) add the NPU draft path's own knobs + // (draft attention backend, topk 1, unquantized draft weights) on + // top of the common trio. + ...(config.isNpuHw(s) ? [ "--speculative-draft-attention-backend ascend", "--speculative-eagle-topk 1", @@ -311,7 +326,7 @@ export const config = { // the Triton decode kernel, the K3 default). It is CUDA-only, and // the PD prefill role opts out — it never runs verify and rejects // the flag at startup. - ...(s.hw !== "a3" && s.pdMode !== "prefill" + ...(!config.isNpuHw(s) && s.pdMode !== "prefill" ? ["--enable-linear-replayssm-spec"] : []), ], @@ -346,8 +361,8 @@ export const config = { { id: "l2", label: "L1+L2 (host)", - disabled: (s) => s.hw === "a3", - disableReason: (s) => (s.hw === "a3" + disabled: (s) => config.isNpuHw(s), + disableReason: (s) => (config.isNpuHw(s) ? "NPU HiCache does not support mamba cache (K3's KDA state is one)." : ""), flags: [ @@ -370,8 +385,8 @@ export const config = { { id: "l3", label: "+ L3 (Mooncake)", - disabled: (s) => s.hw === "a3", - disableReason: (s) => (s.hw === "a3" + disabled: (s) => config.isNpuHw(s), + disableReason: (s) => (config.isNpuHw(s) ? "NPU HiCache does not support mamba cache (K3's KDA state is one)." : ""), flags: [ @@ -413,6 +428,10 @@ export const config = { default: "moonshotai/Kimi-K3", nvfp4: "nvidia/Kimi-K3-NVFP4", a3: "sgl-npu/Kimi-K3-W4A8", + // The 950PR/DT recipe serves the official Moonshot checkpoint (MXFP4) from + // ModelScope — the same id the NVIDIA recipes resolve to, fetched through + // SGLANG_USE_MODELSCOPE. The A3 Series recipe serves the W4A8 build. + a5: "moonshotai/Kimi-K3", }, placeholders: { @@ -468,6 +487,9 @@ export const config = { "b200|nvfp4": "lmsysorg/sglang:dev-dev-kimi-k3-nvfp4", "gb200|nvfp4": "lmsysorg/sglang:dev-dev-kimi-k3-nvfp4", a3: "quay.io/ascend/sglang:main-cann9.0.0-a3", + // 950PR/DT Series builds are CANN 9.1.0 and published from the Ascend SWR + // registry (the quay.io A2/A3 line carries no 950PR/DT tag). + a5: "swr.cn-southwest-2.myhuaweicloud.com/base_image/dockerhub/lmsysorg/sglang:cann9.1.0-950-B070", }, // Pre-selects the issue template's `model` field on "Submit verified cell". github: { @@ -492,6 +514,10 @@ export const config = { when: { hw: ["a3"] }, reason: "Only TP64 is supported on this recipe.", }, + { + when: { hw: ["a5"] }, + reason: "Only TP32 is supported on this recipe.", + }, ]; }, }, { @@ -501,6 +527,10 @@ export const config = { when: { hw: ["a3"] }, reason: "Only TP64 is supported on this recipe.", }, + { + when: { hw: ["a5"] }, + reason: "Only TP32 is supported on this recipe.", + }, { when: { hw: ["b300", "gb300"] }, reason: "TP=16 needs 16 ranks; the B300 and GB300 recipes have 8 ranks.", @@ -513,14 +543,27 @@ export const config = { ]; }, }, { - // A3 Series only: 64 ranks (4 nodes × 8 cards × 2 dies); hidden on the GPU recipes. + // 950PR/DT Series only: 32 ranks (4 nodes × 8 cards, one rank per + // card); hidden on the other recipes. + value: 32, + hide: { hw: ["b300", "gb300", "b200", "gb200", "h200", "h100", "mi350x", "mi355x", "a3"] }, + }, + { + // A3 Series only: 64 ranks (4 nodes × 8 cards × 2 dies); hidden on + // the other recipes. value: 64, - hide: { hw: ["b300", "gb300", "b200", "gb200", "h200", "h100", "mi350x", "mi355x"] }, + hide: { hw: ["b300", "gb300", "b200", "gb200", "h200", "h100", "mi350x", "mi355x", "a5"] }, }, ]}, { id: "dpAttn", label: "DP-Attention", values: [ null, + { + // 950PR/DT Series only: the recipe enables DP-Attention at dp=1 + // (attn-TP 32), unlike A3's dp=4. + value: 1, + hide: { hw: ["b300", "gb300", "b200", "gb200", "h200", "h100", "mi350x", "mi355x", "a3"] }, + }, { value: false, get disable() { return [ @@ -528,6 +571,10 @@ export const config = { when: { hw: ["a3"] }, reason: "Only DP-Attention=4 is supported on this recipe.", }, + { + when: { hw: ["a5"] }, + reason: "Only DP-Attention=1 is supported on this recipe.", + }, ]; }, }, { @@ -537,9 +584,21 @@ export const config = { when: { hw: ["a3"] }, reason: "Only DP-Attention=4 is supported on this recipe.", }, + { + when: { hw: ["a5"] }, + reason: "Only DP-Attention=1 is supported on this recipe.", + }, + ]; }, + }, + { + value: 4, + get disable() { return [ + { + when: { hw: ["a5"] }, + reason: "Only DP-Attention=1 is supported on this recipe.", + }, ]; }, }, - 4, { value: 8, get disable() { return [ @@ -547,6 +606,10 @@ export const config = { when: { hw: ["a3"] }, reason: "Only DP-Attention=4 is supported on this recipe.", }, + { + when: { hw: ["a5"] }, + reason: "Only DP-Attention=1 is supported on this recipe.", + }, { when: { hw: ["b300", "gb300"] }, reason: "On an 8-rank deployment (B300 1×8, GB300 2×4) dp=8 leaves attn_tp=1, so each rank holds the full unsharded MLA KV and OOMs — prefer dp=2/attn_tp=4.", @@ -565,6 +628,10 @@ export const config = { when: { hw: ["a3"] }, reason: "Only DP-Attention=4 is supported on this recipe.", }, + { + when: { hw: ["a5"] }, + reason: "Only DP-Attention=1 is supported on this recipe.", + }, { when: { hw: ["b300", "gb300"] }, reason: "DP-Attention=16 needs 16 TP ranks; the B300 and GB300 recipes have 8.", @@ -595,10 +662,10 @@ export const config = { // Blackwell-only: runs FlashInfer's official trtllm-gen SiTU kernels. { id: "flashinfer_mxfp4", label: "FlashInfer (MXFP4)", flags: ["--moe-runner-backend flashinfer_mxfp4"], requiresHw: ["b200", "b300", "gb200", "gb300"], - disable: [{ when: { hw: ["a3"] }, + disable: [{ when: { hw: ["a3", "a5"] }, reason: "FlashInfer is a CUDA kernel and is not supported on NPU." }] }, { id: "marlin", label: "Marlin (W4A16)", flags: ["--moe-runner-backend marlin"], - disable: [{ when: { hw: ["a3"] }, + disable: [{ when: { hw: ["a3", "a5"] }, reason: "Marlin is a CUDA kernel and is not supported on NPU." }] }, ], }, @@ -614,7 +681,7 @@ export const config = { ], }, ep: { - showWhen: (b) => b.hw !== "a3", + showWhen: (b) => !config.isNpuHw(b), label: "EP", values: [ null, 1, 2, 4, 8, @@ -640,9 +707,10 @@ export const config = { parsers: { items: [ { id: "reasoning", label: "Reasoning Parser", flag: "--reasoning-parser kimi_k3" }, - // Tool calling is not yet supported on the NPU, so the item hides on a3. + // Tool calling is not yet supported on the NPU recipes, so the item + // hides on a3 and a5. { id: "toolCall", label: "Tool Call Parser", flag: "--tool-call-parser kimi_k3", - hide: { hw: ["a3"] } }, + hide: { hw: ["a3", "a5"] } }, ], }, @@ -719,8 +787,8 @@ export const config = { // EAGLE --speculative-num-steps N (chain; topk>1 is a tree) // Only DSPARK is selectable today, so only its form is emitted. id: "proposedDraftTokens", title: "Proposed Draft Tokens", - // The A3 Series recipe pins the shipped block size (7). - showWhen: (b) => b.spec === "dspark" && b.hw !== "a3", + // The NPU recipes pin the shipped block size (7). + showWhen: (b) => b.spec === "dspark" && !config.isNpuHw(b), control: "slider", stripPrefixes: [ "--speculative-dspark-block-size", @@ -743,10 +811,10 @@ export const config = { // Spec-only, so gate the row on DSPARK; every DSPARK recipe (except the PD // prefill role) turns it on in the base, so this row derives to On and // exists mainly as the opt-out. - // Needs the Triton linear-attn decode backend (the K3 default); the A3 Series - // script never sets it. + // Needs the Triton linear-attn decode backend (the K3 default); the NPU + // recipes never enable it. id: "replaySsm", title: "ReplaySSM (spec)", - showWhen: (b) => b.spec === "dspark" && b.hw !== "a3", + showWhen: (b) => b.spec === "dspark" && !config.isNpuHw(b), stripPrefixes: ["--enable-linear-replayssm-spec"], options: [ { id: "off", label: "Off" }, @@ -766,8 +834,8 @@ export const config = { // without --speculative-dspark-sps-table-path (every step still // verifies full width); fails fast with ReplaySSM or DCP > 1. id: "raggedVerify", title: "Ragged Verify Mode (spec)", - // The A3 Series recipe pins static. - showWhen: (b) => b.spec === "dspark" && b.hw !== "a3", + // The NPU recipes pin static. + showWhen: (b) => b.spec === "dspark" && !config.isNpuHw(b), stripEnv: ["SGLANG_RAGGED_VERIFY_MODE"], options: [ { id: "static", label: "Auto (static)" }, @@ -776,7 +844,7 @@ export const config = { }, { id: "kvCacheDtype", title: "KV Cache Precision", - showWhen: (b) => b.hw !== "a3", + showWhen: (b) => !config.isNpuHw(b), stripPrefixes: ["--kv-cache-dtype"], options: [ { id: "auto", label: "Auto (BF16)" }, @@ -785,7 +853,7 @@ export const config = { }, { id: "mambaSsmDtype", title: "KDA State Precision", - showWhen: (b) => b.hw !== "a3", + showWhen: (b) => !config.isNpuHw(b), stripPrefixes: ["--mamba-ssm-dtype"], options: [ { id: "auto", label: "Auto (FP32)" }, @@ -800,7 +868,7 @@ export const config = { // Off suits prefix-free traffic (offline batch, evals): 1 state slot // per request instead of 4-5. id: "prefixCache", title: "Prefix Cache", - showWhen: (b) => b.hw !== "a3", + showWhen: (b) => !config.isNpuHw(b), stripPrefixes: ["--disable-radix-cache"], options: [ { id: "on", label: "On" }, @@ -812,7 +880,7 @@ export const config = { // prefix cache off, so the row hides (and stops emitting) there. // Slot cost per request: extra_buffer 5, extra_buffer_lazy 4. id: "mambaRadix", title: "KDA Radix Cache Strategy", - showWhen: (b, v, d) => (b.hw !== "a3") + showWhen: (b, v, d) => (!config.isNpuHw(b)) && ((((v && v.prefixCache) ?? (d && d.prefixCache)) !== "off")), stripPrefixes: ["--mamba-radix-cache-strategy"], options: [ @@ -827,7 +895,7 @@ export const config = { // (extra_buffer 5→4, extra_buffer_lazy 4→3; no_buffer stays 3). Off by // default. Env var, not a flag, so it emits via env/stripEnv. id: "mambaSlotSaving", title: "KDA Slot Saving (experimental)", - showWhen: (b) => b.hw !== "a3", + showWhen: (b) => !config.isNpuHw(b), stripEnv: ["SGLANG_OPT_MAMBA_SKIP_DECODE_LOCK"], options: [ { id: "off", label: "Off" }, @@ -863,7 +931,7 @@ export const config = { { // Only meaningful with EP a2a on (MoE card or a large-scale preset). id: "eplb", title: "Expert Rebalancing (EPLB)", - showWhen: (b) => b.hw !== "a3", + showWhen: (b) => !config.isNpuHw(b), stripPrefixes: ["--enable-eplb"], options: [ { id: "off", label: "Off" }, @@ -875,7 +943,7 @@ export const config = { // Validated on the no-a2a MXFP4 runner; untested against SBO (EP a2a) // and DP attention. id: "prefillGraph", title: "Prefill CUDA Graph", - showWhen: (b) => b.hw !== "a3", + showWhen: (b) => !config.isNpuHw(b), stripPrefixes: ["--cuda-graph-backend-prefill"], options: [ { id: "auto", label: "Auto (off)" }, @@ -898,7 +966,7 @@ export const config = { // resolves it into the full parallelism shape (tp/ep/dp/dcp; attn-tp = // tp/dp); pool sizing rides the calculator-driven ratio. id: "lsGpus", title: "Cluster Size (large-scale)", - showWhen: (b) => (b.hw !== "a3") && (b.pdMode === undefined || b.pdMode === "unified"), + showWhen: (b) => (!config.isNpuHw(b)) && (b.pdMode === undefined || b.pdMode === "unified"), // Default follows the base cell's own GPU count (tp8 lanes -> 8, // tp16 lanes -> 16), so a preset starts from "same hardware, new shape". default: (b) => @@ -914,7 +982,7 @@ export const config = { }, { id: "lsPreset", title: "Large-Scale Preset", - showWhen: (b) => (b.hw !== "a3") && (b.pdMode === undefined || b.pdMode === "unified"), + showWhen: (b) => (!config.isNpuHw(b)) && (b.pdMode === undefined || b.pdMode === "unified"), stripPrefixes: [ "--tp-size", "--tp", "--tensor-parallel-size", "--ep-size", "--ep", "--expert-parallel-size", @@ -2376,6 +2444,89 @@ export const config = { "--port {{PORT}}", ], }, + { + // Ascend 950PR/DT Series: 4 nodes × 8 cards, one rank per card (TP32). + // Unified PD, Balanced, DSPARK-only, with the product-line kernels armed + // per env: FIAS V2 BSND for the DSpark target-verify/draft attention + // paths and the fine-grained dual-stream MoE overlap. DP-attention runs + // at dp=1 (attn-TP 32), and the shared experts / dense MLP shard across + // attention-TP through the server flags (--shared-experts-tp-size 4). + // Checkpoint: the official Moonshot MXFP4 build (moonshotai/Kimi-K3, + // fetched from ModelScope by SGLANG_USE_MODELSCOPE=1 above). Its routed + // experts declare compressed-tensors "mxfp4-pack-quantized", which the + // loader detects from the checkpoint itself — the NPU MXFP4 MoE scheme + // serves them — so the recipe passes no --quantization flag (attention, + // shared experts and the dense MLP are ignored by the checkpoint and stay + // BF16). A ModelSlim (W4A8) checkpoint would need --quantization modelslim + // and a modelslim-style --model-path instead; the two are not mixable. + // Pool sizing is internal to the Ascend path: the KDA state and MLA KV + // pools are sized by the runtime, so the recipe sets neither + // --mamba-full-memory-ratio nor --max-mamba-cache-size, and the radix + // cache is off (one KDA state slot per request) — the ratio calculator + // does not apply. + // --disable-custom-all-reduce is carried verbatim from the recipe script. + // Ascend has no custom all-reduce kernel, so the NPU backend resolves the + // flag to True on its own (hardware_backend/npu/utils.py); spelling it out + // keeps the panel command identical to the command that was measured. + // The `unset` lines of the source script (ASCEND_CUSTOM_OPP_PATH, + // SGLANG_NPU_FUSED_MOE_MODE, ENABLE_PROFILING, the K3 trace files) are + // launch-script hygiene against stale profiling state, not recipe env — + // a panel command starts clean, so they don't carry over. + match: { hw: "a5", pdMode: "unified", strategy: "balanced" }, + nnodes: 4, + verified: false, + verificationStatus: "in-progress", + env: [ + "SGLANG_USE_MODELSCOPE=1", + "GLOO_SOCKET_IFNAME={{NETWORK_IFACE}}", + "HCCL_SOCKET_IFNAME={{NETWORK_IFACE}}", + "PYTORCH_NPU_ALLOC_CONF=expandable_segments:True", + "SGLANG_NPU_USE_FIAS_V2_BSND=True", + "SGLANG_NPU_FINE_GRAINED_MOE_DUAL_STREAM=True", + "SGLANG_ENABLE_OVERLAP_PLAN_STREAM=1", + "SGLANG_ENABLE_SPEC_V2=1", + "SGLANG_RAGGED_VERIFY_MODE=static", + "SGLANG_DSPARK_FOLDED_PROPOSAL=0", + "SGLANG_DSPARK_FOLDED_SAMPLING=0", + "SGLANG_DSPARK_STACKED_CTX_KV=0", + "SGLANG_DSPARK_EMBED_IN_GRAPH=0", + "STREAMS_PER_DEVICE=32", + "DEEP_NORMAL_MODE_USE_INT8_QUANT=1", + "SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK=128", + "HCCL_BUFFSIZE=2000", + "DEEPEP_NORMAL_LONG_SEQ_ROUND=64", + "DEEPEP_NORMAL_LONG_SEQ_PER_ROUND_TOKENS=512", + "HCCL_OP_EXPANSION_MODE=AIV", + ], + flags: [ + "--trust-remote-code", + "--model-path {{MODEL_NAME}}", + "--tokenizer-path {{MODEL_NAME}}", + "--attention-backend ascend", + "--device npu", + "--dtype bfloat16", + "--tp-size 32", + "--enable-dp-attention", + "--enable-dp-lm-head", + "--mem-fraction-static 0.9", + "--chunked-prefill-size 8192", + "--cuda-graph-bs-decode 32", + "--max-running-requests 32", + "--enable-shared-experts-attn-tp", + "--enable-dense-mlp-attn-tp", + "--shared-experts-tp-size 4", + "--reasoning-parser kimi_k3", + "--moe-a2a-backend deepep", + "--deepep-mode auto", + "--linear-attn-verify-backend triton", + "--disable-radix-cache", + "--disable-custom-all-reduce", + "--watchdog-timeout 9000", + "--model-loader-extra-config '{\"enable_multithread_load\": true}'", + "--host {{HOST_IP}}", + "--port {{PORT}}", + ], + }, ], // Cross-node fabric env (substitute the NIC used by every rank). @@ -2418,5 +2569,15 @@ export const config = { " SGLANG_HOST_IP= # this node's IP on that NIC", "If running outside the official image, source set_env.sh on every node first.", ], + a5: [ + "Run the same command on all four nodes with --node-rank 0/1/2/3.", + "NPU collectives use HCCL. Pin the cross-node NIC on EVERY node:", + " GLOO_SOCKET_IFNAME= # bootstrap interface", + " HCCL_SOCKET_IFNAME= # HCCL transport interface", + " SGLANG_HOST_IP= # this node's IP on that NIC", + "If running outside the official image, source both set_env.sh scripts on every", + "node first: /usr/local/Ascend/ascend-toolkit/set_env.sh and", + "/usr/local/Ascend/nnal/atb/set_env.sh.", + ], }, };