docs: add NVFP4 quantization option to Kimi-K3 deploy panel (#35168)
This commit is contained in:
@@ -40,7 +40,7 @@ For how to launch the image, see [Install → Method 3: Using Docker](../../../d
|
|||||||
|
|
||||||
</Accordion>
|
</Accordion>
|
||||||
|
|
||||||
Pick your hardware, then the deployment shape and operating point. Node count follows the hardware recipe (B200 2×8, GB200 4×4, H100 4×8, B300 1×8, H200 2×8 — 4×8 on Unified High-Throughput, GB300 2×4, MI350X/MI355X 1×8), so it is not a separate choice.
|
Pick your hardware, then the deployment shape and operating point. Node count follows the hardware recipe (B200 2×8, GB200 4×4, H100 4×8, B300 1×8, H200 2×8 — 4×8 on Unified High-Throughput, GB300 2×4, MI350X/MI355X 1×8), so it is not a separate choice. If you serve the NVFP4 checkpoint (`nvidia/Kimi-K3-NVFP4`, the **Quantization** row in the panel below), use the `lmsysorg/sglang:dev-dev-kimi-k3-nvfp4` image.
|
||||||
|
|
||||||
**PD Mode** — `Unified` serves prefill and decode together. `Prefill` / `Decode` split them into dedicated pools (see [PD disaggregation](#3-4-pd-disaggregation)); `Prefill` ships two strategies, both chunked at 16k. On the 8-GPU platforms (B300 1×8, GB300 2×4), `Default` is TP8 and `Long-Context` is `--pp-size 8 --tp-size 1`. On the 16-GPU platforms (B200 2×8, GB200 4×4), both are `--pp-size 16 --tp-size 1` and differ only in `--mem-fraction-static` (0.85 vs 0.90) — deep PP is the throughput shape there, not just the long-context one (see [Deep PP](#deep-pp-for-prefill)).
|
**PD Mode** — `Unified` serves prefill and decode together. `Prefill` / `Decode` split them into dedicated pools (see [PD disaggregation](#3-4-pd-disaggregation)); `Prefill` ships two strategies, both chunked at 16k. On the 8-GPU platforms (B300 1×8, GB300 2×4), `Default` is TP8 and `Long-Context` is `--pp-size 8 --tp-size 1`. On the 16-GPU platforms (B200 2×8, GB200 4×4), both are `--pp-size 16 --tp-size 1` and differ only in `--mem-fraction-static` (0.85 vs 0.90) — deep PP is the throughput shape there, not just the long-context one (see [Deep PP](#deep-pp-for-prefill)).
|
||||||
|
|
||||||
|
|||||||
@@ -160,6 +160,39 @@ export const config = {
|
|||||||
// Orthogonal to the cell grid: the picked option layers flags onto whichever
|
// Orthogonal to the cell grid: the picked option layers flags onto whichever
|
||||||
// cell is showing, so turning speculation on does not triple the cell count.
|
// cell is showing, so turning speculation on does not triple the cell count.
|
||||||
overlayDims: [
|
overlayDims: [
|
||||||
|
{
|
||||||
|
// Checkpoint choice, orthogonal to the cell grid: MXFP4 is the shipping
|
||||||
|
// default, NVFP4 is NVIDIA's ModelOpt mixed checkpoint (NVFP4 SiTU routed
|
||||||
|
// experts + FP8_PB_WO 128x128 block-FP8 attention). NVFP4 swaps the model
|
||||||
|
// slug (modelNames) and pins the TRT-LLM MoE runner — the auto resolution
|
||||||
|
// never engages TRT-LLM deferred finalize and the NVFP4 MoE raises
|
||||||
|
// NotImplementedError at CUDA-graph capture, while flashinfer_cutlass has
|
||||||
|
// no SiTU kernel. The DSPARK overlay needs no change: the same draft
|
||||||
|
// checkpoint serves on top of the NVFP4 base.
|
||||||
|
id: "quant",
|
||||||
|
title: "Quantization",
|
||||||
|
default: "mxfp4",
|
||||||
|
options: [
|
||||||
|
{ id: "mxfp4", label: "MXFP4", subtitle: "Moonshot AI checkpoint" },
|
||||||
|
{
|
||||||
|
id: "nvfp4",
|
||||||
|
label: "NVFP4",
|
||||||
|
subtitle: "NVIDIA checkpoint",
|
||||||
|
// The NVFP4 MoE kernels (FlashInfer TRT-LLM) are Blackwell-only.
|
||||||
|
disabled: (s) => !["b200", "gb200", "b300", "gb300"].includes(s.hw),
|
||||||
|
disableReason:
|
||||||
|
"The nvidia/Kimi-K3-NVFP4 checkpoint needs Blackwell: its routed experts run on FlashInfer TRT-LLM NVFP4 kernels (SiTU), which do not exist for Hopper or AMD.",
|
||||||
|
// B200's Balanced/High-Throughput cells pin flashinfer_mxfp4; Hopper
|
||||||
|
// pins marlin (unreachable here — NVFP4 is Blackwell-gated). Replace
|
||||||
|
// whatever the cell pins with the one working NVFP4 runner.
|
||||||
|
stripPrefixes: ["--moe-runner-backend"],
|
||||||
|
flags: ["--moe-runner-backend flashinfer_trtllm"],
|
||||||
|
hints: [
|
||||||
|
"Use docker image lmsysorg/sglang:dev-dev-kimi-k3-nvfp4 (CUDA 13).",
|
||||||
|
],
|
||||||
|
},
|
||||||
|
],
|
||||||
|
},
|
||||||
{
|
{
|
||||||
id: "mmTransport",
|
id: "mmTransport",
|
||||||
title: "VLM Transport",
|
title: "VLM Transport",
|
||||||
@@ -333,6 +366,7 @@ export const config = {
|
|||||||
|
|
||||||
modelNames: {
|
modelNames: {
|
||||||
default: "moonshotai/Kimi-K3",
|
default: "moonshotai/Kimi-K3",
|
||||||
|
nvfp4: "nvidia/Kimi-K3-NVFP4",
|
||||||
},
|
},
|
||||||
|
|
||||||
placeholders: {
|
placeholders: {
|
||||||
@@ -379,6 +413,12 @@ export const config = {
|
|||||||
gb200: "lmsysorg/sglang:kimi-k3",
|
gb200: "lmsysorg/sglang:kimi-k3",
|
||||||
mi350x: "lmsysorg/sglang-rocm:rocm720-mi35x-k3-20260727",
|
mi350x: "lmsysorg/sglang-rocm:rocm720-mi35x-k3-20260727",
|
||||||
mi355x: "lmsysorg/sglang-rocm:rocm720-mi35x-k3-20260727",
|
mi355x: "lmsysorg/sglang-rocm:rocm720-mi35x-k3-20260727",
|
||||||
|
// NVFP4 needs a build with sgl-project/sglang#35077; the purpose-built dev
|
||||||
|
// image is cut from that PR's head (CUDA 13).
|
||||||
|
"b300|nvfp4": "lmsysorg/sglang:dev-dev-kimi-k3-nvfp4",
|
||||||
|
"gb300|nvfp4": "lmsysorg/sglang:dev-dev-kimi-k3-nvfp4",
|
||||||
|
"b200|nvfp4": "lmsysorg/sglang:dev-dev-kimi-k3-nvfp4",
|
||||||
|
"gb200|nvfp4": "lmsysorg/sglang:dev-dev-kimi-k3-nvfp4",
|
||||||
},
|
},
|
||||||
// Pre-selects the issue template's `model` field on "Submit verified cell".
|
// Pre-selects the issue template's `model` field on "Submit verified cell".
|
||||||
github: {
|
github: {
|
||||||
|
|||||||
Reference in New Issue
Block a user