diff --git a/docs_new/cookbook/autoregressive/DeepSeek/DeepSeek-V4.mdx b/docs_new/cookbook/autoregressive/DeepSeek/DeepSeek-V4.mdx
index 8a26dd402..86f5778d3 100644
--- a/docs_new/cookbook/autoregressive/DeepSeek/DeepSeek-V4.mdx
+++ b/docs_new/cookbook/autoregressive/DeepSeek/DeepSeek-V4.mdx
@@ -1,7 +1,7 @@
---
title: DeepSeek-V4
metatags:
- description: "Deploy DeepSeek-V4 with SGLang — a next-generation MoE model from DeepSeek. Blackwell deployments use the FP4 checkpoint; Hopper deployments use the FP8 checkpoint."
+ description: "Deploy DeepSeek-V4 with SGLang — a next-generation MoE model from DeepSeek."
tag: NEW
---
@@ -35,7 +35,7 @@ tag: NEW
DeepSeek-V4-Pro |
1.6T |
49B |
- high-capacity: B200 8 GPU / GB200 8 GPU (2 nodes) / GB300 4 GPU / H200 16 GPU (2 nodes) |
+ high-capacity: B200 8 GPU / GB200 8 GPU (2 nodes) / GB300 4 GPU / H200 8 GPU(fp4)/16 GPU(fp8) |
@@ -153,9 +153,9 @@ The generator currently picks values on the **conservative** side (mirroring an
**Hopper (H200) note**
-H200 image (`lmsysorg/sglang:deepseek-v4-hopper`) and FP8 checkpoints
-(`sgl-project/DeepSeek-V4-Flash-FP8`, `sgl-project/DeepSeek-V4-Pro-FP8`) are
-publicly available.
+We provide two different options for running DeepSeek-V4 models on Hopper devices (H200)
+- Original FP4 checkpoints: To run original FP4 checkpoints, apply the w4a16 MoE kernels (marlin) as in interactive command generator. For this option we only support TP method. Complete Pro model can be run on a single H200 node with this option.
+- Converted FP8 checkpoints: We also provide pre-converted FP8 checkpoints (`sgl-project/DeepSeek-V4-Flash-FP8`, `sgl-project/DeepSeek-V4-Pro-FP8`), which support more parallelism and features.
PD-Disagg recipes on H200 may require `docker run --privileged --ulimit memlock=-1`
(or `--device /dev/infiniband:/dev/infiniband --cap-add IPC_LOCK`) so mooncake
diff --git a/docs_new/src/snippets/autoregressive/deepseek-v4-deployment.jsx b/docs_new/src/snippets/autoregressive/deepseek-v4-deployment.jsx
index d61b66e44..66488474b 100644
--- a/docs_new/src/snippets/autoregressive/deepseek-v4-deployment.jsx
+++ b/docs_new/src/snippets/autoregressive/deepseek-v4-deployment.jsx
@@ -31,6 +31,7 @@ export const DeepSeekV4Deployment = () => {
{ id: "gb200", label: "GB200 (FP4)", default: false },
{ id: "gb300", label: "GB300 (FP4)", default: false },
{ id: "h200", label: "H200 (FP8)", default: false },
+ { id: "h200-fp4", label: "H200 (FP4)", default: false },
],
},
modelSize: {
@@ -70,7 +71,19 @@ export const DeepSeekV4Deployment = () => {
},
};
- const resolveItems = (option) => option.items;
+ // Recipes that are not supported on the H200 (FP4) Marlin path.
+ const H200_FP4_UNSUPPORTED_RECIPES = new Set(["cp", "pd-disagg"]);
+
+ const resolveItems = (option, vals) => {
+ if (option.name === "recipe" && vals && vals.hardware === "h200-fp4") {
+ return option.items.map((it) =>
+ H200_FP4_UNSUPPORTED_RECIPES.has(it.id)
+ ? { ...it, disabled: true, disabledReason: "Not supported on H200 (FP4)" }
+ : it
+ );
+ }
+ return option.items;
+ };
const getInitialState = () => {
const initialState = {};
@@ -104,7 +117,19 @@ export const DeepSeekV4Deployment = () => {
}, []);
const handleRadioChange = (optionName, value) => {
- setValues((prev) => ({ ...prev, [optionName]: value }));
+ setValues((prev) => {
+ const next = { ...prev, [optionName]: value };
+ // Switching to H200 (FP4) while cp / pd-disagg is selected: fall back
+ // to low-latency since those recipes are not supported on this path.
+ if (
+ optionName === "hardware" &&
+ value === "h200-fp4" &&
+ H200_FP4_UNSUPPORTED_RECIPES.has(next.recipe)
+ ) {
+ next.recipe = "low-latency";
+ }
+ return next;
+ });
};
// ============================================================================
@@ -147,6 +172,11 @@ export const DeepSeekV4Deployment = () => {
// repackagings for both variants.
"h200|small": { slug: "sgl-project/DeepSeek-V4-Flash-FP8", tp: 4, multinode: false },
"h200|big": { slug: "sgl-project/DeepSeek-V4-Pro-FP8", tp: 16, multinode: true, nnodes: 2 },
+ // H200 (FP4) runs the original FP4-mixed Instruct repos through the Marlin
+ // MoE runner: experts are dequantized from FP4 to FP16 at runtime, so a
+ // single-node TP=4 / TP=8 deployment fits Flash / Pro on Hopper.
+ "h200-fp4|small": { slug: "deepseek-ai/DeepSeek-V4-Flash", tp: 4, multinode: false },
+ "h200-fp4|big": { slug: "deepseek-ai/DeepSeek-V4-Pro", tp: 8, multinode: false },
};
// Per (hardware, modelSize) PD role TP (from allinone _PD_SPEC).
const PD_TP_SPEC = {
@@ -202,6 +232,12 @@ export const DeepSeekV4Deployment = () => {
"gb200|big|low-latency",
"gb200|big|balanced",
"gb200|big|max-throughput",
+ "h200-fp4|small|low-latency",
+ "h200-fp4|small|balanced",
+ "h200-fp4|small|max-throughput",
+ "h200-fp4|big|low-latency",
+ "h200-fp4|big|balanced",
+ "h200-fp4|big|max-throughput",
]);
// Recipes whose command is intentionally not yet provided (e.g. blocked by an
// upstream limitation). Showing a minimal placeholder is friendlier to users
@@ -255,6 +291,46 @@ export const DeepSeekV4Deployment = () => {
return buildPDDisaggCommand(hardware, modelSize);
}
+ // H200 (FP4) Marlin path: dedicated branch — Hopper runs the FP4-mixed
+ // Instruct repos through the Marlin MoE runner, so it doesn't share envs
+ // or flags with either the FP8 H200 path or the Blackwell paths.
+ // Flash: TP=4, single node Pro: TP=8, single node
+ // low-latency: MTP 3 / 1 / 4 (steps / topk / draft-tokens)
+ // balanced: MTP 1 / 1 / 2
+ // max-throughput: MTP disabled
+ if (hardware === "h200-fp4") {
+ const verifyKey = `${hardware}|${modelSize}|${recipe}`;
+ if (TBD_RECIPES.has(verifyKey)) return TBD_PLACEHOLDER;
+
+ const fp4Flags = [
+ " --trust-remote-code",
+ ` --model-path ${slug}`,
+ ` --tp ${tp}`,
+ " --moe-runner-backend marlin",
+ ];
+ if (recipe === "low-latency") {
+ fp4Flags.push(" --speculative-algo EAGLE");
+ fp4Flags.push(" --speculative-num-steps 3");
+ fp4Flags.push(" --speculative-eagle-topk 1");
+ fp4Flags.push(" --speculative-num-draft-tokens 4");
+ } else if (recipe === "balanced") {
+ fp4Flags.push(" --speculative-algo EAGLE");
+ fp4Flags.push(" --speculative-num-steps 1");
+ fp4Flags.push(" --speculative-eagle-topk 1");
+ fp4Flags.push(" --speculative-num-draft-tokens 2");
+ }
+ if (isBig) fp4Flags.push(" --mem-fraction-static 0.88");
+ if (toolcall === "enabled") fp4Flags.push(" --tool-call-parser deepseekv4");
+ if (reasoningParser === "enabled") fp4Flags.push(" --reasoning-parser deepseek-v4");
+ fp4Flags.push(" --host 0.0.0.0");
+ fp4Flags.push(" --port 30000");
+
+ const fp4Cmd = `sglang serve \\\n${fp4Flags.join(" \\\n")}`;
+ return VERIFIED_RECIPES.has(verifyKey)
+ ? fp4Cmd
+ : `${BEING_VERIFIED_NOTE}\n${commentOutCommand(fp4Cmd)}`;
+ }
+
// ---- env ----
// _LAUNCH_HEAD always prepends these:
// Per-hardware env (whitelist #1: NVSHMEM removed for B200).
@@ -725,7 +801,7 @@ python3 -m sglang_router.launch_router \\
return (
{Object.entries(options).map(([key, option]) => {
- const items = resolveItems(option);
+ const items = resolveItems(option, values);
return (