diff --git a/docs_new/src/snippets/autoregressive/deepseek-v4-deployment.jsx b/docs_new/src/snippets/autoregressive/deepseek-v4-deployment.jsx index 0f91841e0..cf46a2a4f 100644 --- a/docs_new/src/snippets/autoregressive/deepseek-v4-deployment.jsx +++ b/docs_new/src/snippets/autoregressive/deepseek-v4-deployment.jsx @@ -33,7 +33,7 @@ export const DeepSeekV4Deployment = () => { { id: "gb300", label: "GB300", default: false }, { id: "h200", label: "H200", default: false }, { id: "h100", label: "H100", default: false }, - { id: "sm120", label: "RTX PRO 6000 (SM120)", default: false }, + { id: "rtx6000", label: "RTX PRO 6000", default: false }, ], }, modelSize: { @@ -125,7 +125,7 @@ export const DeepSeekV4Deployment = () => { // low-latency / balanced / cp recipes, and on PD-Disagg (the cookbook's // PD command builder doesn't emit the megamoe backend / env vars yet). const MEGAMOE_UNSUPPORTED_RECIPES = new Set(["low-latency", "balanced", "cp", "pd-disagg"]); - const MEGAMOE_UNSUPPORTED_HARDWARE = new Set(["h100", "h200", "sm120"]); + const MEGAMOE_UNSUPPORTED_HARDWARE = new Set(["h100", "h200", "rtx6000"]); const isMegamoeUnsupported = (vals) => MEGAMOE_UNSUPPORTED_HARDWARE.has(vals.hardware) || MEGAMOE_UNSUPPORTED_RECIPES.has(vals.recipe); @@ -134,7 +134,9 @@ export const DeepSeekV4Deployment = () => { // mooncake_store/README.md), but the cookbook generator doesn't yet emit // the hicache flags into buildPDDisaggCommand. Grey it out for now. const HICACHE_UNSUPPORTED_RECIPES = new Set(["pd-disagg"]); + const HICACHE_UNSUPPORTED_HARDWARE = new Set(["rtx6000"]); const isHicacheUnsupported = (vals) => + HICACHE_UNSUPPORTED_HARDWARE.has(vals.hardware) || HICACHE_UNSUPPORTED_RECIPES.has(vals.recipe); // H100 + SGLang FP8 only ships a Flash variant — Pro FP8 on H100 isn't @@ -172,10 +174,13 @@ export const DeepSeekV4Deployment = () => { ); } if (option.name === "hicache" && vals && isHicacheUnsupported(vals)) { + const reason = HICACHE_UNSUPPORTED_HARDWARE.has(vals.hardware) + ? "HiCache is not supported on RTX PRO 6000" + : "HiCache is not yet wired into the PD-Disagg cookbook command"; return option.items.map((it) => it.id === "disabled" ? it - : { ...it, disabled: true, disabledReason: "HiCache is not yet wired into the PD-Disagg cookbook command" } + : { ...it, disabled: true, disabledReason: reason } ); } if (option.name === "quantization" && vals && !FP8_SUPPORTED_HARDWARE.has(vals.hardware)) { @@ -185,20 +190,20 @@ export const DeepSeekV4Deployment = () => { : it ); } - // SM120: Flash only (Pro doesn't fit in 8× 96 GB). - if (option.name === "modelSize" && vals && vals.hardware === "sm120") { + // RTX PRO 6000: Flash only (Pro doesn't fit in 8× 96 GB). + if (option.name === "modelSize" && vals && vals.hardware === "rtx6000") { return option.items.map((it) => it.id === "big" - ? { ...it, disabled: true, disabledReason: "V4-Pro does not fit on SM120 (8× 96 GB)" } + ? { ...it, disabled: true, disabledReason: "V4-Pro does not fit on RTX PRO 6000 (8× 96 GB)" } : it ); } - // SM120: TP-only, no EP / CP / PD-Disagg. - if (option.name === "recipe" && vals && vals.hardware === "sm120") { - const sm120Unsupported = new Set(["balanced", "max-throughput", "cp", "pd-disagg"]); + // RTX PRO 6000: TP-only, no EP / CP / PD-Disagg. + if (option.name === "recipe" && vals && vals.hardware === "rtx6000") { + const rtx6000Unsupported = new Set(["balanced", "max-throughput", "cp", "pd-disagg"]); return option.items.map((it) => - sm120Unsupported.has(it.id) - ? { ...it, disabled: true, disabledReason: "SM120 supports low-latency (TP-only) recipe" } + rtx6000Unsupported.has(it.id) + ? { ...it, disabled: true, disabledReason: "RTX PRO 6000 supports low-latency (TP-only) recipe" } : it ); } @@ -284,10 +289,10 @@ export const DeepSeekV4Deployment = () => { ) { next.megamoe = "disabled"; } - // Switching to a recipe that doesn't support HiCache (pd-disagg) while + // Switching to a recipe or hardware that doesn't support HiCache while // L2 is selected: fall back to disabled. if ( - optionName === "recipe" && + (optionName === "hardware" || optionName === "recipe") && next.hicache !== "disabled" && isHicacheUnsupported(next) ) { @@ -360,9 +365,9 @@ export const DeepSeekV4Deployment = () => { // the generator. TP=8 single-node uses the same sgl-project FP8 ckpt as // H200; the Flash/balanced/max-throughput recipes use TP=8 DP=8 + DeepEP. "h100-fp8|small": { slug: "sgl-project/DeepSeek-V4-Flash-FP8", tp: 8, multinode: false }, - // SM120 (RTX PRO 6000): Flash only, TP=4 single-node. Uses Marlin MoE runner - // with SM120 Triton fallback kernels. Requires lmsysorg/sglang:dev-cu13. - "sm120|small": { slug: "deepseek-ai/DeepSeek-V4-Flash", tp: 4, multinode: false }, + // RTX PRO 6000: Flash only, TP=4 single-node. Uses Marlin MoE runner + // with RTX PRO 6000 Triton fallback kernels. Requires lmsysorg/sglang:latest. + "rtx6000|small": { slug: "deepseek-ai/DeepSeek-V4-Flash", tp: 4, multinode: false }, }; // Per (hardware, modelSize) PD role TP (from allinone _PD_SPEC). const PD_TP_SPEC = { @@ -433,7 +438,7 @@ export const DeepSeekV4Deployment = () => { "h100-fp8|small|low-latency", "h100-fp8|small|balanced", "h100-fp8|small|max-throughput", - "sm120|small|low-latency", + "rtx6000|small|low-latency", ]); // Recipes whose command is intentionally not yet provided (e.g. blocked by an // upstream limitation). Showing a minimal placeholder is friendlier to users @@ -497,11 +502,11 @@ export const DeepSeekV4Deployment = () => { return buildPDDisaggCommand(hardware, modelSize); } - // SM120 (RTX PRO 6000) path: Flash only, TP=4, Marlin MoE w/ SM120 Triton - // fallback. Requires Docker image lmsysorg/sglang:dev-cu13 (CUDA 13). - if (hardware === "sm120") { + // RTX PRO 6000 path: Flash only, TP=4, Marlin MoE w/ RTX PRO 6000 Triton + // fallback. Requires Docker image lmsysorg/sglang:latest. + if (hardware === "rtx6000") { const verifyKey = `${hardware}|${modelSize}|${recipe}`; - const sm120Flags = [ + const rtx6000Flags = [ " --trust-remote-code", ` --model-path ${slug}`, ` --tp ${tp}`, @@ -509,16 +514,15 @@ export const DeepSeekV4Deployment = () => { " --mem-fraction-static 0.70", " --cuda-graph-max-bs 32", ]; - if (toolcall === "enabled") sm120Flags.push(" --tool-call-parser deepseekv4"); - if (reasoningParser === "enabled") sm120Flags.push(" --reasoning-parser deepseek-v4"); - sm120Flags.push(" --host 0.0.0.0"); - sm120Flags.push(" --port 30000"); + if (toolcall === "enabled") rtx6000Flags.push(" --tool-call-parser deepseekv4"); + if (reasoningParser === "enabled") rtx6000Flags.push(" --reasoning-parser deepseek-v4"); + rtx6000Flags.push(" --host 0.0.0.0"); + rtx6000Flags.push(" --port 30000"); - const sm120Note = "# SM120: use Docker image lmsysorg/sglang:dev-cu13\n"; - const sm120Cmd = `${sm120Note}sglang serve \\\n${sm120Flags.join(" \\\n")}`; + const rtx6000Cmd = `sglang serve \\\n${rtx6000Flags.join(" \\\n")}`; return VERIFIED_RECIPES.has(verifyKey) - ? sm120Cmd - : `${BEING_VERIFIED_NOTE}\n${commentOutCommand(sm120Cmd)}`; + ? rtx6000Cmd + : `${BEING_VERIFIED_NOTE}\n${commentOutCommand(rtx6000Cmd)}`; } // H200 (FP4) path: dedicated branch — Hopper runs the FP4-mixed Instruct