diff --git a/docs_new/cookbook/autoregressive/DeepSeek/DeepSeek-V4.mdx b/docs_new/cookbook/autoregressive/DeepSeek/DeepSeek-V4.mdx
index 8546b4d08..f779ed26a 100644
--- a/docs_new/cookbook/autoregressive/DeepSeek/DeepSeek-V4.mdx
+++ b/docs_new/cookbook/autoregressive/DeepSeek/DeepSeek-V4.mdx
@@ -35,7 +35,7 @@ tag: NEW
DeepSeek-V4-Pro |
1.6T |
49B |
- high-capacity: B200 8 GPU / GB200 8 GPU (2 nodes) / GB300 4 GPU / H200 8 GPU(fp4)/16 GPU(fp8) |
+ high-capacity: B200 8 GPU / GB200 8 GPU (2 nodes) / GB300 4 GPU / H200 8 GPU (FP4) or 16 GPU (SGLang FP8) |
@@ -182,7 +182,7 @@ curl http://localhost:30000/v1/chat/completions \
Enable the `deepseek-v4` reasoning parser (check the box in the [command panel above](#3-model-deployment)) to separate thinking from the final answer into `reasoning_content` vs `content`.
-**Streaming with Thinking Process:**
+
```python Example
from openai import OpenAI
@@ -227,17 +227,36 @@ for chunk in response:
print()
```
-**Output Example:**
+
+
+
```text Output
-Pending update — replace with real server output after deployment.
+We are asked: "What is 15% of 240?" This is a simple percentage problem. I need to provide a step-by-step solution. The user wants the solution explained step by step. I'll calculate 15% of 240: 0.15 * 240 = 36. I'll break it down into steps: understand what percent means, convert percentage to decimal or fraction, then multiply. I'll present the answer clearly.To find 15% of 240, follow these steps:
+
+**Step 1: Understand the meaning of percent**
+"Percent" means "per hundred," so 15% means 15 out of every100, or \( \frac{15}{100} \).
+
+**Step2: Convert the percentage to a decimal or fraction**
+\( 15\% = \frac{15}{100} = 0.15 \)
+
+**Step3: Multiply by the given number**
+Multiply the decimal form by 240:
+\( 0.15 \times 240 \)
+
+**Step4: Perform the multiplication**
+\( 0.15 \times 240 = 36 \)
+
+**Answer:** 15% of 240 is **36**.
```
+
+
#### 4.2.2 Tool Calling
Enable the `deepseekv4` tool-call parser (check the box in the [command panel above](#3-model-deployment)) to surface structured tool calls via `message.tool_calls`.
-**Python Example (with Thinking Process):**
+
```python Example
from openai import OpenAI
@@ -313,12 +332,22 @@ for index, tool_call in sorted(tool_calls_accumulator.items()):
print()
```
-**Output Example:**
+
+
+
```text Output
-Pending update — replace with real server output after deployment.
+The user wants to know the weather in Beijing. I'll use the get_weather function with Beijing as the location. I don't need to specify a unit, so I'll just use the default.
+
+<|DSML|tool_calls>
+<|DSML|invoke name="get_weather">
+<|DSML|parameter name="location" string="true">Beijing|DSML|parameter>
+|DSML|invoke>
+|DSML|tool_calls>
```
+
+
#### 4.2.3 HiCache (Hierarchical KV Caching)
HiCache enables multi-tier KV cache offloading (GPU → CPU → Storage), significantly expanding effective context capacity for long-context and multi-turn scenarios. Combined with UnifiedRadixTree, it provides intelligent prefix caching across all tiers.
@@ -332,68 +361,9 @@ For more details, see the [HiCache documentation](../../../docs/advanced_feature
## 5. Benchmark
-### 5.1 Speed Benchmark on Blackwell
+### 5.1 Accuracy Benchmark
-**Test Environment:**
-
-- Hardware: NVIDIA B200 GPU (4x)
-- Model: DeepSeek-V4-Flash (FP4)
-- Tensor Parallelism: 4
-- sglang version: Pending update
-
-We use SGLang's built-in benchmarking tool to conduct performance evaluation on the [ShareGPT_Vicuna_unfiltered](https://huggingface.co/datasets/anon8231489123/ShareGPT_Vicuna_unfiltered) dataset. This dataset contains real conversation data and can better reflect performance in actual use scenarios. To simulate real-world usage patterns, we configure each request with 1024 input tokens and 1024 output tokens, representing typical medium-length conversations with detailed responses.
-
-#### 5.1.1 Latency-Sensitive Benchmark
-
-- **Model Deployment Command:** see the [command panel above](#3-model-deployment).
-
-- Benchmark Command:
-
-```shell Command
-python3 -m sglang.bench_serving \
- --backend sglang \
- --host 127.0.0.1 \
- --port 30000 \
- --model deepseek-ai/DeepSeek-V4-Flash \
- --random-input-len 1024 \
- --random-output-len 1024 \
- --num-prompts 10 \
- --max-concurrency 1
-```
-
-- **Test Results:**
-
-```text Output
-Pending update — replace with real bench_serving output after the latency run.
-```
-
-#### 5.1.2 Throughput-Sensitive Benchmark
-
-- **Model Deployment Command:** see the [command panel above](#3-model-deployment).
-
-- Benchmark Command:
-
-```shell Command
-python3 -m sglang.bench_serving \
- --backend sglang \
- --host 127.0.0.1 \
- --port 30000 \
- --model deepseek-ai/DeepSeek-V4-Flash \
- --random-input-len 1024 \
- --random-output-len 1024 \
- --num-prompts 1000 \
- --max-concurrency 100
-```
-
-- **Test Results:**
-
-```text Output
-Pending update — replace with real bench_serving output after the throughput run.
-```
-
-### 5.2 Accuracy Benchmark
-
-#### 5.2.1 GSM8K Benchmark
+#### 5.1.1 GSM8K Benchmark
- **Benchmark Command:**
@@ -402,16 +372,18 @@ python3 -m sglang.test.few_shot_gsm8k --num-questions 200 --port 30000
```
- **Test Results:**
- - DeepSeek-V4-Flash (FP4, Blackwell)
+ - DeepSeek-V4-Pro (FP4, B300, low-latency)
```
- Pending update
+ Accuracy: 0.965
+ Invalid: 0.000
```
- - DeepSeek-V4-Flash (FP8, Hopper)
+ - DeepSeek-V4-Pro (FP4, H200, low-latency)
```
- Pending update
+ Accuracy: 0.975
+ Invalid: 0.000
```
-#### 5.2.2 MMLU Benchmark
+#### 5.1.2 MMLU Benchmark
- **Benchmark Command:**
@@ -422,27 +394,53 @@ python3 benchmark/mmlu/bench_sglang.py --nsub 10 --port 30000
```
- **Test Results:**
- - DeepSeek-V4-Flash (FP4, Blackwell)
+ - DeepSeek-V4-Pro (FP4, B300, low-latency)
```
- Pending update
+ subject: abstract_algebra, #q:100, acc: 0.820
+ subject: anatomy, #q:135, acc: 0.881
+ subject: astronomy, #q:152, acc: 0.934
+ subject: business_ethics, #q:100, acc: 0.840
+ subject: clinical_knowledge, #q:265, acc: 0.913
+ subject: college_biology, #q:144, acc: 0.972
+ subject: college_chemistry, #q:100, acc: 0.680
+ subject: college_computer_science, #q:100, acc: 0.890
+ subject: college_mathematics, #q:100, acc: 0.870
+ subject: college_medicine, #q:173, acc: 0.873
+ Total latency: 14.903
+ Average accuracy: 0.879
```
- - DeepSeek-V4-Flash (FP8, Hopper)
+ - DeepSeek-V4-Pro (FP4, H200, low-latency)
```
- Pending update
+ subject: abstract_algebra, #q:100, acc: 0.850
+ subject: anatomy, #q:135, acc: 0.889
+ subject: astronomy, #q:152, acc: 0.947
+ subject: business_ethics, #q:100, acc: 0.860
+ subject: clinical_knowledge, #q:265, acc: 0.932
+ subject: college_biology, #q:144, acc: 0.972
+ subject: college_chemistry, #q:100, acc: 0.710
+ subject: college_computer_science, #q:100, acc: 0.910
+ subject: college_mathematics, #q:100, acc: 0.830
+ subject: college_medicine, #q:173, acc: 0.896
+ Total latency: 42.004
+ Average accuracy: 0.893
```
-### 5.3 Speed Benchmark on Hopper
+### 5.2 Speed Benchmark
+
+We use SGLang's built-in benchmarking tool with its `random` dataset — real prompts sampled from [ShareGPT_Vicuna_unfiltered](https://huggingface.co/datasets/anon8231489123/ShareGPT_Vicuna_unfiltered) and then truncated/padded to a controlled length. This dataset contains real conversation data and can better reflect performance in actual use scenarios. To simulate real-world usage patterns, we configure each request with 1024 input tokens and 1024 output tokens, representing typical medium-length conversations with detailed responses.
+
+#### 5.2.1 Hopper
**Test Environment:**
- Hardware: NVIDIA H200 GPU (4x)
-- Model: DeepSeek-V4-Flash (FP8)
+- Model: DeepSeek-V4-Flash (FP4)
- Tensor Parallelism: 4
-- sglang version: Pending update
+- sglang version: 0.5.12
-#### 5.3.1 Latency-Sensitive Benchmark
+##### Latency-Sensitive Benchmark
-- **Model Deployment Command:** see the [command panel above](#3-model-deployment).
+- **Model Deployment Command:** H200 · DeepSeek-V4-Flash · FP4 · Low-Latency. See the [command panel above](#3-model-deployment).
- Benchmark Command:
@@ -452,6 +450,7 @@ python3 -m sglang.bench_serving \
--host 127.0.0.1 \
--port 30000 \
--model deepseek-ai/DeepSeek-V4-Flash \
+ --dataset-name random \
--random-input-len 1024 \
--random-output-len 1024 \
--num-prompts 10 \
@@ -461,12 +460,49 @@ python3 -m sglang.bench_serving \
- **Test Results:**
```text Output
-Pending update — replace with real bench_serving output after the latency run.
+============ Serving Benchmark Result ============
+Backend: sglang
+Traffic request rate: inf
+Max request concurrency: 1
+Successful requests: 10
+Benchmark duration (s): 15.98
+Total input tokens: 6101
+Total input text tokens: 6101
+Total generated tokens: 4220
+Total generated tokens (retokenized): 4220
+Request throughput (req/s): 0.63
+Input token throughput (tok/s): 381.86
+Output token throughput (tok/s): 264.13
+Peak output token throughput (tok/s): 324.00
+Peak concurrent requests: 3
+Total token throughput (tok/s): 645.98
+Concurrency: 1.00
+Accept length: 2.96
+----------------End-to-End Latency----------------
+Mean E2E Latency (ms): 1596.65
+Median E2E Latency (ms): 1274.48
+P90 E2E Latency (ms): 2950.70
+P99 E2E Latency (ms): 3333.18
+---------------Time to First Token----------------
+Mean TTFT (ms): 147.26
+Median TTFT (ms): 132.22
+P99 TTFT (ms): 181.37
+-----Time per Output Token (excl. 1st token)------
+Mean TPOT (ms): 3.50
+Median TPOT (ms): 3.48
+P99 TPOT (ms): 4.18
+---------------Inter-Token Latency----------------
+Mean ITL (ms): 3.44
+Median ITL (ms): 3.36
+P95 ITL (ms): 5.06
+P99 ITL (ms): 5.15
+Max ITL (ms): 35.31
+==================================================
```
-#### 5.3.2 Throughput-Sensitive Benchmark
+##### Throughput-Sensitive Benchmark
-- **Model Deployment Command:** see the [command panel above](#3-model-deployment).
+- **Model Deployment Command:** H200 · DeepSeek-V4-Flash · FP4 · Max-Throughput. See the [command panel above](#3-model-deployment).
- Benchmark Command:
@@ -476,6 +512,7 @@ python3 -m sglang.bench_serving \
--host 127.0.0.1 \
--port 30000 \
--model deepseek-ai/DeepSeek-V4-Flash \
+ --dataset-name random \
--random-input-len 1024 \
--random-output-len 1024 \
--num-prompts 1000 \
@@ -485,5 +522,173 @@ python3 -m sglang.bench_serving \
- **Test Results:**
```text Output
-Pending update — replace with real bench_serving output after the throughput run.
+============ Serving Benchmark Result ============
+Backend: sglang
+Traffic request rate: inf
+Max request concurrency: 100
+Successful requests: 1000
+Benchmark duration (s): 198.42
+Total input tokens: 512842
+Total input text tokens: 512842
+Total generated tokens: 510855
+Total generated tokens (retokenized): 510765
+Request throughput (req/s): 5.04
+Input token throughput (tok/s): 2584.65
+Output token throughput (tok/s): 2574.64
+Peak output token throughput (tok/s): 4400.00
+Peak concurrent requests: 110
+Total token throughput (tok/s): 5159.28
+Concurrency: 96.21
+----------------End-to-End Latency----------------
+Mean E2E Latency (ms): 19090.29
+Median E2E Latency (ms): 18328.71
+P90 E2E Latency (ms): 35698.68
+P99 E2E Latency (ms): 39161.43
+---------------Time to First Token----------------
+Mean TTFT (ms): 302.41
+Median TTFT (ms): 131.35
+P99 TTFT (ms): 2172.03
+-----Time per Output Token (excl. 1st token)------
+Mean TPOT (ms): 37.46
+Median TPOT (ms): 37.72
+P99 TPOT (ms): 55.72
+---------------Inter-Token Latency----------------
+Mean ITL (ms): 36.85
+Median ITL (ms): 21.75
+P95 ITL (ms): 107.64
+P99 ITL (ms): 134.58
+Max ITL (ms): 1930.74
+==================================================
+```
+
+#### 5.2.2 Blackwell
+
+**Test Environment:**
+
+- Hardware: NVIDIA B200 GPU (4x)
+- Model: DeepSeek-V4-Flash (FP4)
+- Tensor Parallelism: 4
+- sglang version: 0.5.12
+
+##### Latency-Sensitive Benchmark
+
+- **Model Deployment Command:** B200 · DeepSeek-V4-Flash · FP4 · Low-Latency. See the [command panel above](#3-model-deployment).
+
+- Benchmark Command:
+
+```shell Command
+python3 -m sglang.bench_serving \
+ --backend sglang \
+ --host 127.0.0.1 \
+ --port 30000 \
+ --model deepseek-ai/DeepSeek-V4-Flash \
+ --dataset-name random \
+ --random-input-len 1024 \
+ --random-output-len 1024 \
+ --num-prompts 10 \
+ --max-concurrency 1
+```
+
+- **Test Results:**
+
+```text Output
+============ Serving Benchmark Result ============
+Backend: sglang
+Traffic request rate: inf
+Max request concurrency: 1
+Successful requests: 10
+Benchmark duration (s): 15.25
+Total input tokens: 6101
+Total input text tokens: 6101
+Total generated tokens: 4220
+Total generated tokens (retokenized): 4220
+Request throughput (req/s): 0.66
+Input token throughput (tok/s): 400.06
+Output token throughput (tok/s): 276.72
+Peak output token throughput (tok/s): 308.00
+Peak concurrent requests: 2
+Total token throughput (tok/s): 676.78
+Concurrency: 1.00
+Accept length: 2.73
+----------------End-to-End Latency----------------
+Mean E2E Latency (ms): 1523.83
+Median E2E Latency (ms): 1173.50
+P90 E2E Latency (ms): 2770.33
+P99 E2E Latency (ms): 3233.82
+---------------Time to First Token----------------
+Mean TTFT (ms): 102.72
+Median TTFT (ms): 85.94
+P99 TTFT (ms): 134.79
+-----Time per Output Token (excl. 1st token)------
+Mean TPOT (ms): 3.40
+Median TPOT (ms): 3.42
+P99 TPOT (ms): 4.00
+---------------Inter-Token Latency----------------
+Mean ITL (ms): 3.38
+Median ITL (ms): 3.06
+P95 ITL (ms): 4.60
+P99 ITL (ms): 4.95
+Max ITL (ms): 34.64
+==================================================
+```
+
+##### Throughput-Sensitive Benchmark
+
+- **Model Deployment Command:** B200 · DeepSeek-V4-Flash · FP4 · Max-Throughput (MegaMoE W4A4). See the [command panel above](#3-model-deployment) — flip the **MegaMoE** toggle to **W4A4** to reproduce these numbers; the default Max-Throughput recipe uses `--moe-a2a-backend deepep` and runs slower.
+
+- Benchmark Command:
+
+```shell Command
+python3 -m sglang.bench_serving \
+ --backend sglang \
+ --host 127.0.0.1 \
+ --port 30000 \
+ --model deepseek-ai/DeepSeek-V4-Flash \
+ --dataset-name random \
+ --random-input-len 1024 \
+ --random-output-len 1024 \
+ --num-prompts 1000 \
+ --max-concurrency 100
+```
+
+- **Test Results:**
+
+```text Output
+============ Serving Benchmark Result ============
+Backend: sglang
+Traffic request rate: inf
+Max request concurrency: 100
+Successful requests: 1000
+Benchmark duration (s): 105.10
+Total input tokens: 512842
+Total input text tokens: 512842
+Total generated tokens: 510855
+Total generated tokens (retokenized): 510682
+Request throughput (req/s): 9.51
+Input token throughput (tok/s): 4879.44
+Output token throughput (tok/s): 4860.54
+Peak output token throughput (tok/s): 6600.00
+Peak concurrent requests: 117
+Total token throughput (tok/s): 9739.98
+Concurrency: 94.34
+----------------End-to-End Latency----------------
+Mean E2E Latency (ms): 9915.50
+Median E2E Latency (ms): 9521.19
+P90 E2E Latency (ms): 17726.66
+P99 E2E Latency (ms): 24910.72
+---------------Time to First Token----------------
+Mean TTFT (ms): 349.95
+Median TTFT (ms): 68.23
+P99 TTFT (ms): 4581.26
+-----Time per Output Token (excl. 1st token)------
+Mean TPOT (ms): 19.86
+Median TPOT (ms): 17.96
+P99 TPOT (ms): 61.58
+---------------Inter-Token Latency----------------
+Mean ITL (ms): 18.76
+Median ITL (ms): 13.23
+P95 ITL (ms): 44.79
+P99 ITL (ms): 88.25
+Max ITL (ms): 2499.49
+==================================================
```
diff --git a/docs_new/src/snippets/autoregressive/deepseek-v4-deployment.jsx b/docs_new/src/snippets/autoregressive/deepseek-v4-deployment.jsx
index 64a94f86e..cbc22b214 100644
--- a/docs_new/src/snippets/autoregressive/deepseek-v4-deployment.jsx
+++ b/docs_new/src/snippets/autoregressive/deepseek-v4-deployment.jsx
@@ -27,13 +27,12 @@ export const DeepSeekV4Deployment = () => {
name: "hardware",
title: "Hardware Platform",
items: [
- { id: "b200", label: "B200 (FP4)", default: true },
- { id: "b300", label: "B300 (FP4)", default: false },
- { id: "gb200", label: "GB200 (FP4)", default: false },
- { id: "gb300", label: "GB300 (FP4)", default: false },
- { id: "h200", label: "H200 (FP8)", default: false },
- { id: "h200-fp4", label: "H200 (FP4)", default: false },
- { id: "h100", label: "H100 (FP4)", default: false },
+ { id: "b200", label: "B200", default: true },
+ { id: "b300", label: "B300", default: false },
+ { id: "gb200", label: "GB200", default: false },
+ { id: "gb300", label: "GB300", default: false },
+ { id: "h200", label: "H200", default: false },
+ { id: "h100", label: "H100", default: false },
],
},
modelSize: {
@@ -44,6 +43,14 @@ export const DeepSeekV4Deployment = () => {
{ id: "big", label: "Pro", default: false, subtitle: "1.6T" },
],
},
+ quantization: {
+ name: "quantization",
+ title: "Quantization",
+ items: [
+ { id: "fp4", label: "FP4", default: true },
+ { id: "fp8", label: "FP8", default: false, subtitle: "H100/H200 only" },
+ ],
+ },
recipe: {
name: "recipe",
title: "Recipe",
@@ -90,26 +97,58 @@ export const DeepSeekV4Deployment = () => {
},
};
+ // Hopper GPUs supporting the SGLang FP8 repackaging path.
+ const FP8_SUPPORTED_HARDWARE = new Set(["h100", "h200"]);
+
+ // Internal "effective hardware" id used by HW_SIZE_SPEC / VERIFIED_RECIPES.
+ // Combines the user-facing hardware choice with the Quantization axis:
+ // h200 + fp4 → h200-fp4 (Marlin FP4 path on H200)
+ // h200 + fp8 → h200 (sgl-project FP8 ckpts on H200)
+ // h100 + fp4 → h100 (Marlin FP4 path on H100)
+ // h100 + fp8 → h100-fp8 (Flash-only FP8 path on H100)
+ // anything else → hardware unchanged
+ const effHw = (hardware, quantization) => {
+ if (hardware === "h200") return quantization === "fp8" ? "h200" : "h200-fp4";
+ if (hardware === "h100") return quantization === "fp8" ? "h100-fp8" : "h100";
+ return hardware;
+ };
+
// Recipes that are not supported on the Marlin (FP4) Hopper paths
// (H200 FP4, H100 FP4).
const MARLIN_UNSUPPORTED_RECIPES = new Set(["cp", "pd-disagg"]);
- const MARLIN_HARDWARE = new Set(["h200-fp4", "h100"]);
+ const MARLIN_EFFHW = new Set(["h200-fp4", "h100"]);
const MARLIN_LABEL = { "h200-fp4": "H200 (FP4)", h100: "H100 (FP4)" };
// MegaMoE is only supported on Blackwell with DeepEP-based recipes
// (balanced / max-throughput / pd-disagg). It's disabled on Hopper
- // (H100 / H200 / H200-FP4) and on low-latency / cp recipes.
+ // (H100 / H200, both FP4 and FP8) and on low-latency / cp recipes.
const MEGAMOE_UNSUPPORTED_RECIPES = new Set(["low-latency", "cp"]);
- const MEGAMOE_UNSUPPORTED_HARDWARE = new Set(["h100", "h200", "h200-fp4"]);
+ const MEGAMOE_UNSUPPORTED_HARDWARE = new Set(["h100", "h200"]);
const isMegamoeUnsupported = (vals) =>
MEGAMOE_UNSUPPORTED_HARDWARE.has(vals.hardware) ||
MEGAMOE_UNSUPPORTED_RECIPES.has(vals.recipe);
+ // H100 + SGLang FP8 only ships a Flash variant — Pro FP8 on H100 isn't
+ // covered by the generator yet, so the Pro radio is greyed out there.
+ const isProDisabledFp8H100 = (vals) =>
+ vals.hardware === "h100" && vals.quantization === "fp8";
+
const resolveItems = (option, vals) => {
- if (option.name === "recipe" && vals && MARLIN_HARDWARE.has(vals.hardware)) {
+ const eff = vals ? effHw(vals.hardware, vals.quantization) : null;
+ if (option.name === "recipe" && eff && MARLIN_EFFHW.has(eff)) {
return option.items.map((it) =>
MARLIN_UNSUPPORTED_RECIPES.has(it.id)
- ? { ...it, disabled: true, disabledReason: `Not supported on ${MARLIN_LABEL[vals.hardware]}` }
+ ? { ...it, disabled: true, disabledReason: `Not supported on ${MARLIN_LABEL[eff]}` }
+ : it
+ );
+ }
+ if (option.name === "recipe" && eff === "h100-fp8") {
+ // H100 SGLang FP8 only has low-latency / balanced / max-throughput
+ // commands verified — cp and pd-disagg fall back to the Marlin
+ // "not supported" message.
+ return option.items.map((it) =>
+ MARLIN_UNSUPPORTED_RECIPES.has(it.id)
+ ? { ...it, disabled: true, disabledReason: "Not supported on H100 (SGLang FP8)" }
: it
);
}
@@ -121,6 +160,20 @@ export const DeepSeekV4Deployment = () => {
it.id === "disabled" ? it : { ...it, disabled: true, disabledReason: reason }
);
}
+ if (option.name === "quantization" && vals && !FP8_SUPPORTED_HARDWARE.has(vals.hardware)) {
+ return option.items.map((it) =>
+ it.id === "fp8"
+ ? { ...it, disabled: true, disabledReason: "SGLang FP8 is only available on H100 / H200" }
+ : it
+ );
+ }
+ if (option.name === "modelSize" && vals && isProDisabledFp8H100(vals)) {
+ return option.items.map((it) =>
+ it.id === "big"
+ ? { ...it, disabled: true, disabledReason: "H100 SGLang FP8 only ships a Flash variant" }
+ : it
+ );
+ }
return option.items;
};
@@ -158,12 +211,31 @@ export const DeepSeekV4Deployment = () => {
const handleRadioChange = (optionName, value) => {
setValues((prev) => {
const next = { ...prev, [optionName]: value };
+ // Switching to a hardware that doesn't support FP8 while FP8 is
+ // selected: fall back to FP4.
+ if (
+ optionName === "hardware" &&
+ next.quantization === "fp8" &&
+ !FP8_SUPPORTED_HARDWARE.has(value)
+ ) {
+ next.quantization = "fp4";
+ }
+ // H100 + SGLang FP8 only supports Flash; auto-flip Pro → Flash when
+ // entering that combo (via hardware or quantization switch).
+ if (
+ (optionName === "hardware" || optionName === "quantization") &&
+ isProDisabledFp8H100(next) &&
+ next.modelSize === "big"
+ ) {
+ next.modelSize = "small";
+ }
// Switching to a Marlin (FP4) Hopper path while cp / pd-disagg is
// selected: fall back to low-latency since those recipes are not
// supported on Marlin.
+ const nextEff = effHw(next.hardware, next.quantization);
if (
- optionName === "hardware" &&
- MARLIN_HARDWARE.has(value) &&
+ (optionName === "hardware" || optionName === "quantization") &&
+ (MARLIN_EFFHW.has(nextEff) || nextEff === "h100-fp8") &&
MARLIN_UNSUPPORTED_RECIPES.has(next.recipe)
) {
next.recipe = "low-latency";
@@ -240,6 +312,10 @@ export const DeepSeekV4Deployment = () => {
// a higher TP: Flash fits at TP=8 single-node, Pro needs TP=16 across 2 nodes.
"h100|small": { slug: "deepseek-ai/DeepSeek-V4-Flash", tp: 8, multinode: false },
"h100|big": { slug: "deepseek-ai/DeepSeek-V4-Pro", tp: 16, multinode: true, nnodes: 2 },
+ // H100 (SGLang FP8) ships Flash only — Pro FP8 on H100 is not exposed by
+ // the generator. TP=8 single-node uses the same sgl-project FP8 ckpt as
+ // H200; the Flash/balanced/max-throughput recipes use TP=8 DP=8 + DeepEP.
+ "h100-fp8|small": { slug: "sgl-project/DeepSeek-V4-Flash-FP8", tp: 8, multinode: false },
};
// Per (hardware, modelSize) PD role TP (from allinone _PD_SPEC).
const PD_TP_SPEC = {
@@ -307,6 +383,9 @@ export const DeepSeekV4Deployment = () => {
"h100|big|low-latency",
"h100|big|balanced",
"h100|big|max-throughput",
+ "h100-fp8|small|low-latency",
+ "h100-fp8|small|balanced",
+ "h100-fp8|small|max-throughput",
]);
// Recipes whose command is intentionally not yet provided (e.g. blocked by an
// upstream limitation). Showing a minimal placeholder is friendlier to users
@@ -347,10 +426,20 @@ export const DeepSeekV4Deployment = () => {
`${cmd}`;
// === SHARED END ===
+ // Hopper FP8 paths (effHw values that share the sgl-project FP8 codepath):
+ // both H200 FP8 and H100 SGLang FP8 go through the same "general" command
+ // branch, skipping flashinfer_mxfp4 / chunked-prefill / mem-frac flags that
+ // are Blackwell-FP4-specific.
+ const isHopperFp8 = (effHwId) => effHwId === "h200" || effHwId === "h100-fp8";
+
const generateCommand = () => {
- const { hardware: rawHardware, modelSize, recipe, reasoningParser, toolcall, hicache, megamoe } = values;
+ const { hardware: userHardware, modelSize, quantization, recipe, reasoningParser, toolcall, hicache, megamoe } = values;
// B300 usage is identical to B200 — alias so we don't duplicate every spec entry.
- const hardware = rawHardware === "b300" ? "b200" : rawHardware;
+ const rawHardware = userHardware === "b300" ? "b200" : userHardware;
+ // Translate (hardware, quantization) into the internal "effective hw" id
+ // that HW_SIZE_SPEC / VERIFIED_RECIPES are keyed by. See the effHw helper
+ // at the top of this component for the full mapping.
+ const hardware = effHw(rawHardware, quantization);
const specKey = `${hardware}|${modelSize}`;
const spec = HW_SIZE_SPEC[specKey];
const { slug, tp, multinode, nnodes } = spec;
@@ -480,11 +569,12 @@ export const DeepSeekV4Deployment = () => {
// _LAUNCH_HEAD always prepends these:
// Per-hardware env (whitelist #1: NVSHMEM removed for B200).
const HW_ENV = {
- h200: ["SGLANG_DSV4_FP4_EXPERTS=0"], // allinone _ENV_H200
- b200: [], // _ENV_B200 minus NVSHMEM
- gb300: [], // _ENV_GB300
+ h200: ["SGLANG_DSV4_FP4_EXPERTS=0"], // allinone _ENV_H200
+ "h100-fp8": ["SGLANG_DSV4_FP4_EXPERTS=0"], // H100 SGLang FP8 shares H200's FP8 env
+ b200: [], // _ENV_B200 minus NVSHMEM
+ gb300: [], // _ENV_GB300
// GB200 multinode needs NCCL MNNVL for cross-node NVLink communication.
- gb200: multinode ? ["NCCL_MNNVL_ENABLE=1", "NCCL_CUMEM_ENABLE=1"] : [],
+ gb200: multinode ? ["NCCL_MNNVL_ENABLE=1", "NCCL_CUMEM_ENABLE=1"] : [],
}[hardware];
// Recipe-specific env (matches allinone exactly, taking size into account).
@@ -497,7 +587,7 @@ export const DeepSeekV4Deployment = () => {
recipeEnv.push("SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK=256");
}
} else if (recipe === "balanced") {
- if (hardware === "h200") {
+ if (isHopperFp8(hardware)) {
recipeEnv.push(isBig
? "SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK=128"
: "SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK=256");
@@ -509,7 +599,7 @@ export const DeepSeekV4Deployment = () => {
: "SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK=1024");
}
} else if (recipe === "max-throughput") {
- if (hardware === "h200") {
+ if (isHopperFp8(hardware)) {
recipeEnv.push(isBig
? "SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK=128"
: "SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK=256");
@@ -554,7 +644,7 @@ export const DeepSeekV4Deployment = () => {
if (hardware === "h200" && isBig) {
flags.push(" --moe-a2a-backend deepep");
}
- if (hardware !== "h200") {
+ if (!isHopperFp8(hardware)) {
flags.push(" --moe-runner-backend flashinfer_mxfp4");
}
if (hardware === "h200" && isBig) {
@@ -566,14 +656,14 @@ export const DeepSeekV4Deployment = () => {
flags.push(" --speculative-num-steps 3");
flags.push(" --speculative-eagle-topk 1");
flags.push(" --speculative-num-draft-tokens 4");
- if (hardware !== "h200") {
+ if (!isHopperFp8(hardware)) {
// B200/B300 Pro accuracy-verified: chunked-prefill-size 8192
flags.push(isBig ? " --chunked-prefill-size 8192" : " --chunked-prefill-size 4096");
flags.push(" --disable-flashinfer-autotune");
flags.push(" --swa-full-tokens-ratio 0.1");
}
// B200/B300 Pro accuracy-verified: mem-fraction-static 0.90
- if (isBig && hardware !== "h200") {
+ if (isBig && !isHopperFp8(hardware)) {
flags.push(" --mem-fraction-static 0.90");
} else if (isBig) {
flags.push(" --mem-fraction-static 0.88");