[Docs] Add Kimi-K2.6 NVFP4 and update Kimi-K2.5 cookbook guidance (#27714)
Signed-off-by: Mohammad Miadh Angkad <176301910+mmangkad@users.noreply.github.com>
This commit is contained in:
@@ -37,7 +37,7 @@ import { KimiK25Deployment } from '/src/snippets/autoregressive/kimi-k25-deploym
|
|||||||
|
|
||||||
### 3.2 Configuration Tips
|
### 3.2 Configuration Tips
|
||||||
|
|
||||||
- **Memory**: Requires GPUs with >=140GB each. Supported platforms: H200 (8x, TP=8), B300 (8x, TP=8), MI300X/MI325X (4x, TP=4), MI350X/MI355X (4x, TP=4). Use `--context-length 128000` to conserve memory.
|
- **Memory**: Requires GPUs with >=140GB each. Supported platforms: H200 (8x, TP=8), B300 (8x, TP=8), GB300 (4x, TP=4), MI300X/MI325X (4x, TP=4), MI350X/MI355X (4x, TP=4). Use `--context-length 128000` to conserve memory.
|
||||||
- **AMD GPU TP Constraint**: On AMD GPUs, TP must be <= 4 (not 8). Kimi-K2.5 has 64 attention heads; the AITER MLA kernel requires `heads_per_gpu % 16 == 0`. With TP=4, each GPU gets 16 heads (valid). With TP=8, each GPU gets 8 heads (invalid).
|
- **AMD GPU TP Constraint**: On AMD GPUs, TP must be <= 4 (not 8). Kimi-K2.5 has 64 attention heads; the AITER MLA kernel requires `heads_per_gpu % 16 == 0`. With TP=4, each GPU gets 16 heads (valid). With TP=8, each GPU gets 8 heads (invalid).
|
||||||
- **AMD Docker Image**: Use `lmsysorg/sglang:v0.5.9-rocm700-mi35x` for MI350X/MI355X and `lmsysorg/sglang:v0.5.9-rocm700-mi30x` for MI300X/MI325X. The ROCm 7.2 images (`rocm720`) have an AITER compatibility issue.
|
- **AMD Docker Image**: Use `lmsysorg/sglang:v0.5.9-rocm700-mi35x` for MI350X/MI355X and `lmsysorg/sglang:v0.5.9-rocm700-mi30x` for MI300X/MI325X. The ROCm 7.2 images (`rocm720`) have an AITER compatibility issue.
|
||||||
- **DP Attention**: Enable with `--dp <N> --enable-dp-attention` for production throughput. A common choice is to set `--dp` equal to `--tp`, but this is not required.
|
- **DP Attention**: Enable with `--dp <N> --enable-dp-attention` for production throughput. A common choice is to set `--dp` equal to `--tp`, but this is not required.
|
||||||
@@ -440,10 +440,10 @@ Let me search for this product and similar items:
|
|||||||
|
|
||||||
**Nvidia**
|
**Nvidia**
|
||||||
|
|
||||||
Deploy Kimi-K2.5 with the following command (H200/B200, all features enabled):
|
Deploy Kimi-K2.5 with the following command (H200/B300, all features enabled):
|
||||||
|
|
||||||
```shell Command
|
```shell Command
|
||||||
SGLANG_ENABLE_SPEC_V2=1 sglang serve \
|
sglang serve \
|
||||||
--model-path moonshotai/Kimi-K2.5 \
|
--model-path moonshotai/Kimi-K2.5 \
|
||||||
--tp 8 \
|
--tp 8 \
|
||||||
--reasoning-parser kimi_k2 \
|
--reasoning-parser kimi_k2 \
|
||||||
@@ -452,16 +452,16 @@ SGLANG_ENABLE_SPEC_V2=1 sglang serve \
|
|||||||
--speculative-num-steps 3 \
|
--speculative-num-steps 3 \
|
||||||
--speculative-eagle-topk 1 \
|
--speculative-eagle-topk 1 \
|
||||||
--speculative-num-draft-tokens 4 \
|
--speculative-num-draft-tokens 4 \
|
||||||
--speculative-draft-model-path lightseekorg/kimi-k2.5-eagle3 \
|
--speculative-draft-model-path lightseekorg/kimi-k2.5-eagle3-mla \
|
||||||
--trust-remote-code \
|
--trust-remote-code \
|
||||||
--host 0.0.0.0 \
|
--host 0.0.0.0 \
|
||||||
--port 30000
|
--port 30000
|
||||||
```
|
```
|
||||||
|
|
||||||
Deploy Kimi-K2.5-NVFP4 with the following command (B200, all features enabled):
|
Deploy Kimi-K2.5-NVFP4 with the following command (B300, all features enabled):
|
||||||
|
|
||||||
```shell Command
|
```shell Command
|
||||||
SGLANG_ENABLE_SPEC_V2=1 sglang serve \
|
sglang serve \
|
||||||
--model-path nvidia/Kimi-K2.5-NVFP4 \
|
--model-path nvidia/Kimi-K2.5-NVFP4 \
|
||||||
--tp 8 \
|
--tp 8 \
|
||||||
--reasoning-parser kimi_k2 \
|
--reasoning-parser kimi_k2 \
|
||||||
@@ -471,12 +471,14 @@ SGLANG_ENABLE_SPEC_V2=1 sglang serve \
|
|||||||
--speculative-num-steps 3 \
|
--speculative-num-steps 3 \
|
||||||
--speculative-eagle-topk 1 \
|
--speculative-eagle-topk 1 \
|
||||||
--speculative-num-draft-tokens 4 \
|
--speculative-num-draft-tokens 4 \
|
||||||
--speculative-draft-model-path lightseekorg/kimi-k2.5-eagle3 \
|
--speculative-draft-model-path lightseekorg/kimi-k2.5-eagle3-mla \
|
||||||
--trust-remote-code \
|
--trust-remote-code \
|
||||||
--host 0.0.0.0 \
|
--host 0.0.0.0 \
|
||||||
--port 30000
|
--port 30000
|
||||||
```
|
```
|
||||||
|
|
||||||
|
For GB300, use `--tp 4`.
|
||||||
|
|
||||||
## 5. Benchmark
|
## 5. Benchmark
|
||||||
|
|
||||||
### 5.1 Accuracy Benchmark
|
### 5.1 Accuracy Benchmark
|
||||||
@@ -1029,7 +1031,7 @@ Max ITL (ms): 7778.47
|
|||||||
- **Model Deployment:**
|
- **Model Deployment:**
|
||||||
|
|
||||||
```bash Command
|
```bash Command
|
||||||
SGLANG_ENABLE_SPEC_V2=1 sglang serve \
|
sglang serve \
|
||||||
--model-path moonshotai/Kimi-K2.5 \
|
--model-path moonshotai/Kimi-K2.5 \
|
||||||
--tp 8 \
|
--tp 8 \
|
||||||
--reasoning-parser kimi_k2 \
|
--reasoning-parser kimi_k2 \
|
||||||
@@ -1038,7 +1040,7 @@ SGLANG_ENABLE_SPEC_V2=1 sglang serve \
|
|||||||
--speculative-num-steps 3 \
|
--speculative-num-steps 3 \
|
||||||
--speculative-eagle-topk 1 \
|
--speculative-eagle-topk 1 \
|
||||||
--speculative-num-draft-tokens 4 \
|
--speculative-num-draft-tokens 4 \
|
||||||
--speculative-draft-model-path lightseekorg/kimi-k2.5-eagle3 \
|
--speculative-draft-model-path lightseekorg/kimi-k2.5-eagle3-mla \
|
||||||
--trust-remote-code \
|
--trust-remote-code \
|
||||||
--host 0.0.0.0 \
|
--host 0.0.0.0 \
|
||||||
--port 30000
|
--port 30000
|
||||||
|
|||||||
@@ -66,7 +66,12 @@ tag: NEW
|
|||||||
- Thinking Mode: `temperature=1.0`, `top_p=0.95`
|
- Thinking Mode: `temperature=1.0`, `top_p=0.95`
|
||||||
- Instant Mode: `temperature=0.6`, `top_p=0.95`
|
- Instant Mode: `temperature=0.6`, `top_p=0.95`
|
||||||
|
|
||||||
**License:** Modified MIT
|
**Available Models:**
|
||||||
|
|
||||||
|
- **INT4 (native checkpoint)**: [moonshotai/Kimi-K2.6](https://huggingface.co/moonshotai/Kimi-K2.6)
|
||||||
|
- **NVFP4 (4-bit quantized, NVIDIA Blackwell)**: [nvidia/Kimi-K2.6-NVFP4](https://huggingface.co/nvidia/Kimi-K2.6-NVFP4)
|
||||||
|
|
||||||
|
**License:** Modified MIT for the native checkpoint. The NVIDIA NVFP4 checkpoint is governed by the [NVIDIA Open Model License](https://www.nvidia.com/en-us/agreements/enterprise-software/nvidia-open-model-license/).
|
||||||
|
|
||||||
For details, see [official documentation](https://huggingface.co/moonshotai/Kimi-K2.6) and [tech blog](https://kimi.com/blog/kimi-k2-6).
|
For details, see [official documentation](https://huggingface.co/moonshotai/Kimi-K2.6) and [tech blog](https://kimi.com/blog/kimi-k2-6).
|
||||||
|
|
||||||
@@ -86,7 +91,8 @@ import { KimiK26Deployment } from '/src/snippets/autoregressive/kimi-k26-deploym
|
|||||||
|
|
||||||
### 3.2 Configuration Tips
|
### 3.2 Configuration Tips
|
||||||
|
|
||||||
- **Memory**: Requires GPUs with ≥140GB each. Supported platforms: H200 (8×, TP=8), B200 (8×, TP=8), B300 (8×, TP=8), GB200 (4×, TP=4), GB300 (4×, TP=4), MI300X/MI325X (4×, TP=4), MI350X/MI355X (4×, TP=4). Use `--context-length 128000` to conserve memory.
|
- **Memory**: Requires GPUs with ≥140GB each. The native INT4 checkpoint supports H200 (8×, TP=8), B300 (8×, TP=8), GB300 (4×, TP=4), MI300X/MI325X (4×, TP=4), and MI350X/MI355X (4×, TP=4). Use `--context-length 128000` to conserve memory.
|
||||||
|
- **NVFP4**: Use `nvidia/Kimi-K2.6-NVFP4` with `--quantization modelopt_fp4` on NVIDIA Blackwell. Use `tp=8` on B300 and `tp=4` on GB300.
|
||||||
- **AMD GPU TP Constraint**: On AMD GPUs, TP must be ≤ 4 (not 8). Kimi-K2.6 has 64 attention heads; the AITER MLA kernel requires `heads_per_gpu % 16 == 0`. With TP=4, each GPU gets 16 heads (valid). With TP=8, each GPU gets 8 heads (invalid).
|
- **AMD GPU TP Constraint**: On AMD GPUs, TP must be ≤ 4 (not 8). Kimi-K2.6 has 64 attention heads; the AITER MLA kernel requires `heads_per_gpu % 16 == 0`. With TP=4, each GPU gets 16 heads (valid). With TP=8, each GPU gets 8 heads (invalid).
|
||||||
- **AMD Docker Image**: Use `lmsysorg/sglang:v0.5.9-rocm700-mi35x` for MI350X/MI355X and `lmsysorg/sglang:v0.5.9-rocm700-mi30x` for MI300X/MI325X.
|
- **AMD Docker Image**: Use `lmsysorg/sglang:v0.5.9-rocm700-mi35x` for MI350X/MI355X and `lmsysorg/sglang:v0.5.9-rocm700-mi30x` for MI300X/MI325X.
|
||||||
- **DP Attention**: Enable with `--dp <N> --enable-dp-attention` for production throughput. A common choice is to set `--dp` equal to `--tp`, but this is not required.
|
- **DP Attention**: Enable with `--dp <N> --enable-dp-attention` for production throughput. A common choice is to set `--dp` equal to `--tp`, but this is not required.
|
||||||
@@ -469,6 +475,48 @@ Let me search for this product and similar items for you.
|
|||||||
Arguments: {"query":"Auntie Anne's Cinnamon Sugar Pretzel"}
|
Arguments: {"query":"Auntie Anne's Cinnamon Sugar Pretzel"}
|
||||||
```
|
```
|
||||||
|
|
||||||
|
#### 4.2.5 Speculative Decoding
|
||||||
|
|
||||||
|
**NVIDIA**
|
||||||
|
|
||||||
|
Deploy Kimi-K2.6 with the following command (H200/B300, all features enabled):
|
||||||
|
|
||||||
|
```shell Command
|
||||||
|
sglang serve \
|
||||||
|
--model-path moonshotai/Kimi-K2.6 \
|
||||||
|
--tp 8 \
|
||||||
|
--reasoning-parser kimi_k2 \
|
||||||
|
--tool-call-parser kimi_k2 \
|
||||||
|
--speculative-algorithm EAGLE3 \
|
||||||
|
--speculative-num-steps 3 \
|
||||||
|
--speculative-eagle-topk 1 \
|
||||||
|
--speculative-num-draft-tokens 4 \
|
||||||
|
--speculative-draft-model-path lightseekorg/kimi-k2.6-eagle3.1-mla \
|
||||||
|
--trust-remote-code \
|
||||||
|
--host 0.0.0.0 \
|
||||||
|
--port 30000
|
||||||
|
```
|
||||||
|
|
||||||
|
Deploy Kimi-K2.6-NVFP4 with the following command (B300, all features enabled):
|
||||||
|
|
||||||
|
```shell Command
|
||||||
|
sglang serve \
|
||||||
|
--model-path nvidia/Kimi-K2.6-NVFP4 \
|
||||||
|
--tp 8 \
|
||||||
|
--quantization modelopt_fp4 \
|
||||||
|
--reasoning-parser kimi_k2 \
|
||||||
|
--tool-call-parser kimi_k2 \
|
||||||
|
--speculative-algorithm EAGLE3 \
|
||||||
|
--speculative-num-steps 3 \
|
||||||
|
--speculative-eagle-topk 1 \
|
||||||
|
--speculative-num-draft-tokens 4 \
|
||||||
|
--speculative-draft-model-path lightseekorg/kimi-k2.6-eagle3.1-mla \
|
||||||
|
--trust-remote-code \
|
||||||
|
--host 0.0.0.0 \
|
||||||
|
--port 30000
|
||||||
|
```
|
||||||
|
|
||||||
|
For GB300, use `--tp 4`.
|
||||||
|
|
||||||
## 5. Benchmark
|
## 5. Benchmark
|
||||||
|
|
||||||
@@ -483,6 +531,44 @@ Let me search for this product and similar items for you.
|
|||||||
- Reasoning Parser: `kimi_k2`
|
- Reasoning Parser: `kimi_k2`
|
||||||
- Tool Call Parser: `kimi_k2`
|
- Tool Call Parser: `kimi_k2`
|
||||||
|
|
||||||
|
#### NVIDIA NVFP4 Accuracy Reference
|
||||||
|
|
||||||
|
NVIDIA reports the following accuracy results for `nvidia/Kimi-K2.6-NVFP4` compared with the native INT4 baseline, using `temperature=1.0`, `top_p=0.95`, and max tokens 128,000:
|
||||||
|
|
||||||
|
<table>
|
||||||
|
<thead>
|
||||||
|
<tr>
|
||||||
|
<th>Precision</th>
|
||||||
|
<th>GPQA Diamond</th>
|
||||||
|
<th>SciCode</th>
|
||||||
|
<th>τ²-Bench Telecom</th>
|
||||||
|
<th>MMMU Pro</th>
|
||||||
|
<th>AA-LCR</th>
|
||||||
|
<th>IFBench</th>
|
||||||
|
</tr>
|
||||||
|
</thead>
|
||||||
|
<tbody>
|
||||||
|
<tr>
|
||||||
|
<td>Baseline (INT4)</td>
|
||||||
|
<td>90.9</td>
|
||||||
|
<td>52.6</td>
|
||||||
|
<td>98.2</td>
|
||||||
|
<td>75.6</td>
|
||||||
|
<td>71.0</td>
|
||||||
|
<td>73.9</td>
|
||||||
|
</tr>
|
||||||
|
<tr>
|
||||||
|
<td>NVFP4</td>
|
||||||
|
<td>90.4</td>
|
||||||
|
<td>54.4</td>
|
||||||
|
<td>98.0</td>
|
||||||
|
<td>76.5</td>
|
||||||
|
<td>71.8</td>
|
||||||
|
<td>73.9</td>
|
||||||
|
</tr>
|
||||||
|
</tbody>
|
||||||
|
</table>
|
||||||
|
|
||||||
#### 5.1.1 K2-Vendor-Verifier (Tool Calling)
|
#### 5.1.1 K2-Vendor-Verifier (Tool Calling)
|
||||||
|
|
||||||
- Dataset: [K2-Vendor-Verifier](https://github.com/MoonshotAI/K2-Vendor-Verifier) tool-calls dataset (2,000 requests)
|
- Dataset: [K2-Vendor-Verifier](https://github.com/MoonshotAI/K2-Vendor-Verifier) tool-calls dataset (2,000 requests)
|
||||||
|
|||||||
@@ -4,13 +4,14 @@ export const KimiK25Deployment = () => {
|
|||||||
// GPU requirements:
|
// GPU requirements:
|
||||||
// H200: tp=8
|
// H200: tp=8
|
||||||
// B300: tp=8
|
// B300: tp=8
|
||||||
|
// GB300: tp=4
|
||||||
// MI300X: tp=4 (64 heads / 4 = 16 heads per GPU, AITER MLA requires heads_per_gpu % 16 == 0)
|
// MI300X: tp=4 (64 heads / 4 = 16 heads per GPU, AITER MLA requires heads_per_gpu % 16 == 0)
|
||||||
// MI325X: tp=4 (same constraint as MI300X)
|
// MI325X: tp=4 (same constraint as MI300X)
|
||||||
// MI350X: tp=4 (same constraint as MI300X)
|
// MI350X: tp=4 (same constraint as MI300X)
|
||||||
// MI355X: tp=4 (same constraint as MI300X)
|
// MI355X: tp=4 (same constraint as MI300X)
|
||||||
//
|
//
|
||||||
// NVFP4 quantization is only supported on NVIDIA Blackwell (B300).
|
// NVFP4 quantization is only supported on NVIDIA Blackwell (B300/GB300).
|
||||||
// Speculative decoding is only supported on H200 and B300.
|
// Speculative decoding is only supported on H200, B300, and GB300.
|
||||||
const options = {
|
const options = {
|
||||||
hardware: {
|
hardware: {
|
||||||
name: 'hardware',
|
name: 'hardware',
|
||||||
@@ -18,6 +19,7 @@ export const KimiK25Deployment = () => {
|
|||||||
items: [
|
items: [
|
||||||
{ id: 'h200', label: 'H200', default: true },
|
{ id: 'h200', label: 'H200', default: true },
|
||||||
{ id: 'b300', label: 'B300', default: false },
|
{ id: 'b300', label: 'B300', default: false },
|
||||||
|
{ id: 'gb300', label: 'GB300', default: false },
|
||||||
{ id: 'mi300x', label: 'MI300X', default: false },
|
{ id: 'mi300x', label: 'MI300X', default: false },
|
||||||
{ id: 'mi325x', label: 'MI325X', default: false },
|
{ id: 'mi325x', label: 'MI325X', default: false },
|
||||||
{ id: 'mi350x', label: 'MI350X', default: false },
|
{ id: 'mi350x', label: 'MI350X', default: false },
|
||||||
@@ -29,10 +31,10 @@ export const KimiK25Deployment = () => {
|
|||||||
title: 'Quantization',
|
title: 'Quantization',
|
||||||
getDynamicItems: (values) => {
|
getDynamicItems: (values) => {
|
||||||
const hw = values.hardware;
|
const hw = values.hardware;
|
||||||
const isB300 = hw === 'b300';
|
const isBlackwell = hw === 'b300' || hw === 'gb300';
|
||||||
return [
|
return [
|
||||||
{ id: 'int4', label: 'INT4', subtitle: 'initial model', default: true },
|
{ id: 'int4', label: 'INT4', subtitle: 'initial model', default: true },
|
||||||
{ id: 'nvfp4', label: 'NVFP4', subtitle: 'Blackwell only', default: false, disabled: !isB300, disabledReason: 'NVFP4 only on B300' }
|
{ id: 'nvfp4', label: 'NVFP4', subtitle: 'Blackwell only', default: false, disabled: !isBlackwell, disabledReason: 'NVFP4 only on B300/GB300' }
|
||||||
];
|
];
|
||||||
}
|
}
|
||||||
},
|
},
|
||||||
@@ -63,7 +65,7 @@ export const KimiK25Deployment = () => {
|
|||||||
speculative: {
|
speculative: {
|
||||||
name: 'speculative',
|
name: 'speculative',
|
||||||
title: 'Speculative Decoding',
|
title: 'Speculative Decoding',
|
||||||
condition: (values) => values.hardware === 'h200' || values.hardware === 'b300',
|
condition: (values) => values.hardware === 'h200' || values.hardware === 'b300' || values.hardware === 'gb300',
|
||||||
items: [
|
items: [
|
||||||
{ id: 'disabled', label: 'Disabled', default: true },
|
{ id: 'disabled', label: 'Disabled', default: true },
|
||||||
{ id: 'enabled', label: 'Enabled', default: false }
|
{ id: 'enabled', label: 'Enabled', default: false }
|
||||||
@@ -74,6 +76,7 @@ export const KimiK25Deployment = () => {
|
|||||||
const modelConfigs = {
|
const modelConfigs = {
|
||||||
h200: { tp: 8 },
|
h200: { tp: 8 },
|
||||||
b300: { tp: 8 },
|
b300: { tp: 8 },
|
||||||
|
gb300: { tp: 4 },
|
||||||
mi300x: { tp: 4 },
|
mi300x: { tp: 4 },
|
||||||
mi325x: { tp: 4 },
|
mi325x: { tp: 4 },
|
||||||
mi350x: { tp: 4 },
|
mi350x: { tp: 4 },
|
||||||
@@ -112,7 +115,7 @@ export const KimiK25Deployment = () => {
|
|||||||
return () => observer.disconnect();
|
return () => observer.disconnect();
|
||||||
}, []);
|
}, []);
|
||||||
|
|
||||||
// When hardware changes, re-resolve quantization defaults (NVFP4 only on B300).
|
// When hardware changes, re-resolve quantization defaults (NVFP4 only on B300/GB300).
|
||||||
useEffect(() => {
|
useEffect(() => {
|
||||||
setValues(prev => {
|
setValues(prev => {
|
||||||
const next = { ...prev };
|
const next = { ...prev };
|
||||||
@@ -138,14 +141,14 @@ export const KimiK25Deployment = () => {
|
|||||||
const { hardware, quantization, speculative } = values;
|
const { hardware, quantization, speculative } = values;
|
||||||
const isAMD = hardware === 'mi300x' || hardware === 'mi325x' || hardware === 'mi350x' || hardware === 'mi355x';
|
const isAMD = hardware === 'mi300x' || hardware === 'mi325x' || hardware === 'mi350x' || hardware === 'mi355x';
|
||||||
|
|
||||||
// NVFP4 is only supported on NVIDIA Blackwell (B300)
|
// NVFP4 is only supported on NVIDIA Blackwell (B300/GB300)
|
||||||
if (quantization === 'nvfp4' && hardware !== 'b300') {
|
if (quantization === 'nvfp4' && hardware !== 'b300' && hardware !== 'gb300') {
|
||||||
return '# NVFP4 quantization is only supported on NVIDIA Blackwell GPUs (B300)';
|
return '# NVFP4 quantization is only supported on NVIDIA Blackwell GPUs (B300/GB300)';
|
||||||
}
|
}
|
||||||
|
|
||||||
// Speculative decoding only supported on H200 and B300
|
// Speculative decoding only supported on H200, B300, and GB300
|
||||||
if (speculative === 'enabled' && hardware !== 'h200' && hardware !== 'b300') {
|
if (speculative === 'enabled' && hardware !== 'h200' && hardware !== 'b300' && hardware !== 'gb300') {
|
||||||
return '# Speculative Decoding for Kimi-K2.5 is only supported on H200 and B300';
|
return '# Speculative Decoding for Kimi-K2.5 is only supported on H200, B300, and GB300';
|
||||||
}
|
}
|
||||||
|
|
||||||
// Model path depends on quantization
|
// Model path depends on quantization
|
||||||
@@ -163,13 +166,8 @@ export const KimiK25Deployment = () => {
|
|||||||
cmd += 'SGLANG_USE_AITER=1 SGLANG_ROCM_FUSED_DECODE_MLA=0 ';
|
cmd += 'SGLANG_USE_AITER=1 SGLANG_ROCM_FUSED_DECODE_MLA=0 ';
|
||||||
}
|
}
|
||||||
|
|
||||||
// Speculative decoding env var
|
|
||||||
if (speculative === 'enabled') {
|
|
||||||
cmd += 'SGLANG_ENABLE_SPEC_V2=1 ';
|
|
||||||
}
|
|
||||||
|
|
||||||
// If we added any env vars above, break to a new line for readability
|
// If we added any env vars above, break to a new line for readability
|
||||||
if (isAMD || speculative === 'enabled') {
|
if (isAMD) {
|
||||||
cmd += '\\\n';
|
cmd += '\\\n';
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -198,8 +196,8 @@ export const KimiK25Deployment = () => {
|
|||||||
cmd += ' \\\n --speculative-algorithm EAGLE3 \\\n --speculative-num-steps 3 \\\n --speculative-eagle-topk 1 \\\n --speculative-num-draft-tokens 4 \\\n --speculative-draft-model-path lightseekorg/kimi-k2.5-eagle3-mla';
|
cmd += ' \\\n --speculative-algorithm EAGLE3 \\\n --speculative-num-steps 3 \\\n --speculative-eagle-topk 1 \\\n --speculative-num-draft-tokens 4 \\\n --speculative-draft-model-path lightseekorg/kimi-k2.5-eagle3-mla';
|
||||||
}
|
}
|
||||||
|
|
||||||
// Blackwell (B300): tokenspeed MLA attention backend
|
// Blackwell (B300/GB300): tokenspeed MLA attention backend
|
||||||
if (hardware === 'b300') {
|
if (hardware === 'b300' || hardware === 'gb300') {
|
||||||
cmd += ' \\\n --attention-backend tokenspeed_mla';
|
cmd += ' \\\n --attention-backend tokenspeed_mla';
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -1,14 +1,20 @@
|
|||||||
export const KimiK26Deployment = () => {
|
export const KimiK26Deployment = () => {
|
||||||
// Config mirrors sgl-cookbook src/components/autoregressive/KimiK26ConfigGenerator/index.js.
|
// Config mirrors sgl-cookbook src/components/autoregressive/KimiK26ConfigGenerator/index.js.
|
||||||
|
//
|
||||||
|
// INT4:
|
||||||
|
// H200/B300: tp=8
|
||||||
|
// GB300/AMD: tp=4
|
||||||
|
//
|
||||||
|
// NVFP4:
|
||||||
|
// B300: tp=8
|
||||||
|
// GB300: tp=4
|
||||||
const options = {
|
const options = {
|
||||||
hardware: {
|
hardware: {
|
||||||
name: 'hardware',
|
name: 'hardware',
|
||||||
title: 'Hardware Platform',
|
title: 'Hardware Platform',
|
||||||
items: [
|
items: [
|
||||||
{ id: 'h200', label: 'H200', default: true },
|
{ id: 'h200', label: 'H200', default: true },
|
||||||
{ id: 'b200', label: 'B200', default: false },
|
|
||||||
{ id: 'b300', label: 'B300', default: false },
|
{ id: 'b300', label: 'B300', default: false },
|
||||||
{ id: 'gb200', label: 'GB200', default: false },
|
|
||||||
{ id: 'gb300', label: 'GB300', default: false },
|
{ id: 'gb300', label: 'GB300', default: false },
|
||||||
{ id: 'mi300x', label: 'MI300X', default: false },
|
{ id: 'mi300x', label: 'MI300X', default: false },
|
||||||
{ id: 'mi325x', label: 'MI325X', default: false },
|
{ id: 'mi325x', label: 'MI325X', default: false },
|
||||||
@@ -16,6 +22,18 @@ export const KimiK26Deployment = () => {
|
|||||||
{ id: 'mi355x', label: 'MI355X', default: false },
|
{ id: 'mi355x', label: 'MI355X', default: false },
|
||||||
],
|
],
|
||||||
},
|
},
|
||||||
|
quantization: {
|
||||||
|
name: 'quantization',
|
||||||
|
title: 'Quantization',
|
||||||
|
getDynamicItems: (values) => {
|
||||||
|
const hw = values.hardware;
|
||||||
|
const isBlackwell = ['b300', 'gb300'].includes(hw);
|
||||||
|
return [
|
||||||
|
{ id: 'int4', label: 'INT4', subtitle: 'Base checkpoint', default: !isBlackwell },
|
||||||
|
{ id: 'nvfp4', label: 'NVFP4', subtitle: 'Blackwell FP4', default: isBlackwell, disabled: !isBlackwell, disabledReason: !isBlackwell ? 'NVFP4 only on NVIDIA Blackwell' : '' },
|
||||||
|
];
|
||||||
|
},
|
||||||
|
},
|
||||||
reasoning: {
|
reasoning: {
|
||||||
name: 'reasoning',
|
name: 'reasoning',
|
||||||
title: 'Reasoning Parser',
|
title: 'Reasoning Parser',
|
||||||
@@ -40,13 +58,20 @@ export const KimiK26Deployment = () => {
|
|||||||
{ id: 'enabled', label: 'Enabled', subtitle: 'High Throughput', default: false },
|
{ id: 'enabled', label: 'Enabled', subtitle: 'High Throughput', default: false },
|
||||||
],
|
],
|
||||||
},
|
},
|
||||||
|
speculative: {
|
||||||
|
name: 'speculative',
|
||||||
|
title: 'Speculative Decoding',
|
||||||
|
condition: (values) => !['mi300x', 'mi325x', 'mi350x', 'mi355x'].includes(values.hardware),
|
||||||
|
items: [
|
||||||
|
{ id: 'disabled', label: 'Disabled', default: true },
|
||||||
|
{ id: 'enabled', label: 'Enabled', default: false },
|
||||||
|
],
|
||||||
|
},
|
||||||
};
|
};
|
||||||
|
|
||||||
const modelConfigs = {
|
const modelConfigs = {
|
||||||
h200: { tp: 8 },
|
h200: { tp: 8 },
|
||||||
b200: { tp: 8 },
|
|
||||||
b300: { tp: 8 },
|
b300: { tp: 8 },
|
||||||
gb200: { tp: 4 },
|
|
||||||
gb300: { tp: 4 },
|
gb300: { tp: 4 },
|
||||||
mi300x: { tp: 4 },
|
mi300x: { tp: 4 },
|
||||||
mi325x: { tp: 4 },
|
mi325x: { tp: 4 },
|
||||||
@@ -54,6 +79,11 @@ export const KimiK26Deployment = () => {
|
|||||||
mi355x: { tp: 4 },
|
mi355x: { tp: 4 },
|
||||||
};
|
};
|
||||||
|
|
||||||
|
const nvfp4ModelConfigs = {
|
||||||
|
b300: { tp: 8 },
|
||||||
|
gb300: { tp: 4 },
|
||||||
|
};
|
||||||
|
|
||||||
const resolveItems = (option, values) =>
|
const resolveItems = (option, values) =>
|
||||||
typeof option.getDynamicItems === 'function' ? option.getDynamicItems(values) : option.items || [];
|
typeof option.getDynamicItems === 'function' ? option.getDynamicItems(values) : option.items || [];
|
||||||
|
|
||||||
@@ -89,15 +119,43 @@ export const KimiK26Deployment = () => {
|
|||||||
return () => observer.disconnect();
|
return () => observer.disconnect();
|
||||||
}, []);
|
}, []);
|
||||||
|
|
||||||
|
useEffect(() => {
|
||||||
|
setValues((prev) => {
|
||||||
|
const next = { ...prev };
|
||||||
|
for (const [key, option] of Object.entries(options)) {
|
||||||
|
if (typeof option.condition === 'function' && !option.condition(next)) {
|
||||||
|
const items = resolveItems(option, next);
|
||||||
|
const fallback = items.find((item) => item.default && !item.disabled) || items.find((item) => !item.disabled);
|
||||||
|
if (fallback) next[key] = fallback.id;
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if (typeof option.getDynamicItems !== 'function') continue;
|
||||||
|
const items = option.getDynamicItems(next);
|
||||||
|
const current = items.find((item) => item.id === next[key]);
|
||||||
|
if (!current || current.disabled) {
|
||||||
|
const fallback = items.find((item) => item.default && !item.disabled) || items.find((item) => !item.disabled);
|
||||||
|
if (fallback) next[key] = fallback.id;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return next;
|
||||||
|
});
|
||||||
|
}, [values.hardware]);
|
||||||
|
|
||||||
const handleRadioChange = (optionName, value) => {
|
const handleRadioChange = (optionName, value) => {
|
||||||
setValues((prev) => ({ ...prev, [optionName]: value }));
|
setValues((prev) => ({ ...prev, [optionName]: value }));
|
||||||
};
|
};
|
||||||
|
|
||||||
const generateCommand = () => {
|
const generateCommand = () => {
|
||||||
const { hardware, reasoning, toolcall, dpattention } = values;
|
const { hardware, quantization, reasoning, toolcall, dpattention, speculative } = values;
|
||||||
const isAMD = hardware === 'mi300x' || hardware === 'mi325x' || hardware === 'mi350x' || hardware === 'mi355x';
|
const isAMD = hardware === 'mi300x' || hardware === 'mi325x' || hardware === 'mi350x' || hardware === 'mi355x';
|
||||||
const hwConfig = modelConfigs[hardware];
|
const isNVFP4 = quantization === 'nvfp4';
|
||||||
|
const hwConfig = isNVFP4 ? nvfp4ModelConfigs[hardware] : modelConfigs[hardware];
|
||||||
|
if (!hwConfig) return '# NVFP4 is only supported on NVIDIA Blackwell hardware.';
|
||||||
|
if (speculative === 'enabled' && isAMD) {
|
||||||
|
return '# Speculative Decoding for Kimi-K2.6 is only supported on NVIDIA GPUs (H200/B300/GB300)';
|
||||||
|
}
|
||||||
const tpValue = hwConfig.tp;
|
const tpValue = hwConfig.tp;
|
||||||
|
const modelName = isNVFP4 ? 'nvidia/Kimi-K2.6-NVFP4' : 'moonshotai/Kimi-K2.6';
|
||||||
|
|
||||||
let cmd = '';
|
let cmd = '';
|
||||||
|
|
||||||
@@ -106,8 +164,11 @@ export const KimiK26Deployment = () => {
|
|||||||
}
|
}
|
||||||
|
|
||||||
cmd += 'sglang serve \\\n';
|
cmd += 'sglang serve \\\n';
|
||||||
cmd += ' --model-path moonshotai/Kimi-K2.6';
|
cmd += ` --model-path ${modelName}`;
|
||||||
cmd += ` \\\n --tp ${tpValue}`;
|
cmd += ` \\\n --tp ${tpValue}`;
|
||||||
|
if (isNVFP4) {
|
||||||
|
cmd += ' \\\n --quantization modelopt_fp4';
|
||||||
|
}
|
||||||
if (isAMD) {
|
if (isAMD) {
|
||||||
cmd += ' \\\n --mem-fraction-static 0.8';
|
cmd += ' \\\n --mem-fraction-static 0.8';
|
||||||
}
|
}
|
||||||
@@ -125,6 +186,18 @@ export const KimiK26Deployment = () => {
|
|||||||
cmd += ' \\\n --tool-call-parser kimi_k2';
|
cmd += ' \\\n --tool-call-parser kimi_k2';
|
||||||
}
|
}
|
||||||
|
|
||||||
|
if (speculative === 'enabled') {
|
||||||
|
cmd += ' \\\n --speculative-algorithm EAGLE3';
|
||||||
|
cmd += ' \\\n --speculative-num-steps 3';
|
||||||
|
cmd += ' \\\n --speculative-eagle-topk 1';
|
||||||
|
cmd += ' \\\n --speculative-num-draft-tokens 4';
|
||||||
|
cmd += ' \\\n --speculative-draft-model-path lightseekorg/kimi-k2.6-eagle3.1-mla';
|
||||||
|
}
|
||||||
|
|
||||||
|
if (hardware === 'b300' || hardware === 'gb300') {
|
||||||
|
cmd += ' \\\n --attention-backend tokenspeed_mla';
|
||||||
|
}
|
||||||
|
|
||||||
if (isAMD) {
|
if (isAMD) {
|
||||||
cmd += ' \\\n --kv-cache-dtype fp8_e4m3';
|
cmd += ' \\\n --kv-cache-dtype fp8_e4m3';
|
||||||
}
|
}
|
||||||
@@ -146,6 +219,7 @@ export const KimiK26Deployment = () => {
|
|||||||
return (
|
return (
|
||||||
<div style={containerStyle} className="not-prose">
|
<div style={containerStyle} className="not-prose">
|
||||||
{Object.entries(options).map(([key, option]) => {
|
{Object.entries(options).map(([key, option]) => {
|
||||||
|
if (typeof option.condition === 'function' && !option.condition(values)) return null;
|
||||||
const items = resolveItems(option, values);
|
const items = resolveItems(option, values);
|
||||||
return (
|
return (
|
||||||
<div key={key} style={cardStyle}>
|
<div key={key} style={cardStyle}>
|
||||||
|
|||||||
Reference in New Issue
Block a user