[Docs] Inkling cookbook: mark B300/GB300 recipes verified, tune B300 MTP mem fractions (#31550)
Co-authored-by: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Fable 5
parent
d67aa05697
commit
8f765bc1c9
@@ -297,6 +297,6 @@ The **Long Context** deploy strategy adds `--kv-cache-dtype mxfp8` on top of the
|
|||||||
|
|
||||||
**Blackwell only.** MXFP8 KV cache requires Blackwell (B200 / B300 / GB200 / GB300), it's not offered on Hopper (H200).
|
**Blackwell only.** MXFP8 KV cache requires Blackwell (B200 / B300 / GB200 / GB300), it's not offered on Hopper (H200).
|
||||||
|
|
||||||
The tradeoff is a ~5% decode latency penalty from the extra quantize/dequantize work versus BF16 KV, so treat it as a capacity lever, not a speed one — stay on **Balanced** if you have headroom in the memory pool and just want lower latency.
|
The tradeoff is not just a ~5% decode latency penalty from the extra quantize/dequantize work versus BF16 KV — storing KV in MXFP8 also introduces some accuracy loss at long context lengths. Treat it as a capacity lever, not a speed one — stay on **Balanced** if you have headroom in the memory pool and just want lower latency or maximum output quality.
|
||||||
|
|
||||||
To try it, select the **Long Context** strategy in the Deploy panel above for any NVFP4 cell; the panel regenerates the launch command with `--kv-cache-dtype mxfp8` inserted. Verified end-to-end on B200.
|
To try it, select the **Long Context** strategy in the Deploy panel above for any NVFP4 cell; the panel regenerates the launch command with `--kv-cache-dtype mxfp8` inserted. Verified end-to-end on B200, B300, and GB300.
|
||||||
|
|||||||
@@ -6,9 +6,9 @@
|
|||||||
export const config = {
|
export const config = {
|
||||||
modelName: "Inkling",
|
modelName: "Inkling",
|
||||||
|
|
||||||
// Full platform list. Actively validated targets: B200 + H200 (verified end-to-end),
|
// Full platform list. Verified end-to-end: B200, B300, GB300, H200, and AMD
|
||||||
// GB300 and AMD (MI350X / MI355X) in progress. The other NVIDIA cells are same-arch
|
// (MI350X / MI355X). The remaining cells (GB200, B200 BF16 multi-node) are
|
||||||
// extrapolations (Blackwell SM100 / Hopper SM90) and stay unverified until re-checked.
|
// same-arch extrapolations and stay unverified until re-checked.
|
||||||
supportedHardware: [
|
supportedHardware: [
|
||||||
"h200", "b200", "b300", "gb200", "gb300",
|
"h200", "b200", "b300", "gb200", "gb300",
|
||||||
"mi350x", "mi355x",
|
"mi350x", "mi355x",
|
||||||
@@ -183,7 +183,7 @@ export const config = {
|
|||||||
cells: [
|
cells: [
|
||||||
// ====================================================================
|
// ====================================================================
|
||||||
// NVIDIA Blackwell (SM100) + NVFP4 — FlashInfer TRT-LLM routed FP4 experts.
|
// NVIDIA Blackwell (SM100) + NVFP4 — FlashInfer TRT-LLM routed FP4 experts.
|
||||||
// B200 verified; B300 / GB200 / GB300 same-arch (GB300 in active validation).
|
// B200 / B300 / GB300 verified; GB200 same-arch extrapolation.
|
||||||
// ====================================================================
|
// ====================================================================
|
||||||
{
|
{
|
||||||
match: { hw: "b200", variant: "default", quant: "nvfp4", strategy: "balanced", nodes: "single" },
|
match: { hw: "b200", variant: "default", quant: "nvfp4", strategy: "balanced", nodes: "single" },
|
||||||
@@ -214,6 +214,7 @@ export const config = {
|
|||||||
},
|
},
|
||||||
{
|
{
|
||||||
match: { hw: "b300", variant: "default", quant: "nvfp4", strategy: "balanced", nodes: "single" },
|
match: { hw: "b300", variant: "default", quant: "nvfp4", strategy: "balanced", nodes: "single" },
|
||||||
|
verified: true,
|
||||||
env: [
|
env: [
|
||||||
"SGLANG_ENABLE_UNIFIED_RADIX_TREE=1",
|
"SGLANG_ENABLE_UNIFIED_RADIX_TREE=1",
|
||||||
],
|
],
|
||||||
@@ -266,6 +267,7 @@ export const config = {
|
|||||||
},
|
},
|
||||||
{
|
{
|
||||||
match: { hw: "gb300", variant: "default", quant: "nvfp4", strategy: "balanced", nodes: "single" },
|
match: { hw: "gb300", variant: "default", quant: "nvfp4", strategy: "balanced", nodes: "single" },
|
||||||
|
verified: true,
|
||||||
env: [
|
env: [
|
||||||
"SGLANG_ENABLE_UNIFIED_RADIX_TREE=1",
|
"SGLANG_ENABLE_UNIFIED_RADIX_TREE=1",
|
||||||
],
|
],
|
||||||
@@ -381,8 +383,8 @@ export const config = {
|
|||||||
// Long Context (MXFP8 KV) — block-scaled KV cache shrinks the per-token
|
// Long Context (MXFP8 KV) — block-scaled KV cache shrinks the per-token
|
||||||
// KV footprint, raising how many tokens fit in the memory pool (longer
|
// KV footprint, raising how many tokens fit in the memory pool (longer
|
||||||
// context / more concurrent sequences) vs the default BF16 KV. Same base
|
// context / more concurrent sequences) vs the default BF16 KV. Same base
|
||||||
// command as Balanced + `--kv-cache-dtype mxfp8`. B200 verified
|
// command as Balanced + `--kv-cache-dtype mxfp8`. B200 / B300 / GB300
|
||||||
// end-to-end.
|
// verified end-to-end.
|
||||||
// ====================================================================
|
// ====================================================================
|
||||||
{
|
{
|
||||||
match: { hw: "b200", variant: "default", quant: "nvfp4", strategy: "long_context", nodes: "single" },
|
match: { hw: "b200", variant: "default", quant: "nvfp4", strategy: "long_context", nodes: "single" },
|
||||||
@@ -414,6 +416,7 @@ export const config = {
|
|||||||
},
|
},
|
||||||
{
|
{
|
||||||
match: { hw: "b300", variant: "default", quant: "nvfp4", strategy: "long_context", nodes: "single" },
|
match: { hw: "b300", variant: "default", quant: "nvfp4", strategy: "long_context", nodes: "single" },
|
||||||
|
verified: true,
|
||||||
env: [
|
env: [
|
||||||
"SGLANG_ENABLE_UNIFIED_RADIX_TREE=1",
|
"SGLANG_ENABLE_UNIFIED_RADIX_TREE=1",
|
||||||
],
|
],
|
||||||
@@ -468,6 +471,7 @@ export const config = {
|
|||||||
},
|
},
|
||||||
{
|
{
|
||||||
match: { hw: "gb300", variant: "default", quant: "nvfp4", strategy: "long_context", nodes: "single" },
|
match: { hw: "gb300", variant: "default", quant: "nvfp4", strategy: "long_context", nodes: "single" },
|
||||||
|
verified: true,
|
||||||
env: [
|
env: [
|
||||||
"SGLANG_ENABLE_UNIFIED_RADIX_TREE=1",
|
"SGLANG_ENABLE_UNIFIED_RADIX_TREE=1",
|
||||||
],
|
],
|
||||||
@@ -498,7 +502,7 @@ export const config = {
|
|||||||
// MTP (speculative decoding) — Inkling's multi-layer MTP draft head.
|
// MTP (speculative decoding) — Inkling's multi-layer MTP draft head.
|
||||||
// --enable-multi-layer-eagle is REQUIRED (without it the standard EAGLE
|
// --enable-multi-layer-eagle is REQUIRED (without it the standard EAGLE
|
||||||
// worker runs against the multi-layer draft and outputs garbage).
|
// worker runs against the multi-layer draft and outputs garbage).
|
||||||
// B200 verified end-to-end; H200 from the same validated command set.
|
// B200 / B300 / GB300 / H200 verified end-to-end.
|
||||||
// ====================================================================
|
// ====================================================================
|
||||||
{
|
{
|
||||||
match: { hw: "b200", variant: "default", quant: "nvfp4", strategy: "mtp", nodes: "single" },
|
match: { hw: "b200", variant: "default", quant: "nvfp4", strategy: "mtp", nodes: "single" },
|
||||||
@@ -535,6 +539,7 @@ export const config = {
|
|||||||
},
|
},
|
||||||
{
|
{
|
||||||
match: { hw: "b300", variant: "default", quant: "nvfp4", strategy: "mtp", nodes: "single" },
|
match: { hw: "b300", variant: "default", quant: "nvfp4", strategy: "mtp", nodes: "single" },
|
||||||
|
verified: true,
|
||||||
env: [
|
env: [
|
||||||
"SGLANG_ENABLE_UNIFIED_RADIX_TREE=1",
|
"SGLANG_ENABLE_UNIFIED_RADIX_TREE=1",
|
||||||
],
|
],
|
||||||
@@ -549,7 +554,7 @@ export const config = {
|
|||||||
"--moe-runner-backend flashinfer_trtllm_routed",
|
"--moe-runner-backend flashinfer_trtllm_routed",
|
||||||
"--enable-torch-symm-mem",
|
"--enable-torch-symm-mem",
|
||||||
"--mamba-radix-cache-strategy extra_buffer",
|
"--mamba-radix-cache-strategy extra_buffer",
|
||||||
"--mem-fraction-static 0.75",
|
"--mem-fraction-static 0.70",
|
||||||
"--swa-full-tokens-ratio 0.1",
|
"--swa-full-tokens-ratio 0.1",
|
||||||
"--mamba-full-memory-ratio 0.1",
|
"--mamba-full-memory-ratio 0.1",
|
||||||
"--enable-multimodal",
|
"--enable-multimodal",
|
||||||
@@ -599,6 +604,7 @@ export const config = {
|
|||||||
},
|
},
|
||||||
{
|
{
|
||||||
match: { hw: "gb300", variant: "default", quant: "nvfp4", strategy: "mtp", nodes: "single" },
|
match: { hw: "gb300", variant: "default", quant: "nvfp4", strategy: "mtp", nodes: "single" },
|
||||||
|
verified: true,
|
||||||
env: [
|
env: [
|
||||||
"SGLANG_ENABLE_UNIFIED_RADIX_TREE=1",
|
"SGLANG_ENABLE_UNIFIED_RADIX_TREE=1",
|
||||||
],
|
],
|
||||||
@@ -631,6 +637,7 @@ export const config = {
|
|||||||
},
|
},
|
||||||
{
|
{
|
||||||
match: { hw: "h200", variant: "default", quant: "nvfp4", strategy: "mtp", nodes: "single" },
|
match: { hw: "h200", variant: "default", quant: "nvfp4", strategy: "mtp", nodes: "single" },
|
||||||
|
verified: true,
|
||||||
env: [
|
env: [
|
||||||
"SGLANG_ENABLE_UNIFIED_RADIX_TREE=1",
|
"SGLANG_ENABLE_UNIFIED_RADIX_TREE=1",
|
||||||
],
|
],
|
||||||
@@ -669,6 +676,7 @@ export const config = {
|
|||||||
// ====================================================================
|
// ====================================================================
|
||||||
{
|
{
|
||||||
match: { hw: "gb300", variant: "default", quant: "bf16", strategy: "balanced", nodes: "multi-2" },
|
match: { hw: "gb300", variant: "default", quant: "bf16", strategy: "balanced", nodes: "multi-2" },
|
||||||
|
verified: true,
|
||||||
env: [
|
env: [
|
||||||
"NCCL_MNNVL_ENABLE=1",
|
"NCCL_MNNVL_ENABLE=1",
|
||||||
"NCCL_NVLS_ENABLE=1",
|
"NCCL_NVLS_ENABLE=1",
|
||||||
@@ -698,6 +706,7 @@ export const config = {
|
|||||||
},
|
},
|
||||||
{
|
{
|
||||||
match: { hw: "gb300", variant: "default", quant: "bf16", strategy: "mtp", nodes: "multi-2" },
|
match: { hw: "gb300", variant: "default", quant: "bf16", strategy: "mtp", nodes: "multi-2" },
|
||||||
|
verified: true,
|
||||||
env: [
|
env: [
|
||||||
"NCCL_MNNVL_ENABLE=1",
|
"NCCL_MNNVL_ENABLE=1",
|
||||||
"NCCL_NVLS_ENABLE=1",
|
"NCCL_NVLS_ENABLE=1",
|
||||||
@@ -733,6 +742,7 @@ export const config = {
|
|||||||
},
|
},
|
||||||
{
|
{
|
||||||
match: { hw: "b300", variant: "default", quant: "bf16", strategy: "balanced", nodes: "single" },
|
match: { hw: "b300", variant: "default", quant: "bf16", strategy: "balanced", nodes: "single" },
|
||||||
|
verified: true,
|
||||||
env: [
|
env: [
|
||||||
"SGLANG_ENABLE_UNIFIED_RADIX_TREE=1",
|
"SGLANG_ENABLE_UNIFIED_RADIX_TREE=1",
|
||||||
],
|
],
|
||||||
@@ -757,6 +767,7 @@ export const config = {
|
|||||||
},
|
},
|
||||||
{
|
{
|
||||||
match: { hw: "b300", variant: "default", quant: "bf16", strategy: "mtp", nodes: "single" },
|
match: { hw: "b300", variant: "default", quant: "bf16", strategy: "mtp", nodes: "single" },
|
||||||
|
verified: true,
|
||||||
env: [
|
env: [
|
||||||
"SGLANG_ENABLE_UNIFIED_RADIX_TREE=1",
|
"SGLANG_ENABLE_UNIFIED_RADIX_TREE=1",
|
||||||
],
|
],
|
||||||
@@ -769,7 +780,7 @@ export const config = {
|
|||||||
"--moe-runner-backend flashinfer_trtllm_routed",
|
"--moe-runner-backend flashinfer_trtllm_routed",
|
||||||
"--enable-torch-symm-mem",
|
"--enable-torch-symm-mem",
|
||||||
"--mamba-radix-cache-strategy extra_buffer",
|
"--mamba-radix-cache-strategy extra_buffer",
|
||||||
"--mem-fraction-static 0.85",
|
"--mem-fraction-static 0.87",
|
||||||
"--swa-full-tokens-ratio 0.1",
|
"--swa-full-tokens-ratio 0.1",
|
||||||
"--mamba-full-memory-ratio 0.1",
|
"--mamba-full-memory-ratio 0.1",
|
||||||
"--enable-multimodal",
|
"--enable-multimodal",
|
||||||
@@ -842,7 +853,7 @@ export const config = {
|
|||||||
// ====================================================================
|
// ====================================================================
|
||||||
// LoRA serving. Prefill CUDA graphs auto-disable under --enable-lora.
|
// LoRA serving. Prefill CUDA graphs auto-disable under --enable-lora.
|
||||||
// Set MAX_LORAS to the number of distinct adapters served (1 is fastest
|
// Set MAX_LORAS to the number of distinct adapters served (1 is fastest
|
||||||
// for single-adapter serving). All three cells verified end-to-end
|
// for single-adapter serving). All cells except GB200 verified end-to-end
|
||||||
// (coherence + trainer-logprob parity + throughput).
|
// (coherence + trainer-logprob parity + throughput).
|
||||||
// ====================================================================
|
// ====================================================================
|
||||||
{
|
{
|
||||||
@@ -882,6 +893,7 @@ export const config = {
|
|||||||
},
|
},
|
||||||
{
|
{
|
||||||
match: { hw: "b300", variant: "lora", quant: "nvfp4", strategy: "balanced", nodes: "single" },
|
match: { hw: "b300", variant: "lora", quant: "nvfp4", strategy: "balanced", nodes: "single" },
|
||||||
|
verified: true,
|
||||||
env: [
|
env: [
|
||||||
"SGLANG_ENABLE_UNIFIED_RADIX_TREE=1",
|
"SGLANG_ENABLE_UNIFIED_RADIX_TREE=1",
|
||||||
"SGLANG_EXPERIMENTAL_LORA_OPTI=1",
|
"SGLANG_EXPERIMENTAL_LORA_OPTI=1",
|
||||||
@@ -950,6 +962,7 @@ export const config = {
|
|||||||
},
|
},
|
||||||
{
|
{
|
||||||
match: { hw: "gb300", variant: "lora", quant: "nvfp4", strategy: "balanced", nodes: "single" },
|
match: { hw: "gb300", variant: "lora", quant: "nvfp4", strategy: "balanced", nodes: "single" },
|
||||||
|
verified: true,
|
||||||
env: [
|
env: [
|
||||||
"SGLANG_ENABLE_UNIFIED_RADIX_TREE=1",
|
"SGLANG_ENABLE_UNIFIED_RADIX_TREE=1",
|
||||||
"SGLANG_EXPERIMENTAL_LORA_OPTI=1",
|
"SGLANG_EXPERIMENTAL_LORA_OPTI=1",
|
||||||
|
|||||||
Reference in New Issue
Block a user