Make the remaining DeepSeek-V4.1 NVIDIA cells start (#38861)
Co-authored-by: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 5
parent
209654c420
commit
4b7331fb77
@@ -1,8 +1,16 @@
|
|||||||
// Single `export const config` literal — no spreads/calls/IIFE (Mintlify re-evals at hydration).
|
// Single `export const config` literal — no spreads/calls/IIFE (Mintlify re-evals at hydration).
|
||||||
//
|
//
|
||||||
// Cells marked `verified` are transcribed from recorded runs on that hardware with
|
// Cells marked `verified` are transcribed from recorded runs on that hardware with
|
||||||
// real weights. DP-Attention, DeepEP and MegaMoE are absent by design: they have
|
// real weights.
|
||||||
// never been enabled on this model. EP is set equal to TP on every shape here.
|
//
|
||||||
|
// Every DSpark cell caps --cuda-graph-max-bs-decode: the derived batch list does
|
||||||
|
// not fit while capturing the DSpark decode graphs, on any NVIDIA platform. The
|
||||||
|
// MI350X cell already carried the equivalent --cuda-graph-max-bs. H200 is the
|
||||||
|
// tightest board at 140 GiB and needs the cap on both cells plus a lower memory
|
||||||
|
// fraction; the three other High-Throughput cells start without either.
|
||||||
|
//
|
||||||
|
// DP-Attention, DeepEP and MegaMoE are absent by design: they have never been
|
||||||
|
// enabled on this model. EP is set equal to TP on every shape here.
|
||||||
|
|
||||||
export const config = {
|
export const config = {
|
||||||
modelName: "DeepSeek-V4.1",
|
modelName: "DeepSeek-V4.1",
|
||||||
@@ -182,6 +190,8 @@ export const config = {
|
|||||||
"--mem-fraction-static 0.8",
|
"--mem-fraction-static 0.8",
|
||||||
"--speculative-algorithm DSPARK",
|
"--speculative-algorithm DSPARK",
|
||||||
"--speculative-dspark-block-size 5",
|
"--speculative-dspark-block-size 5",
|
||||||
|
// Decode CUDA graphs: the derived batch list does not fit here.
|
||||||
|
"--cuda-graph-max-bs-decode 64",
|
||||||
"--reasoning-parser auto",
|
"--reasoning-parser auto",
|
||||||
"--tool-call-parser auto",
|
"--tool-call-parser auto",
|
||||||
"--host {{HOST_IP}}",
|
"--host {{HOST_IP}}",
|
||||||
@@ -209,20 +219,24 @@ export const config = {
|
|||||||
],
|
],
|
||||||
},
|
},
|
||||||
|
|
||||||
// ---------- H200: 8x H200, TP8 + EP8. No MXFP8 dense path on Hopper;
|
// ---------- H200: 8x H200, TP8 + EP8. No MXFP8 dense path on Hopper. The
|
||||||
// verification round open. ----------
|
// tightest board here at 140 GiB, so both cells also need the memory
|
||||||
|
// fraction pulled back; the cap alone still leaves the graphs short. ------
|
||||||
{
|
{
|
||||||
match: { hw: "h200", strategy: "low-latency" },
|
match: { hw: "h200", strategy: "low-latency" },
|
||||||
nnodes: 1,
|
nnodes: 1,
|
||||||
verificationStatus: "in-progress",
|
verified: true,
|
||||||
flags: [
|
flags: [
|
||||||
"--trust-remote-code",
|
"--trust-remote-code",
|
||||||
"--model-path {{MODEL_NAME}}",
|
"--model-path {{MODEL_NAME}}",
|
||||||
"--tp 8",
|
"--tp 8",
|
||||||
"--ep-size 8",
|
"--ep-size 8",
|
||||||
|
"--mem-fraction-static 0.8",
|
||||||
"--attention-backend dsv4",
|
"--attention-backend dsv4",
|
||||||
"--moe-runner-backend flashinfer_mxfp4",
|
"--moe-runner-backend flashinfer_mxfp4",
|
||||||
"--enable-decoder-swa-bounded-replay",
|
"--enable-decoder-swa-bounded-replay",
|
||||||
|
// Decode CUDA graphs: the derived batch list does not fit here.
|
||||||
|
"--cuda-graph-max-bs-decode 64",
|
||||||
"--reasoning-parser auto",
|
"--reasoning-parser auto",
|
||||||
"--tool-call-parser auto",
|
"--tool-call-parser auto",
|
||||||
"--host {{HOST_IP}}",
|
"--host {{HOST_IP}}",
|
||||||
@@ -232,15 +246,18 @@ export const config = {
|
|||||||
{
|
{
|
||||||
match: { hw: "h200", strategy: "high-throughput" },
|
match: { hw: "h200", strategy: "high-throughput" },
|
||||||
nnodes: 1,
|
nnodes: 1,
|
||||||
verificationStatus: "in-progress",
|
verified: true,
|
||||||
flags: [
|
flags: [
|
||||||
"--trust-remote-code",
|
"--trust-remote-code",
|
||||||
"--model-path {{MODEL_NAME}}",
|
"--model-path {{MODEL_NAME}}",
|
||||||
"--tp 8",
|
"--tp 8",
|
||||||
"--ep-size 8",
|
"--ep-size 8",
|
||||||
|
"--mem-fraction-static 0.8",
|
||||||
"--attention-backend dsv4",
|
"--attention-backend dsv4",
|
||||||
"--moe-runner-backend flashinfer_mxfp4",
|
"--moe-runner-backend flashinfer_mxfp4",
|
||||||
"--max-running-requests 256",
|
"--max-running-requests 256",
|
||||||
|
// Decode CUDA graphs: the derived batch list does not fit here.
|
||||||
|
"--cuda-graph-max-bs-decode 64",
|
||||||
"--reasoning-parser auto",
|
"--reasoning-parser auto",
|
||||||
"--tool-call-parser auto",
|
"--tool-call-parser auto",
|
||||||
"--host {{HOST_IP}}",
|
"--host {{HOST_IP}}",
|
||||||
@@ -248,12 +265,12 @@ export const config = {
|
|||||||
],
|
],
|
||||||
},
|
},
|
||||||
|
|
||||||
// ---------- B200: verification round open. Mirrors the GB300 recipe because
|
// ---------- B200: 4x B200, TP4 + EP4. Mirrors the GB300 recipe — the
|
||||||
// the kernels dispatch by architecture family. ----------
|
// kernels dispatch by architecture family. ----------
|
||||||
{
|
{
|
||||||
match: { hw: "b200", strategy: "low-latency" },
|
match: { hw: "b200", strategy: "low-latency" },
|
||||||
nnodes: 1,
|
nnodes: 1,
|
||||||
verificationStatus: "in-progress",
|
verified: true,
|
||||||
flags: [
|
flags: [
|
||||||
"--trust-remote-code",
|
"--trust-remote-code",
|
||||||
"--model-path {{MODEL_NAME}}",
|
"--model-path {{MODEL_NAME}}",
|
||||||
@@ -262,6 +279,8 @@ export const config = {
|
|||||||
"--mem-fraction-static 0.8",
|
"--mem-fraction-static 0.8",
|
||||||
"--speculative-algorithm DSPARK",
|
"--speculative-algorithm DSPARK",
|
||||||
"--speculative-dspark-block-size 5",
|
"--speculative-dspark-block-size 5",
|
||||||
|
// Decode CUDA graphs: the derived batch list does not fit here.
|
||||||
|
"--cuda-graph-max-bs-decode 64",
|
||||||
"--reasoning-parser auto",
|
"--reasoning-parser auto",
|
||||||
"--tool-call-parser auto",
|
"--tool-call-parser auto",
|
||||||
"--host {{HOST_IP}}",
|
"--host {{HOST_IP}}",
|
||||||
@@ -271,7 +290,7 @@ export const config = {
|
|||||||
{
|
{
|
||||||
match: { hw: "b200", strategy: "high-throughput" },
|
match: { hw: "b200", strategy: "high-throughput" },
|
||||||
nnodes: 1,
|
nnodes: 1,
|
||||||
verificationStatus: "in-progress",
|
verified: true,
|
||||||
flags: [
|
flags: [
|
||||||
"--trust-remote-code",
|
"--trust-remote-code",
|
||||||
"--model-path {{MODEL_NAME}}",
|
"--model-path {{MODEL_NAME}}",
|
||||||
@@ -286,8 +305,7 @@ export const config = {
|
|||||||
},
|
},
|
||||||
|
|
||||||
// ---------- B300: 4x B300, TP4 + EP4. Same recipe as GB300 — the kernels
|
// ---------- B300: 4x B300, TP4 + EP4. Same recipe as GB300 — the kernels
|
||||||
// dispatch by architecture family — with one extra memory knob on
|
// dispatch by architecture family. ----------
|
||||||
// Low-Latency that GB300 does not need. ----------
|
|
||||||
{
|
{
|
||||||
match: { hw: "b300", strategy: "low-latency" },
|
match: { hw: "b300", strategy: "low-latency" },
|
||||||
nnodes: 1,
|
nnodes: 1,
|
||||||
@@ -300,8 +318,7 @@ export const config = {
|
|||||||
"--mem-fraction-static 0.8",
|
"--mem-fraction-static 0.8",
|
||||||
"--speculative-algorithm DSPARK",
|
"--speculative-algorithm DSPARK",
|
||||||
"--speculative-dspark-block-size 5",
|
"--speculative-dspark-block-size 5",
|
||||||
// GB300 does not need this, B300 does: the derived value (256) runs
|
// Decode CUDA graphs: the derived batch list does not fit here.
|
||||||
// out of memory capturing the DSpark decode graphs at this fraction.
|
|
||||||
"--cuda-graph-max-bs-decode 64",
|
"--cuda-graph-max-bs-decode 64",
|
||||||
"--reasoning-parser auto",
|
"--reasoning-parser auto",
|
||||||
"--tool-call-parser auto",
|
"--tool-call-parser auto",
|
||||||
|
|||||||
Reference in New Issue
Block a user