[Docs] Add RTX 5090 DeepSeek-V4 recipe (#33157)

This commit is contained in:
Baizhou Zhang
2026-07-31 19:04:23 -07:00
committed by GitHub
parent bae8eb8d6c
commit e1964da451
@@ -8,16 +8,18 @@ export const config = {
supportedHardware: [
"h100", "h200", "b200", "b300", "gb200", "gb300",
"rtx6000",
"rtx6000", "rtx5090",
// AMD ROCm — MI300X (Flash FP8) + MI355X (Flash/Pro, FP4/FP8).
"mi300x", "mi355x",
],
// Model-specific GPUs the shared HARDWARE_CATALOG doesn't carry — the engine
// merges these in, so a model-specific GPU is config data, not an engine edit.
// RTX PRO 6000 (SM120 / Blackwell Desktop) is a workstation card, not datacenter.
// RTX PRO 6000 and RTX 5090 (SM120 / Blackwell Desktop) are workstation and
// consumer cards, not datacenter GPUs.
hardware: [
{ id: "rtx6000", label: "RTX PRO 6000", vram: "96GB", vendor: "blackwell" },
{ id: "rtx5090", label: "RTX 5090", vram: "32GB", vendor: "blackwell" },
],
variants: [
@@ -1784,6 +1786,25 @@ sgl-eval run aime25 \\
],
},
// ====================================================================
// RTX 5090 (SM120 / Blackwell Desktop) — Flash Official + low-latency
// ====================================================================
{
match: { hw: "rtx5090", variant: "flash-official", quant: "fp4", strategy: "low-latency", nodes: "single" },
verified: false,
env: [],
flags: [
"--trust-remote-code",
"--model-path {{MODEL_NAME}}",
"--tp 8",
"--moe-runner-backend flashinfer_mxfp4",
"--mem-fraction-static 0.90",
"--cuda-graph-max-bs-decode 32",
"--host {{HOST_IP}}",
"--port {{PORT}}",
],
},
// ====================================================================
// AMD ROCm (MI300X / MI355X)
// --------------------------------------------------------------------