[Docs] Update MegaMoE handling and rerun benchmarks (#27726)
This commit is contained in:
@@ -7,13 +7,12 @@
|
||||
// `config.playgroundFeatures` — a keyed map where each present key opts that
|
||||
// axis in. Recognised axes:
|
||||
// attention — TP/CP/DP-Attention knobs
|
||||
// moe — backend + EP
|
||||
// moe — backend (+ MegaMoE quantization sub-select) + EP
|
||||
// parsers — per-item toggle flags
|
||||
// speculative — single-select preset
|
||||
// pdDisagg — role + transfer backend + IB device + optional router
|
||||
// hicache — enable + backend + write policy
|
||||
// hisparse — enable + host ratio (decode-only)
|
||||
// megamoe — single-select, Blackwell-only
|
||||
//
|
||||
// Adding an axis = one entry in AXIS_HANDLERS below; nothing else switches on
|
||||
// an axis id. Each handler implements initState / revertHidden / apply /
|
||||
@@ -354,18 +353,25 @@ export const Playground = ({ config }) => {
|
||||
},
|
||||
|
||||
// ---- Axis: MoE Parallelism ----------------------------------------------
|
||||
// Backend single-select + EP numeric knob; either is optional.
|
||||
// Backend single-select + EP numeric knob; either is optional. Picking the
|
||||
// "megamoe" backend reveals a Quantization sub-select (W4A8 / W4A4) in the same
|
||||
// row — W4A4 adds the FP4-activations env vars.
|
||||
moe: {
|
||||
initState: () => ({ backend: null, ep: null }),
|
||||
initState: () => ({ backend: null, ep: null, mmQuant: null }),
|
||||
|
||||
// Prefer --moe-a2a-backend over --moe-runner-backend when both present.
|
||||
// mmQuant is derived from the base env (FP4 activations present → W4A4).
|
||||
deriveFromBase: (cell, fc, h) => {
|
||||
const flags = (cell && cell.flags) || [];
|
||||
const baseEnv = (cell && cell.env) || [];
|
||||
const a2a = h.findFlagArg(flags, "--moe-a2a-backend");
|
||||
const runner = h.findFlagArg(flags, "--moe-runner-backend");
|
||||
const fp4Acts = baseEnv.some(
|
||||
(e) => e.startsWith("SGLANG_OPT_DEEPGEMM_MEGA_MOE_USE_FP4_ACTS"));
|
||||
return {
|
||||
backend: a2a || runner || null,
|
||||
ep: h.parseIntFlag(flags, "--ep"),
|
||||
mmQuant: fp4Acts ? "w4a4" : "w4a8",
|
||||
};
|
||||
},
|
||||
|
||||
@@ -376,6 +382,15 @@ export const Playground = ({ config }) => {
|
||||
&& h.isHidden(fc.backend.options, next.backend, base)) {
|
||||
next.backend = null; changed = true;
|
||||
}
|
||||
// MegaMoE backend availability — gated by its option's requiresHw /
|
||||
// excludesStrategy (this model gates by hw only; the check is generic).
|
||||
const mmOpt = (fc.backend?.options || []).find((o) => o.id === "megamoe");
|
||||
const mmAvail = !!mmOpt
|
||||
&& (!mmOpt.requiresHw || mmOpt.requiresHw.includes(base.hw))
|
||||
&& (!mmOpt.excludesStrategy || !mmOpt.excludesStrategy.includes(base.strategy));
|
||||
if (next.backend === "megamoe" && !mmAvail) {
|
||||
next.backend = null; changed = true;
|
||||
}
|
||||
if (next.ep !== null && fc.ep?.values
|
||||
&& h.isHidden(fc.ep.values, next.ep, base)) {
|
||||
next.ep = null; changed = true;
|
||||
@@ -383,7 +398,7 @@ export const Playground = ({ config }) => {
|
||||
return changed ? next : value;
|
||||
},
|
||||
|
||||
apply: ({ flags, env, value, fc, h }) => {
|
||||
apply: ({ flags, env, value, fc, h, derived }) => {
|
||||
if (value.backend !== null) {
|
||||
flags = h.stripFlagsByFirstToken(flags, [
|
||||
"--moe-a2a-backend", "--moe-runner-backend",
|
||||
@@ -393,6 +408,28 @@ export const Playground = ({ config }) => {
|
||||
flags = h.insertAfter(flags, h.ANCHOR_NEAR_DPATTN, opt.flags);
|
||||
}
|
||||
}
|
||||
// MegaMoE owns the MoE path: when the effective backend is megamoe, strip the
|
||||
// DeepEP dispatch + any prior megamoe env, then re-add the selected quant's
|
||||
// env. When the backend is explicitly switched away from megamoe, only drop
|
||||
// the megamoe quant env (leave DeepEP dispatch intact).
|
||||
const mq = fc.megamoeQuant;
|
||||
if (mq) {
|
||||
const quantKeys = [];
|
||||
for (const o of (mq.options || [])) {
|
||||
for (const e of (o.env || [])) quantKeys.push(e.split("=")[0]);
|
||||
}
|
||||
const effBackend = value.backend !== null
|
||||
? value.backend : (derived && derived.backend);
|
||||
if (effBackend === "megamoe") {
|
||||
env = h.stripEnvByPrefix(env, [...(mq.stripEnv || []), ...quantKeys]);
|
||||
const quant = value.mmQuant != null
|
||||
? value.mmQuant : ((derived && derived.mmQuant) || "w4a8");
|
||||
const opt = (mq.options || []).find((o) => o.id === quant);
|
||||
if (opt?.env?.length) env = [...env, ...opt.env];
|
||||
} else if (value.backend !== null) {
|
||||
env = h.stripEnvByPrefix(env, quantKeys);
|
||||
}
|
||||
}
|
||||
if (value.ep !== null) {
|
||||
flags = h.stripFlagsByFirstToken(flags, ["--ep"]);
|
||||
if (value.ep > 1) {
|
||||
@@ -416,6 +453,12 @@ export const Playground = ({ config }) => {
|
||||
const d = derived ? derived[k] : null;
|
||||
return (d !== null && d !== undefined) ? [null] : [];
|
||||
};
|
||||
// Hide the MegaMoE backend option where its requiresHw / excludesStrategy exclude this base.
|
||||
const mmOpt = (fc.backend?.options || []).find((o) => o.id === "megamoe");
|
||||
const mmAvail = !!mmOpt
|
||||
&& (!mmOpt.requiresHw || mmOpt.requiresHw.includes(base.hw))
|
||||
&& (!mmOpt.excludesStrategy || !mmOpt.excludesStrategy.includes(base.strategy));
|
||||
const backendIsMega = slotDisplay("backend") === "megamoe";
|
||||
return (
|
||||
<div key={axisId} style={s.card}>
|
||||
<div style={s.compactRow}>
|
||||
@@ -425,7 +468,16 @@ export const Playground = ({ config }) => {
|
||||
<span style={s.fieldLabel}>Backend</span>
|
||||
{renderSelect(slotDisplay("backend"), fc.backend.options || [],
|
||||
(v) => setSlot("backend", v), base, undefined,
|
||||
{ hideValues: hideNull("backend") })}
|
||||
{ hideValues: [...hideNull("backend"), ...(mmAvail ? [] : ["megamoe"])] })}
|
||||
</span>
|
||||
)}
|
||||
{fc.megamoeQuant && backendIsMega && (
|
||||
<span style={s.field}>
|
||||
<span style={s.fieldLabel}>Quantization</span>
|
||||
{renderSelect(
|
||||
value.mmQuant != null ? value.mmQuant : ((derived && derived.mmQuant) || "w4a8"),
|
||||
fc.megamoeQuant.options || [],
|
||||
(v) => setSlot("mmQuant", v), base)}
|
||||
</span>
|
||||
)}
|
||||
{fc.ep && (
|
||||
@@ -863,60 +915,6 @@ export const Playground = ({ config }) => {
|
||||
},
|
||||
},
|
||||
|
||||
// ---- Axis: MegaMoE ------------------------------------------------------
|
||||
// Single-select with axis-level gating (requiresHw / excludesStrategy)
|
||||
// plus per-option hide constraints and env mutation (stripEnv + option.env).
|
||||
megamoe: {
|
||||
initState: () => "disabled",
|
||||
|
||||
revertHidden: (value, fc, base, h) => {
|
||||
const hwGate = !fc.requiresHw || fc.requiresHw.includes(base.hw);
|
||||
const stratGate = !fc.excludesStrategy || !fc.excludesStrategy.includes(base.strategy);
|
||||
if (!hwGate || !stratGate) {
|
||||
return value === "disabled" ? value : "disabled";
|
||||
}
|
||||
if (value !== "disabled" && h.isHidden(fc.options || [], value, base)) {
|
||||
return "disabled";
|
||||
}
|
||||
return value;
|
||||
},
|
||||
|
||||
apply: ({ flags, env, value, fc, h }) => {
|
||||
if (!value || value === "disabled") return { flags, env };
|
||||
const opt = (fc.options || []).find((o) => o.id === value);
|
||||
if (!opt) return { flags, env };
|
||||
flags = h.stripFlagsByFirstToken(flags, [
|
||||
"--moe-a2a-backend", "--moe-runner-backend",
|
||||
]);
|
||||
if (opt.flags?.length) {
|
||||
flags = h.insertAfter(flags, h.ANCHOR_NEAR_DPATTN, opt.flags);
|
||||
}
|
||||
const ownedEnvKeys = [...(fc.stripEnv || [])];
|
||||
for (const o of (fc.options || [])) {
|
||||
for (const e of (o.env || [])) ownedEnvKeys.push(e.split("=")[0]);
|
||||
}
|
||||
env = h.stripEnvByPrefix(env, ownedEnvKeys);
|
||||
if (opt.env?.length) env = [...env, ...opt.env];
|
||||
return { flags, env };
|
||||
},
|
||||
|
||||
render: ({ axisId, value, setValue, fc, base, s, renderSelect }) => {
|
||||
const hwGate = !fc.requiresHw || fc.requiresHw.includes(base.hw);
|
||||
const stratGate = !fc.excludesStrategy || !fc.excludesStrategy.includes(base.strategy);
|
||||
if (!hwGate || !stratGate) return null;
|
||||
return (
|
||||
<div key={axisId} style={s.card}>
|
||||
<div style={s.compactRow}>
|
||||
<span style={s.axisTitle}>MegaMoE</span>
|
||||
<span style={s.field}>
|
||||
{renderSelect(value, fc.options || [], setValue, base)}
|
||||
</span>
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
},
|
||||
},
|
||||
|
||||
};
|
||||
|
||||
// ==========================================================================
|
||||
@@ -1527,8 +1525,7 @@ export const Playground = ({ config }) => {
|
||||
const dpAttnOn = (effDpAttn === true)
|
||||
|| (typeof effDpAttn === "number" && effDpAttn > 0);
|
||||
const pdMode = (deltas.pdDisagg && deltas.pdDisagg.mode) || "off";
|
||||
const megamoeOn = !!(deltas.megamoe && deltas.megamoe !== "disabled");
|
||||
const constraintBase = { ...base, dpAttnOn, pdMode, megamoeOn };
|
||||
const constraintBase = { ...base, dpAttnOn, pdMode };
|
||||
|
||||
let baseCommand = "";
|
||||
let playgroundCommand = "";
|
||||
|
||||
@@ -10,9 +10,9 @@ export const benchmarks = [
|
||||
sglang_version: "0.5.12.post1",
|
||||
speed: [
|
||||
{ workload: { dataset: "random", isl: 8192, osl: 1024, max_concurrency: 1 },
|
||||
ttft_ms: 428, tpot_ms: 3.53, tokens_per_sec_per_gpu: 44 },
|
||||
ttft_ms: 87, tpot_ms: 3.68, tokens_per_sec_per_gpu: 65 },
|
||||
{ workload: { dataset: "random", isl: 8192, osl: 1024, max_concurrency: 16 },
|
||||
ttft_ms: 3111, tpot_ms: 23.82, tokens_per_sec_per_gpu: 121 },
|
||||
ttft_ms: 290, tpot_ms: 6.21, tokens_per_sec_per_gpu: 489 },
|
||||
],
|
||||
},
|
||||
{
|
||||
@@ -30,9 +30,9 @@ export const benchmarks = [
|
||||
sglang_version: "0.5.12.post1",
|
||||
speed: [
|
||||
{ workload: { dataset: "random", isl: 8192, osl: 1024, max_concurrency: 1024 },
|
||||
ttft_ms: 105918, tpot_ms: 70.73, tokens_per_sec_per_gpu: 881 },
|
||||
ttft_ms: 99949, tpot_ms: 67.46, tokens_per_sec_per_gpu: 939 },
|
||||
{ workload: { dataset: "random", isl: 8192, osl: 1024, max_concurrency: 4096 },
|
||||
ttft_ms: 273356, tpot_ms: 71.61, tokens_per_sec_per_gpu: 889 },
|
||||
ttft_ms: 253310, tpot_ms: 66.11, tokens_per_sec_per_gpu: 964 },
|
||||
],
|
||||
},
|
||||
{
|
||||
@@ -59,9 +59,9 @@ export const benchmarks = [
|
||||
sglang_version: "0.5.12.post1",
|
||||
speed: [
|
||||
{ workload: { dataset: "random", isl: 8192, osl: 1024, max_concurrency: 1 },
|
||||
ttft_ms: 205, tpot_ms: 3.43, tokens_per_sec_per_gpu: 54 },
|
||||
ttft_ms: 88, tpot_ms: 3.67, tokens_per_sec_per_gpu: 66 },
|
||||
{ workload: { dataset: "random", isl: 8192, osl: 1024, max_concurrency: 16 },
|
||||
ttft_ms: 1856, tpot_ms: 14.82, tokens_per_sec_per_gpu: 205 },
|
||||
ttft_ms: 266, tpot_ms: 6.06, tokens_per_sec_per_gpu: 495 },
|
||||
],
|
||||
},
|
||||
{
|
||||
@@ -79,9 +79,9 @@ export const benchmarks = [
|
||||
sglang_version: "0.5.12.post1",
|
||||
speed: [
|
||||
{ workload: { dataset: "random", isl: 8192, osl: 1024, max_concurrency: 1024 },
|
||||
ttft_ms: 82556, tpot_ms: 55.37, tokens_per_sec_per_gpu: 1130 },
|
||||
ttft_ms: 97028, tpot_ms: 65.09, tokens_per_sec_per_gpu: 966 },
|
||||
{ workload: { dataset: "random", isl: 8192, osl: 1024, max_concurrency: 4096 },
|
||||
ttft_ms: 207987, tpot_ms: 54.05, tokens_per_sec_per_gpu: 1171 },
|
||||
ttft_ms: 243335, tpot_ms: 63.98, tokens_per_sec_per_gpu: 998 },
|
||||
],
|
||||
},
|
||||
{
|
||||
@@ -89,9 +89,9 @@ export const benchmarks = [
|
||||
sglang_version: "0.5.12.post1",
|
||||
speed: [
|
||||
{ workload: { dataset: "random", isl: 8192, osl: 1024, max_concurrency: 1 },
|
||||
ttft_ms: 239, tpot_ms: 5.04, tokens_per_sec_per_gpu: 24 },
|
||||
ttft_ms: 261, tpot_ms: 5.01, tokens_per_sec_per_gpu: 23 },
|
||||
{ workload: { dataset: "random", isl: 8192, osl: 1024, max_concurrency: 16 },
|
||||
ttft_ms: 830, tpot_ms: 15.55, tokens_per_sec_per_gpu: 101 },
|
||||
ttft_ms: 364, tpot_ms: 11.37, tokens_per_sec_per_gpu: 137 },
|
||||
],
|
||||
},
|
||||
{
|
||||
@@ -106,26 +106,12 @@ export const benchmarks = [
|
||||
},
|
||||
{
|
||||
match: { hw: "b300", variant: "pro", quant: "fp4", strategy: "high-throughput", nodes: "single" },
|
||||
sglang_version: "0.5.12.post1",
|
||||
speed: [
|
||||
{ workload: { dataset: "random", isl: 8192, osl: 1024, max_concurrency: 1024 },
|
||||
ttft_ms: 99139, tpot_ms: 44.37, tokens_per_sec_per_gpu: 476 },
|
||||
{ workload: { dataset: "random", isl: 8192, osl: 1024, max_concurrency: 4096 },
|
||||
ttft_ms: 241544, tpot_ms: 43.51, tokens_per_sec_per_gpu: 492 },
|
||||
],
|
||||
},
|
||||
// ====================================================================
|
||||
// GB200 + FP4
|
||||
// ====================================================================
|
||||
{
|
||||
match: { hw: "gb200", variant: "flash", quant: "fp4", strategy: "low-latency", nodes: "single" },
|
||||
sglang_version: "0.5.12.post1",
|
||||
speed: [
|
||||
{ workload: { dataset: "random", isl: 8192, osl: 1024, max_concurrency: 1 },
|
||||
ttft_ms: 335, tpot_ms: 3.67, tokens_per_sec_per_gpu: 47 },
|
||||
{ workload: { dataset: "random", isl: 8192, osl: 1024, max_concurrency: 16 },
|
||||
ttft_ms: 2440, tpot_ms: 15.95, tokens_per_sec_per_gpu: 163 },
|
||||
],
|
||||
},
|
||||
{
|
||||
match: { hw: "gb200", variant: "flash", quant: "fp4", strategy: "balanced", nodes: "single" },
|
||||
@@ -139,23 +125,9 @@ export const benchmarks = [
|
||||
},
|
||||
{
|
||||
match: { hw: "gb200", variant: "flash", quant: "fp4", strategy: "high-throughput", nodes: "single" },
|
||||
sglang_version: "0.5.12.post1",
|
||||
speed: [
|
||||
{ workload: { dataset: "random", isl: 8192, osl: 1024, max_concurrency: 1024 },
|
||||
ttft_ms: 128397, tpot_ms: 84.95, tokens_per_sec_per_gpu: 757 },
|
||||
{ workload: { dataset: "random", isl: 8192, osl: 1024, max_concurrency: 4096 },
|
||||
ttft_ms: 330479, tpot_ms: 86.7, tokens_per_sec_per_gpu: 741 },
|
||||
],
|
||||
},
|
||||
{
|
||||
match: { hw: "gb200", variant: "pro", quant: "fp4", strategy: "low-latency", nodes: "multi-2" },
|
||||
sglang_version: "0.5.12.post1",
|
||||
speed: [
|
||||
{ workload: { dataset: "random", isl: 8192, osl: 1024, max_concurrency: 1 },
|
||||
ttft_ms: 343, tpot_ms: 6.47, tokens_per_sec_per_gpu: 18 },
|
||||
{ workload: { dataset: "random", isl: 8192, osl: 1024, max_concurrency: 16 },
|
||||
ttft_ms: 1345, tpot_ms: 23.85, tokens_per_sec_per_gpu: 65 },
|
||||
],
|
||||
},
|
||||
{
|
||||
match: { hw: "gb200", variant: "pro", quant: "fp4", strategy: "balanced", nodes: "multi-2" },
|
||||
@@ -168,13 +140,6 @@ export const benchmarks = [
|
||||
// ====================================================================
|
||||
{
|
||||
match: { hw: "gb300", variant: "flash", quant: "fp4", strategy: "low-latency", nodes: "single" },
|
||||
sglang_version: "0.5.12.post1",
|
||||
speed: [
|
||||
{ workload: { dataset: "random", isl: 8192, osl: 1024, max_concurrency: 1 },
|
||||
ttft_ms: 380, tpot_ms: 4.4, tokens_per_sec_per_gpu: 38 },
|
||||
{ workload: { dataset: "random", isl: 8192, osl: 1024, max_concurrency: 16 },
|
||||
ttft_ms: 2960, tpot_ms: 21.26, tokens_per_sec_per_gpu: 125 },
|
||||
],
|
||||
},
|
||||
{
|
||||
match: { hw: "gb300", variant: "flash", quant: "fp4", strategy: "balanced", nodes: "single" },
|
||||
@@ -191,20 +156,13 @@ export const benchmarks = [
|
||||
sglang_version: "0.5.12.post1",
|
||||
speed: [
|
||||
{ workload: { dataset: "random", isl: 8192, osl: 1024, max_concurrency: 1024 },
|
||||
ttft_ms: 146954, tpot_ms: 97.24, tokens_per_sec_per_gpu: 662 },
|
||||
ttft_ms: 154868, tpot_ms: 104.84, tokens_per_sec_per_gpu: 621 },
|
||||
{ workload: { dataset: "random", isl: 8192, osl: 1024, max_concurrency: 4096 },
|
||||
ttft_ms: 368557, tpot_ms: 99.33, tokens_per_sec_per_gpu: 651 },
|
||||
ttft_ms: 386489, tpot_ms: 103.37, tokens_per_sec_per_gpu: 627 },
|
||||
],
|
||||
},
|
||||
{
|
||||
match: { hw: "gb300", variant: "pro", quant: "fp4", strategy: "low-latency", nodes: "single" },
|
||||
sglang_version: "0.5.12.post1",
|
||||
speed: [
|
||||
{ workload: { dataset: "random", isl: 8192, osl: 1024, max_concurrency: 1 },
|
||||
ttft_ms: 363, tpot_ms: 6.53, tokens_per_sec_per_gpu: 36 },
|
||||
{ workload: { dataset: "random", isl: 8192, osl: 1024, max_concurrency: 16 },
|
||||
ttft_ms: 1275, tpot_ms: 20.75, tokens_per_sec_per_gpu: 152 },
|
||||
],
|
||||
},
|
||||
{
|
||||
match: { hw: "gb300", variant: "pro", quant: "fp4", strategy: "balanced", nodes: "single" },
|
||||
|
||||
@@ -68,7 +68,8 @@ export const config = {
|
||||
--model {{MODEL_NAME}} \\
|
||||
--dataset-name {{DATASET}} \\
|
||||
--random-input-len {{ISL}} --random-output-len {{OSL}} \\
|
||||
--num-prompts {{NUM_PROMPTS}} --max-concurrency {{MAX_CONCURRENCY}}`,
|
||||
--num-prompts {{NUM_PROMPTS}} --max-concurrency {{MAX_CONCURRENCY}} \\
|
||||
--warmup-requests 64`,
|
||||
accuracy: {
|
||||
gsm8k_pct:
|
||||
`# To install sgl-eval: pip install git+https://github.com/sgl-project/sgl-eval
|
||||
@@ -112,7 +113,7 @@ sgl-eval run aime25 \\
|
||||
--base-url http://{{CURL_HOST}}:{{CURL_PORT}}/v1`,
|
||||
},
|
||||
},
|
||||
numPromptsByConc: { 1: 8, 16: 32, 64: 128, 256: 512, 1024: 2048, 4096: 4096 },
|
||||
numPromptsByConc: { 1: 32, 16: 32, 64: 128, 256: 512, 1024: 2048, 4096: 4096 },
|
||||
},
|
||||
|
||||
// Per-variant accuracy applied to every cell; per-cell `accuracy` overrides.
|
||||
@@ -182,21 +183,29 @@ sgl-eval run aime25 \\
|
||||
options: [
|
||||
{ id: null, label: "Inherited" },
|
||||
{ id: "deepep", label: "DeepEP",
|
||||
flags: ["--moe-a2a-backend deepep"],
|
||||
disable: { megamoeOn: [true] },
|
||||
disableReason: "MegaMoE owns the MoE backend — turn MegaMoE off to pick one." },
|
||||
flags: ["--moe-a2a-backend deepep"] },
|
||||
// Blackwell-only; no strategy gate — the Playground allows MegaMoE on any
|
||||
// strategy for experimentation (docs recommend it on high-throughput).
|
||||
{ id: "megamoe", label: "MegaMoE",
|
||||
flags: ["--moe-a2a-backend megamoe"],
|
||||
disable: { megamoeOn: [true] },
|
||||
disableReason: "MegaMoE owns the MoE backend — turn MegaMoE off to pick one." },
|
||||
requiresHw: ["b200", "b300", "gb200", "gb300"] },
|
||||
{ id: "flashinfer_mxfp4", label: "FlashInfer (MXFP4)",
|
||||
flags: ["--moe-runner-backend flashinfer_mxfp4"],
|
||||
disable: { megamoeOn: [true] },
|
||||
disableReason: "MegaMoE owns the MoE backend — turn MegaMoE off to pick one." },
|
||||
flags: ["--moe-runner-backend flashinfer_mxfp4"] },
|
||||
{ id: "marlin", label: "Marlin (W4A16)",
|
||||
flags: ["--moe-runner-backend marlin"],
|
||||
disable: { megamoeOn: [true] },
|
||||
disableReason: "MegaMoE owns the MoE backend — turn MegaMoE off to pick one." },
|
||||
flags: ["--moe-runner-backend marlin"] },
|
||||
],
|
||||
},
|
||||
megamoeQuant: {
|
||||
stripEnv: ["SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK"],
|
||||
options: [
|
||||
{ id: "w4a8", label: "W4A8",
|
||||
env: ["SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK=8320"] },
|
||||
{ id: "w4a4", label: "W4A4",
|
||||
env: [
|
||||
"SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK=8320",
|
||||
"SGLANG_OPT_DEEPGEMM_MEGA_MOE_USE_FP4_ACTS=1",
|
||||
"SGLANG_OPT_DEEPGEMM_MEGA_MOE_USE_MXF4_KIND=1",
|
||||
] },
|
||||
],
|
||||
},
|
||||
ep: { label: "EP", values: [
|
||||
@@ -305,27 +314,6 @@ sgl-eval run aime25 \\
|
||||
],
|
||||
defaultHostRatio: 10,
|
||||
},
|
||||
|
||||
// ----- Card 8: "MegaMoE" -----
|
||||
// Blackwell-only, high-throughput only (see requiresHw / excludesStrategy).
|
||||
megamoe: {
|
||||
requiresHw: ["b200", "b300", "gb200", "gb300"],
|
||||
excludesStrategy: ["low-latency", "balanced"],
|
||||
stripEnv: ["SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK"],
|
||||
options: [
|
||||
{ id: "disabled", label: "Disabled" },
|
||||
{ id: "w4a8", label: "W4A8",
|
||||
flags: ["--moe-a2a-backend megamoe"],
|
||||
env: ["SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK=8320"] },
|
||||
{ id: "w4a4", label: "W4A4",
|
||||
flags: ["--moe-a2a-backend megamoe"],
|
||||
env: [
|
||||
"SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK=8320",
|
||||
"SGLANG_OPT_DEEPGEMM_MEGA_MOE_USE_FP4_ACTS=1",
|
||||
"SGLANG_OPT_DEEPGEMM_MEGA_MOE_USE_MXF4_KIND=1",
|
||||
] },
|
||||
],
|
||||
},
|
||||
},
|
||||
|
||||
cells: [
|
||||
@@ -377,8 +365,6 @@ sgl-eval run aime25 \\
|
||||
verified: true,
|
||||
env: [
|
||||
"SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK=8320",
|
||||
"SGLANG_OPT_DEEPGEMM_MEGA_MOE_USE_FP4_ACTS=1",
|
||||
"SGLANG_OPT_DEEPGEMM_MEGA_MOE_USE_MXF4_KIND=1",
|
||||
],
|
||||
flags: [
|
||||
"--trust-remote-code",
|
||||
@@ -442,8 +428,6 @@ sgl-eval run aime25 \\
|
||||
verified: true,
|
||||
env: [
|
||||
"SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK=8320",
|
||||
"SGLANG_OPT_DEEPGEMM_MEGA_MOE_USE_FP4_ACTS=1",
|
||||
"SGLANG_OPT_DEEPGEMM_MEGA_MOE_USE_MXF4_KIND=1",
|
||||
],
|
||||
flags: [
|
||||
"--trust-remote-code",
|
||||
@@ -508,8 +492,6 @@ sgl-eval run aime25 \\
|
||||
verified: true,
|
||||
env: [
|
||||
"SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK=8320",
|
||||
"SGLANG_OPT_DEEPGEMM_MEGA_MOE_USE_FP4_ACTS=1",
|
||||
"SGLANG_OPT_DEEPGEMM_MEGA_MOE_USE_MXF4_KIND=1",
|
||||
],
|
||||
flags: [
|
||||
"--trust-remote-code",
|
||||
@@ -573,8 +555,6 @@ sgl-eval run aime25 \\
|
||||
verified: true,
|
||||
env: [
|
||||
"SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK=8320",
|
||||
"SGLANG_OPT_DEEPGEMM_MEGA_MOE_USE_FP4_ACTS=1",
|
||||
"SGLANG_OPT_DEEPGEMM_MEGA_MOE_USE_MXF4_KIND=1",
|
||||
],
|
||||
flags: [
|
||||
"--trust-remote-code",
|
||||
@@ -642,8 +622,6 @@ sgl-eval run aime25 \\
|
||||
verified: true,
|
||||
env: [
|
||||
"SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK=8320",
|
||||
"SGLANG_OPT_DEEPGEMM_MEGA_MOE_USE_FP4_ACTS=1",
|
||||
"SGLANG_OPT_DEEPGEMM_MEGA_MOE_USE_MXF4_KIND=1",
|
||||
],
|
||||
flags: [
|
||||
"--trust-remote-code",
|
||||
@@ -778,8 +756,6 @@ sgl-eval run aime25 \\
|
||||
verified: true,
|
||||
env: [
|
||||
"SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK=8320",
|
||||
"SGLANG_OPT_DEEPGEMM_MEGA_MOE_USE_FP4_ACTS=1",
|
||||
"SGLANG_OPT_DEEPGEMM_MEGA_MOE_USE_MXF4_KIND=1",
|
||||
],
|
||||
flags: [
|
||||
"--trust-remote-code",
|
||||
@@ -841,8 +817,6 @@ sgl-eval run aime25 \\
|
||||
verified: true,
|
||||
env: [
|
||||
"SGLANG_OPT_DEEPGEMM_MEGA_MOE_NUM_MAX_TOKENS_PER_RANK=8320",
|
||||
"SGLANG_OPT_DEEPGEMM_MEGA_MOE_USE_FP4_ACTS=1",
|
||||
"SGLANG_OPT_DEEPGEMM_MEGA_MOE_USE_MXF4_KIND=1",
|
||||
],
|
||||
flags: [
|
||||
"--trust-remote-code",
|
||||
|
||||
Reference in New Issue
Block a user