docs: fix Qwen3.8-27B mamba ratio calculator for speculative decoding (#35064)
Co-authored-by: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Fable 5
parent
9be3044b9c
commit
07a28ec5cf
@@ -73,9 +73,17 @@ ratio = (S + D) x state_bytes / (L x kv_bytes_per_token)
|
||||
```
|
||||
|
||||
- `S` — state slots per running request: `extra_buffer=5` (default),
|
||||
`extra_buffer_lazy=4`, `no_buffer=3`, disabled radix cache `=1`.
|
||||
`extra_buffer_lazy=4`, `no_buffer=3`, disabled radix cache `=1`. For the two
|
||||
`extra_buffer` strategies, `SGLANG_OPT_MAMBA_SKIP_DECODE_LOCK=1` frees one
|
||||
slot, and `extra_buffer` frees one more with the overlap scheduler off; the
|
||||
calculator reads both knobs.
|
||||
- `D` — verify intermediate states under speculative decoding:
|
||||
`--speculative-num-draft-tokens` (4 at the recommended EAGLE 3/1/4), 0 otherwise.
|
||||
`--speculative-num-draft-tokens` for EAGLE/MTP (4 at the recommended 3/1/4);
|
||||
`--speculative-dspark-block-size + 1` for DSPARK, where the block size falls
|
||||
back to the draft checkpoint's `block_size` when the flag is omitted (7 for
|
||||
`RadixArk/Qwen3.8-27B-DSpark`, so `D = 8`); 0 with speculation off or with
|
||||
`--enable-linear-replayssm-spec`, which keeps the verify intermediates on a
|
||||
fixed ring instead of per-request slots.
|
||||
- `state_bytes` — one state slot, from the fixed geometry
|
||||
(48 GDN layers x 48 heads x 128 x 128 at `--mamba-ssm-dtype`, plus bf16 conv
|
||||
state): 153.9 MB at fp32, 78.4 MB at bf16.
|
||||
@@ -83,10 +91,12 @@ ratio = (S + D) x state_bytes / (L x kv_bytes_per_token)
|
||||
32.8 KB at fp8, 65.5 KB at bf16.
|
||||
- `L` — average total request length in tokens: input + output.
|
||||
|
||||
`--max-mamba-cache-size = target_concurrency x (S + D)` is the equivalent
|
||||
explicit pin and overrides the ratio; the calculator emits it alongside. After
|
||||
boot, verify with the `max_running_requests` line in the server log — it should
|
||||
not be capped below your target concurrency.
|
||||
`--max-mamba-cache-size = target_concurrency x S` is the equivalent explicit
|
||||
pin and overrides the ratio; the calculator emits it alongside. `D` is not a
|
||||
term here: the engine divides the state pool by `S` alone and sizes the
|
||||
speculative verify buffer separately, so folding `D` into the pin would
|
||||
over-provision the pool. After boot, verify with the `max_running_requests`
|
||||
line in the server log — it should not be capped below your target concurrency.
|
||||
|
||||
</Accordion>
|
||||
|
||||
|
||||
@@ -73,7 +73,7 @@ export const Qwen38MambaRatioCalculator = () => {
|
||||
//
|
||||
// r = (S + D) x state_bytes / (L x kv_bytes_per_token)
|
||||
//
|
||||
const derive = (flags) => {
|
||||
const derive = (flags, env) => {
|
||||
const flagArg = (name) => {
|
||||
for (const f of flags) {
|
||||
const parts = f.split(/\s+/);
|
||||
@@ -115,21 +115,42 @@ export const Qwen38MambaRatioCalculator = () => {
|
||||
? strategyFlag
|
||||
: "extra_buffer";
|
||||
|
||||
// S mirrors kv_cache_configurator._calculate_mamba_ratio (single GPU,
|
||||
// overlap scheduler on): extra_buffer=5, extra_buffer_lazy=4,
|
||||
// no_buffer=3, radix cache disabled=1.
|
||||
// S mirrors kv_cache_configurator._calculate_mamba_ratio: base 3, minus 1
|
||||
// when SGLANG_OPT_MAMBA_SKIP_DECODE_LOCK=1, plus the ping-pong track
|
||||
// buffer (2 under the overlap scheduler, 1 for extra_buffer_lazy or with
|
||||
// overlap off). no_buffer has no track buffer and no decode lock, so it
|
||||
// stays 3; with the radix cache disabled S is 1. At the stock defaults:
|
||||
// extra_buffer=5, extra_buffer_lazy=4, no_buffer=3, radix off=1.
|
||||
const skipLock = env.some((e) =>
|
||||
e.startsWith("SGLANG_OPT_MAMBA_SKIP_DECODE_LOCK=1"));
|
||||
const overlapOff =
|
||||
hasFlag("--disable-overlap-schedule") ||
|
||||
(Number(flagArg("--pp-size")) || 1) > 1;
|
||||
const slots = radixOff
|
||||
? 1
|
||||
: strategy === "no_buffer"
|
||||
? 3
|
||||
: strategy === "extra_buffer_lazy"
|
||||
? 4
|
||||
: 5;
|
||||
: 3 -
|
||||
(skipLock ? 1 : 0) +
|
||||
(overlapOff || strategy === "extra_buffer_lazy" ? 1 : 2);
|
||||
|
||||
// Verify intermediates under speculative decoding: the draft-token count
|
||||
// (4 at the recommended EAGLE/MTP 3/1/4), 0 when spec is off.
|
||||
// D: verify intermediate states under speculative decoding, 0 with spec
|
||||
// off.
|
||||
const specOn = hasFlag("--speculative-algorithm");
|
||||
const drafts = specOn ? Number(flagArg("--speculative-num-draft-tokens")) || 4 : 0;
|
||||
const algo = (flagArg("--speculative-algorithm") || "").toUpperCase();
|
||||
// ReplaySSM spec-verify keeps the verify intermediates on a fixed ring
|
||||
// rather than per-request state slots, so D is 0 even with spec on.
|
||||
const replaySpec = hasFlag("--enable-linear-replayssm-spec");
|
||||
// DSPARK takes no --speculative-num-draft-tokens: its verify window is
|
||||
// --speculative-dspark-block-size (gamma) + 1, and gamma is read from the
|
||||
// draft checkpoint when the flag is omitted — block_size 7 for
|
||||
// RadixArk/Qwen3.8-27B-DSpark, so D = 8.
|
||||
const dsparkBlock = Number(flagArg("--speculative-dspark-block-size")) || 7;
|
||||
const drafts = !specOn || replaySpec
|
||||
? 0
|
||||
: algo === "DSPARK"
|
||||
? dsparkBlock + 1
|
||||
: Number(flagArg("--speculative-num-draft-tokens")) || 4;
|
||||
|
||||
// Fixed Qwen3.8-27B geometry (TP1):
|
||||
// GDN: 48 layers, 48 value heads x 128 x 128 SSM state (--mamba-ssm-dtype),
|
||||
@@ -147,14 +168,23 @@ export const Qwen38MambaRatioCalculator = () => {
|
||||
|
||||
// Two evaluations: `eff` matches the Playground's composed command, `bs`
|
||||
// matches the Deploy command (cell + overlays only).
|
||||
const eff = derive(cfg.flags);
|
||||
const bs = derive(cfg.baseFlags.length ? cfg.baseFlags : cfg.flags);
|
||||
const eff = derive(cfg.flags, cfg.env);
|
||||
const bs = derive(
|
||||
cfg.baseFlags.length ? cfg.baseFlags : cfg.flags,
|
||||
// The env rides with the command it belongs to: a base command with an
|
||||
// empty env must not inherit the Playground's overlay env.
|
||||
cfg.baseFlags.length ? cfg.baseEnv : cfg.env,
|
||||
);
|
||||
const { ratio, tp, kvDtype, ssmDtype, radixOff, strategy, slots, specOn,
|
||||
drafts, stateBytesPerSlot, kvBytesPerToken } = eff;
|
||||
|
||||
const valid = Number.isFinite(ratio) && ratio > 0 && L > 0 && tp === 1;
|
||||
const baseValid = Number.isFinite(bs.ratio) && bs.ratio > 0 && L > 0 && bs.tp === 1;
|
||||
const pin = Math.ceil(C * (slots + drafts));
|
||||
// The engine divides the state pool by S alone (kv_cache_configurator.py:
|
||||
// mamba_cap = max_mamba_cache_size // _calculate_mamba_ratio()) and sizes
|
||||
// the speculative verify buffer separately from D, so the pin is C x S,
|
||||
// not C x (S + D).
|
||||
const pin = Math.ceil(C * slots);
|
||||
const pinValid = valid && Number.isFinite(pin) && pin > 0 && C > 0;
|
||||
|
||||
const formatRatio = (value) => (Math.round(value * 100) / 100).toString();
|
||||
@@ -350,8 +380,9 @@ export const Qwen38MambaRatioCalculator = () => {
|
||||
)}
|
||||
<div style={{ color: colors.muted, fontSize: "11px" }}>
|
||||
state/slot {(stateBytesPerSlot / 1e6).toFixed(1)} MB · KV/token{" "}
|
||||
{(kvBytesPerToken / 1e3).toFixed(1)} KB · {slots + drafts} state slots per
|
||||
request, so {targetConcurrency || "N"} concurrent requests need{" "}
|
||||
{(kvBytesPerToken / 1e3).toFixed(1)} KB · the ratio prices{" "}
|
||||
{slots + drafts} state slots per request; the pin counts S = {slots},
|
||||
so {targetConcurrency || "N"} concurrent requests pin{" "}
|
||||
{pinValid ? pin : "—"} slots.
|
||||
</div>
|
||||
</div>
|
||||
|
||||
Reference in New Issue
Block a user