[diffusion] chore: retire released warmup and decoder flags (#33850)
This commit is contained in:
@@ -207,7 +207,7 @@ Notes on launch behavior:
|
||||
- **CPU offload flags are optional.** `--dit-cpu-offload`, `--text-encoder-cpu-offload`, and `--image-encoder-cpu-offload` are available; defaults are auto-adjusted from GPU memory (GPUs under 30 GB get more aggressive offloading).
|
||||
- **Multi-GPU realtime.** Prefer explicit sequence parallelism (`--sp-degree` equal to the number of GPUs for a single session). Do not enable CFG parallel for the realtime profile: the default realtime request uses `guidance_scale=1.0`, while CFG parallel requires active cond/uncond branches.
|
||||
- **FSDP.** Use `--use-fsdp-inference` only when you specifically need weight sharding for memory. For the low-latency realtime profile, prefer keeping components resident and using SP first.
|
||||
- **Warmup.** Server warmup is **automatically skipped** for the realtime pipeline — a synthetic warmup request has no WebSocket session, so the server detects the registered realtime adapter and skips it. No `--warmup` flag is needed.
|
||||
- **Warmup.** Server warmup is **automatically skipped** for the realtime pipeline — a synthetic warmup request has no WebSocket session, so the server detects the registered realtime adapter and skips it. No explicit `--warmup-mode` setting is needed.
|
||||
|
||||
Once up, the realtime WebSocket endpoint lives at `ws://127.0.0.1:30000/v1/realtime_video/generate` (use the Python client in §7 to connect — plain `curl` does not speak the `ws://` upgrade).
|
||||
|
||||
|
||||
@@ -254,7 +254,7 @@ Then, apply the quantization config from yaml. Please also enable torch.compile
|
||||
sglang generate \
|
||||
--backend diffusers \
|
||||
--model-path Qwen/Qwen-Image \
|
||||
--warmup \
|
||||
--warmup-mode request \
|
||||
--cache-dit-config quantize.yaml \
|
||||
--enable-torch-compile \
|
||||
--dit-cpu-offload false \
|
||||
|
||||
@@ -273,10 +273,6 @@ Set `--disagg-p2p-hostname` to the actual IP on each machine. For multi-machine,
|
||||
<td><code>--decoder-sp</code></td>
|
||||
<td>Decoder sequence parallelism</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td><code>--decoder-tp</code></td>
|
||||
<td>Deprecated alias for <code>--decoder-sp</code></td>
|
||||
</tr>
|
||||
</tbody>
|
||||
</table>
|
||||
|
||||
|
||||
@@ -122,7 +122,7 @@ SGLANG_CACHE_DIT_FN=2 SGLANG_CACHE_DIT_BN=1 SGLANG_CACHE_DIT_WARMUP=4 SGLANG_CAC
|
||||
SGLANG_CACHE_DIT_MC=4 SGLANG_CACHE_DIT_TAYLORSEER=true SGLANG_CACHE_DIT_TS_ORDER=2 \
|
||||
SGLANG_CACHE_DIT_ENABLED=true sglang generate --model-path /path/to/zai-org/GLM-Image/ \
|
||||
--prompt "A curious raccoon" --height 1920 --width 1088 --num-inference-steps 50 --num-gpus 4 \
|
||||
--sp-degree 4 --srt-encoder-url "http://${HOST}:${PORT}" --warmup
|
||||
--sp-degree 4 --srt-encoder-url "http://${HOST}:${PORT}" --warmup-mode request
|
||||
```
|
||||
Result:
|
||||
```bash
|
||||
|
||||
+11
-11
@@ -292,7 +292,7 @@ sglang generate \
|
||||
--prompt="At night, while their owner sleeps in a bedroom, three cats march in loudly playing tiny brass instruments, then abruptly file out." \
|
||||
--seed=1101 --num-gpus=4 --tp-size=2 --ulysses-degree=2 \
|
||||
--performance-mode=speed --enable-torch-compile=false \
|
||||
--save-output --warmup \
|
||||
--save-output --warmup-mode request \
|
||||
--perf-dump-path="${BENCH_DIR}/minimax-h3-t2va-baseline.json"
|
||||
```
|
||||
|
||||
@@ -344,7 +344,7 @@ sglang generate \
|
||||
--width=768 --height=512 \
|
||||
--num-frames=121 \
|
||||
--seed=42 --num-gpus=2 --enable-cfg-parallel \
|
||||
--save-output --enable-torch-compile --warmup
|
||||
--save-output --enable-torch-compile --warmup-mode request
|
||||
```
|
||||
|
||||
`LTX2TwoStagePipeline` is a native path. The spatial upsampler and distilled
|
||||
@@ -361,7 +361,7 @@ sglang generate \
|
||||
--width=768 --height=512 \
|
||||
--num-frames=121 \
|
||||
--seed=42 --num-gpus=2 --cfg-parallel-size=2 \
|
||||
--save-output --enable-torch-compile --warmup
|
||||
--save-output --enable-torch-compile --warmup-mode request
|
||||
```
|
||||
|
||||
This matches the nightly comparison case `ltx2.3_twostage_ti2v_2gpus`.
|
||||
@@ -377,7 +377,7 @@ sglang generate \
|
||||
--num-frames=121 --fps=24 \
|
||||
--num-inference-steps=30 --guidance-scale=3.0 \
|
||||
--seed=1234 --num-gpus=2 \
|
||||
--save-output --enable-torch-compile --warmup
|
||||
--save-output --enable-torch-compile --warmup-mode request
|
||||
```
|
||||
|
||||
Use this when you want the native `LTX2Pipeline` baseline for `LTX-2.3` at the
|
||||
@@ -395,7 +395,7 @@ sglang generate \
|
||||
--num-frames=121 --fps=24 \
|
||||
--num-inference-steps=30 --guidance-scale=3.0 \
|
||||
--seed=1234 --num-gpus=2 \
|
||||
--save-output --enable-torch-compile --warmup
|
||||
--save-output --enable-torch-compile --warmup-mode request
|
||||
```
|
||||
|
||||
This matches the skill-only `ltx23-two-stage` preset. Use it as a
|
||||
@@ -413,7 +413,7 @@ sglang generate \
|
||||
--num-inference-steps=40 --guidance-scale=4.0 \
|
||||
--num-gpus=2 --enable-cfg-parallel --ulysses-degree=1 \
|
||||
--dit-layerwise-offload false --dit-cpu-offload false \
|
||||
--save-output --enable-torch-compile --warmup
|
||||
--save-output --enable-torch-compile --warmup-mode request
|
||||
```
|
||||
|
||||
### Manual command example: FireRed Image Edit
|
||||
@@ -428,7 +428,7 @@ sglang generate \
|
||||
--num-inference-steps=40 --guidance-scale=4.0 \
|
||||
--num-gpus=2 --enable-cfg-parallel --ulysses-degree=1 \
|
||||
--dit-layerwise-offload false --dit-cpu-offload false \
|
||||
--save-output --enable-torch-compile --warmup
|
||||
--save-output --enable-torch-compile --warmup-mode request
|
||||
```
|
||||
|
||||
Use `FireRedTeam/FireRed-Image-Edit-1.0` in the same command when comparing the
|
||||
@@ -453,7 +453,7 @@ sglang generate \
|
||||
--config="${CONFIG_DIR}/hunyuan3d-shape.json" \
|
||||
--num-inference-steps=50 --guidance-scale=5.0 \
|
||||
--dit-layerwise-offload false --dit-cpu-offload false \
|
||||
--save-output --enable-torch-compile --warmup
|
||||
--save-output --enable-torch-compile --warmup-mode request
|
||||
```
|
||||
|
||||
For Hunyuan3D, compare the denoise stage separately from mesh export and paint
|
||||
@@ -473,7 +473,7 @@ sglang generate \
|
||||
--seed=42 --save-output \
|
||||
--num-gpus=4 --enable-cfg-parallel --ulysses-degree=2 \
|
||||
--text-encoder-cpu-offload --pin-cpu-memory \
|
||||
--warmup --enable-torch-compile
|
||||
--warmup-mode request --enable-torch-compile
|
||||
```
|
||||
|
||||
`Wan2.2-I2V-A14B` uses the 720p max-area config by default, and explicit
|
||||
@@ -485,7 +485,7 @@ reference-image aspect ratio.
|
||||
For every benchmark run, write a perf dump JSON:
|
||||
|
||||
```bash
|
||||
sglang generate ... --warmup --perf-dump-path "${BENCH_DIR}/<result>.json"
|
||||
sglang generate ... --warmup-mode request --perf-dump-path "${BENCH_DIR}/<result>.json"
|
||||
```
|
||||
|
||||
Before/after comparison:
|
||||
@@ -542,7 +542,7 @@ sglang generate \
|
||||
--model-path=black-forest-labs/FLUX.1-dev \
|
||||
--prompt="A futuristic cyberpunk city at night" \
|
||||
--width=1024 --height=1024 --num-inference-steps=50 \
|
||||
--seed=42 --enable-torch-compile --warmup \
|
||||
--seed=42 --enable-torch-compile --warmup-mode request \
|
||||
--profile
|
||||
```
|
||||
|
||||
|
||||
+2
-2
@@ -665,7 +665,7 @@ def _expected_nightly_cli_args(case: dict) -> dict[str, str]:
|
||||
serve_args = shlex.split(case["frameworks"]["sglang"].get("serve_args", ""))
|
||||
parsed_serve_args = _parse_cli_args(serve_args)
|
||||
for flag, value in parsed_serve_args.items():
|
||||
if flag in {"enable-torch-compile", "warmup"}:
|
||||
if flag in {"enable-torch-compile", "warmup-mode"}:
|
||||
continue
|
||||
expected[flag] = _normalize_cli_value(value)
|
||||
|
||||
@@ -805,7 +805,7 @@ def build_sglang_cmd(
|
||||
if save_output:
|
||||
cmd.append("--save-output")
|
||||
if warmup:
|
||||
cmd.append("--warmup")
|
||||
cmd.extend(["--warmup-mode", "request"])
|
||||
if torch_compile and not cfg.get("force_eager", False):
|
||||
cmd.append("--enable-torch-compile")
|
||||
if perf_dump_path:
|
||||
|
||||
@@ -32,8 +32,8 @@ These options are intended to preserve output quality. In practice, some paths (
|
||||
|
||||
| Option | CLI Flag / Env Var | What It Does | Speedup | Limitations / Notes |
|
||||
|---|---|---|---|---|
|
||||
| **torch.compile** | `--enable-torch-compile` | Applies `torch.compile` to the DiT forward pass, fusing ops and reducing kernel launch overhead. | ~1.2–1.5x on denoising | First request is slow (compilation). May cause minor precision drifts due to [PyTorch issue #145213](https://github.com/pytorch/pytorch/issues/145213). Pair with `--warmup` for best results. |
|
||||
| **Warmup** | `--warmup` | Runs dummy forward passes to warm up CUDA caches, JIT, and `torch.compile`. Eliminates cold-start penalty. | Removes first-request latency spike | Adds startup time. Without `--warmup-resolutions`, warmup happens on first request. |
|
||||
| **torch.compile** | `--enable-torch-compile` | Applies `torch.compile` to the DiT forward pass, fusing ops and reducing kernel launch overhead. | ~1.2–1.5x on denoising | First request is slow (compilation). May cause minor precision drifts due to [PyTorch issue #145213](https://github.com/pytorch/pytorch/issues/145213). Pair with `--warmup-mode request` for best results. |
|
||||
| **Warmup** | `--warmup-mode request` | Runs dummy forward passes to warm up CUDA caches, JIT, and `torch.compile`. Eliminates cold-start penalty. | Removes first-request latency spike | Adds startup time. Without `--warmup-resolutions`, warmup happens on first request. |
|
||||
| **Warmup Resolutions** | `--warmup-resolutions 256x256 720x720` | Pre-compiles and warms up specific resolutions at server startup (instead of lazily on first request). | Faster first request per resolution | Each resolution adds to startup time. Serving mode only; useful when you know your target resolutions in advance. |
|
||||
| **Multi-GPU (SP)** | `--num-gpus N --ulysses-degree N` | Sequence parallelism across GPUs. Shards sequence tokens (not frames) to minimize padding. | Near-linear scaling with N GPUs | Requires NCCL; inter-GPU bandwidth matters. `ulysses_degree * ring_degree = sp_degree`. For Wan2.2 video, start by benchmarking pure Ulysses before assuming a mixed Ulysses/Ring layout is fastest. |
|
||||
| **CFG Parallel** | `--enable-cfg-parallel` | Runs conditional and unconditional CFG branches in parallel across GPUs. For CFG models on multi-GPU, benchmark this against pure Ulysses on your topology instead of assuming one always wins. | Often faster than pure SP for CFG models | Requires `num_gpus >= 2`. Halves the Ulysses group size (e.g. 8 GPU → two 4-GPU groups). Only for models that use CFG. Nightly coverage configs may intentionally use smaller Ulysses groups to keep ring behavior exercised; that does not automatically make them the lowest-latency choice. |
|
||||
@@ -127,7 +127,7 @@ Current H3 restrictions:
|
||||
```bash
|
||||
sglang generate --model-path Wan-AI/Wan2.2-T2V-A14B-Diffusers \
|
||||
--num-gpus 8 --enable-cfg-parallel --ulysses-degree 4 \
|
||||
--enable-torch-compile --warmup \
|
||||
--enable-torch-compile --warmup-mode request \
|
||||
--text-encoder-cpu-offload true \
|
||||
--prompt "..." --save-output
|
||||
```
|
||||
@@ -149,7 +149,7 @@ sglang generate --model-path Lightricks/LTX-2 \
|
||||
--width 768 --height 512 \
|
||||
--num-frames 121 \
|
||||
--seed 42 --num-gpus 2 --enable-cfg-parallel \
|
||||
--enable-torch-compile --warmup --save-output
|
||||
--enable-torch-compile --warmup-mode request --save-output
|
||||
```
|
||||
|
||||
Note: LTX-2 is a current-source benchmark preset rather than a nightly
|
||||
@@ -168,7 +168,7 @@ sglang generate --model-path Lightricks/LTX-2.3 \
|
||||
--width 768 --height 512 \
|
||||
--num-frames 121 \
|
||||
--seed 42 --num-gpus 2 --cfg-parallel-size 2 \
|
||||
--enable-torch-compile --warmup --save-output
|
||||
--enable-torch-compile --warmup-mode request --save-output
|
||||
```
|
||||
|
||||
Note: this matches the nightly comparison case `ltx2.3_twostage_ti2v_2gpus`. The nightly config omits explicit steps and guidance, so this command omits them too and uses runtime defaults. Download `${ASSET_DIR}/cat.png` with the benchmark/profile skill before running it.
|
||||
@@ -183,7 +183,7 @@ sglang generate --model-path Lightricks/LTX-2.3 \
|
||||
--num-frames 121 --fps 24 \
|
||||
--num-inference-steps 30 --guidance-scale 3.0 \
|
||||
--seed 1234 --num-gpus 2 \
|
||||
--enable-torch-compile --warmup --save-output
|
||||
--enable-torch-compile --warmup-mode request --save-output
|
||||
```
|
||||
|
||||
Note: use this as the native `LTX2Pipeline` baseline for `LTX-2.3`. It keeps the validated one-stage resolution and explicit `LTX-2.3` sampling defaults, and matches the `ltx23-one-stage` benchmark preset in `sglang-diffusion-benchmark-profile`.
|
||||
@@ -199,7 +199,7 @@ sglang generate --model-path Lightricks/LTX-2.3 \
|
||||
--num-frames 121 --fps 24 \
|
||||
--num-inference-steps 30 --guidance-scale 3.0 \
|
||||
--seed 1234 --num-gpus 2 \
|
||||
--enable-torch-compile --warmup --save-output
|
||||
--enable-torch-compile --warmup-mode request --save-output
|
||||
```
|
||||
|
||||
Note: this is a high-resolution stress target for the native `LTX-2.3` two-stage path. It matches the skill-only `ltx23-two-stage` benchmark preset, not a nightly comparison case.
|
||||
@@ -208,7 +208,7 @@ Note: this is a high-resolution stress target for the native `LTX-2.3` two-stage
|
||||
|
||||
```bash
|
||||
sglang generate --model-path <IMAGE_MODEL> \
|
||||
--enable-torch-compile --warmup \
|
||||
--enable-torch-compile --warmup-mode request \
|
||||
--dit-layerwise-offload false \
|
||||
--dit-cpu-offload false \
|
||||
--prompt "..." --save-output
|
||||
@@ -227,7 +227,7 @@ sglang generate --backend=sglang \
|
||||
--num-inference-steps 40 --guidance-scale 4.0 \
|
||||
--num-gpus 2 --enable-cfg-parallel --ulysses-degree 1 \
|
||||
--dit-layerwise-offload false --dit-cpu-offload false \
|
||||
--enable-torch-compile --warmup --save-output
|
||||
--enable-torch-compile --warmup-mode request --save-output
|
||||
```
|
||||
|
||||
```bash
|
||||
@@ -239,7 +239,7 @@ sglang generate --backend=sglang \
|
||||
--num-inference-steps 40 --guidance-scale 4.0 \
|
||||
--num-gpus 2 --enable-cfg-parallel --ulysses-degree 1 \
|
||||
--dit-layerwise-offload false --dit-cpu-offload false \
|
||||
--enable-torch-compile --warmup --save-output
|
||||
--enable-torch-compile --warmup-mode request --save-output
|
||||
```
|
||||
|
||||
Use `FireRedTeam/FireRed-Image-Edit-1.0` in the same command when comparing
|
||||
@@ -264,7 +264,7 @@ sglang generate --backend=sglang \
|
||||
--config "${CONFIG_DIR}/hunyuan3d-shape.json" \
|
||||
--num-inference-steps 50 --guidance-scale 5.0 \
|
||||
--dit-layerwise-offload false --dit-cpu-offload false \
|
||||
--enable-torch-compile --warmup --save-output
|
||||
--enable-torch-compile --warmup-mode request --save-output
|
||||
```
|
||||
|
||||
For Hunyuan3D, treat `Hunyuan3DShapeDenoisingStage` as the primary latency
|
||||
@@ -275,7 +275,7 @@ drive DiT optimization decisions.
|
||||
|
||||
```bash
|
||||
sglang generate --model-path <MODEL> \
|
||||
--enable-torch-compile --warmup \
|
||||
--enable-torch-compile --warmup-mode request \
|
||||
--dit-layerwise-offload --dit-offload-prefetch-size 0.1 \
|
||||
--text-encoder-cpu-offload true --vae-cpu-offload true \
|
||||
--prompt "..." --save-output
|
||||
@@ -287,7 +287,7 @@ sglang generate --model-path <MODEL> \
|
||||
SGLANG_CACHE_DIT_ENABLED=true sglang generate --model-path <MODEL> \
|
||||
--attention-backend sage_attn \
|
||||
--dit-layerwise-offload false \
|
||||
--enable-torch-compile --warmup \
|
||||
--enable-torch-compile --warmup-mode request \
|
||||
--prompt "..." --save-output
|
||||
```
|
||||
|
||||
@@ -314,22 +314,22 @@ Use these as first commands to benchmark, not as universal winners.
|
||||
| Model family | First performance shape | Starting flags | Notes |
|
||||
|---|---|---|---|
|
||||
| MiniMax-H3 | 1344x768 resolved canvas, 5 seconds / 124 frames at 24 fps, 50 joint video/audio steps | H200: `--num-gpus 4 --ulysses-degree 4 --performance-mode speed --enable-torch-compile false`; H100: TP2 + Ulysses2 | Root ID plus `--model-variant fl2va` for T2VA/FL2VA or `ref2va` for Ref2VA. Ulysses only; no Ring/CFG/SageAttention. Preserve tiled video-VAE decode. Profile joint denoise, video VAE, audio VAE/vocoder, encoder, and collectives separately. |
|
||||
| FLUX.1 / FLUX.2 image | 1024x1024, runtime-default steps/guidance, 1 GPU | `--enable-torch-compile --warmup --dit-layerwise-offload false` | `black-forest-labs/FLUX.*` repos are gated; for FP8/NVFP4 use validated `--transformer-path` or `--transformer-weights-path` flows from the quant skill. |
|
||||
| FLUX.2 Klein / Klein Base | 1024x1024, runtime-default steps/guidance, 1 GPU | `--enable-torch-compile --warmup --dit-layerwise-offload false` | Current registry has `black-forest-labs/FLUX.2-klein-4B`, `FLUX.2-klein-9B`, and base variants. Klein is step-distilled; Klein Base is not. |
|
||||
| Qwen-Image / Qwen-Image-Edit | 1024x1024, runtime-default steps/guidance, 1 GPU | `--enable-torch-compile --warmup`; optionally native `SGLANG_CACHE_DIT_ENABLED=true` | Cache-DiT is lossy. For edit tasks, keep reference image, seed, and output size fixed. |
|
||||
| Z-Image / Z-Image-Turbo | 1024x1024, runtime-default steps/guidance, 1 GPU | `--enable-torch-compile --warmup` | Keep base Z-Image separate from Turbo: base uses 50-step CFG defaults, Turbo uses 9-step zero-CFG defaults. Mainline has bf16-native Triton RMSNorm scale and tanh-residual fusions. |
|
||||
| FLUX.1 / FLUX.2 image | 1024x1024, runtime-default steps/guidance, 1 GPU | `--enable-torch-compile --warmup-mode request --dit-layerwise-offload false` | `black-forest-labs/FLUX.*` repos are gated; for FP8/NVFP4 use validated `--transformer-path` or `--transformer-weights-path` flows from the quant skill. |
|
||||
| FLUX.2 Klein / Klein Base | 1024x1024, runtime-default steps/guidance, 1 GPU | `--enable-torch-compile --warmup-mode request --dit-layerwise-offload false` | Current registry has `black-forest-labs/FLUX.2-klein-4B`, `FLUX.2-klein-9B`, and base variants. Klein is step-distilled; Klein Base is not. |
|
||||
| Qwen-Image / Qwen-Image-Edit | 1024x1024, runtime-default steps/guidance, 1 GPU | `--enable-torch-compile --warmup-mode request`; optionally native `SGLANG_CACHE_DIT_ENABLED=true` | Cache-DiT is lossy. For edit tasks, keep reference image, seed, and output size fixed. |
|
||||
| Z-Image / Z-Image-Turbo | 1024x1024, runtime-default steps/guidance, 1 GPU | `--enable-torch-compile --warmup-mode request` | Keep base Z-Image separate from Turbo: base uses 50-step CFG defaults, Turbo uses 9-step zero-CFG defaults. Mainline has bf16-native Triton RMSNorm scale and tanh-residual fusions. |
|
||||
| Wan2.2 A14B T2V/I2V | 1280x720, 81 frames | Nightly: `--num-gpus 4 --enable-cfg-parallel --ulysses-degree 2 --text-encoder-cpu-offload --pin-cpu-memory` | For lowest latency, also benchmark pure Ulysses on the same GPUs. |
|
||||
| Wan2.2 TI2V 5B | 1280x720, 81 frames, 1 GPU | `--enable-torch-compile --warmup` | Keep the input image and motion prompt fixed when comparing sparse attention or Cache-DiT. |
|
||||
| Wan2.1 / FastWan / TurboWan variants | 480p or 720p video, family defaults | `--enable-torch-compile --warmup`; add `--ulysses-degree` / CFG parallel only after measuring | Current registry includes Wan2.1, FastWan2.1, FastWan2.2 TI2V, TurboWan2.1, TurboWan2.2 I2V, and Wan2.1-Fun InP. Use the compatibility matrix and benchmark presets before choosing topology. |
|
||||
| Cosmos3 Nano / Super | T2I: 1024x1024 with `--num-frames 1`; T2V/I2V: 480p/720p video | `SGLANG_DISABLE_COSMOS3_GUARDRAILS=1` for benchmark isolation; `--enable-torch-compile --warmup` | One checkpoint serves T2I/T2V/I2V. Mode is request-driven: `num_frames == 1` means T2I, `--image-path` means I2V. |
|
||||
| Ideogram 4 FP8/NVFP4 | 1024x1024, native preset defaults | `--enable-torch-compile --warmup` | Do not set `--num-inference-steps` or `--guidance-scale` directly unless you also update the Ideogram preset; sampling params derive them from `preset`. |
|
||||
| ERNIE-Image / GLM-Image / SANA / SD3 | 1024-class image, family defaults | `--enable-torch-compile --warmup`; disable offload only after checking VRAM | Treat these as current native image families. Start with benchmark/profile presets for ERNIE, GLM, and SANA; use registry/config defaults for SD3 unless you add a new preset. |
|
||||
| LTX-2 / LTX-2.3 | 768x512 or HQ 1920x1088, 121 frames | `--pipeline-class-name LTX2TwoStagePipeline --enable-torch-compile --warmup`; HQ uses `LTX2TwoStageHQPipeline` | Use benchmark/profile presets for nightly alignment, one-stage, high-resolution stress, and HQ. Device mode choices are `original` and `resident`; `resident` is fastest but uses more VRAM. `snapshot` is a deprecated alias for `original`, so do not use it in new commands. |
|
||||
| HunyuanVideo | 848x480 or 720p class video | `--text-encoder-cpu-offload --pin-cpu-memory --enable-torch-compile --warmup` | Check VAE decode separately. GroupNorm+SiLU is default-eligible in mainline when wrapper guards pass; use `bench_group_norm_silu.py` when VAE residual blocks are hot. |
|
||||
| JoyAI-Image-Edit | 1024-class TI2I, 40 steps, guidance 4.0 | `--backend=sglang --num-gpus 2 --enable-cfg-parallel --ulysses-degree 1 --enable-torch-compile --warmup --dit-layerwise-offload false --dit-cpu-offload false` | Newly supported image-edit path. Keep the input image, prompt, seed, and output size fixed; 2-GPU CFG parallel is the validated H100 starting point. |
|
||||
| FireRed-Image-Edit 1.0 / 1.1 | 1024x1024 image edit, 40 steps, guidance 4.0 | `--backend=sglang --num-gpus 2 --enable-cfg-parallel --ulysses-degree 1 --enable-torch-compile --warmup --dit-layerwise-offload false --dit-cpu-offload false` | Uses the native `QwenImageEditPlusPipeline` path. 2-GPU CFG parallel is the validated H100 starting point; benchmark 1.0 and 1.1 separately because checkpoint differences can change denoise latency. |
|
||||
| Hunyuan3D-2 shape | Shape generation, 50 steps, guidance 5.0 | `--backend=sglang --enable-torch-compile --warmup --dit-layerwise-offload false --dit-cpu-offload false` | Focus on `Hunyuan3DShapeDenoisingStage`; keep mesh export/paint timings separate from denoise. |
|
||||
| MOVA / Helios / LingBot World | Use the benchmark/profile presets or server test cases first | `--enable-torch-compile --warmup`; pin offload and topology flags explicitly | These video/realtime families have model-specific stages and condition handling. Keep prompt/image/action inputs fixed and prefer perf dumps over wall time alone. |
|
||||
| Wan2.2 TI2V 5B | 1280x720, 81 frames, 1 GPU | `--enable-torch-compile --warmup-mode request` | Keep the input image and motion prompt fixed when comparing sparse attention or Cache-DiT. |
|
||||
| Wan2.1 / FastWan / TurboWan variants | 480p or 720p video, family defaults | `--enable-torch-compile --warmup-mode request`; add `--ulysses-degree` / CFG parallel only after measuring | Current registry includes Wan2.1, FastWan2.1, FastWan2.2 TI2V, TurboWan2.1, TurboWan2.2 I2V, and Wan2.1-Fun InP. Use the compatibility matrix and benchmark presets before choosing topology. |
|
||||
| Cosmos3 Nano / Super | T2I: 1024x1024 with `--num-frames 1`; T2V/I2V: 480p/720p video | `SGLANG_DISABLE_COSMOS3_GUARDRAILS=1` for benchmark isolation; `--enable-torch-compile --warmup-mode request` | One checkpoint serves T2I/T2V/I2V. Mode is request-driven: `num_frames == 1` means T2I, `--image-path` means I2V. |
|
||||
| Ideogram 4 FP8/NVFP4 | 1024x1024, native preset defaults | `--enable-torch-compile --warmup-mode request` | Do not set `--num-inference-steps` or `--guidance-scale` directly unless you also update the Ideogram preset; sampling params derive them from `preset`. |
|
||||
| ERNIE-Image / GLM-Image / SANA / SD3 | 1024-class image, family defaults | `--enable-torch-compile --warmup-mode request`; disable offload only after checking VRAM | Treat these as current native image families. Start with benchmark/profile presets for ERNIE, GLM, and SANA; use registry/config defaults for SD3 unless you add a new preset. |
|
||||
| LTX-2 / LTX-2.3 | 768x512 or HQ 1920x1088, 121 frames | `--pipeline-class-name LTX2TwoStagePipeline --enable-torch-compile --warmup-mode request`; HQ uses `LTX2TwoStageHQPipeline` | Use benchmark/profile presets for nightly alignment, one-stage, high-resolution stress, and HQ. Device mode choices are `original` and `resident`; `resident` is fastest but uses more VRAM. `snapshot` is a deprecated alias for `original`, so do not use it in new commands. |
|
||||
| HunyuanVideo | 848x480 or 720p class video | `--text-encoder-cpu-offload --pin-cpu-memory --enable-torch-compile --warmup-mode request` | Check VAE decode separately. GroupNorm+SiLU is default-eligible in mainline when wrapper guards pass; use `bench_group_norm_silu.py` when VAE residual blocks are hot. |
|
||||
| JoyAI-Image-Edit | 1024-class TI2I, 40 steps, guidance 4.0 | `--backend=sglang --num-gpus 2 --enable-cfg-parallel --ulysses-degree 1 --enable-torch-compile --warmup-mode request --dit-layerwise-offload false --dit-cpu-offload false` | Newly supported image-edit path. Keep the input image, prompt, seed, and output size fixed; 2-GPU CFG parallel is the validated H100 starting point. |
|
||||
| FireRed-Image-Edit 1.0 / 1.1 | 1024x1024 image edit, 40 steps, guidance 4.0 | `--backend=sglang --num-gpus 2 --enable-cfg-parallel --ulysses-degree 1 --enable-torch-compile --warmup-mode request --dit-layerwise-offload false --dit-cpu-offload false` | Uses the native `QwenImageEditPlusPipeline` path. 2-GPU CFG parallel is the validated H100 starting point; benchmark 1.0 and 1.1 separately because checkpoint differences can change denoise latency. |
|
||||
| Hunyuan3D-2 shape | Shape generation, 50 steps, guidance 5.0 | `--backend=sglang --enable-torch-compile --warmup-mode request --dit-layerwise-offload false --dit-cpu-offload false` | Focus on `Hunyuan3DShapeDenoisingStage`; keep mesh export/paint timings separate from denoise. |
|
||||
| MOVA / Helios / LingBot World | Use the benchmark/profile presets or server test cases first | `--enable-torch-compile --warmup-mode request`; pin offload and topology flags explicitly | These video/realtime families have model-specific stages and condition handling. Keep prompt/image/action inputs fixed and prefer perf dumps over wall time alone. |
|
||||
|
||||
## Historical PR Watchlist
|
||||
|
||||
@@ -343,7 +343,7 @@ about whether the work has merged:
|
||||
|
||||
## Tips
|
||||
|
||||
- **Benchmarking**: always use `--warmup` and look for the line ending with `(with warmup excluded)` for accurate timing.
|
||||
- **Benchmarking**: always use `--warmup-mode request` and look for the line ending with `(with warmup excluded)` for accurate timing.
|
||||
- **Perf dump**: use `--perf-dump-path result.json` to save structured metrics, then compare with `python python/sglang/multimodal_gen/benchmarks/compare_perf.py baseline.json result.json`.
|
||||
- **Offload tuning**: after the first request, the runtime logs peak GPU memory and which components could stay resident. Use this to decide which `--*-cpu-offload` flags to disable.
|
||||
- **Backend selection**: `--backend sglang` (default, auto-detected) enables native optimizations (fused kernels, SP, native Cache-DiT env knobs, etc.). `--backend diffusers` falls back to Diffusers pipelines and is the path that accepts `--cache-dit-config` plus diffusers attention backend names.
|
||||
|
||||
@@ -472,7 +472,7 @@ class DiffGenerator:
|
||||
def _log_summary(self, results: list[GenerationResult]) -> None:
|
||||
if not results:
|
||||
return
|
||||
if self.server_args.warmup:
|
||||
if self.server_args.warmup_mode != "off":
|
||||
total_duration_ms = results[0].metrics.get("total_duration_ms", 0)
|
||||
logger.info(
|
||||
f"Warmed-up request processed in {GREEN}%.2f{RESET} seconds (with warmup excluded)",
|
||||
|
||||
@@ -118,7 +118,7 @@ async def lifespan(app: FastAPI):
|
||||
# 2. Start the ZMQ Broker in the background to handle offline requests
|
||||
broker_task = asyncio.create_task(run_zeromq_broker(server_args))
|
||||
warmup_task = None
|
||||
if server_args.server_warmup:
|
||||
if server_args.warmup_mode == "server":
|
||||
warmup_task = asyncio.create_task(
|
||||
_run_server_warmup_after_http_ready(server_args, warmup_done)
|
||||
)
|
||||
|
||||
@@ -411,8 +411,7 @@ def launch_pool_disagg_server(
|
||||
"pool_work_endpoint": work_eps[inst_idx],
|
||||
"pool_result_endpoint": result_ep,
|
||||
"num_gpus": num_role_gpus,
|
||||
"warmup": role_type == RoleType.ENCODER,
|
||||
"server_warmup": False,
|
||||
"warmup_mode": "request" if role_type == RoleType.ENCODER else "off",
|
||||
"scheduler_port": find_port(port_cursor),
|
||||
"master_port": find_port(port_cursor + 100),
|
||||
# Per-role parallelism (None = auto-derive from num_gpus)
|
||||
@@ -692,8 +691,7 @@ def launch_disagg_role(server_args: ServerArgs):
|
||||
"disagg_mode": True,
|
||||
"pool_work_endpoint": work_endpoint,
|
||||
"pool_result_endpoint": result_endpoint,
|
||||
"warmup": role_type == RoleType.ENCODER,
|
||||
"server_warmup": False,
|
||||
"warmup_mode": "request" if role_type == RoleType.ENCODER else "off",
|
||||
"scheduler_port": internal_scheduler_port,
|
||||
# Per-role parallelism (None = auto-derive from num_gpus)
|
||||
"tp_size": role_par["tp_size"],
|
||||
|
||||
@@ -358,7 +358,7 @@ class DenoisingStage(PipelineStage, RolloutDenoisingMixin):
|
||||
if (
|
||||
not args.enable_torch_compile
|
||||
or not args.offload_during_compile
|
||||
or not args.warmup
|
||||
or args.warmup_mode == "off"
|
||||
or not self._owns_compile_warmup_lifecycle()
|
||||
or args.use_fsdp_inference
|
||||
or self._cache_dit_requested()
|
||||
|
||||
@@ -234,9 +234,3 @@ class DisaggServerArgsMixin:
|
||||
default=cls.decoder_sp,
|
||||
help="Sequence parallelism for decoder role.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--decoder-tp",
|
||||
type=int,
|
||||
default=cls.decoder_tp,
|
||||
help="Deprecated alias for --decoder-sp.",
|
||||
)
|
||||
|
||||
@@ -315,21 +315,15 @@ class ServerArgs(DisaggServerArgsMixin):
|
||||
# NVTX profiling
|
||||
enable_layerwise_nvtx_marker: bool = False
|
||||
|
||||
# warmup
|
||||
# `warmup_mode` is the canonical knob: one of WARMUP_MODES
|
||||
# Warmup is controlled by the canonical `warmup_mode` knob: one of WARMUP_MODES.
|
||||
# - "off": no warmup.
|
||||
# - "server": server-based warmup — a synthetic request right after the
|
||||
# server is ready, before real traffic
|
||||
# - "request": request-based warmup — warm on the first real request(s).
|
||||
# This is a BENCHMARK aid
|
||||
# existing consumers keep working) and as deprecated CLI aliases. None means
|
||||
# "derive the mode from the legacy booleans"; _adjust_warmup resolves it.
|
||||
# This is a BENCHMARK aid.
|
||||
# None is resolved by _adjust_warmup from the selected runtime features.
|
||||
warmup_mode: str | None = None
|
||||
|
||||
# deprecated: warmup and server_warmup
|
||||
warmup: bool = False
|
||||
server_warmup: bool = False
|
||||
|
||||
warmup_resolutions: list[str] = None
|
||||
warmup_steps: int = 1
|
||||
|
||||
@@ -411,7 +405,6 @@ class ServerArgs(DisaggServerArgsMixin):
|
||||
denoiser_ulysses: int | None = None
|
||||
denoiser_ring: int | None = None
|
||||
decoder_sp: int | None = None
|
||||
decoder_tp: int | None = None
|
||||
pool_work_endpoint: str | None = None
|
||||
pool_result_endpoint: str | None = None
|
||||
pool_control_endpoint: str | None = None
|
||||
@@ -457,7 +450,6 @@ class ServerArgs(DisaggServerArgsMixin):
|
||||
"""set defaults and normalize values."""
|
||||
auto_tuner = ServerArgsAutoTuner(self)
|
||||
auto_tuner.adjust_based_on_performance_mode()
|
||||
self._adjust_disagg_parallelism_aliases()
|
||||
if auto_tuner.could_override_server_args():
|
||||
self._adjust_offload()
|
||||
auto_tuner.maybe_adjust_auto_default_layerwise_offload()
|
||||
@@ -480,21 +472,6 @@ class ServerArgs(DisaggServerArgsMixin):
|
||||
auto_tuner.finalize_auto_flags()
|
||||
self.adjust_pipeline_config()
|
||||
|
||||
def _adjust_disagg_parallelism_aliases(self):
|
||||
if self.decoder_tp is None:
|
||||
return
|
||||
if self.decoder_sp is not None and self.decoder_sp != self.decoder_tp:
|
||||
raise ValueError(
|
||||
"decoder_tp is deprecated in favor of decoder_sp; "
|
||||
"please set only one of them or keep the same value."
|
||||
)
|
||||
if self.decoder_sp is None:
|
||||
logger.warning(
|
||||
"decoder_tp is deprecated and is treated as decoder_sp for "
|
||||
"decoder/VAE parallel decode. Please use decoder_sp instead."
|
||||
)
|
||||
self.decoder_sp = self.decoder_tp
|
||||
|
||||
def _validate_parameters(self):
|
||||
"""check consistency and raise errors for invalid configs"""
|
||||
self._validate_pipeline()
|
||||
@@ -889,64 +866,37 @@ class ServerArgs(DisaggServerArgsMixin):
|
||||
return None, None
|
||||
|
||||
def _adjust_warmup(self):
|
||||
# --warmup-mode > --warmup/--server-warmup
|
||||
mode_explicit = self.is_arg_explicitly_set("warmup_mode")
|
||||
legacy_explicit = self.is_arg_explicitly_set(
|
||||
"warmup"
|
||||
) or self.is_arg_explicitly_set("server_warmup")
|
||||
if self.warmup_mode is not None:
|
||||
if self.warmup_mode not in WARMUP_MODES:
|
||||
raise ValueError(
|
||||
f"Invalid --warmup-mode {self.warmup_mode!r}; "
|
||||
f"expected one of {WARMUP_MODES}."
|
||||
)
|
||||
if mode_explicit and legacy_explicit:
|
||||
logger.warning(
|
||||
"Both --warmup-mode and the deprecated --warmup/--server-warmup "
|
||||
"were set; --warmup-mode=%s takes precedence.",
|
||||
self.warmup_mode,
|
||||
)
|
||||
if mode_explicit or not legacy_explicit:
|
||||
self.warmup = self.warmup_mode != "off"
|
||||
self.server_warmup = self.warmup_mode == "server"
|
||||
elif self.warmup:
|
||||
self.server_warmup = self.server_warmup or self.warmup_mode == "server"
|
||||
if self.warmup_mode is not None and self.warmup_mode not in WARMUP_MODES:
|
||||
raise ValueError(
|
||||
f"Invalid --warmup-mode {self.warmup_mode!r}; "
|
||||
f"expected one of {WARMUP_MODES}."
|
||||
)
|
||||
|
||||
# Explicit resolutions imply warmup is on (request-based).
|
||||
if self.warmup_resolutions is not None:
|
||||
self.warmup = True
|
||||
|
||||
if (
|
||||
self.enable_torch_compile
|
||||
and self.warmup_mode is None
|
||||
and not mode_explicit
|
||||
and not legacy_explicit
|
||||
):
|
||||
self.warmup = True
|
||||
self.server_warmup = True
|
||||
if self.enable_torch_compile and self.warmup_mode is None:
|
||||
self.warmup_mode = "server"
|
||||
logger.info(
|
||||
"Automatically enabled server warmup for torch.compile so first "
|
||||
"real requests do not pay compile latency. Set --warmup-mode off "
|
||||
"to disable this behavior."
|
||||
)
|
||||
|
||||
# Explicit resolutions need a request path unless an existing server
|
||||
# default already supplies the synthetic startup request.
|
||||
if self.warmup_resolutions is not None and self.warmup_mode in (None, "off"):
|
||||
self.warmup_mode = "request"
|
||||
|
||||
# BCG captures every graph during a synthetic warmup forward at startup
|
||||
# so that serving never records a fresh graph. That requires
|
||||
# server-based warmup (a real warmup request issued at startup), not
|
||||
# request-based warmup which runs no forward until the first request.
|
||||
# so serving never records a fresh graph.
|
||||
if self.enable_breakable_cuda_graph and self.disagg_role == RoleType.MONOLITHIC:
|
||||
self.warmup = True
|
||||
self.server_warmup = True
|
||||
self.warmup_mode = "server"
|
||||
|
||||
if self.disagg_role != RoleType.MONOLITHIC:
|
||||
self.server_warmup = False
|
||||
# Disaggregated roles do not host the HTTP startup request. Preserve
|
||||
# warmup intent, but schedule it on the first request instead.
|
||||
if self.disagg_role != RoleType.MONOLITHIC and self.warmup_mode == "server":
|
||||
self.warmup_mode = "request"
|
||||
|
||||
if not self.warmup:
|
||||
self.server_warmup = False
|
||||
|
||||
self.warmup_mode = (
|
||||
"off" if not self.warmup else "server" if self.server_warmup else "request"
|
||||
)
|
||||
if self.warmup_mode is None:
|
||||
self.warmup_mode = "off"
|
||||
|
||||
@staticmethod
|
||||
def _require_port(port: int, name: str) -> None:
|
||||
@@ -1622,28 +1572,16 @@ class ServerArgs(DisaggServerArgsMixin):
|
||||
choices=list(WARMUP_MODES),
|
||||
default=ServerArgs.warmup_mode,
|
||||
help=(
|
||||
"Warmup mode (canonical knob). One of: "
|
||||
"`off` (no warmup); `request` (request-based: warm on real "
|
||||
"incoming requests); `server` (server-based: a synthetic warmup "
|
||||
"request right after the server is ready, before traffic). "
|
||||
"Takes precedence over the deprecated --warmup/--server-warmup. "
|
||||
"`sglang serve` defaults to `server`; other entrypoints default "
|
||||
"Warmup mode. One of: `off` (no warmup); `request` "
|
||||
"(request-based: warm on real incoming requests); `server` "
|
||||
"(server-based: a synthetic warmup request right after the server "
|
||||
"is ready, before traffic). `sglang serve` defaults to `server`; "
|
||||
"other entrypoints default "
|
||||
"to request-based when warmup is enabled. When enabled, look for "
|
||||
"the line ending with `(with warmup excluded)` for actual "
|
||||
"processing time."
|
||||
),
|
||||
)
|
||||
parser.add_argument(
|
||||
"--warmup",
|
||||
action=StoreBoolean,
|
||||
default=ServerArgs.warmup,
|
||||
help=(
|
||||
"[DEPRECATED: use --warmup-mode] Perform warmup before normal "
|
||||
"traffic. Maps to --warmup-mode request (or server, combined "
|
||||
"with --server-warmup). Recommended when benchmarking for fair "
|
||||
"comparison and best performance."
|
||||
),
|
||||
)
|
||||
parser.add_argument(
|
||||
"--warmup-resolutions",
|
||||
type=str,
|
||||
@@ -1657,16 +1595,6 @@ class ServerArgs(DisaggServerArgsMixin):
|
||||
default=ServerArgs.warmup_steps,
|
||||
help="The number of warmup steps to perform for each resolution.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--server-warmup",
|
||||
action=StoreBoolean,
|
||||
default=ServerArgs.server_warmup,
|
||||
help=(
|
||||
"[DEPRECATED: use --warmup-mode server] Send a synthetic warmup "
|
||||
"request after the server is ready (server-based warmup)."
|
||||
),
|
||||
)
|
||||
|
||||
# layerwise offload
|
||||
parser.add_argument(
|
||||
"--dit-cpu-offload",
|
||||
@@ -2238,6 +2166,7 @@ class ServerArgs(DisaggServerArgsMixin):
|
||||
@classmethod
|
||||
def from_dict(cls, kwargs: dict[str, Any]) -> "ServerArgs":
|
||||
"""Create a ServerArgs object from a dictionary."""
|
||||
cls._reject_retired_args(kwargs)
|
||||
attrs = [attr.name for attr in dataclasses.fields(cls) if attr.init]
|
||||
server_args_kwargs: dict[str, Any] = {}
|
||||
explicit_arg_names = kwargs.get("_explicit_arg_names")
|
||||
@@ -2264,6 +2193,20 @@ class ServerArgs(DisaggServerArgsMixin):
|
||||
|
||||
return cls(**server_args_kwargs)
|
||||
|
||||
@staticmethod
|
||||
def _reject_retired_args(kwargs: dict[str, Any]) -> None:
|
||||
retired_args = {
|
||||
"decoder_tp": "decoder_sp for decoder/VAE parallel decode",
|
||||
"warmup": "warmup_mode=request or warmup_mode=off",
|
||||
"server_warmup": "warmup_mode=server or warmup_mode=off",
|
||||
}
|
||||
removed = [name for name in retired_args if name in kwargs]
|
||||
if removed:
|
||||
replacements = "; ".join(
|
||||
f"{name} -> {retired_args[name]}" for name in removed
|
||||
)
|
||||
raise ValueError(f"Removed server argument(s): {replacements}")
|
||||
|
||||
@staticmethod
|
||||
def load_config_file(config_file: str) -> dict[str, Any]:
|
||||
"""Load a config file."""
|
||||
@@ -2285,6 +2228,7 @@ class ServerArgs(DisaggServerArgsMixin):
|
||||
|
||||
@classmethod
|
||||
def from_kwargs(cls, **kwargs: Any) -> "ServerArgs":
|
||||
cls._reject_retired_args(kwargs)
|
||||
explicit_arg_names = set(kwargs)
|
||||
|
||||
# Convert backend string to enum if necessary
|
||||
|
||||
@@ -69,7 +69,7 @@ def should_return_warmup_result(req_or_group: Any) -> bool:
|
||||
|
||||
|
||||
def should_run_server_warmup(server_args: ServerArgs) -> bool:
|
||||
return server_args.warmup and server_args.server_warmup
|
||||
return server_args.warmup_mode == "server"
|
||||
|
||||
|
||||
def is_realtime_serving(server_args: ServerArgs) -> bool:
|
||||
@@ -95,7 +95,7 @@ def should_run_synthetic_server_warmup(server_args: ServerArgs) -> bool:
|
||||
|
||||
def should_run_explicit_client_warmup(server_args: ServerArgs) -> bool:
|
||||
return (
|
||||
server_args.warmup
|
||||
server_args.warmup_mode != "off"
|
||||
and server_args.warmup_resolutions is not None
|
||||
and supports_synthetic_warmup(server_args)
|
||||
)
|
||||
@@ -298,10 +298,9 @@ class SchedulerWarmupMixin:
|
||||
) -> list[tuple[bytes, Any]]:
|
||||
if (
|
||||
self.req_based_warmup_scheduled
|
||||
or not self.server_args.warmup
|
||||
or self.server_args.warmup_mode != "request"
|
||||
or not recv_reqs
|
||||
or self.server_args.warmup_resolutions is not None
|
||||
or self.server_args.server_warmup
|
||||
):
|
||||
return recv_reqs
|
||||
|
||||
|
||||
@@ -68,8 +68,8 @@ def _build_server_extra_args(case: DiffusionTestCase) -> str:
|
||||
if server_args.lora_path:
|
||||
a += f" --lora-path {server_args.lora_path}"
|
||||
|
||||
# default warmup
|
||||
a += " --warmup"
|
||||
# request-based warmup keeps the first measured generation out of the baseline
|
||||
a += " --warmup-mode request"
|
||||
|
||||
for extra_arg in server_args.extras:
|
||||
a += f" {extra_arg}"
|
||||
|
||||
@@ -457,7 +457,7 @@ ONE_GPU_CASES: list[DiffusionTestCase] = [
|
||||
modality="video",
|
||||
num_gpus=1,
|
||||
extras=[
|
||||
"--pipeline-class-name LingBotWorldCausalDMDPipeline --warmup false"
|
||||
"--pipeline-class-name LingBotWorldCausalDMDPipeline --warmup-mode off"
|
||||
],
|
||||
text_encoder_cpu_offload=True,
|
||||
),
|
||||
|
||||
@@ -77,7 +77,7 @@ def pi05_generator():
|
||||
kwargs = {
|
||||
"model_path": _MODEL_PATH,
|
||||
"num_gpus": num_gpus,
|
||||
"warmup": False,
|
||||
"warmup_mode": "off",
|
||||
"trust_remote_code": False,
|
||||
}
|
||||
if num_gpus > 1:
|
||||
|
||||
@@ -64,12 +64,11 @@ def _make_bare_scheduler(enable_cfg_parallel: bool) -> Scheduler:
|
||||
scheduler = object.__new__(Scheduler)
|
||||
|
||||
server_args = MagicMock()
|
||||
server_args.warmup = True
|
||||
server_args.warmup_mode = "request"
|
||||
server_args.warmup_steps = 1
|
||||
server_args.warmup_resolutions = ["512x512"]
|
||||
server_args.enable_cfg_parallel = enable_cfg_parallel
|
||||
server_args.enable_torch_compile = False
|
||||
server_args.server_warmup = False
|
||||
server_args.is_arg_explicitly_set.return_value = False
|
||||
|
||||
task_type = MagicMock()
|
||||
@@ -235,7 +234,7 @@ class TestWarmupReqCfgParallel(unittest.TestCase):
|
||||
def test_req_based_warmup_remains_explicit_legacy_entry(self):
|
||||
scheduler = _make_bare_scheduler(enable_cfg_parallel=False)
|
||||
scheduler.server_args.warmup_resolutions = None
|
||||
scheduler.server_args.server_warmup = False
|
||||
scheduler.server_args.warmup_mode = "request"
|
||||
|
||||
req = _make_generation_req()
|
||||
recv_reqs = [(b"0", req)]
|
||||
@@ -253,7 +252,7 @@ class TestWarmupReqCfgParallel(unittest.TestCase):
|
||||
def test_req_based_warmup_skips_default_server_warmup_path(self):
|
||||
scheduler = _make_bare_scheduler(enable_cfg_parallel=False)
|
||||
scheduler.server_args.warmup_resolutions = None
|
||||
scheduler.server_args.server_warmup = True
|
||||
scheduler.server_args.warmup_mode = "server"
|
||||
|
||||
recv_reqs = [(b"0", _make_generation_req())]
|
||||
processed = scheduler.process_received_reqs_with_req_based_warmup(recv_reqs)
|
||||
@@ -265,7 +264,7 @@ class TestWarmupReqCfgParallel(unittest.TestCase):
|
||||
def test_diff_generator_runs_explicit_warmup_through_scheduler_client(self):
|
||||
generator = object.__new__(DiffGenerator)
|
||||
server_args = MagicMock()
|
||||
server_args.warmup = True
|
||||
server_args.warmup_mode = "request"
|
||||
server_args.warmup_resolutions = ["832x480"]
|
||||
server_args.warmup_steps = 1
|
||||
server_args.enable_cfg_parallel = False
|
||||
@@ -693,8 +692,7 @@ class TestWarmupReqCfgParallel(unittest.TestCase):
|
||||
|
||||
def test_action_pipeline_disables_synthetic_warmup(self):
|
||||
server_args = MagicMock()
|
||||
server_args.warmup = True
|
||||
server_args.server_warmup = True
|
||||
server_args.warmup_mode = "server"
|
||||
server_args.warmup_resolutions = ["512x512"]
|
||||
server_args.pipeline_config.task_type = ModelTaskType.VLA_ACTION
|
||||
|
||||
@@ -704,8 +702,7 @@ class TestWarmupReqCfgParallel(unittest.TestCase):
|
||||
|
||||
def test_mesh_pipeline_builds_image_conditioned_warmup(self):
|
||||
server_args = MagicMock()
|
||||
server_args.warmup = True
|
||||
server_args.server_warmup = True
|
||||
server_args.warmup_mode = "server"
|
||||
server_args.warmup_steps = 1
|
||||
server_args.warmup_resolutions = None
|
||||
server_args.enable_cfg_parallel = False
|
||||
|
||||
@@ -421,12 +421,10 @@ class TestServerArgsPathExpansion(unittest.TestCase):
|
||||
execute_serve_cmd(args, unknown_args)
|
||||
|
||||
server_args = dispatch_launch.call_args.args[0]
|
||||
self.assertTrue(server_args.warmup)
|
||||
self.assertTrue(server_args.server_warmup)
|
||||
self.assertFalse(server_args.is_arg_explicitly_set("warmup"))
|
||||
self.assertFalse(server_args.is_arg_explicitly_set("server_warmup"))
|
||||
self.assertEqual(server_args.warmup_mode, "server")
|
||||
self.assertFalse(server_args.is_arg_explicitly_set("warmup_mode"))
|
||||
|
||||
def test_serve_cli_preserves_explicit_warmup_false(self):
|
||||
def test_serve_cli_preserves_explicit_warmup_mode_off(self):
|
||||
from sglang.multimodal_gen.runtime.entrypoints.cli.serve import (
|
||||
add_multimodal_gen_serve_args,
|
||||
execute_serve_cmd,
|
||||
@@ -437,8 +435,8 @@ class TestServerArgsPathExpansion(unittest.TestCase):
|
||||
argv = [
|
||||
"--model-path",
|
||||
"/fake",
|
||||
"--warmup",
|
||||
"false",
|
||||
"--warmup-mode",
|
||||
"off",
|
||||
]
|
||||
|
||||
with (
|
||||
@@ -454,18 +452,17 @@ class TestServerArgsPathExpansion(unittest.TestCase):
|
||||
execute_serve_cmd(args, unknown_args)
|
||||
|
||||
server_args = dispatch_launch.call_args.args[0]
|
||||
self.assertFalse(server_args.warmup)
|
||||
self.assertFalse(server_args.server_warmup)
|
||||
self.assertTrue(server_args.is_arg_explicitly_set("warmup"))
|
||||
self.assertEqual(server_args.warmup_mode, "off")
|
||||
self.assertTrue(server_args.is_arg_explicitly_set("warmup_mode"))
|
||||
|
||||
def test_serve_cli_preserves_config_warmup_false(self):
|
||||
def test_serve_cli_preserves_config_warmup_mode_off(self):
|
||||
from sglang.multimodal_gen.runtime.entrypoints.cli.serve import (
|
||||
add_multimodal_gen_serve_args,
|
||||
execute_serve_cmd,
|
||||
)
|
||||
|
||||
with tempfile.NamedTemporaryFile("w", suffix=".json") as config_file:
|
||||
json.dump({"model_path": "/fake", "warmup": False}, config_file)
|
||||
json.dump({"model_path": "/fake", "warmup_mode": "off"}, config_file)
|
||||
config_file.flush()
|
||||
|
||||
parser = FlexibleArgumentParser()
|
||||
@@ -490,9 +487,18 @@ class TestServerArgsPathExpansion(unittest.TestCase):
|
||||
execute_serve_cmd(args, unknown_args)
|
||||
|
||||
server_args = dispatch_launch.call_args.args[0]
|
||||
self.assertFalse(server_args.warmup)
|
||||
self.assertFalse(server_args.server_warmup)
|
||||
self.assertTrue(server_args.is_arg_explicitly_set("warmup"))
|
||||
self.assertEqual(server_args.warmup_mode, "off")
|
||||
self.assertTrue(server_args.is_arg_explicitly_set("warmup_mode"))
|
||||
|
||||
def test_retired_warmup_config_is_rejected(self):
|
||||
with self.assertRaisesRegex(ValueError, "warmup.*warmup_mode"):
|
||||
_from_dict_without_model_resolution(
|
||||
{"model_path": "/fake", "warmup": False}
|
||||
)
|
||||
|
||||
def test_retired_warmup_kwargs_are_rejected(self):
|
||||
with self.assertRaisesRegex(ValueError, "warmup.*warmup_mode"):
|
||||
ServerArgs.from_kwargs(model_path="/fake", warmup=False)
|
||||
|
||||
def test_disagg_role_disables_server_warmup(self):
|
||||
with patch.object(
|
||||
@@ -501,156 +507,82 @@ class TestServerArgsPathExpansion(unittest.TestCase):
|
||||
server_args = ServerArgs.from_dict(
|
||||
{
|
||||
"model_path": "/fake",
|
||||
"warmup": True,
|
||||
"server_warmup": True,
|
||||
"warmup_mode": "server",
|
||||
"disagg_role": "server",
|
||||
}
|
||||
)
|
||||
|
||||
self.assertTrue(server_args.warmup)
|
||||
self.assertFalse(server_args.server_warmup)
|
||||
self.assertEqual(server_args.warmup_mode, "request")
|
||||
|
||||
|
||||
class TestWarmupModeNormalization(unittest.TestCase):
|
||||
"""`_adjust_warmup` resolves the canonical warmup_mode and its derived booleans."""
|
||||
"""`_adjust_warmup` resolves the canonical warmup mode."""
|
||||
|
||||
def _resolve(
|
||||
self,
|
||||
*,
|
||||
warmup_mode=None,
|
||||
warmup=False,
|
||||
server_warmup=False,
|
||||
warmup_resolutions=None,
|
||||
enable_torch_compile=False,
|
||||
enable_breakable_cuda_graph=False,
|
||||
disagg_role=None,
|
||||
explicit=(),
|
||||
):
|
||||
from sglang.multimodal_gen.runtime.disaggregation.roles import RoleType
|
||||
|
||||
sa = ServerArgs.__new__(ServerArgs)
|
||||
sa.warmup_mode = warmup_mode
|
||||
sa.warmup = warmup
|
||||
sa.server_warmup = server_warmup
|
||||
sa.warmup_resolutions = warmup_resolutions
|
||||
sa.enable_torch_compile = enable_torch_compile
|
||||
sa.enable_breakable_cuda_graph = enable_breakable_cuda_graph
|
||||
sa.disagg_role = RoleType.MONOLITHIC if disagg_role is None else disagg_role
|
||||
sa._explicit_arg_names = set(explicit)
|
||||
sa._adjust_warmup()
|
||||
return sa
|
||||
|
||||
def test_explicit_mode_off_disables_all(self):
|
||||
sa = self._resolve(warmup_mode="off", explicit=("warmup_mode",))
|
||||
sa = self._resolve(warmup_mode="off")
|
||||
self.assertEqual(sa.warmup_mode, "off")
|
||||
self.assertFalse(sa.warmup)
|
||||
self.assertFalse(sa.server_warmup)
|
||||
|
||||
def test_explicit_mode_request(self):
|
||||
sa = self._resolve(warmup_mode="request", explicit=("warmup_mode",))
|
||||
sa = self._resolve(warmup_mode="request")
|
||||
self.assertEqual(sa.warmup_mode, "request")
|
||||
self.assertTrue(sa.warmup)
|
||||
self.assertFalse(sa.server_warmup)
|
||||
|
||||
def test_explicit_mode_server(self):
|
||||
sa = self._resolve(warmup_mode="server", explicit=("warmup_mode",))
|
||||
self.assertEqual(sa.warmup_mode, "server")
|
||||
self.assertTrue(sa.warmup)
|
||||
self.assertTrue(sa.server_warmup)
|
||||
|
||||
def test_explicit_mode_overrides_explicit_legacy(self):
|
||||
sa = self._resolve(
|
||||
warmup_mode="request",
|
||||
warmup=True,
|
||||
server_warmup=True,
|
||||
explicit=("warmup_mode", "warmup", "server_warmup"),
|
||||
)
|
||||
self.assertEqual(sa.warmup_mode, "request")
|
||||
self.assertTrue(sa.warmup)
|
||||
self.assertFalse(sa.server_warmup)
|
||||
|
||||
def test_explicit_legacy_false_beats_defaulted_mode(self):
|
||||
# serve defaults warmup_mode="server" (not explicit); `--warmup false` wins.
|
||||
sa = self._resolve(
|
||||
warmup_mode="server",
|
||||
warmup=False,
|
||||
server_warmup=False,
|
||||
explicit=("warmup",),
|
||||
)
|
||||
self.assertEqual(sa.warmup_mode, "off")
|
||||
self.assertFalse(sa.warmup)
|
||||
self.assertFalse(sa.server_warmup)
|
||||
|
||||
def test_defaulted_mode_applies_without_legacy_flags(self):
|
||||
# bare `sglang serve`: warmup_mode="server" defaulted, no legacy override.
|
||||
sa = self._resolve(warmup_mode="server")
|
||||
self.assertEqual(sa.warmup_mode, "server")
|
||||
self.assertTrue(sa.warmup)
|
||||
self.assertTrue(sa.server_warmup)
|
||||
|
||||
def test_legacy_only_maps_to_request(self):
|
||||
sa = self._resolve(warmup_mode=None, warmup=True, explicit=("warmup",))
|
||||
self.assertEqual(sa.warmup_mode, "request")
|
||||
self.assertTrue(sa.warmup)
|
||||
self.assertFalse(sa.server_warmup)
|
||||
def test_defaulted_mode_applies_without_legacy_flags(self):
|
||||
# Bare `sglang serve` defaults to server-based warmup.
|
||||
sa = self._resolve(warmup_mode="server")
|
||||
self.assertEqual(sa.warmup_mode, "server")
|
||||
|
||||
def test_resolutions_force_warmup_on(self):
|
||||
sa = self._resolve(
|
||||
warmup_mode="off",
|
||||
warmup_resolutions=["512x512"],
|
||||
explicit=("warmup_mode",),
|
||||
)
|
||||
self.assertTrue(sa.warmup)
|
||||
self.assertFalse(sa.server_warmup)
|
||||
self.assertEqual(sa.warmup_mode, "request")
|
||||
|
||||
def test_torch_compile_defaults_to_server_warmup(self):
|
||||
sa = self._resolve(enable_torch_compile=True)
|
||||
|
||||
self.assertEqual(sa.warmup_mode, "server")
|
||||
self.assertTrue(sa.warmup)
|
||||
self.assertTrue(sa.server_warmup)
|
||||
|
||||
def test_legacy_warmup_on_uses_defaulted_server_mode(self):
|
||||
# `serve --warmup` (legacy ON, mode defaulted to "server" but not
|
||||
# explicit) must resolve to server-based warmup, not silently downgrade
|
||||
# to request mode.
|
||||
sa = self._resolve(warmup_mode="server", warmup=True, explicit=("warmup",))
|
||||
|
||||
self.assertEqual(sa.warmup_mode, "server")
|
||||
self.assertTrue(sa.warmup)
|
||||
self.assertTrue(sa.server_warmup)
|
||||
|
||||
def test_torch_compile_respects_explicit_warmup_off(self):
|
||||
sa = self._resolve(
|
||||
warmup_mode="off",
|
||||
enable_torch_compile=True,
|
||||
explicit=("warmup_mode",),
|
||||
)
|
||||
self.assertEqual(sa.warmup_mode, "off")
|
||||
self.assertFalse(sa.warmup)
|
||||
self.assertFalse(sa.server_warmup)
|
||||
|
||||
def test_torch_compile_uses_server_warmup_for_explicit_resolutions(self):
|
||||
sa = self._resolve(
|
||||
warmup_resolutions=["1024x1024"],
|
||||
enable_torch_compile=True,
|
||||
explicit=("warmup_resolutions",),
|
||||
)
|
||||
self.assertEqual(sa.warmup_mode, "server")
|
||||
self.assertTrue(sa.warmup)
|
||||
self.assertTrue(sa.server_warmup)
|
||||
|
||||
def test_legacy_warmup_with_resolutions_runs_server_warmup(self):
|
||||
# Dead-zone regression: `serve --warmup --warmup-resolutions X` must run
|
||||
# server-based (synthetic) warmup, not end up with no warmup at all
|
||||
# (request-based warmup bails out when warmup_resolutions is set).
|
||||
sa = self._resolve(
|
||||
warmup_mode="server",
|
||||
warmup=True,
|
||||
warmup_resolutions=["1024x1024"],
|
||||
explicit=("warmup",),
|
||||
)
|
||||
self.assertTrue(sa.warmup)
|
||||
self.assertTrue(sa.server_warmup)
|
||||
def test_breakable_cuda_graph_forces_server_warmup(self):
|
||||
sa = self._resolve(enable_breakable_cuda_graph=True)
|
||||
self.assertEqual(sa.warmup_mode, "server")
|
||||
|
||||
def test_disagg_role_disables_server_warmup(self):
|
||||
@@ -659,10 +591,7 @@ class TestWarmupModeNormalization(unittest.TestCase):
|
||||
sa = self._resolve(
|
||||
warmup_mode="server",
|
||||
disagg_role=RoleType.DENOISER,
|
||||
explicit=("warmup_mode",),
|
||||
)
|
||||
self.assertTrue(sa.warmup)
|
||||
self.assertFalse(sa.server_warmup)
|
||||
self.assertEqual(sa.warmup_mode, "request")
|
||||
|
||||
def test_torch_compile_server_warmup_disabled_for_disagg_role(self):
|
||||
@@ -670,12 +599,10 @@ class TestWarmupModeNormalization(unittest.TestCase):
|
||||
|
||||
sa = self._resolve(enable_torch_compile=True, disagg_role=RoleType.DENOISER)
|
||||
self.assertEqual(sa.warmup_mode, "request")
|
||||
self.assertTrue(sa.warmup)
|
||||
self.assertFalse(sa.server_warmup)
|
||||
|
||||
def test_invalid_mode_raises(self):
|
||||
with self.assertRaises(ValueError):
|
||||
self._resolve(warmup_mode="bogus", explicit=("warmup_mode",))
|
||||
self._resolve(warmup_mode="bogus")
|
||||
|
||||
|
||||
class TestWarmupImageIsModelValid(unittest.TestCase):
|
||||
@@ -1962,24 +1889,9 @@ class TestPerRoleParallelism(unittest.TestCase):
|
||||
self.assertIsNone(par["ulysses_degree"])
|
||||
self.assertIsNone(par["ring_degree"])
|
||||
|
||||
def test_decoder_tp_is_alias_of_decoder_sp(self):
|
||||
args = self._from_dict({"model_path": "/fake", "decoder_tp": 2})
|
||||
from sglang.multimodal_gen.runtime.disaggregation.roles import RoleType
|
||||
|
||||
self.assertEqual(args.decoder_sp, 2)
|
||||
par = args.get_role_parallelism(RoleType.DECODER)
|
||||
self.assertIsNone(par["tp_size"])
|
||||
self.assertEqual(par["sp_degree"], 2)
|
||||
|
||||
def test_conflicting_decoder_tp_and_decoder_sp_raise(self):
|
||||
with self.assertRaisesRegex(ValueError, "decoder_tp is deprecated"):
|
||||
self._from_dict(
|
||||
{
|
||||
"model_path": "/fake",
|
||||
"decoder_tp": 2,
|
||||
"decoder_sp": 4,
|
||||
}
|
||||
)
|
||||
def test_removed_decoder_tp_is_rejected(self):
|
||||
with self.assertRaisesRegex(ValueError, "decoder_tp.*decoder_sp"):
|
||||
self._from_dict({"model_path": "/fake", "decoder_tp": 2})
|
||||
|
||||
def test_monolithic_returns_all_none(self):
|
||||
args = self._from_dict({"model_path": "/fake", "encoder_tp": 2})
|
||||
@@ -2090,7 +2002,6 @@ class TestPerRoleParallelism(unittest.TestCase):
|
||||
self.assertEqual(args.denoiser_ring, 2)
|
||||
self.assertEqual(args.encoder_tp, 1)
|
||||
self.assertEqual(args.decoder_sp, 8)
|
||||
self.assertIsNone(args.decoder_tp)
|
||||
|
||||
|
||||
class TestPipelineResolutionCliOverride(unittest.TestCase):
|
||||
|
||||
Reference in New Issue
Block a user