[diffusion] doc: update ltx2 multi-gpu deployment guide (#24682)

This commit is contained in:
Mick
2026-05-08 18:38:05 +08:00
committed by GitHub
parent 7f8e7a9130
commit 17888fa92a
5 changed files with 76 additions and 18 deletions
+8
View File
@@ -0,0 +1,8 @@
<svg width="940" height="525" viewBox="0 0 940 525" fill="none" xmlns="http://www.w3.org/2000/svg" role="img" aria-label="LTX logo">
<rect width="940" height="525" rx="28" fill="transparent"/>
<g transform="translate(251 169) scale(5.86)" fill="#111111">
<path d="M0 30.0087V7.50057C0 7.09765 0.154254 6.69973 0.460162 6.43869C0.708822 6.22729 0.987356 6.12029 1.29316 6.12029H8.2339C8.63671 6.12029 9.03463 6.26205 9.31316 6.55308C9.53944 6.79174 9.65133 7.07777 9.65133 7.41345V23.198C9.65133 23.6108 9.98462 23.944 10.3974 23.944H21.4638C21.8666 23.944 22.267 24.0858 22.5431 24.3767C22.7668 24.6155 22.8812 24.9015 22.8812 25.2372V30.5856C22.8812 31.0457 22.6823 31.4982 22.3018 31.7569C22.078 31.9086 21.8244 31.9832 21.5383 31.9832L1.99199 31.9956C0.890348 31.9981 0 31.1078 0 30.0087Z"/>
<path d="M36.5888 31.9926C34.4062 31.9876 32.492 31.6543 30.8413 30.9878C29.1906 30.3214 27.9104 29.2346 26.9981 27.7227C26.0856 26.2132 25.6333 24.2137 25.6382 21.7269L25.6532 13.7194L21.7016 13.7119C21.3486 13.7119 21.0528 13.5876 20.8116 13.3365C20.5705 13.0853 20.4512 12.7819 20.4537 12.4164L20.4636 7.39299C20.4636 7.02744 20.5854 6.72154 20.8265 6.47288C21.0677 6.22422 21.3635 6.09983 21.7165 6.10233L25.6681 6.10983L25.6779 1.29066C25.6779 0.925114 25.7998 0.619208 26.041 0.370548C26.2821 0.121887 26.5779 0 26.9309 0L33.9065 0.0124913C34.2595 0.0124913 34.5554 0.136772 34.7965 0.387931C35.0376 0.639089 35.1569 0.944995 35.1545 1.30805L35.1445 6.12721L41.2078 6.13959C41.5608 6.13959 41.8566 6.26398 42.0977 6.51514C42.3389 6.76629 42.4582 7.0722 42.4557 7.43525L42.4458 12.4586C42.4458 12.8242 42.3239 13.1301 42.0829 13.3787C41.8417 13.6274 41.5434 13.7518 41.1928 13.7493L35.1296 13.7368L35.1171 20.8988C35.1171 21.8613 35.3061 22.6148 35.6914 23.1618C36.0767 23.7089 36.6833 23.985 37.5186 23.985L41.5608 23.9925C41.9138 23.9925 42.2096 24.1168 42.4507 24.368C42.6919 24.6192 42.8113 24.9251 42.8088 25.2881L42.7988 30.7093C42.7988 31.075 42.677 31.3808 42.4358 31.6294C42.1947 31.8782 41.8963 32.0025 41.5459 32L36.5913 31.9901L36.5888 31.9926Z"/>
<path d="M47.5486 31.9851C47.2282 31.9851 46.965 31.8682 46.7589 31.6369C46.5503 31.4056 46.4485 31.1395 46.4485 30.841C46.4485 30.7416 46.4634 30.6248 46.4957 30.4929C46.5279 30.3611 46.5926 30.2268 46.6869 30.0951L54.3506 18.9342C54.4648 18.7675 54.4673 18.5463 54.3556 18.3771L47.4543 8.01457C47.3896 7.91517 47.335 7.79827 47.2854 7.6664C47.2382 7.53463 47.2133 7.40036 47.2133 7.26859C47.2133 6.97017 47.3251 6.70403 47.5486 6.47275C47.7722 6.24147 48.0279 6.12458 48.316 6.12458H55.6444C56.0914 6.12458 56.4267 6.23158 56.6501 6.44787C56.8737 6.66426 57.0327 6.85328 57.1295 7.01993L60.3082 11.8169C60.5043 12.1128 60.939 12.1128 61.1352 11.8169L64.3139 7.01993C64.4405 6.85328 64.6094 6.66426 64.8156 6.44787C65.0216 6.23158 65.3494 6.12458 65.7964 6.12458H72.7896C73.0778 6.12458 73.331 6.24147 73.557 6.47275C73.7805 6.70403 73.8922 6.95268 73.8922 7.21883C73.8922 7.38547 73.8748 7.53463 73.8451 7.6664C73.8128 7.79827 73.7482 7.91517 73.6539 8.01457L66.6159 18.3747C66.4992 18.5463 66.5017 18.77 66.6209 18.9392L74.4212 30.0975C74.5181 30.2293 74.5801 30.3636 74.6124 30.4954C74.6448 30.6273 74.6596 30.744 74.6596 30.8435C74.6596 31.142 74.5479 31.4081 74.3244 31.6394C74.1008 31.8707 73.8451 31.9874 73.557 31.9874H65.8934C65.4786 31.9874 65.1756 31.888 64.9844 31.689C64.7932 31.4901 64.6317 31.3086 64.5051 31.142L60.9886 25.9544C60.7924 25.671 60.3753 25.6685 60.1766 25.9471L56.4118 31.1395C56.3149 31.3061 56.1634 31.4876 55.9573 31.6865C55.7488 31.8855 55.4383 31.9851 55.0236 31.9851H47.5486Z"/>
</g>
</svg>

After

Width:  |  Height:  |  Size: 3.5 KiB

@@ -1,5 +1,5 @@
---
title: LTX
title: LTX2 & LTX2.3
description: Run LTX-2 and LTX-2.3 video generation pipelines with SGLang Diffusion.
metatags:
description: "Deploy and use LTX-2 and LTX-2.3 video generation models with SGLang Diffusion, including one-stage, two-stage, HQ, TI2V, and LoRA examples."
@@ -21,7 +21,7 @@ Use `Lightricks/LTX-2` or `Lightricks/LTX-2.3` as `--model-path`. For two-stage
Install SGLang with diffusion dependencies:
```bash Command
```bash
uv pip install "sglang[diffusion]" --prerelease=allow
```
@@ -35,7 +35,7 @@ This section provides deployment configurations optimized for different LTX pipe
The LTX series supports one-stage and two-stage pipelines. LTX-2.3 also supports the HQ two-stage pipeline. The recommended launch configuration depends on whether the target GPU can keep both two-stage DiTs resident.
**Interactive Command Generator**: Use the configuration selector below to generate a deployment command. The default selection targets a single NVIDIA H200 with `resident` two-stage mode, which is the fastest startup path for the specified high-memory environment.
**Interactive Command Generator**: Use the configuration selector below to generate a deployment command. The default selection targets a single NVIDIA H200 with `resident` two-stage mode. For multi-GPU serving, start from the 2-GPU or 4-GPU presets and only change parallelism if you need more memory headroom.
<LTXDeployment />
@@ -74,6 +74,45 @@ Other deployment flags:
For native LTX-2.3 two-stage serving without a user LoRA, `resident` is the fastest high-VRAM path. When you pass `--lora-path`, SGLang still applies the user LoRA during the two-stage switch, so use `resident` on H200-class GPUs for enough VRAM, but do not expect the same premerged-stage2 benefit as the no-user-LoRA path.
</Note>
### 3.3 Fast multi-GPU presets
For latency-oriented LTX serving, prefer CFG parallel over sequence parallelism. CFG parallel splits guidance branches across GPUs, while SP/Ulysses is mainly a memory/long-sequence tool for LTX.
| Target | Recommended server flags | Notes |
| --- | --- | --- |
| 1 high-VRAM GPU | `--ltx2-two-stage-device-mode resident` | Fastest two-stage setup when both DiTs fit. |
| 1 standard GPU | `--ltx2-two-stage-device-mode snapshot` | Lower VRAM than `resident`; use this when H100-class memory is tight. |
| 2 GPUs | `--num-gpus 2 --enable-cfg-parallel --ltx2-two-stage-device-mode resident` | Fastest common 2-GPU setup. |
| 4 GPUs | `--num-gpus 4 --tp-size 2 --enable-cfg-parallel --ltx2-two-stage-device-mode resident` | Fastest common 4-GPU layout: TP2 inside each CFG branch. |
| Official comparison | `--ltx2-two-stage-device-mode original` | Use this only when matching the original stage-switch semantics matters. |
Use `--enable-cfg-parallel` for degree-2 CFG parallel. Use `--cfg-parallel-size` only when you explicitly need a different CFG branch count. If `resident` exceeds available VRAM, keep the same parallelism preset and switch only the device mode to `snapshot`.
On high-VRAM GPUs, add `--text-encoder-cpu-offload false` if text encoding latency matters and you have enough memory.
#### 3.3.1 Two GPUs
```bash
sglang serve \
--model-path Lightricks/LTX-2.3 \
--pipeline-class-name LTX2TwoStagePipeline \
--num-gpus 2 \
--enable-cfg-parallel \
--ltx2-two-stage-device-mode resident
```
#### 3.3.2 Four GPUs
```bash
sglang serve \
--model-path Lightricks/LTX-2.3 \
--pipeline-class-name LTX2TwoStagePipeline \
--num-gpus 4 \
--tp-size 2 \
--enable-cfg-parallel \
--ltx2-two-stage-device-mode resident
```
## 4. Model Invocation
### 4.1 Basic Usage
@@ -88,7 +127,7 @@ The examples below spell out the current SGLang sampling defaults for reproducib
#### 4.1.1 LTX-2 one-stage text-to-video
```bash Command
```bash
sglang generate \
--model-path Lightricks/LTX-2 \
--pipeline-class-name LTX2Pipeline \
@@ -98,7 +137,7 @@ sglang generate \
#### 4.1.2 LTX-2.3 one-stage text-to-video
```bash Command
```bash
sglang generate \
--model-path Lightricks/LTX-2.3 \
--pipeline-class-name LTX2Pipeline \
@@ -108,7 +147,7 @@ sglang generate \
#### 4.1.3 LTX-2 two-stage text-to-video
```bash Command
```bash
sglang generate \
--model-path Lightricks/LTX-2 \
--pipeline-class-name LTX2TwoStagePipeline \
@@ -118,7 +157,7 @@ sglang generate \
#### 4.1.4 LTX-2.3 two-stage text-to-video
```bash Command
```bash
sglang generate \
--model-path Lightricks/LTX-2.3 \
--pipeline-class-name LTX2TwoStagePipeline \
@@ -128,7 +167,7 @@ sglang generate \
#### 4.1.5 LTX-2.3 HQ text-to-video
```bash Command
```bash
sglang generate \
--model-path Lightricks/LTX-2.3 \
--pipeline-class-name LTX2TwoStageHQPipeline \
@@ -140,7 +179,7 @@ sglang generate \
Pass one image to `--image-path` for image-conditioned generation:
```bash Command
```bash
sglang generate \
--model-path Lightricks/LTX-2.3 \
--pipeline-class-name LTX2TwoStagePipeline \
@@ -153,7 +192,7 @@ sglang generate \
Pass two images to `--image-path` for transition-style TI2V. The first image is used as the starting condition and the second image is used as the ending condition.
```bash Command
```bash
sglang generate \
--model-path Lightricks/LTX-2.3 \
--pipeline-class-name LTX2TwoStagePipeline \
@@ -170,7 +209,7 @@ Use `--lora-path` to load a LoRA adapter. If the Hugging Face repo contains mult
The following example uses [`valiantcat/LTX-2.3-Transition-LORA`](https://huggingface.co/valiantcat/LTX-2.3-Transition-LORA):
```bash Command
```bash
sglang generate \
--model-path Lightricks/LTX-2.3 \
--pipeline-class-name LTX2TwoStagePipeline \
@@ -182,7 +221,7 @@ sglang generate \
You can combine the Transition LoRA with two reference images:
```bash Command
```bash
sglang generate \
--model-path Lightricks/LTX-2.3 \
--pipeline-class-name LTX2TwoStagePipeline \
+2 -2
View File
@@ -22,8 +22,8 @@ metatags:
<Card
title="LTX"
mode="card"
href="/cookbook/diffusion/LTX/LTX"
img="/cards/Diffusion-card.png"
href="/cookbook/diffusion/LTX/LTX2 & LTX2.3"
img="/cards/logos/ltx.svg"
/>
<Card
title="Qwen-Image"
+1 -1
View File
@@ -1107,7 +1107,7 @@
{
"group": "LTX",
"pages": [
"cookbook/diffusion/LTX/LTX"
"cookbook/diffusion/LTX/LTX2 & LTX2.3"
]
},
{
@@ -2,9 +2,11 @@ export const LTXDeployment = () => {
const options = {
hardware: {
name: 'hardware',
title: 'Hardware Platform',
title: 'Deployment Target',
items: [
{ id: 'h200', label: 'H200', subtitle: 'Fastest, resident', default: true },
{ id: 'h200', label: '1x H200', subtitle: 'resident', default: true },
{ id: 'h200-2gpu', label: '2 GPUs', subtitle: 'CFG parallel', default: false },
{ id: 'h200-4gpu', label: '4 GPUs', subtitle: 'TP2 + CFG', default: false },
{ id: 'standard', label: 'Standard CUDA', subtitle: 'Snapshot mode', default: false },
{ id: 'official', label: 'Official Match', subtitle: 'Original switching', default: false },
],
@@ -113,7 +115,7 @@ export const LTXDeployment = () => {
};
const getDeviceMode = () => {
if (values.hardware === 'h200') {
if (values.hardware.startsWith('h200')) {
return 'resident';
}
if (values.hardware === 'official') {
@@ -122,6 +124,14 @@ export const LTXDeployment = () => {
return 'snapshot';
};
const getParallelFlags = () => {
const parallelFlagsMap = {
'h200-2gpu': ` \\\n --num-gpus 2 \\\n --enable-cfg-parallel`,
'h200-4gpu': ` \\\n --num-gpus 4 \\\n --tp-size 2 \\\n --enable-cfg-parallel`,
};
return parallelFlagsMap[values.hardware] || '';
};
const generateCommand = () => {
const config = modelConfigs[values.model];
const pipelineClass = config.pipelines[values.pipeline];
@@ -130,6 +140,7 @@ export const LTXDeployment = () => {
}
let command = `sglang serve \\\n --model-path ${config.repoId} \\\n --pipeline-class-name ${pipelineClass}`;
command += getParallelFlags();
if (values.pipeline !== 'one-stage') {
command += ` \\\n --ltx2-two-stage-device-mode ${getDeviceMode()}`;
}