[diffusion] docs: surface diffusion AR and PE guides (#32932)
This commit is contained in:
@@ -1326,6 +1326,8 @@
|
|||||||
"pages": [
|
"pages": [
|
||||||
"docs/sglang-diffusion/api/cli",
|
"docs/sglang-diffusion/api/cli",
|
||||||
"docs/sglang-diffusion/api/openai_api",
|
"docs/sglang-diffusion/api/openai_api",
|
||||||
|
"docs/sglang-diffusion/models_with_ar",
|
||||||
|
"docs/sglang-diffusion/models_with_pe",
|
||||||
"docs/sglang-diffusion/api/post_processing"
|
"docs/sglang-diffusion/api/post_processing"
|
||||||
]
|
]
|
||||||
},
|
},
|
||||||
|
|||||||
@@ -92,10 +92,10 @@ Use `sglang generate --help` and `sglang serve --help` for the full argument lis
|
|||||||
- `--attention-backend {BACKEND}`: attention backend for native SGLang and diffusers pipelines
|
- `--attention-backend {BACKEND}`: attention backend for native SGLang and diffusers pipelines
|
||||||
- `--component-attention-backends {MAP}`: per-component attention backend overrides, for example `text_encoder=torch_sdpa,transformer=fa`
|
- `--component-attention-backends {MAP}`: per-component attention backend overrides, for example `text_encoder=torch_sdpa,transformer=fa`
|
||||||
- `--attention-backend-config {CONFIG}`: attention backend configuration
|
- `--attention-backend-config {CONFIG}`: attention backend configuration
|
||||||
- `--srt-encoder-url {HTTPADDRESS}`: address of SGLang srt server with AR model for GLM-Image like models
|
- `--srt-encoder-url {HTTPADDRESS}`: address of SGLang srt server with AR model for GLM-Image like models. See [Models with AR Stage](../models_with_ar).
|
||||||
- `--srt-encoder-timeout {SECONDS}`: Timeout in seconds for HTTP requests to the SGLang encoder server
|
- `--srt-encoder-timeout {SECONDS}`: Timeout in seconds for HTTP requests to the SGLang encoder server
|
||||||
- `--srt-encoder-connection-timeout {SECONDS}`: TCP connection timeout in seconds for SGLang encoder server
|
- `--srt-encoder-connection-timeout {SECONDS}`: TCP connection timeout in seconds for SGLang encoder server
|
||||||
- `--pe-server-url {HTTPADDRESS}`: url of SGLang server hosting the PE model (e.g., for ERNIE-Image)
|
- `--pe-server-url {HTTPADDRESS}`: url of SGLang server hosting the PE model (e.g., for ERNIE-Image). See [Models with Prompt Enhancement](../models_with_pe).
|
||||||
|
|
||||||
### Sampling and output
|
### Sampling and output
|
||||||
|
|
||||||
|
|||||||
@@ -41,6 +41,8 @@ sglang serve --model-path Qwen/Qwen-Image --port 30010
|
|||||||
## Additional Documentation
|
## Additional Documentation
|
||||||
|
|
||||||
- [Post-Processing](/docs/sglang-diffusion/api/post_processing): frame interpolation and upscaling
|
- [Post-Processing](/docs/sglang-diffusion/api/post_processing): frame interpolation and upscaling
|
||||||
|
- [Models with AR Stage](/docs/sglang-diffusion/models_with_ar): run hybrid diffusion pipelines like GLM-Image with a separate AR encoder server
|
||||||
|
- [Models with Prompt Enhancement](/docs/sglang-diffusion/models_with_pe): run ERNIE-Image with either built-in PE or a separate PE server
|
||||||
- [Deployment and Performance Modes](/docs/sglang-diffusion/deployment_cookbook): choose `--performance-mode`, offload, FSDP, CFG parallelism, SP, and TP
|
- [Deployment and Performance Modes](/docs/sglang-diffusion/deployment_cookbook): choose `--performance-mode`, offload, FSDP, CFG parallelism, SP, and TP
|
||||||
- [Attention Backends](/docs/sglang-diffusion/attention_backends): choose the best backend for your model and hardware
|
- [Attention Backends](/docs/sglang-diffusion/attention_backends): choose the best backend for your model and hardware
|
||||||
- [Sequence Parallelism](/docs/sglang-diffusion/ring_sp_performance): configure SP, Ulysses, and ring-based splitting for long sequences
|
- [Sequence Parallelism](/docs/sglang-diffusion/ring_sp_performance): configure SP, Ulysses, and ring-based splitting for long sequences
|
||||||
|
|||||||
@@ -1,5 +1,6 @@
|
|||||||
---
|
---
|
||||||
title: "Diffusion models with AR stage like GLM-Image"
|
title: "Diffusion models with AR stage like GLM-Image"
|
||||||
|
description: "Run diffusion pipelines that delegate an AR stage to a separate SGLang server, such as GLM-Image."
|
||||||
---
|
---
|
||||||
|
|
||||||
## Quick Start
|
## Quick Start
|
||||||
@@ -27,7 +28,9 @@ sglang serve --model-path /path/to/zai-org/GLM-Image/vision_language_encoder/ \
|
|||||||
```
|
```
|
||||||
```bash
|
```bash
|
||||||
# Terminal 2 : launch server with Diffusion model
|
# Terminal 2 : launch server with Diffusion model
|
||||||
sglang serve --model-path /path/to/zai-org/GLM-Image/ --srt-encoder-url "http://${HOST}:${AR_PORT}"
|
sglang serve --model-path /path/to/zai-org/GLM-Image/ \
|
||||||
|
--srt-encoder-url "http://${HOST}:${AR_PORT}" \
|
||||||
|
--port ${PORT}
|
||||||
```
|
```
|
||||||
```bash
|
```bash
|
||||||
# Terminal 3 : launch client
|
# Terminal 3 : launch client
|
||||||
|
|||||||
@@ -1,5 +1,6 @@
|
|||||||
---
|
---
|
||||||
title: "Diffusion Models with Prompt Enhancement (PE)"
|
title: "Diffusion Models with Prompt Enhancement (PE)"
|
||||||
|
description: "Run ERNIE-Image with built-in prompt enhancement or a separate SGLang-served PE model."
|
||||||
---
|
---
|
||||||
|
|
||||||
## Quick Start
|
## Quick Start
|
||||||
@@ -34,7 +35,9 @@ sglang serve --model-path /path/to/baidu/ERNIE-Image/pe/ --port ${PE_PORT}
|
|||||||
```
|
```
|
||||||
```bash
|
```bash
|
||||||
# Terminal 2: launch diffusion model server with PE server
|
# Terminal 2: launch diffusion model server with PE server
|
||||||
sglang serve --model-path /path/to/baidu/ERNIE-Image/ --pe-server-url "http://${HOST}:${PE_PORT}"
|
sglang serve --model-path /path/to/baidu/ERNIE-Image/ \
|
||||||
|
--pe-server-url "http://${HOST}:${PE_PORT}" \
|
||||||
|
--port ${PORT}
|
||||||
```
|
```
|
||||||
```bash
|
```bash
|
||||||
# Terminal 3: launch client
|
# Terminal 3: launch client
|
||||||
@@ -58,4 +61,4 @@ curl -X POST http://${HOST}:${PORT}/v1/images/generations \
|
|||||||
|
|
||||||
## Ascend NPU Environment
|
## Ascend NPU Environment
|
||||||
|
|
||||||
<a href="https://github.com/sgl-project/sglang/tree/main/docs_new/docs/sglang-diffusion/models_with_ar.mdx#ascend-npu-env">Check here.</a>
|
See [Diffusion models with AR stage like GLM-Image](/docs/sglang-diffusion/models_with_ar#ascend-npu-env).
|
||||||
|
|||||||
Reference in New Issue
Block a user