Author SHA1 Message Date
sglang-bot 3d8df4b127 chore: update CI test est_time from recent run data 2026-09-28 00:00:30 +00:00
1126 changed files with 1243 additions and 8680 deletions
-57
View File
@@ -1,57 +0,0 @@
# Gitea Actions 自动构建 sglang 镜像(海外节点,原版源)
# 基底 lmsysorg/sglang:dev-dsv41(docker.io),依赖走 pypi.org 默认源。
# 触发:push 到 dsv41-pd 分支。
#
# 前置条件(一次性,在 Gitea 实例上配置):
# 1. 实例已注册 act_runner(Gitea Actions runner,标签含 ubuntu-latest)
# 2. 仓库 Settings → Secrets 添加:
# REGISTRY_USERNAME / REGISTRY_PASSWORD(推送镜像的账号,如 Gitea 访问令牌)
# 3. 可选:Settings → Actions → Variables 添加 REGISTRY(默认 git.agentwithu.com,
# 即 Gitea 自带容器 registry;也可填 docker.io 等)
name: build-sglang-image
on:
push:
branches: [dsv41-pd, dsv41-pd-visioncp]
# compose/部署配置 与 workflow 自身的改动不触发镜像构建(省 runner 与推送带宽)。
# 改了 workflow 想重建镜像时,需伴随任意代码改动或手动触发。
paths-ignore:
- 'deploy/**'
- '.gitea/**'
env:
# 直接写死 Gitea 自带 registry(vars context 在该实例上求值异常会导致回退 docker.io)
REGISTRY: git.agentwithu.com
IMAGE_NAME: minke.yu/sglang
jobs:
build:
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v4
- name: Resolve tag
id: meta
run: |
SHA9=$(git rev-parse --short=9 HEAD)
echo "tag=${{ github.ref_name }}-${SHA9}-ci-$(date -u +%Y%m%d-%H%M)" >> "$GITHUB_OUTPUT"
- uses: docker/setup-buildx-action@v3
- uses: docker/login-action@v3
with:
registry: ${{ env.REGISTRY }}
username: ${{ secrets.REGISTRY_USERNAME }}
password: ${{ secrets.REGISTRY_PASSWORD }}
- uses: docker/build-push-action@v6
with:
context: .
file: Dockerfile.gitea
push: true
# cache-from: type=registry
# cache-to: type=registry,mode=max
tags: |
${{ env.REGISTRY }}/${{ env.IMAGE_NAME }}:${{ steps.meta.outputs.tag }}
${{ env.REGISTRY }}/${{ env.IMAGE_NAME }}:${{ github.ref_name }}-latest
-11
View File
@@ -1,11 +0,0 @@
# Gitea Actions 用(海外节点):原版 docker.io 基底 + 原版 pypi
# 与 b300 离线版(/data/ymk/build/Dockerfile)的区别:
# - 基底直用 docker.io 的 lmsysorg/sglang:dev-dsv41(不走 umirror/DaoCloud)
# - 不设 PIP_INDEX_URL,用默认 pypi.org
# - 源码由 CI checkout 后经 COPY 进镜像(不用 in-image clone)
FROM lmsysorg/sglang:dev-dsv41
# checkout 含 .git,setuptools-scm 可打戳(main 系分支 version 显示 dev 属正常)
COPY . /sgl-workspace/sglang
RUN pip install --no-cache-dir -e /sgl-workspace/sglang/python
-21
View File
@@ -1,21 +0,0 @@
# 优化版(2026-09-24):解决清华源限流下每次构建重下 torch 的问题。
# 要点:
# 1. pyproject build-system.requires 含 torch==2.13.0,pip 隔离构建环境每次都重下 ~6GB。
# 基底镜像已装 torch 2.13 → 用 --no-build-isolation 复用,构建环境零下载。
# 2. 基底缺 setuptools-scm(版本打戳要用),单层预装 + pip cache mount,只下载一次。
# 3. RUN 的 pip cache mount 持久化在宿主机,依赖解析命中的包不再重复下载。
FROM uhub.service.ucloud.cn/umirror/sglang:dev-dsv41
ARG PIP_INDEX=https://pypi.tuna.tsinghua.edu.cn/simple
ENV PIP_INDEX_URL=${PIP_INDEX}
# 构建后端(pyproject build-system.requires,torch 除外——基底已有)
RUN --mount=type=cache,target=/root/.cache/pip \
pip install "setuptools>=61" "setuptools-scm>=8" "setuptools-rust>=1.11" wheel
# 整个源码树(含 .git,用于 setuptools-scm 版本打戳)
COPY sglang/ /sgl-workspace/sglang/
# editable 安装,全量依赖解析(pyproject.toml 变化时走这个)
RUN --mount=type=cache,target=/root/.cache/pip \
pip install --no-build-isolation -e /sgl-workspace/sglang/python
-15
View File
@@ -1,15 +0,0 @@
# fast 路径:python/pyproject.toml 未变化时使用(build.sh 自动判断)。
# --no-deps 跳过全部依赖解析(连 index 元数据请求都省掉),纯源码迭代秒级完成。
# 依赖有变化时必须走 Dockerfile 全量路径(build.sh 按 pyproject sha256 自动切换)。
FROM uhub.service.ucloud.cn/umirror/sglang:dev-dsv41
ARG PIP_INDEX=https://pypi.tuna.tsinghua.edu.cn/simple
ENV PIP_INDEX_URL=${PIP_INDEX}
RUN --mount=type=cache,target=/root/.cache/pip \
pip install "setuptools>=61" "setuptools-scm>=8" "setuptools-rust>=1.11" wheel
COPY sglang/ /sgl-workspace/sglang/
RUN --mount=type=cache,target=/root/.cache/pip \
pip install --no-deps --no-build-isolation -e /sgl-workspace/sglang/python
-33
View File
@@ -1,33 +0,0 @@
#!/bin/bash
# 用法: /data/ymk/build/build.sh [tag]
# 默认 tag: ymkymx/sglang:<分支>-<sha9>-local-<UTC构建日期时间>(命名规则见 AGENTS.md)
# 另打 <分支>-latest-local 别名。
#
# 依赖路径自动选择:
# python/pyproject.toml 的 sha256 与上次成功构建一致 → Dockerfile.fast(--no-deps,秒级)
# 不一致(或 FULL=1)→ Dockerfile 全量依赖解析,成功后记录新 hash
set -euo pipefail
cd /data/ymk
SHA=$(git -C sglang rev-parse --short=9 HEAD)
BR=$(git -C sglang rev-parse --abbrev-ref HEAD)
DT=$(date -u +%Y%m%d-%H%M)
TAG=${1:-ymkymx/sglang:$BR-$SHA-local-$DT}
HASH_FILE=build/.last_pyproject_sha256
CUR=$(sha256sum sglang/python/pyproject.toml | cut -d' ' -f1)
PREV=$(cat "$HASH_FILE" 2>/dev/null || echo none)
DOCKERFILE=build/Dockerfile
if [ "$CUR" = "$PREV" ] && [ "${FULL:-0}" != "1" ]; then
DOCKERFILE=build/Dockerfile.fast
echo "== pyproject 未变,走 fast 路径(--no-deps)"
else
echo "== pyproject 有变化或 FULL=1,走全量依赖解析"
fi
echo "== building $TAG from sglang@$(git -C sglang log --oneline -1) [$DOCKERFILE]"
docker build -f "$DOCKERFILE" -t "$TAG" .
# 构建成功才记录 hash / 打别名
echo "$CUR" > "$HASH_FILE"
docker tag "$TAG" "ymkymx/sglang:$BR-latest-local"
echo "== done: $TAG (别名 $BR-latest-local)"
-33
View File
@@ -1,33 +0,0 @@
# B300 ds41 部署 compose 档案
b300-01 上 DeepSeek-V4.1-Flash 各部署方案的 docker compose 存档。
**本目录是 source of truth**;b300 上的运行目录 `/data/ymk/ds41/` 是工作副本。
方案说明、参数矩阵、特殊 env、已知坑见飞书部署手册(组内共享):
《B300 sglang 部署手册》 https://u04wb5irxz.feishu.cn/docx/FOi8dbnybodHr0xIfkPcQ6gtnRx
## 同步流程
```bash
# 本机(D:\B300\sglang)
git add deploy/ && git commit -m "deploy: ..."
git push origin dsv41-pd-visioncp # Gitea(paths-ignore,不触发镜像构建)
git push b300 dsv41-pd-visioncp # b300 裸仓库
# b300-01
git -C /data/ymk/sglang pull # deploy/ 出现在 /data/ymk/sglang/deploy/b300-ds41/
```
## CI 说明
`.gitea/workflows/build-image.yaml` 的 push 触发器带 `paths-ignore: ['deploy/**', '.gitea/**']`,
改本目录不会触发镜像构建。注意:workflow 自身的改动也不再自动触发,需要重建镜像时
伴随代码改动 push 或手动触发。
## 目录规则
- 命名:`dockerserve-<拓扑>-<特性>-<卡位>.yml`(如 `-b4` = 后 4 卡)
- 退役文件不要留在主目录:移入 `archive-YYYYMMDD/`(b300 侧同样执行)
- 禁止提交 `.bak` 备份文件
- `mooncake-store.json` 是 hicache L3 mooncake store 配置,被
`dockerserve-pd2-p-r1/r2.yml` 通过 `SGLANG_HICACHE_MOONCAKE_CONFIG_PATH` 引用
-171
View File
@@ -1,171 +0,0 @@
# 3x cp2-P + 1x dp2-D PD 分离 + L3 mooncake 互联缓存 + engram host table
# 2026-09-23 实验;基线见 archive-20260923/ 与 D:\B300\experiments\3cp2-pd\baseline-20260923.md
# GPU: P1=0,1 P2=2,3 P3=4,5 D=6,7
# 端口: P1=30000 P2=30010 P3=30020 D=30001 router=30002
# bootstrap: P1=8998 P2=8999 P3=9000 D=9001
x-sglang-common: &sglang-common
image: ymkymx/sglang:dsv41-pd-visioncp-db7d2cb7d-fix
shm_size: "32gb"
ipc: host
pid: host
privileged: true
network_mode: host
volumes:
- /data:/data
- /data/ymk/cache/sglang:/root/.cache/sglang
deploy:
resources:
reservations:
devices:
- driver: nvidia
count: all
capabilities: [gpu]
x-p-environment: &p-environment
SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE: "1"
SGLANG_RAGGED_VERIFY_MODE: static
MC_INTRANODE_NVLINK: "true"
MC_INTRA_NVLINK: "true"
SGLANG_MOONCAKE_SEND_AUX_TCP: "1"
SGLANG_HICACHE_MOONCAKE_CONFIG_PATH: /data/ymk/ds41/mooncake-store.json
SGLANG_DISAGGREGATION_QUEUE_SIZE: "16"
SGLANG_DISAGGREGATION_THREAD_POOL_SIZE: "32"
services:
p1:
<<: *sglang-common
container_name: ds41-cp2-p1
environment:
<<: *p-environment
CUDA_VISIBLE_DEVICES: "0,1"
command: >
sglang serve
--trust-remote-code
--model-path /data/models/DeepSeek-V4.1-Flash
--tp 2 --ep-size 2
--enable-prefill-cp --cp-strategy interleave
--mem-fraction-static 0.75
--attention-backend dsv4
--moe-runner-backend flashinfer_mxfp4
--cuda-graph-max-bs-decode 32
--reasoning-parser auto --tool-call-parser auto
--default-chat-template-kwargs '{"thinking": true, "reasoning_effort": "high"}'
--max-running-requests 64
--tokenizer-worker-num 8
--host 0.0.0.0 --port 30000
--enable-cache-report --enable-metrics
--speculative-algorithm DSPARK --speculative-dspark-block-size 5
--enable-hierarchical-cache
--hicache-ratio 2
--hicache-mem-layout page_first_direct
--hicache-io-backend direct
--hicache-write-policy write_through
--hicache-storage-backend mooncake
--hicache-storage-prefetch-policy wait_complete
--hicache-size 0
--disaggregation-mode prefill
--disaggregation-transfer-backend mooncake
--disaggregation-bootstrap-port 8998
p2:
<<: *sglang-common
container_name: ds41-cp2-p2
environment:
<<: *p-environment
CUDA_VISIBLE_DEVICES: "2,3"
command: >
sglang serve
--trust-remote-code
--model-path /data/models/DeepSeek-V4.1-Flash
--tp 2 --ep-size 2
--enable-prefill-cp --cp-strategy interleave
--mem-fraction-static 0.75
--attention-backend dsv4
--moe-runner-backend flashinfer_mxfp4
--cuda-graph-max-bs-decode 32
--reasoning-parser auto --tool-call-parser auto
--default-chat-template-kwargs '{"thinking": true, "reasoning_effort": "high"}'
--max-running-requests 64
--tokenizer-worker-num 8
--host 0.0.0.0 --port 30010
--enable-cache-report --enable-metrics
--speculative-algorithm DSPARK --speculative-dspark-block-size 5
--enable-hierarchical-cache
--hicache-ratio 2
--hicache-mem-layout page_first_direct
--hicache-io-backend direct
--hicache-write-policy write_through
--hicache-storage-backend mooncake
--hicache-storage-prefetch-policy wait_complete
--hicache-size 0
--disaggregation-mode prefill
--disaggregation-transfer-backend mooncake
--disaggregation-bootstrap-port 8999
p3:
<<: *sglang-common
container_name: ds41-cp2-p3
environment:
<<: *p-environment
CUDA_VISIBLE_DEVICES: "4,5"
command: >
sglang serve
--trust-remote-code
--model-path /data/models/DeepSeek-V4.1-Flash
--tp 2 --ep-size 2
--enable-prefill-cp --cp-strategy interleave
--mem-fraction-static 0.75
--attention-backend dsv4
--moe-runner-backend flashinfer_mxfp4
--cuda-graph-max-bs-decode 32
--reasoning-parser auto --tool-call-parser auto
--default-chat-template-kwargs '{"thinking": true, "reasoning_effort": "high"}'
--max-running-requests 64
--tokenizer-worker-num 8
--host 0.0.0.0 --port 30020
--enable-cache-report --enable-metrics
--speculative-algorithm DSPARK --speculative-dspark-block-size 5
--enable-hierarchical-cache
--hicache-ratio 2
--hicache-mem-layout page_first_direct
--hicache-io-backend direct
--hicache-write-policy write_through
--hicache-storage-backend mooncake
--hicache-storage-prefetch-policy wait_complete
--hicache-size 0
--disaggregation-mode prefill
--disaggregation-transfer-backend mooncake
--disaggregation-bootstrap-port 9000
d:
<<: *sglang-common
container_name: ds41-cp2-d
environment:
SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE: "1"
SGLANG_RAGGED_VERIFY_MODE: static
MC_INTRANODE_NVLINK: "true"
MC_INTRA_NVLINK: "true"
SGLANG_MOONCAKE_SEND_AUX_TCP: "1"
SGLANG_DISAGGREGATION_QUEUE_SIZE: "16"
SGLANG_DISAGGREGATION_THREAD_POOL_SIZE: "32"
CUDA_VISIBLE_DEVICES: "6,7"
command: >
sglang serve
--trust-remote-code
--model-path /data/models/DeepSeek-V4.1-Flash
--tp 2 --ep-size 2 --dp-size 2
--enable-dp-attention --enable-dp-lm-head
--mem-fraction-static 0.80
--attention-backend dsv4
--moe-runner-backend flashinfer_mxfp4
--cuda-graph-max-bs-decode 32
--reasoning-parser auto --tool-call-parser auto
--default-chat-template-kwargs '{"thinking": true, "reasoning_effort": "high"}'
--max-running-requests 64
--tokenizer-worker-num 8
--host 0.0.0.0 --port 30001
--enable-cache-report --enable-metrics
--speculative-algorithm DSPARK --speculative-dspark-block-size 5
--disaggregation-mode decode
--disaggregation-transfer-backend mooncake
--disaggregation-bootstrap-port 9001
@@ -1,195 +0,0 @@
# 3x cp2-P + 1x dp2-D(cg256) PD 分离 + L3 mooncake + engram + OTLP trace(2026-09-27)
# 基于 dockerserve-3p1d-soak.yml 增加: --enable-trace --trace-modules request,mooncake
# --otlp-traces-endpoint 127.0.0.1:4317 --otlp-service-name ds41-3p1d-<实例名>
# 依赖: b300-01 k8s cw jaeger-trace(jaegertracing/all-in-one:1.76.0,badger 6h TTL,hostNetwork 4317)
# 安全性: BatchSpanProcessor 导出失败只丢 span;Jaeger 宕机不影响推理服务
# 调优旋钮(默认未开): SGLANG_TRACE_ASYNC=1(异步 span 创建), SGLANG_TRACE_LEVEL(默认 3)
# router(ds41-router-3p1d,rust)无 OTLP 支持,不产生 span,追踪从各 sglang HTTP 层开始
# 2026-09-23 实验;基线见 archive-20260923/ 与 D:\B300\experiments\3cp2-pd\baseline-20260923.md
# GPU: P1=0,1 P2=2,3 P3=4,5 D=6,7
# 端口: P1=30000 P2=30010 P3=30020 D=30001 router=30002
# bootstrap: P1=8998 P2=8999 P3=9000 D=9001
x-sglang-common: &sglang-common
image: ymkymx/sglang:dsv41-pd-visioncp-db7d2cb7d-fix
shm_size: "32gb"
ipc: host
pid: host
privileged: true
network_mode: host
volumes:
- /data:/data
- /data/ymk/cache/sglang:/root/.cache/sglang
deploy:
resources:
reservations:
devices:
- driver: nvidia
count: all
capabilities: [gpu]
x-p-environment: &p-environment
SGLANG_TORCH_PROFILER_DIR: /data/ymk/prof/torch
SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE: "1"
SGLANG_RAGGED_VERIFY_MODE: static
MC_INTRANODE_NVLINK: "true"
MC_INTRA_NVLINK: "true"
SGLANG_MOONCAKE_SEND_AUX_TCP: "1"
SGLANG_HICACHE_MOONCAKE_CONFIG_PATH: /data/ymk/ds41/mooncake-store.json
SGLANG_DISAGGREGATION_QUEUE_SIZE: "16"
SGLANG_DISAGGREGATION_THREAD_POOL_SIZE: "32"
services:
p1:
<<: *sglang-common
container_name: ds41-3p1d-p1
environment:
<<: *p-environment
CUDA_VISIBLE_DEVICES: "0,1"
command: >
sglang serve
--trust-remote-code
--model-path /data/models/DeepSeek-V4.1-Flash
--tp 2 --ep-size 2
--enable-prefill-cp --cp-strategy interleave --chunked-prefill-size 32768 --max-prefill-tokens 32768
--mem-fraction-static 0.75
--attention-backend dsv4
--moe-runner-backend flashinfer_mxfp4
--cuda-graph-max-bs-decode 32
--reasoning-parser auto --tool-call-parser auto
--default-chat-template-kwargs '{"thinking": true, "reasoning_effort": "high"}'
--max-running-requests 256
--tokenizer-worker-num 8
--host 0.0.0.0 --port 30000
--enable-cache-report --enable-metrics
--speculative-algorithm DSPARK --speculative-dspark-block-size 5
--enable-hierarchical-cache
--hicache-ratio 2
--hicache-mem-layout page_first_direct
--hicache-io-backend direct
--hicache-write-policy write_through
--hicache-storage-backend mooncake
--hicache-storage-prefetch-policy wait_complete
--hicache-size 0
--disaggregation-mode prefill
--disaggregation-transfer-backend mooncake
--disaggregation-bootstrap-port 8998
--enable-trace
--trace-modules request,mooncake
--otlp-traces-endpoint 127.0.0.1:4317
--otlp-service-name ds41-3p1d-p1
p2:
<<: *sglang-common
container_name: ds41-3p1d-p2
environment:
<<: *p-environment
CUDA_VISIBLE_DEVICES: "2,3"
command: >
sglang serve
--trust-remote-code
--model-path /data/models/DeepSeek-V4.1-Flash
--tp 2 --ep-size 2
--enable-prefill-cp --cp-strategy interleave --chunked-prefill-size 32768 --max-prefill-tokens 32768
--mem-fraction-static 0.75
--attention-backend dsv4
--moe-runner-backend flashinfer_mxfp4
--cuda-graph-max-bs-decode 32
--reasoning-parser auto --tool-call-parser auto
--default-chat-template-kwargs '{"thinking": true, "reasoning_effort": "high"}'
--max-running-requests 256
--tokenizer-worker-num 8
--host 0.0.0.0 --port 30010
--enable-cache-report --enable-metrics
--speculative-algorithm DSPARK --speculative-dspark-block-size 5
--enable-hierarchical-cache
--hicache-ratio 2
--hicache-mem-layout page_first_direct
--hicache-io-backend direct
--hicache-write-policy write_through
--hicache-storage-backend mooncake
--hicache-storage-prefetch-policy wait_complete
--hicache-size 0
--disaggregation-mode prefill
--disaggregation-transfer-backend mooncake
--disaggregation-bootstrap-port 8999
--enable-trace
--trace-modules request,mooncake
--otlp-traces-endpoint 127.0.0.1:4317
--otlp-service-name ds41-3p1d-p2
p3:
<<: *sglang-common
container_name: ds41-3p1d-p3
environment:
<<: *p-environment
CUDA_VISIBLE_DEVICES: "4,5"
command: >
sglang serve
--trust-remote-code
--model-path /data/models/DeepSeek-V4.1-Flash
--tp 2 --ep-size 2
--enable-prefill-cp --cp-strategy interleave --chunked-prefill-size 32768 --max-prefill-tokens 32768
--mem-fraction-static 0.75
--attention-backend dsv4
--moe-runner-backend flashinfer_mxfp4
--cuda-graph-max-bs-decode 32
--reasoning-parser auto --tool-call-parser auto
--default-chat-template-kwargs '{"thinking": true, "reasoning_effort": "high"}'
--max-running-requests 256
--tokenizer-worker-num 8
--host 0.0.0.0 --port 30020
--enable-cache-report --enable-metrics
--speculative-algorithm DSPARK --speculative-dspark-block-size 5
--enable-hierarchical-cache
--hicache-ratio 2
--hicache-mem-layout page_first_direct
--hicache-io-backend direct
--hicache-write-policy write_through
--hicache-storage-backend mooncake
--hicache-storage-prefetch-policy wait_complete
--hicache-size 0
--disaggregation-mode prefill
--disaggregation-transfer-backend mooncake
--disaggregation-bootstrap-port 9000
--enable-trace
--trace-modules request,mooncake
--otlp-traces-endpoint 127.0.0.1:4317
--otlp-service-name ds41-3p1d-p3
d:
<<: *sglang-common
container_name: ds41-3p1d-d1
environment:
SGLANG_TORCH_PROFILER_DIR: /data/ymk/prof/torch
SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE: "1"
SGLANG_RAGGED_VERIFY_MODE: static
MC_INTRANODE_NVLINK: "true"
MC_INTRA_NVLINK: "true"
SGLANG_MOONCAKE_SEND_AUX_TCP: "1"
SGLANG_DISAGGREGATION_QUEUE_SIZE: "16"
SGLANG_DISAGGREGATION_THREAD_POOL_SIZE: "32"
CUDA_VISIBLE_DEVICES: "6,7"
command: >
sglang serve
--trust-remote-code
--model-path /data/models/DeepSeek-V4.1-Flash
--tp 2 --ep-size 2 --dp-size 2
--enable-dp-attention --enable-dp-lm-head
--mem-fraction-static 0.80
--attention-backend dsv4
--moe-runner-backend flashinfer_mxfp4
--cuda-graph-max-bs-decode 256
--reasoning-parser auto --tool-call-parser auto
--default-chat-template-kwargs '{"thinking": true, "reasoning_effort": "high"}'
--max-running-requests 256
--tokenizer-worker-num 8
--host 0.0.0.0 --port 30001
--enable-cache-report --enable-metrics
--speculative-algorithm DSPARK --speculative-dspark-block-size 5
--disaggregation-mode decode
--disaggregation-transfer-backend mooncake
--disaggregation-bootstrap-port 9001
--enable-trace
--trace-modules request,mooncake
--otlp-traces-endpoint 127.0.0.1:4317
--otlp-service-name ds41-3p1d-d1
-195
View File
@@ -1,195 +0,0 @@
# 3x cp2-P + 1x dp2-D(cg256) PD 分离 + L3 mooncake + engram + OTLP trace(2026-09-27)
# 基于 dockerserve-3p1d-soak.yml 增加: --enable-trace --trace-modules request,mooncake
# --otlp-traces-endpoint 127.0.0.1:4317 --otlp-service-name ds41-3p1d-<实例名>
# 依赖: b300-01 k8s cw jaeger-trace(jaegertracing/all-in-one:1.76.0,badger 6h TTL,hostNetwork 4317)
# 安全性: BatchSpanProcessor 导出失败只丢 span;Jaeger 宕机不影响推理服务
# 调优旋钮(默认未开): SGLANG_TRACE_ASYNC=1(异步 span 创建), SGLANG_TRACE_LEVEL(默认 3)
# router(ds41-router-3p1d,rust)无 OTLP 支持,不产生 span,追踪从各 sglang HTTP 层开始
# 2026-09-23 实验;基线见 archive-20260923/ 与 D:\B300\experiments\3cp2-pd\baseline-20260923.md
# GPU: P1=0,1 P2=2,3 P3=4,5 D=6,7
# 端口: P1=30000 P2=30010 P3=30020 D=30001 router=30002
# bootstrap: P1=8998 P2=8999 P3=9000 D=9001
x-sglang-common: &sglang-common
image: ymkymx/sglang:dsv41-pd-visioncp-db7d2cb7d-fix
shm_size: "32gb"
ipc: host
pid: host
privileged: true
network_mode: host
volumes:
- /data:/data
- /data/ymk/cache/sglang:/root/.cache/sglang
deploy:
resources:
reservations:
devices:
- driver: nvidia
count: all
capabilities: [gpu]
x-p-environment: &p-environment
SGLANG_TORCH_PROFILER_DIR: /data/ymk/prof/torch
SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE: "1"
SGLANG_RAGGED_VERIFY_MODE: static
MC_INTRANODE_NVLINK: "true"
MC_INTRA_NVLINK: "true"
SGLANG_MOONCAKE_SEND_AUX_TCP: "1"
SGLANG_HICACHE_MOONCAKE_CONFIG_PATH: /data/ymk/ds41/mooncake-store.json
SGLANG_DISAGGREGATION_QUEUE_SIZE: "16"
SGLANG_DISAGGREGATION_THREAD_POOL_SIZE: "32"
services:
p1:
<<: *sglang-common
container_name: ds41-3p1d-p1
environment:
<<: *p-environment
CUDA_VISIBLE_DEVICES: "0,1"
command: >
sglang serve
--trust-remote-code
--model-path /data/models/DeepSeek-V4.1-Flash
--tp 2 --ep-size 2
--enable-prefill-cp --cp-strategy interleave
--mem-fraction-static 0.75
--attention-backend dsv4
--moe-runner-backend flashinfer_mxfp4
--cuda-graph-max-bs-decode 32
--reasoning-parser auto --tool-call-parser auto
--default-chat-template-kwargs '{"thinking": true, "reasoning_effort": "high"}'
--max-running-requests 256
--tokenizer-worker-num 8
--host 0.0.0.0 --port 30000
--enable-cache-report --enable-metrics
--speculative-algorithm DSPARK --speculative-dspark-block-size 5
--enable-hierarchical-cache
--hicache-ratio 2
--hicache-mem-layout page_first_direct
--hicache-io-backend direct
--hicache-write-policy write_through
--hicache-storage-backend mooncake
--hicache-storage-prefetch-policy wait_complete
--hicache-size 0
--disaggregation-mode prefill
--disaggregation-transfer-backend mooncake
--disaggregation-bootstrap-port 8998
--enable-trace
--trace-modules request,mooncake
--otlp-traces-endpoint 127.0.0.1:4317
--otlp-service-name ds41-3p1d-p1
p2:
<<: *sglang-common
container_name: ds41-3p1d-p2
environment:
<<: *p-environment
CUDA_VISIBLE_DEVICES: "2,3"
command: >
sglang serve
--trust-remote-code
--model-path /data/models/DeepSeek-V4.1-Flash
--tp 2 --ep-size 2
--enable-prefill-cp --cp-strategy interleave
--mem-fraction-static 0.75
--attention-backend dsv4
--moe-runner-backend flashinfer_mxfp4
--cuda-graph-max-bs-decode 32
--reasoning-parser auto --tool-call-parser auto
--default-chat-template-kwargs '{"thinking": true, "reasoning_effort": "high"}'
--max-running-requests 256
--tokenizer-worker-num 8
--host 0.0.0.0 --port 30010
--enable-cache-report --enable-metrics
--speculative-algorithm DSPARK --speculative-dspark-block-size 5
--enable-hierarchical-cache
--hicache-ratio 2
--hicache-mem-layout page_first_direct
--hicache-io-backend direct
--hicache-write-policy write_through
--hicache-storage-backend mooncake
--hicache-storage-prefetch-policy wait_complete
--hicache-size 0
--disaggregation-mode prefill
--disaggregation-transfer-backend mooncake
--disaggregation-bootstrap-port 8999
--enable-trace
--trace-modules request,mooncake
--otlp-traces-endpoint 127.0.0.1:4317
--otlp-service-name ds41-3p1d-p2
p3:
<<: *sglang-common
container_name: ds41-3p1d-p3
environment:
<<: *p-environment
CUDA_VISIBLE_DEVICES: "4,5"
command: >
sglang serve
--trust-remote-code
--model-path /data/models/DeepSeek-V4.1-Flash
--tp 2 --ep-size 2
--enable-prefill-cp --cp-strategy interleave
--mem-fraction-static 0.75
--attention-backend dsv4
--moe-runner-backend flashinfer_mxfp4
--cuda-graph-max-bs-decode 32
--reasoning-parser auto --tool-call-parser auto
--default-chat-template-kwargs '{"thinking": true, "reasoning_effort": "high"}'
--max-running-requests 256
--tokenizer-worker-num 8
--host 0.0.0.0 --port 30020
--enable-cache-report --enable-metrics
--speculative-algorithm DSPARK --speculative-dspark-block-size 5
--enable-hierarchical-cache
--hicache-ratio 2
--hicache-mem-layout page_first_direct
--hicache-io-backend direct
--hicache-write-policy write_through
--hicache-storage-backend mooncake
--hicache-storage-prefetch-policy wait_complete
--hicache-size 0
--disaggregation-mode prefill
--disaggregation-transfer-backend mooncake
--disaggregation-bootstrap-port 9000
--enable-trace
--trace-modules request,mooncake
--otlp-traces-endpoint 127.0.0.1:4317
--otlp-service-name ds41-3p1d-p3
d:
<<: *sglang-common
container_name: ds41-3p1d-d1
environment:
SGLANG_TORCH_PROFILER_DIR: /data/ymk/prof/torch
SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE: "1"
SGLANG_RAGGED_VERIFY_MODE: static
MC_INTRANODE_NVLINK: "true"
MC_INTRA_NVLINK: "true"
SGLANG_MOONCAKE_SEND_AUX_TCP: "1"
SGLANG_DISAGGREGATION_QUEUE_SIZE: "16"
SGLANG_DISAGGREGATION_THREAD_POOL_SIZE: "32"
CUDA_VISIBLE_DEVICES: "6,7"
command: >
sglang serve
--trust-remote-code
--model-path /data/models/DeepSeek-V4.1-Flash
--tp 2 --ep-size 2 --dp-size 2
--enable-dp-attention --enable-dp-lm-head
--mem-fraction-static 0.80
--attention-backend dsv4
--moe-runner-backend flashinfer_mxfp4
--cuda-graph-max-bs-decode 256
--reasoning-parser auto --tool-call-parser auto
--default-chat-template-kwargs '{"thinking": true, "reasoning_effort": "high"}'
--max-running-requests 256
--tokenizer-worker-num 8
--host 0.0.0.0 --port 30001
--enable-cache-report --enable-metrics
--speculative-algorithm DSPARK --speculative-dspark-block-size 5
--disaggregation-mode decode
--disaggregation-transfer-backend mooncake
--disaggregation-bootstrap-port 9001
--enable-trace
--trace-modules request,mooncake
--otlp-traces-endpoint 127.0.0.1:4317
--otlp-service-name ds41-3p1d-d1
-173
View File
@@ -1,173 +0,0 @@
# 3x cp2-P + 1x dp2-D(cg256) PD 分离 + L3 mooncake + engram(2026-09-25,mr256,nsys on p1/d1)
# 2026-09-23 实验;基线见 archive-20260923/ 与 D:\B300\experiments\3cp2-pd\baseline-20260923.md
# GPU: P1=0,1 P2=2,3 P3=4,5 D=6,7
# 端口: P1=30000 P2=30010 P3=30020 D=30001 router=30002
# bootstrap: P1=8998 P2=8999 P3=9000 D=9001
x-sglang-common: &sglang-common
image: ymkymx/sglang:dsv41-pd-visioncp-db7d2cb7d-fix
shm_size: "32gb"
ipc: host
pid: host
privileged: true
network_mode: host
volumes:
- /data:/data
- /data/ymk/cache/sglang:/root/.cache/sglang
deploy:
resources:
reservations:
devices:
- driver: nvidia
count: all
capabilities: [gpu]
x-p-environment: &p-environment
SGLANG_TORCH_PROFILER_DIR: /data/ymk/prof/torch
SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE: "1"
SGLANG_RAGGED_VERIFY_MODE: static
MC_INTRANODE_NVLINK: "true"
MC_INTRA_NVLINK: "true"
SGLANG_MOONCAKE_SEND_AUX_TCP: "1"
SGLANG_HICACHE_MOONCAKE_CONFIG_PATH: /data/ymk/ds41/mooncake-store.json
SGLANG_DISAGGREGATION_QUEUE_SIZE: "16"
SGLANG_DISAGGREGATION_THREAD_POOL_SIZE: "32"
services:
p1:
<<: *sglang-common
container_name: ds41-3p1d-prof-p1
environment:
<<: *p-environment
CUDA_VISIBLE_DEVICES: "0,1"
command: >
nsys launch --session-new=p1prof --cuda-graph-trace=node sglang serve
--trust-remote-code
--model-path /data/models/DeepSeek-V4.1-Flash
--tp 2 --ep-size 2
--enable-prefill-cp --cp-strategy interleave
--mem-fraction-static 0.75
--attention-backend dsv4
--moe-runner-backend flashinfer_mxfp4
--cuda-graph-max-bs-decode 32
--reasoning-parser auto --tool-call-parser auto
--default-chat-template-kwargs '{"thinking": true, "reasoning_effort": "high"}'
--max-running-requests 256
--tokenizer-worker-num 8
--host 0.0.0.0 --port 30000
--enable-cache-report --enable-metrics
--speculative-algorithm DSPARK --speculative-dspark-block-size 5
--enable-hierarchical-cache
--hicache-ratio 2
--hicache-mem-layout page_first_direct
--hicache-io-backend direct
--hicache-write-policy write_through
--hicache-storage-backend mooncake
--hicache-storage-prefetch-policy wait_complete
--hicache-size 0
--disaggregation-mode prefill
--disaggregation-transfer-backend mooncake
--disaggregation-bootstrap-port 8998
p2:
<<: *sglang-common
container_name: ds41-3p1d-prof-p2
environment:
<<: *p-environment
CUDA_VISIBLE_DEVICES: "2,3"
command: >
sglang serve
--trust-remote-code
--model-path /data/models/DeepSeek-V4.1-Flash
--tp 2 --ep-size 2
--enable-prefill-cp --cp-strategy interleave
--mem-fraction-static 0.75
--attention-backend dsv4
--moe-runner-backend flashinfer_mxfp4
--cuda-graph-max-bs-decode 32
--reasoning-parser auto --tool-call-parser auto
--default-chat-template-kwargs '{"thinking": true, "reasoning_effort": "high"}'
--max-running-requests 256
--tokenizer-worker-num 8
--host 0.0.0.0 --port 30010
--enable-cache-report --enable-metrics
--speculative-algorithm DSPARK --speculative-dspark-block-size 5
--enable-hierarchical-cache
--hicache-ratio 2
--hicache-mem-layout page_first_direct
--hicache-io-backend direct
--hicache-write-policy write_through
--hicache-storage-backend mooncake
--hicache-storage-prefetch-policy wait_complete
--hicache-size 0
--disaggregation-mode prefill
--disaggregation-transfer-backend mooncake
--disaggregation-bootstrap-port 8999
p3:
<<: *sglang-common
container_name: ds41-3p1d-prof-p3
environment:
<<: *p-environment
CUDA_VISIBLE_DEVICES: "4,5"
command: >
sglang serve
--trust-remote-code
--model-path /data/models/DeepSeek-V4.1-Flash
--tp 2 --ep-size 2
--enable-prefill-cp --cp-strategy interleave
--mem-fraction-static 0.75
--attention-backend dsv4
--moe-runner-backend flashinfer_mxfp4
--cuda-graph-max-bs-decode 32
--reasoning-parser auto --tool-call-parser auto
--default-chat-template-kwargs '{"thinking": true, "reasoning_effort": "high"}'
--max-running-requests 256
--tokenizer-worker-num 8
--host 0.0.0.0 --port 30020
--enable-cache-report --enable-metrics
--speculative-algorithm DSPARK --speculative-dspark-block-size 5
--enable-hierarchical-cache
--hicache-ratio 2
--hicache-mem-layout page_first_direct
--hicache-io-backend direct
--hicache-write-policy write_through
--hicache-storage-backend mooncake
--hicache-storage-prefetch-policy wait_complete
--hicache-size 0
--disaggregation-mode prefill
--disaggregation-transfer-backend mooncake
--disaggregation-bootstrap-port 9000
d:
<<: *sglang-common
container_name: ds41-3p1d-prof-d1
environment:
SGLANG_TORCH_PROFILER_DIR: /data/ymk/prof/torch
SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE: "1"
SGLANG_RAGGED_VERIFY_MODE: static
MC_INTRANODE_NVLINK: "true"
MC_INTRA_NVLINK: "true"
SGLANG_MOONCAKE_SEND_AUX_TCP: "1"
SGLANG_DISAGGREGATION_QUEUE_SIZE: "16"
SGLANG_DISAGGREGATION_THREAD_POOL_SIZE: "32"
CUDA_VISIBLE_DEVICES: "6,7"
command: >
nsys launch --session-new=d1prof --cuda-graph-trace=node sglang serve
--trust-remote-code
--model-path /data/models/DeepSeek-V4.1-Flash
--tp 2 --ep-size 2 --dp-size 2
--enable-dp-attention --enable-dp-lm-head
--mem-fraction-static 0.80
--attention-backend dsv4
--moe-runner-backend flashinfer_mxfp4
--cuda-graph-max-bs-decode 256
--reasoning-parser auto --tool-call-parser auto
--default-chat-template-kwargs '{"thinking": true, "reasoning_effort": "high"}'
--max-running-requests 256
--tokenizer-worker-num 8
--host 0.0.0.0 --port 30001
--enable-cache-report --enable-metrics
--speculative-algorithm DSPARK --speculative-dspark-block-size 5
--disaggregation-mode decode
--disaggregation-transfer-backend mooncake
--disaggregation-bootstrap-port 9001
-173
View File
@@ -1,173 +0,0 @@
# 3x cp2-P + 1x dp2-D(cg256) PD 分离 + L3 mooncake + engram(2026-09-25,mr256,nsys on p1/d1)
# 2026-09-23 实验;基线见 archive-20260923/ 与 D:\B300\experiments\3cp2-pd\baseline-20260923.md
# GPU: P1=0,1 P2=2,3 P3=4,5 D=6,7
# 端口: P1=30000 P2=30010 P3=30020 D=30001 router=30002
# bootstrap: P1=8998 P2=8999 P3=9000 D=9001
x-sglang-common: &sglang-common
image: ymkymx/sglang:dsv41-pd-visioncp-db7d2cb7d-fix
shm_size: "32gb"
ipc: host
pid: host
privileged: true
network_mode: host
volumes:
- /data:/data
- /data/ymk/cache/sglang:/root/.cache/sglang
deploy:
resources:
reservations:
devices:
- driver: nvidia
count: all
capabilities: [gpu]
x-p-environment: &p-environment
SGLANG_TORCH_PROFILER_DIR: /data/ymk/prof/torch
SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE: "1"
SGLANG_RAGGED_VERIFY_MODE: static
MC_INTRANODE_NVLINK: "true"
MC_INTRA_NVLINK: "true"
SGLANG_MOONCAKE_SEND_AUX_TCP: "1"
SGLANG_HICACHE_MOONCAKE_CONFIG_PATH: /data/ymk/ds41/mooncake-store.json
SGLANG_DISAGGREGATION_QUEUE_SIZE: "16"
SGLANG_DISAGGREGATION_THREAD_POOL_SIZE: "32"
services:
p1:
<<: *sglang-common
container_name: ds41-3p1d-p1
environment:
<<: *p-environment
CUDA_VISIBLE_DEVICES: "0,1"
command: >
sglang serve
--trust-remote-code
--model-path /data/models/DeepSeek-V4.1-Flash
--tp 2 --ep-size 2
--enable-prefill-cp --cp-strategy interleave
--mem-fraction-static 0.75
--attention-backend dsv4
--moe-runner-backend flashinfer_mxfp4
--cuda-graph-max-bs-decode 32
--reasoning-parser auto --tool-call-parser auto
--default-chat-template-kwargs '{"thinking": true, "reasoning_effort": "high"}'
--max-running-requests 256
--tokenizer-worker-num 8
--host 0.0.0.0 --port 30000
--enable-cache-report --enable-metrics
--speculative-algorithm DSPARK --speculative-dspark-block-size 5
--enable-hierarchical-cache
--hicache-ratio 2
--hicache-mem-layout page_first_direct
--hicache-io-backend direct
--hicache-write-policy write_through
--hicache-storage-backend mooncake
--hicache-storage-prefetch-policy wait_complete
--hicache-size 0
--disaggregation-mode prefill
--disaggregation-transfer-backend mooncake
--disaggregation-bootstrap-port 8998
p2:
<<: *sglang-common
container_name: ds41-3p1d-p2
environment:
<<: *p-environment
CUDA_VISIBLE_DEVICES: "2,3"
command: >
sglang serve
--trust-remote-code
--model-path /data/models/DeepSeek-V4.1-Flash
--tp 2 --ep-size 2
--enable-prefill-cp --cp-strategy interleave
--mem-fraction-static 0.75
--attention-backend dsv4
--moe-runner-backend flashinfer_mxfp4
--cuda-graph-max-bs-decode 32
--reasoning-parser auto --tool-call-parser auto
--default-chat-template-kwargs '{"thinking": true, "reasoning_effort": "high"}'
--max-running-requests 256
--tokenizer-worker-num 8
--host 0.0.0.0 --port 30010
--enable-cache-report --enable-metrics
--speculative-algorithm DSPARK --speculative-dspark-block-size 5
--enable-hierarchical-cache
--hicache-ratio 2
--hicache-mem-layout page_first_direct
--hicache-io-backend direct
--hicache-write-policy write_through
--hicache-storage-backend mooncake
--hicache-storage-prefetch-policy wait_complete
--hicache-size 0
--disaggregation-mode prefill
--disaggregation-transfer-backend mooncake
--disaggregation-bootstrap-port 8999
p3:
<<: *sglang-common
container_name: ds41-3p1d-p3
environment:
<<: *p-environment
CUDA_VISIBLE_DEVICES: "4,5"
command: >
sglang serve
--trust-remote-code
--model-path /data/models/DeepSeek-V4.1-Flash
--tp 2 --ep-size 2
--enable-prefill-cp --cp-strategy interleave
--mem-fraction-static 0.75
--attention-backend dsv4
--moe-runner-backend flashinfer_mxfp4
--cuda-graph-max-bs-decode 32
--reasoning-parser auto --tool-call-parser auto
--default-chat-template-kwargs '{"thinking": true, "reasoning_effort": "high"}'
--max-running-requests 256
--tokenizer-worker-num 8
--host 0.0.0.0 --port 30020
--enable-cache-report --enable-metrics
--speculative-algorithm DSPARK --speculative-dspark-block-size 5
--enable-hierarchical-cache
--hicache-ratio 2
--hicache-mem-layout page_first_direct
--hicache-io-backend direct
--hicache-write-policy write_through
--hicache-storage-backend mooncake
--hicache-storage-prefetch-policy wait_complete
--hicache-size 0
--disaggregation-mode prefill
--disaggregation-transfer-backend mooncake
--disaggregation-bootstrap-port 9000
d:
<<: *sglang-common
container_name: ds41-3p1d-d1
environment:
SGLANG_TORCH_PROFILER_DIR: /data/ymk/prof/torch
SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE: "1"
SGLANG_RAGGED_VERIFY_MODE: static
MC_INTRANODE_NVLINK: "true"
MC_INTRA_NVLINK: "true"
SGLANG_MOONCAKE_SEND_AUX_TCP: "1"
SGLANG_DISAGGREGATION_QUEUE_SIZE: "16"
SGLANG_DISAGGREGATION_THREAD_POOL_SIZE: "32"
CUDA_VISIBLE_DEVICES: "6,7"
command: >
sglang serve
--trust-remote-code
--model-path /data/models/DeepSeek-V4.1-Flash
--tp 2 --ep-size 2 --dp-size 2
--enable-dp-attention --enable-dp-lm-head
--mem-fraction-static 0.80
--attention-backend dsv4
--moe-runner-backend flashinfer_mxfp4
--cuda-graph-max-bs-decode 256
--reasoning-parser auto --tool-call-parser auto
--default-chat-template-kwargs '{"thinking": true, "reasoning_effort": "high"}'
--max-running-requests 256
--tokenizer-worker-num 8
--host 0.0.0.0 --port 30001
--enable-cache-report --enable-metrics
--speculative-algorithm DSPARK --speculative-dspark-block-size 5
--disaggregation-mode decode
--disaggregation-transfer-backend mooncake
--disaggregation-bootstrap-port 9001
-43
View File
@@ -1,43 +0,0 @@
services:
sglang:
image: ymkymx/sglang:main-ee5fcdf0d-20260920-0142
container_name: ds41-sglang-b4
shm_size: "32gb"
ipc: host
privileged: true
ports:
- "30001:30000"
environment:
- CUDA_VISIBLE_DEVICES=4,5,6,7
- PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True
volumes:
- /data:/data
- /data/ymk/cache/sglang:/root/.cache/sglang
deploy:
resources:
reservations:
devices:
- driver: nvidia
count: all
capabilities: [gpu]
command: >-
sglang serve
--trust-remote-code
--model-path /data/models/DeepSeek-V4.1-Flash
--tp 4
--ep-size 4
--mem-fraction-static 0.60
--attention-backend dsv4
--moe-runner-backend flashinfer_mxfp4
--cuda-graph-max-bs-decode 64
--reasoning-parser auto
--tool-call-parser auto
--max-running-requests 64
--host 0.0.0.0
--port 30000
--dp-size 4
--enable-dp-attention
--enable-cache-report
--enable-metrics
--json-model-override-args '{"vision_n_layers": 0}'
-43
View File
@@ -1,43 +0,0 @@
services:
sglang:
image: ymkymx/sglang:main-ee5fcdf0d-20260920-0142
container_name: ds41-sglang-b4
shm_size: "32gb"
ipc: host
privileged: true
ports:
- "30001:30000"
environment:
- CUDA_VISIBLE_DEVICES=4,5,6,7
- PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True
volumes:
- /data:/data
- /data/ymk/cache/sglang:/root/.cache/sglang
deploy:
resources:
reservations:
devices:
- driver: nvidia
count: all
capabilities: [gpu]
command: >-
sglang serve
--trust-remote-code
--model-path /data/models/DeepSeek-V4.1-Flash
--tp 4
--ep-size 4
--mem-fraction-static 0.60
--attention-backend dsv4
--moe-runner-backend flashinfer_mxfp4
--enable-decoder-swa-bounded-replay
--cuda-graph-max-bs-decode 64
--reasoning-parser auto
--tool-call-parser auto
--max-running-requests 64
--host 0.0.0.0
--port 30000
--enable-cache-report
--enable-metrics
--json-model-override-args '{"vision_n_layers": 0}'
-42
View File
@@ -1,42 +0,0 @@
services:
sglang:
image: ymkymx/sglang:main-ee5fcdf0d-20260920-0142
container_name: ds41-sglang-b4
shm_size: "32gb"
ipc: host
privileged: true
ports:
- "30001:30000"
environment:
- CUDA_VISIBLE_DEVICES=4,5,6,7
volumes:
- /data:/data
- /data/ymk/cache/sglang:/root/.cache/sglang
deploy:
resources:
reservations:
devices:
- driver: nvidia
count: all
capabilities: [gpu]
command: >-
sglang serve
--trust-remote-code
--model-path /data/models/DeepSeek-V4.1-Flash
--tp 4
--ep-size 4
--mem-fraction-static 0.75
--attention-backend dsv4
--moe-runner-backend flashinfer_mxfp4
--enable-decoder-swa-bounded-replay
--cuda-graph-max-bs-decode 64
--reasoning-parser auto
--tool-call-parser auto
--max-running-requests 64
--host 0.0.0.0
--port 30000
--enable-prefill-cp
--cp-strategy interleave
--enable-cache-report
--enable-metrics
--json-model-override-args '{"vision_n_layers": 0}'
-160
View File
@@ -1,160 +0,0 @@
# 4×cp2 独立实例(非 PD)+ hicache L3(mooncake) + dspark + engram host table 卸载(2026-09-23 晚)
# 用户指定组合:4 cp2 hicache l3 dspark roundrobin loadbalance engram offload
# 每实例:tp2 ep2 + interleave CP2 + dspark b5 + hicache L3 write_through(size 0) + engram host table
# (tp2 权重必须靠 engram 卸载才放得下,见 goals-track G1.9 反转)
# GPU: a=0,1 b=2,3 c=4,5 d=6,7;端口: 30000/30010/30020/30030;rr router=30002
# 参考:dockerserve-cpdspark-l3.yml(hicache 参数)、dockerserve-3cp2-pd.yml(cp2/engram 参数)
x-common: &common
image: ymkymx/sglang:dsv41-pd-visioncp-db7d2cb7d-fix
shm_size: "32gb"
ipc: host
privileged: true
network_mode: host
volumes:
- /data:/data
- /data/ymk/cache/sglang:/root/.cache/sglang
# 热补丁(2026-09-23):CP 对称 idle 检查,修 health-check × hicache drain 死锁
# 源文件:/data/ymk/sglang(dsv41-pd-visioncp 分支工作区已改);补丁脚本见
# D:\B300\experiments\cp2x4\patch_healthcheck_cp_idle.py
- /data/ymk/sglang/python/sglang/srt/managers/scheduler.py:/sgl-workspace/sglang/python/sglang/srt/managers/scheduler.py
environment: &env
SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE: "1"
SGLANG_RAGGED_VERIFY_MODE: static
MC_MS_AUTO_DISC: "0"
MOONCAKE_MASTER: 127.0.0.1:50051
MOONCAKE_TE_META_DATA_SERVER: P2PHANDSHAKE
MOONCAKE_PROTOCOL: tcp
MOONCAKE_GLOBAL_SEGMENT_SIZE: 300gb
deploy:
resources:
reservations:
devices:
- driver: nvidia
count: all
capabilities: [gpu]
services:
a:
<<: *common
container_name: ds41-cp2-a
environment:
<<: *env
CUDA_VISIBLE_DEVICES: "0,1"
command: >
sglang serve
--trust-remote-code
--model-path /data/models/DeepSeek-V4.1-Flash
--tp 2 --ep-size 2
--enable-prefill-cp --cp-strategy interleave
--mem-fraction-static 0.75
--attention-backend dsv4
--moe-runner-backend flashinfer_mxfp4
--cuda-graph-max-bs-decode 32
--reasoning-parser auto --tool-call-parser auto
--default-chat-template-kwargs '{"thinking": true, "reasoning_effort": "high"}'
--max-running-requests 64
--tokenizer-worker-num 8
--host 0.0.0.0 --port 30000
--enable-cache-report --enable-metrics
--speculative-algorithm DSPARK --speculative-dspark-block-size 5
--enable-hierarchical-cache
--hicache-ratio 2
--hicache-mem-layout page_first_direct
--hicache-io-backend direct
--hicache-write-policy write_through
--hicache-storage-backend mooncake
--hicache-storage-prefetch-policy wait_complete
--hicache-size 0
b:
<<: *common
container_name: ds41-cp2-b
environment:
<<: *env
CUDA_VISIBLE_DEVICES: "2,3"
command: >
sglang serve
--trust-remote-code
--model-path /data/models/DeepSeek-V4.1-Flash
--tp 2 --ep-size 2
--enable-prefill-cp --cp-strategy interleave
--mem-fraction-static 0.75
--attention-backend dsv4
--moe-runner-backend flashinfer_mxfp4
--cuda-graph-max-bs-decode 32
--reasoning-parser auto --tool-call-parser auto
--default-chat-template-kwargs '{"thinking": true, "reasoning_effort": "high"}'
--max-running-requests 64
--tokenizer-worker-num 8
--host 0.0.0.0 --port 30010
--enable-cache-report --enable-metrics
--speculative-algorithm DSPARK --speculative-dspark-block-size 5
--enable-hierarchical-cache
--hicache-ratio 2
--hicache-mem-layout page_first_direct
--hicache-io-backend direct
--hicache-write-policy write_through
--hicache-storage-backend mooncake
--hicache-storage-prefetch-policy wait_complete
--hicache-size 0
c:
<<: *common
container_name: ds41-cp2-c
environment:
<<: *env
CUDA_VISIBLE_DEVICES: "4,5"
command: >
sglang serve
--trust-remote-code
--model-path /data/models/DeepSeek-V4.1-Flash
--tp 2 --ep-size 2
--enable-prefill-cp --cp-strategy interleave
--mem-fraction-static 0.75
--attention-backend dsv4
--moe-runner-backend flashinfer_mxfp4
--cuda-graph-max-bs-decode 32
--reasoning-parser auto --tool-call-parser auto
--default-chat-template-kwargs '{"thinking": true, "reasoning_effort": "high"}'
--max-running-requests 64
--tokenizer-worker-num 8
--host 0.0.0.0 --port 30020
--enable-cache-report --enable-metrics
--speculative-algorithm DSPARK --speculative-dspark-block-size 5
--enable-hierarchical-cache
--hicache-ratio 2
--hicache-mem-layout page_first_direct
--hicache-io-backend direct
--hicache-write-policy write_through
--hicache-storage-backend mooncake
--hicache-storage-prefetch-policy wait_complete
--hicache-size 0
d:
<<: *common
container_name: ds41-cp2-d2
environment:
<<: *env
CUDA_VISIBLE_DEVICES: "6,7"
command: >
sglang serve
--trust-remote-code
--model-path /data/models/DeepSeek-V4.1-Flash
--tp 2 --ep-size 2
--enable-prefill-cp --cp-strategy interleave
--mem-fraction-static 0.75
--attention-backend dsv4
--moe-runner-backend flashinfer_mxfp4
--cuda-graph-max-bs-decode 32
--reasoning-parser auto --tool-call-parser auto
--default-chat-template-kwargs '{"thinking": true, "reasoning_effort": "high"}'
--max-running-requests 64
--tokenizer-worker-num 8
--host 0.0.0.0 --port 30030
--enable-cache-report --enable-metrics
--speculative-algorithm DSPARK --speculative-dspark-block-size 5
--enable-hierarchical-cache
--hicache-ratio 2
--hicache-mem-layout page_first_direct
--hicache-io-backend direct
--hicache-write-policy write_through
--hicache-storage-backend mooncake
--hicache-storage-prefetch-policy wait_complete
--hicache-size 0
-44
View File
@@ -1,44 +0,0 @@
# cp8 单机非 PD 测试(2026-09-23 晚)
# 背景:dp8 单机 dspark 健康但长上下文 prefill 结构性慢(单请求只落 1 rank),
# 换 cp8(tp8 + interleave prefill CP8,单请求 prefill 切到 8 卡)对照。
# 关 engram、无 hicache L3、无 PD。cp+dspark 无 replay 是已知健康组合
# (cp4+dspark+replay 三元组必崩,本配置不开 replay)。
# GPU 0-7,端口 30000,容器 ds41-cp8,project cp8
services:
cp8:
image: ymkymx/sglang:dsv41-pd-visioncp-db7d2cb7d-fix
container_name: ds41-cp8
shm_size: "32gb"
ipc: host
privileged: true
network_mode: host
environment:
SGLANG_RAGGED_VERIFY_MODE: "static"
CUDA_VISIBLE_DEVICES: "0,1,2,3,4,5,6,7"
volumes:
- /data:/data
- /data/ymk/cache/sglang:/root/.cache/sglang
deploy:
resources:
reservations:
devices:
- driver: nvidia
count: all
capabilities: [gpu]
command: >
sglang serve
--trust-remote-code
--model-path /data/models/DeepSeek-V4.1-Flash
--tp 8 --ep-size 8
--enable-prefill-cp --cp-strategy interleave
--mem-fraction-static 0.80
--attention-backend dsv4
--moe-runner-backend flashinfer_mxfp4
--cuda-graph-max-bs-decode 64
--reasoning-parser auto --tool-call-parser auto
--default-chat-template-kwargs '{"thinking": true, "reasoning_effort": "high"}'
--max-running-requests 64
--tokenizer-worker-num 8
--host 0.0.0.0 --port 30000
--enable-cache-report --enable-metrics
--speculative-algorithm DSPARK --speculative-dspark-block-size 5
@@ -1,50 +0,0 @@
services:
sglang:
image: ymkymx/sglang:dsv41-pd-ddf520763-local-20260923-0536
container_name: ds41-cpdspark-l3-b
shm_size: "32gb"
ipc: host
privileged: true
network_mode: host
environment:
- CUDA_VISIBLE_DEVICES=4,5,6,7
- MC_MS_AUTO_DISC=0
- MOONCAKE_MASTER=127.0.0.1:50051
- MOONCAKE_TE_META_DATA_SERVER=P2PHANDSHAKE
- MOONCAKE_PROTOCOL=tcp
- MOONCAKE_GLOBAL_SEGMENT_SIZE=400gb
volumes:
- /data:/data
- /data/ymk/cache/sglang:/root/.cache/sglang
deploy:
resources:
reservations:
devices:
- driver: nvidia
count: all
capabilities: [gpu]
command: >-
sglang serve
--trust-remote-code
--model-path /data/models/DeepSeek-V4.1-Flash
--tp 4 --ep-size 4
--mem-fraction-static 0.75
--attention-backend dsv4
--moe-runner-backend flashinfer_mxfp4
--cuda-graph-max-bs-decode 64
--reasoning-parser auto --tool-call-parser auto
--tokenizer-worker-num 8
--host 0.0.0.0 --port 30001
--enable-cache-report --enable-metrics
--speculative-algorithm DSPARK
--speculative-dspark-block-size 5
--enable-prefill-cp --cp-strategy interleave
--enable-hierarchical-cache
--hicache-ratio 2
--hicache-mem-layout page_first_direct
--hicache-io-backend direct
--hicache-write-policy write_through
--hicache-storage-backend mooncake
--hicache-storage-prefetch-policy wait_complete
--hicache-size 0
--json-model-override-args '{"vision_n_layers": 0}'
@@ -1,51 +0,0 @@
services:
sglang:
image: ymkymx/sglang:dsv41-pd-ddf520763-local-20260923-0536
container_name: ds41-cpdspark-l3-b
shm_size: "32gb"
ipc: host
privileged: true
network_mode: host
environment:
- CUDA_VISIBLE_DEVICES=4,5,6,7
- MC_MS_AUTO_DISC=0
- MOONCAKE_MASTER=127.0.0.1:50051
- MOONCAKE_TE_META_DATA_SERVER=P2PHANDSHAKE
- MOONCAKE_PROTOCOL=tcp
- MOONCAKE_GLOBAL_SEGMENT_SIZE=400gb
volumes:
- /data:/data
- /data/ymk/cache/sglang:/root/.cache/sglang
deploy:
resources:
reservations:
devices:
- driver: nvidia
count: all
capabilities: [gpu]
command: >-
sglang serve
--trust-remote-code
--model-path /data/models/DeepSeek-V4.1-Flash
--tp 4 --ep-size 4
--mem-fraction-static 0.75
--attention-backend dsv4
--moe-runner-backend flashinfer_mxfp4
--cuda-graph-max-bs-decode 64
--reasoning-parser auto --tool-call-parser auto
--max-running-requests 64
--tokenizer-worker-num 8
--host 0.0.0.0 --port 30001
--enable-cache-report --enable-metrics
--speculative-algorithm DSPARK
--speculative-dspark-block-size 5
--enable-prefill-cp --cp-strategy interleave
--enable-hierarchical-cache
--hicache-ratio 2
--hicache-mem-layout page_first_direct
--hicache-io-backend direct
--hicache-write-policy write_through
--hicache-storage-backend mooncake
--hicache-storage-prefetch-policy wait_complete
--hicache-size 0
--json-model-override-args '{"vision_n_layers": 0}'
@@ -1,50 +0,0 @@
services:
sglang:
image: ymkymx/sglang:dsv41-pd-ddf520763-local-20260923-0536
container_name: ds41-cpdspark-l3-a
shm_size: "32gb"
ipc: host
privileged: true
network_mode: host
environment:
- CUDA_VISIBLE_DEVICES=0,1,2,3
- MC_MS_AUTO_DISC=0
- MOONCAKE_MASTER=127.0.0.1:50051
- MOONCAKE_TE_META_DATA_SERVER=P2PHANDSHAKE
- MOONCAKE_PROTOCOL=tcp
- MOONCAKE_GLOBAL_SEGMENT_SIZE=400gb
volumes:
- /data:/data
- /data/ymk/cache/sglang:/root/.cache/sglang
deploy:
resources:
reservations:
devices:
- driver: nvidia
count: all
capabilities: [gpu]
command: >-
sglang serve
--trust-remote-code
--model-path /data/models/DeepSeek-V4.1-Flash
--tp 4 --ep-size 4
--mem-fraction-static 0.75
--attention-backend dsv4
--moe-runner-backend flashinfer_mxfp4
--cuda-graph-max-bs-decode 64
--reasoning-parser auto --tool-call-parser auto
--tokenizer-worker-num 8
--host 0.0.0.0 --port 30000
--enable-cache-report --enable-metrics
--speculative-algorithm DSPARK
--speculative-dspark-block-size 5
--enable-prefill-cp --cp-strategy interleave
--enable-hierarchical-cache
--hicache-ratio 2
--hicache-mem-layout page_first_direct
--hicache-io-backend direct
--hicache-write-policy write_through
--hicache-storage-backend mooncake
--hicache-storage-prefetch-policy wait_complete
--hicache-size 0
--json-model-override-args '{"vision_n_layers": 0}'
@@ -1,51 +0,0 @@
services:
sglang:
image: ymkymx/sglang:dsv41-pd-ddf520763-local-20260923-0536
container_name: ds41-cpdspark-l3-a
shm_size: "32gb"
ipc: host
privileged: true
network_mode: host
environment:
- CUDA_VISIBLE_DEVICES=0,1,2,3
- MC_MS_AUTO_DISC=0
- MOONCAKE_MASTER=127.0.0.1:50051
- MOONCAKE_TE_META_DATA_SERVER=P2PHANDSHAKE
- MOONCAKE_PROTOCOL=tcp
- MOONCAKE_GLOBAL_SEGMENT_SIZE=400gb
volumes:
- /data:/data
- /data/ymk/cache/sglang:/root/.cache/sglang
deploy:
resources:
reservations:
devices:
- driver: nvidia
count: all
capabilities: [gpu]
command: >-
sglang serve
--trust-remote-code
--model-path /data/models/DeepSeek-V4.1-Flash
--tp 4 --ep-size 4
--mem-fraction-static 0.75
--attention-backend dsv4
--moe-runner-backend flashinfer_mxfp4
--cuda-graph-max-bs-decode 64
--reasoning-parser auto --tool-call-parser auto
--max-running-requests 64
--tokenizer-worker-num 8
--host 0.0.0.0 --port 30000
--enable-cache-report --enable-metrics
--speculative-algorithm DSPARK
--speculative-dspark-block-size 5
--enable-prefill-cp --cp-strategy interleave
--enable-hierarchical-cache
--hicache-ratio 2
--hicache-mem-layout page_first_direct
--hicache-io-backend direct
--hicache-write-policy write_through
--hicache-storage-backend mooncake
--hicache-storage-prefetch-policy wait_complete
--hicache-size 0
--json-model-override-args '{"vision_n_layers": 0}'
@@ -1,44 +0,0 @@
services:
sglang:
image: ymkymx/sglang:dsv41-pd-92632a60b-20260922-2100
container_name: ds41-cpdspark-nr-b4
shm_size: "32gb"
ipc: host
privileged: true
ports:
- "30001:30001"
environment:
- CUDA_VISIBLE_DEVICES=4,5,6,7
volumes:
- /data:/data
- /data/ymk/cache/sglang:/root/.cache/sglang
deploy:
resources:
reservations:
devices:
- driver: nvidia
count: all
capabilities: [gpu]
command: >-
sglang serve
--trust-remote-code
--model-path /data/models/DeepSeek-V4.1-Flash
--tp 4
--ep-size 4
--mem-fraction-static 0.75
--attention-backend dsv4
--moe-runner-backend flashinfer_mxfp4
--cuda-graph-max-bs-decode 64
--reasoning-parser auto
--tool-call-parser auto
--max-running-requests 64
--host 0.0.0.0
--port 30000
--enable-cache-report
--enable-metrics
--speculative-algorithm DSPARK
--speculative-dspark-block-size 5
--enable-prefill-cp
--cp-strategy interleave
--json-model-override-args '{"vision_n_layers": 0}'
@@ -1,44 +0,0 @@
services:
sglang:
image: ymkymx/sglang:dsv41-pd-92632a60b-20260922-2100
container_name: ds41-cpdspark-nr
shm_size: "32gb"
ipc: host
privileged: true
ports:
- "30000:30000"
environment:
- CUDA_VISIBLE_DEVICES=0,1,2,3
volumes:
- /data:/data
- /data/ymk/cache/sglang:/root/.cache/sglang
deploy:
resources:
reservations:
devices:
- driver: nvidia
count: all
capabilities: [gpu]
command: >-
sglang serve
--trust-remote-code
--model-path /data/models/DeepSeek-V4.1-Flash
--tp 4
--ep-size 4
--mem-fraction-static 0.75
--attention-backend dsv4
--moe-runner-backend flashinfer_mxfp4
--cuda-graph-max-bs-decode 64
--reasoning-parser auto
--tool-call-parser auto
--max-running-requests 64
--host 0.0.0.0
--port 30000
--enable-cache-report
--enable-metrics
--speculative-algorithm DSPARK
--speculative-dspark-block-size 5
--enable-prefill-cp
--cp-strategy interleave
--json-model-override-args '{"vision_n_layers": 0}'
-44
View File
@@ -1,44 +0,0 @@
services:
sglang:
image: ymkymx/sglang:main-ee5fcdf0d-20260920-0142
container_name: ds41-cpdspark
shm_size: "32gb"
ipc: host
privileged: true
ports:
- "30000:30000"
environment:
- CUDA_VISIBLE_DEVICES=0,1,2,3
volumes:
- /data:/data
- /data/ymk/cache/sglang:/root/.cache/sglang
deploy:
resources:
reservations:
devices:
- driver: nvidia
count: all
capabilities: [gpu]
command: >-
sglang serve
--trust-remote-code
--model-path /data/models/DeepSeek-V4.1-Flash
--tp 4
--ep-size 4
--mem-fraction-static 0.75
--attention-backend dsv4
--moe-runner-backend flashinfer_mxfp4
--cuda-graph-max-bs-decode 64
--reasoning-parser auto
--tool-call-parser auto
--max-running-requests 64
--host 0.0.0.0
--port 30000
--enable-cache-report
--enable-metrics
--speculative-algorithm DSPARK
--speculative-dspark-block-size 5
--enable-prefill-cp
--cp-strategy interleave
--enable-decoder-swa-bounded-replay
--json-model-override-args '{"vision_n_layers": 0}'
-62
View File
@@ -1,62 +0,0 @@
# dp2 对照(2026-09-24 上午):验证 dp2(dp-attention)下 dspark 是否正常。
# 与 dockerserve-tp2.yml 逐参数对齐,唯一差异 = 加 --dp-size 2 --enable-dp-attention --enable-dp-lm-head。
# GPU 6-7,端口 30030 直连。镜像/补丁挂载/L3/engram/tok8 全部相同。
# 注:dp+dspark 不能开 --enable-decoder-swa-bounded-replay(本配置本来就没开)。
x-common: &common
image: ymkymx/sglang:dsv41-pd-visioncp-db7d2cb7d-fix
shm_size: "32gb"
ipc: host
privileged: true
network_mode: host
volumes:
- /data:/data
- /data/ymk/cache/sglang:/root/.cache/sglang
- /data/ymk/sglang/python/sglang/srt/managers/scheduler.py:/sgl-workspace/sglang/python/sglang/srt/managers/scheduler.py
environment: &env
SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE: "1"
SGLANG_RAGGED_VERIFY_MODE: static
MC_MS_AUTO_DISC: "0"
MOONCAKE_MASTER: 127.0.0.1:50051
MOONCAKE_TE_META_DATA_SERVER: P2PHANDSHAKE
MOONCAKE_PROTOCOL: tcp
MOONCAKE_GLOBAL_SEGMENT_SIZE: 300gb
deploy:
resources:
reservations:
devices:
- driver: nvidia
count: all
capabilities: [gpu]
services:
dp2:
<<: *common
container_name: ds41-dp2
environment:
<<: *env
CUDA_VISIBLE_DEVICES: "6,7"
command: >
sglang serve
--trust-remote-code
--model-path /data/models/DeepSeek-V4.1-Flash
--tp 2 --ep-size 2
--dp-size 2 --enable-dp-attention --enable-dp-lm-head
--mem-fraction-static 0.75
--attention-backend dsv4
--moe-runner-backend flashinfer_mxfp4
--cuda-graph-max-bs-decode 32
--reasoning-parser auto --tool-call-parser auto
--default-chat-template-kwargs '{"thinking": true, "reasoning_effort": "high"}'
--max-running-requests 64
--tokenizer-worker-num 8
--host 0.0.0.0 --port 30030
--enable-cache-report --enable-metrics
--speculative-algorithm DSPARK --speculative-dspark-block-size 5
--enable-hierarchical-cache
--hicache-ratio 2
--hicache-mem-layout page_first_direct
--hicache-io-backend direct
--hicache-write-policy write_through
--hicache-storage-backend mooncake
--hicache-storage-prefetch-policy wait_complete
--hicache-size 0
-42
View File
@@ -1,42 +0,0 @@
services:
sglang:
image: ymkymx/sglang:main-ee5fcdf0d-20260920-0142
container_name: ds41-dp4test
shm_size: "32gb"
ipc: host
privileged: true
ports:
- "30000:30000"
environment:
- CUDA_VISIBLE_DEVICES=0,1,2,3
volumes:
- /data:/data
- /data/ymk/cache/sglang:/root/.cache/sglang
deploy:
resources:
reservations:
devices:
- driver: nvidia
count: all
capabilities: [gpu]
command: >-
sglang serve
--trust-remote-code
--model-path /data/models/DeepSeek-V4.1-Flash
--tp 4
--ep-size 4
--mem-fraction-static 0.75
--attention-backend dsv4
--moe-runner-backend flashinfer_mxfp4
--enable-decoder-swa-bounded-replay
--cuda-graph-max-bs-decode 64
--reasoning-parser auto
--tool-call-parser auto
--max-running-requests 64
--host 0.0.0.0
--port 30000
--enable-prefill-cp
--cp-strategy interleave
--enable-cache-report
--enable-metrics
--json-model-override-args '{"vision_n_layers": 0}'
-44
View File
@@ -1,44 +0,0 @@
# dp8+tp8 单机非 PD 对照测试(2026-09-23 晚)
# 背景:3cp2+dp2 PD 部署 dspark accept 回归(32k ctx accept len 3.14→1.31,
# 见 experiments/3cp2-pd/bench-20260923.md「未决问题」)。本配置退回单机验证
# dspark 健康度:engram host table 关闭、无 hicache L3、无 PD、无 mooncake。
# 参数基准:dockerserve-3cp2-pd.yml 的 d 服务(dp 路径),去掉 PD/L3/engram 相关。
# GPU 0-7,端口 30000,容器 ds41-dp8,project dp8
services:
dp8:
image: ymkymx/sglang:dsv41-pd-visioncp-db7d2cb7d-fix
container_name: ds41-dp8
shm_size: "32gb"
ipc: host
privileged: true
network_mode: host
environment:
SGLANG_RAGGED_VERIFY_MODE: "static"
CUDA_VISIBLE_DEVICES: "0,1,2,3,4,5,6,7"
volumes:
- /data:/data
- /data/ymk/cache/sglang:/root/.cache/sglang
deploy:
resources:
reservations:
devices:
- driver: nvidia
count: all
capabilities: [gpu]
command: >
sglang serve
--trust-remote-code
--model-path /data/models/DeepSeek-V4.1-Flash
--tp 8 --ep-size 8 --dp-size 8
--enable-dp-attention --enable-dp-lm-head
--mem-fraction-static 0.80
--attention-backend dsv4
--moe-runner-backend flashinfer_mxfp4
--cuda-graph-max-bs-decode 32
--reasoning-parser auto --tool-call-parser auto
--default-chat-template-kwargs '{"thinking": true, "reasoning_effort": "high"}'
--max-running-requests 64
--tokenizer-worker-num 8
--host 0.0.0.0 --port 30000
--enable-cache-report --enable-metrics
--speculative-algorithm DSPARK --speculative-dspark-block-size 5
-47
View File
@@ -1,47 +0,0 @@
services:
sglang:
image: ymkymx/sglang:main-ee5fcdf0d-20260920-0142
container_name: ds41-prefill
shm_size: "32gb"
ipc: host
privileged: true
ports:
- "30000:30000"
environment:
- CUDA_VISIBLE_DEVICES=0,1,2,3
volumes:
- /data:/data
- /data/ymk/cache/sglang:/root/.cache/sglang
deploy:
resources:
reservations:
devices:
- driver: nvidia
count: all
capabilities: [gpu]
command: >-
sglang serve
--trust-remote-code
--model-path /data/models/DeepSeek-V4.1-Flash
--tp 4
--ep-size 4
--mem-fraction-static 0.75
--attention-backend dsv4
--moe-runner-backend flashinfer_mxfp4
--cuda-graph-max-bs-decode 64
--reasoning-parser auto
--tool-call-parser auto
--max-running-requests 64
--host 0.0.0.0
--port 30000
--enable-cache-report
--enable-metrics
--speculative-algorithm DSPARK
--speculative-dspark-block-size 5
--enable-decoder-swa-bounded-replay
--enable-hierarchical-cache
--hicache-ratio 2.5
--hicache-write-policy write_back
--disaggregation-mode prefill
--optimistic-prefill-attempts 4
--json-model-override-args '{"vision_n_layers": 0}'
@@ -1,51 +0,0 @@
services:
sglang:
image: ymkymx/sglang:dsv41-pd-visioncp-db7d2cb7d-fix
container_name: ds41-pd-decode
shm_size: "32gb"
ipc: host
pid: host
privileged: true
network_mode: host
environment:
- CUDA_VISIBLE_DEVICES=4,5,6,7
- SGLANG_RAGGED_VERIFY_MODE=static
- MC_INTRANODE_NVLINK=true
- MC_INTRA_NVLINK=true
- SGLANG_MOONCAKE_SEND_AUX_TCP=1
volumes:
- /data:/data
- /data/ymk/cache/sglang:/root/.cache/sglang
deploy:
resources:
reservations:
devices:
- driver: nvidia
count: all
capabilities: [gpu]
command: >-
sglang serve
--trust-remote-code
--model-path /data/models/DeepSeek-V4.1-Flash
--tp 4
--ep-size 4
--dp-size 4
--enable-dp-attention
--enable-dp-lm-head
--mem-fraction-static 0.75
--attention-backend dsv4
--moe-runner-backend flashinfer_mxfp4
--cuda-graph-max-bs-decode 64
--reasoning-parser auto
--tool-call-parser auto
--default-chat-template-kwargs '{"thinking": true, "reasoning_effort": "high"}'
--tokenizer-worker-num 8
--host 0.0.0.0
--port 30001
--enable-cache-report
--enable-metrics
--speculative-algorithm DSPARK
--speculative-dspark-block-size 5
--disaggregation-mode decode
--disaggregation-transfer-backend mooncake
-50
View File
@@ -1,50 +0,0 @@
services:
sglang:
image: ymkymx/sglang:dsv41-pd-ddf520763-local-20260923-0536
container_name: ds41-pd-decode
shm_size: "32gb"
ipc: host
pid: host
privileged: true
network_mode: host
environment:
- CUDA_VISIBLE_DEVICES=4,5,6,7
- SGLANG_RAGGED_VERIFY_MODE=static
- MC_INTRANODE_NVLINK=true
- MC_INTRA_NVLINK=true
- SGLANG_MOONCAKE_SEND_AUX_TCP=1
volumes:
- /data:/data
- /data/ymk/cache/sglang:/root/.cache/sglang
deploy:
resources:
reservations:
devices:
- driver: nvidia
count: all
capabilities: [gpu]
command: >-
sglang serve
--trust-remote-code
--model-path /data/models/DeepSeek-V4.1-Flash
--tp 4
--ep-size 4
--dp-size 4
--enable-dp-attention
--enable-dp-lm-head
--mem-fraction-static 0.75
--attention-backend dsv4
--moe-runner-backend flashinfer_mxfp4
--cuda-graph-max-bs-decode 64
--reasoning-parser auto
--tool-call-parser auto
--tokenizer-worker-num 8
--host 0.0.0.0
--port 30001
--enable-cache-report
--speculative-algorithm DSPARK
--speculative-dspark-block-size 5
--disaggregation-mode decode
--disaggregation-transfer-backend mooncake
--json-model-override-args '{"vision_n_layers": 0}'
-55
View File
@@ -1,55 +0,0 @@
services:
sglang:
image: ymkymx/sglang:dsv41-pd-c74a4037f-20260921-1300
container_name: ds41-pd-prefill-dp
shm_size: "32gb"
ipc: host
pid: host
privileged: true
network_mode: host
environment:
- CUDA_VISIBLE_DEVICES=0,1,2,3
- SGLANG_RAGGED_VERIFY_MODE=static
- MC_INTRANODE_NVLINK=true
- MC_INTRA_NVLINK=true
- SGLANG_MOONCAKE_SEND_AUX_TCP=1
volumes:
- /data:/data
- /data/ymk/cache/sglang:/root/.cache/sglang
deploy:
resources:
reservations:
devices:
- driver: nvidia
count: all
capabilities: [gpu]
command: >-
sglang serve
--trust-remote-code
--model-path /data/models/DeepSeek-V4.1-Flash
--tp 4
--ep-size 4
--dp-size 4
--enable-dp-attention
--enable-dp-lm-head
--load-balance-method total_tokens
--mem-fraction-static 0.75
--attention-backend dsv4
--moe-runner-backend flashinfer_mxfp4
--cuda-graph-max-bs-decode 64
--reasoning-parser auto
--tool-call-parser auto
--max-running-requests 64
--host 0.0.0.0
--port 30010
--enable-cache-report
--speculative-algorithm DSPARK
--speculative-dspark-block-size 5
--enable-hierarchical-cache
--hicache-ratio 2.5
--hicache-write-policy write_back
--disaggregation-mode prefill
--disaggregation-bootstrap-port 8918
--disaggregation-transfer-backend mooncake
--json-model-override-args '{"vision_n_layers": 0}'
@@ -1,53 +0,0 @@
services:
sglang:
image: ymkymx/sglang:dsv41-pd-visioncp-db7d2cb7d-fix
container_name: ds41-pd-prefill
shm_size: "32gb"
ipc: host
pid: host
privileged: true
network_mode: host
environment:
- CUDA_VISIBLE_DEVICES=0,1,2,3
- SGLANG_RAGGED_VERIFY_MODE=static
- MC_INTRANODE_NVLINK=true
- MC_INTRA_NVLINK=true
- SGLANG_MOONCAKE_SEND_AUX_TCP=1
volumes:
- /data:/data
- /data/ymk/cache/sglang:/root/.cache/sglang
deploy:
resources:
reservations:
devices:
- driver: nvidia
count: all
capabilities: [gpu]
command: >-
sglang serve
--trust-remote-code
--model-path /data/models/DeepSeek-V4.1-Flash
--tp 4
--ep-size 4
--mem-fraction-static 0.75
--attention-backend dsv4
--moe-runner-backend flashinfer_mxfp4
--cuda-graph-max-bs-decode 64
--reasoning-parser auto
--tool-call-parser auto
--default-chat-template-kwargs '{"thinking": true, "reasoning_effort": "high"}'
--tokenizer-worker-num 8
--host 0.0.0.0
--port 30000
--enable-cache-report
--enable-metrics
--speculative-algorithm DSPARK
--speculative-dspark-block-size 5
--enable-hierarchical-cache
--hicache-ratio 2.5
--hicache-write-policy write_back
--enable-prefill-cp
--cp-strategy interleave
--disaggregation-mode prefill
--disaggregation-transfer-backend mooncake
-52
View File
@@ -1,52 +0,0 @@
services:
sglang:
image: ymkymx/sglang:dsv41-pd-ddf520763-local-20260923-0536
container_name: ds41-pd-prefill
shm_size: "32gb"
ipc: host
pid: host
privileged: true
network_mode: host
environment:
- CUDA_VISIBLE_DEVICES=0,1,2,3
- SGLANG_RAGGED_VERIFY_MODE=static
- MC_INTRANODE_NVLINK=true
- MC_INTRA_NVLINK=true
- SGLANG_MOONCAKE_SEND_AUX_TCP=1
volumes:
- /data:/data
- /data/ymk/cache/sglang:/root/.cache/sglang
deploy:
resources:
reservations:
devices:
- driver: nvidia
count: all
capabilities: [gpu]
command: >-
sglang serve
--trust-remote-code
--model-path /data/models/DeepSeek-V4.1-Flash
--tp 4
--ep-size 4
--mem-fraction-static 0.75
--attention-backend dsv4
--moe-runner-backend flashinfer_mxfp4
--cuda-graph-max-bs-decode 64
--reasoning-parser auto
--tool-call-parser auto
--tokenizer-worker-num 8
--host 0.0.0.0
--port 30000
--enable-cache-report
--speculative-algorithm DSPARK
--speculative-dspark-block-size 5
--enable-hierarchical-cache
--hicache-ratio 2.5
--hicache-write-policy write_back
--enable-prefill-cp
--cp-strategy interleave
--disaggregation-mode prefill
--disaggregation-transfer-backend mooncake
--json-model-override-args '{"vision_n_layers": 0}'
-57
View File
@@ -1,57 +0,0 @@
# PD2 二分 Round 2 D 侧(2026-09-24):干净 D + 嫌疑项②
# SGLANG_DISAGGREGATION_QUEUE_SIZE=16 + SGLANG_DISAGGREGATION_THREAD_POOL_SIZE=32
services:
sglang:
image: ymkymx/sglang:dsv41-pd-visioncp-db7d2cb7d-fix
container_name: ds41-pd2-decode
shm_size: "32gb"
ipc: host
pid: host
privileged: true
network_mode: host
environment:
- CUDA_VISIBLE_DEVICES=6,7
- SGLANG_RAGGED_VERIFY_MODE=static
- MC_INTRANODE_NVLINK=true
- MC_INTRA_NVLINK=true
- SGLANG_MOONCAKE_SEND_AUX_TCP=1
- SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE=1
- SGLANG_DISAGGREGATION_QUEUE_SIZE=16
- SGLANG_DISAGGREGATION_THREAD_POOL_SIZE=32
volumes:
- /data:/data
- /data/ymk/cache/sglang:/root/.cache/sglang
- /data/ymk/sglang/python/sglang/srt/managers/scheduler.py:/sgl-workspace/sglang/python/sglang/srt/managers/scheduler.py
deploy:
resources:
reservations:
devices:
- driver: nvidia
count: all
capabilities: [gpu]
command: >-
sglang serve
--trust-remote-code
--model-path /data/models/DeepSeek-V4.1-Flash
--tp 2
--ep-size 2
--dp-size 2
--enable-dp-attention
--enable-dp-lm-head
--mem-fraction-static 0.75
--attention-backend dsv4
--moe-runner-backend flashinfer_mxfp4
--cuda-graph-max-bs-decode 32
--reasoning-parser auto
--tool-call-parser auto
--default-chat-template-kwargs '{"thinking": true, "reasoning_effort": "high"}'
--tokenizer-worker-num 8
--host 0.0.0.0
--port 30030
--enable-cache-report
--enable-metrics
--speculative-algorithm DSPARK
--speculative-dspark-block-size 5
--disaggregation-mode decode
--disaggregation-transfer-backend mooncake
-54
View File
@@ -1,54 +0,0 @@
# PD 干净对照实验 D 节点(2026-09-24 上午):dp2 decode,配 dockerserve-pd2-p.yml 使用。
services:
sglang:
image: ymkymx/sglang:dsv41-pd-visioncp-db7d2cb7d-fix
container_name: ds41-pd2-decode
shm_size: "32gb"
ipc: host
pid: host
privileged: true
network_mode: host
environment:
- CUDA_VISIBLE_DEVICES=6,7
- SGLANG_RAGGED_VERIFY_MODE=static
- MC_INTRANODE_NVLINK=true
- MC_INTRA_NVLINK=true
- SGLANG_MOONCAKE_SEND_AUX_TCP=1
- SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE=1
volumes:
- /data:/data
- /data/ymk/cache/sglang:/root/.cache/sglang
- /data/ymk/sglang/python/sglang/srt/managers/scheduler.py:/sgl-workspace/sglang/python/sglang/srt/managers/scheduler.py
deploy:
resources:
reservations:
devices:
- driver: nvidia
count: all
capabilities: [gpu]
command: >-
sglang serve
--trust-remote-code
--model-path /data/models/DeepSeek-V4.1-Flash
--tp 2
--ep-size 2
--dp-size 2
--enable-dp-attention
--enable-dp-lm-head
--mem-fraction-static 0.75
--attention-backend dsv4
--moe-runner-backend flashinfer_mxfp4
--cuda-graph-max-bs-decode 32
--reasoning-parser auto
--tool-call-parser auto
--default-chat-template-kwargs '{"thinking": true, "reasoning_effort": "high"}'
--tokenizer-worker-num 8
--host 0.0.0.0
--port 30030
--enable-cache-report
--enable-metrics
--speculative-algorithm DSPARK
--speculative-dspark-block-size 5
--disaggregation-mode decode
--disaggregation-transfer-backend mooncake
-67
View File
@@ -1,67 +0,0 @@
# PD2 二分 Round 1(2026-09-24):干净基线 + 嫌疑项①「P 侧 L3 hicache 块 + mooncake-store.json」
# 相对 dockerserve-pd2-p.yml 的差异:
# env 加 SGLANG_HICACHE_MOONCAKE_CONFIG_PATH=/data/ymk/ds41/mooncake-store.json
# hicache 参数块换成坏部署同款:page_first_direct + direct + write_through + mooncake +
# wait_complete + size 0(替代 write_back L2)
# D 侧不变(坏部署 D 本无 L3)。探针:bs16 low_entropy decode 看 accept len。
services:
sglang:
image: ymkymx/sglang:dsv41-pd-visioncp-db7d2cb7d-fix
container_name: ds41-pd2-prefill
shm_size: "32gb"
ipc: host
pid: host
privileged: true
network_mode: host
environment:
- CUDA_VISIBLE_DEVICES=4,5
- SGLANG_RAGGED_VERIFY_MODE=static
- MC_INTRANODE_NVLINK=true
- MC_INTRA_NVLINK=true
- SGLANG_MOONCAKE_SEND_AUX_TCP=1
- SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE=1
- SGLANG_HICACHE_MOONCAKE_CONFIG_PATH=/data/ymk/ds41/mooncake-store.json
volumes:
- /data:/data
- /data/ymk/cache/sglang:/root/.cache/sglang
- /data/ymk/sglang/python/sglang/srt/managers/scheduler.py:/sgl-workspace/sglang/python/sglang/srt/managers/scheduler.py
deploy:
resources:
reservations:
devices:
- driver: nvidia
count: all
capabilities: [gpu]
command: >-
sglang serve
--trust-remote-code
--model-path /data/models/DeepSeek-V4.1-Flash
--tp 2
--ep-size 2
--mem-fraction-static 0.75
--attention-backend dsv4
--moe-runner-backend flashinfer_mxfp4
--cuda-graph-max-bs-decode 32
--reasoning-parser auto
--tool-call-parser auto
--default-chat-template-kwargs '{"thinking": true, "reasoning_effort": "high"}'
--tokenizer-worker-num 8
--host 0.0.0.0
--port 30020
--enable-cache-report
--enable-metrics
--speculative-algorithm DSPARK
--speculative-dspark-block-size 5
--enable-hierarchical-cache
--hicache-ratio 2
--hicache-mem-layout page_first_direct
--hicache-io-backend direct
--hicache-write-policy write_through
--hicache-storage-backend mooncake
--hicache-storage-prefetch-policy wait_complete
--hicache-size 0
--enable-prefill-cp
--cp-strategy interleave
--disaggregation-mode prefill
--disaggregation-transfer-backend mooncake
-65
View File
@@ -1,65 +0,0 @@
# PD2 二分 Round 2 P 侧(2026-09-24):R1(L3 块)之上再加嫌疑项②
# SGLANG_DISAGGREGATION_QUEUE_SIZE=16 + SGLANG_DISAGGREGATION_THREAD_POOL_SIZE=32
services:
sglang:
image: ymkymx/sglang:dsv41-pd-visioncp-db7d2cb7d-fix
container_name: ds41-pd2-prefill
shm_size: "32gb"
ipc: host
pid: host
privileged: true
network_mode: host
environment:
- CUDA_VISIBLE_DEVICES=4,5
- SGLANG_RAGGED_VERIFY_MODE=static
- MC_INTRANODE_NVLINK=true
- MC_INTRA_NVLINK=true
- SGLANG_MOONCAKE_SEND_AUX_TCP=1
- SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE=1
- SGLANG_HICACHE_MOONCAKE_CONFIG_PATH=/data/ymk/ds41/mooncake-store.json
- SGLANG_DISAGGREGATION_QUEUE_SIZE=16
- SGLANG_DISAGGREGATION_THREAD_POOL_SIZE=32
volumes:
- /data:/data
- /data/ymk/cache/sglang:/root/.cache/sglang
- /data/ymk/sglang/python/sglang/srt/managers/scheduler.py:/sgl-workspace/sglang/python/sglang/srt/managers/scheduler.py
deploy:
resources:
reservations:
devices:
- driver: nvidia
count: all
capabilities: [gpu]
command: >-
sglang serve
--trust-remote-code
--model-path /data/models/DeepSeek-V4.1-Flash
--tp 2
--ep-size 2
--mem-fraction-static 0.75
--attention-backend dsv4
--moe-runner-backend flashinfer_mxfp4
--cuda-graph-max-bs-decode 32
--reasoning-parser auto
--tool-call-parser auto
--default-chat-template-kwargs '{"thinking": true, "reasoning_effort": "high"}'
--tokenizer-worker-num 8
--host 0.0.0.0
--port 30020
--enable-cache-report
--enable-metrics
--speculative-algorithm DSPARK
--speculative-dspark-block-size 5
--enable-hierarchical-cache
--hicache-ratio 2
--hicache-mem-layout page_first_direct
--hicache-io-backend direct
--hicache-write-policy write_through
--hicache-storage-backend mooncake
--hicache-storage-prefetch-policy wait_complete
--hicache-size 0
--enable-prefill-cp
--cp-strategy interleave
--disaggregation-mode prefill
--disaggregation-transfer-backend mooncake
-59
View File
@@ -1,59 +0,0 @@
# PD 干净对照实验(2026-09-24 上午):验证「cp2-P → dp2-D」传输路径下 dspark 是否正常。
# 用户疑问:单机六形态健康不等于 PD 链路健康,可能两边传输没对齐。
# 以已验证可用的 dockerserve-pd-{p,d}-vision.yml 为底,缩到 2+2 卡;P 加 engram host table
# (2 卡权重放不下,见 G1.9)。P=GPU4-5 端口 30020,D=GPU6-7 端口 30030,mini_lb=30004。
services:
sglang:
image: ymkymx/sglang:dsv41-pd-visioncp-db7d2cb7d-fix
container_name: ds41-pd2-prefill
shm_size: "32gb"
ipc: host
pid: host
privileged: true
network_mode: host
environment:
- CUDA_VISIBLE_DEVICES=4,5
- SGLANG_RAGGED_VERIFY_MODE=static
- MC_INTRANODE_NVLINK=true
- MC_INTRA_NVLINK=true
- SGLANG_MOONCAKE_SEND_AUX_TCP=1
- SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE=1
volumes:
- /data:/data
- /data/ymk/cache/sglang:/root/.cache/sglang
- /data/ymk/sglang/python/sglang/srt/managers/scheduler.py:/sgl-workspace/sglang/python/sglang/srt/managers/scheduler.py
deploy:
resources:
reservations:
devices:
- driver: nvidia
count: all
capabilities: [gpu]
command: >-
sglang serve
--trust-remote-code
--model-path /data/models/DeepSeek-V4.1-Flash
--tp 2
--ep-size 2
--mem-fraction-static 0.75
--attention-backend dsv4
--moe-runner-backend flashinfer_mxfp4
--cuda-graph-max-bs-decode 32
--reasoning-parser auto
--tool-call-parser auto
--default-chat-template-kwargs '{"thinking": true, "reasoning_effort": "high"}'
--tokenizer-worker-num 8
--host 0.0.0.0
--port 30020
--enable-cache-report
--enable-metrics
--speculative-algorithm DSPARK
--speculative-dspark-block-size 5
--enable-hierarchical-cache
--hicache-ratio 2.5
--hicache-write-policy write_back
--enable-prefill-cp
--cp-strategy interleave
--disaggregation-mode prefill
--disaggregation-transfer-backend mooncake
@@ -1,42 +0,0 @@
services:
sglang:
image: ymkymx/sglang:main-ee5fcdf0d-20260920-0142
container_name: ds41-replay-off
shm_size: "32gb"
ipc: host
privileged: true
ports:
- "30000:30000"
environment:
- CUDA_VISIBLE_DEVICES=0,1,2,3
volumes:
- /data:/data
- /data/ymk/cache/sglang:/root/.cache/sglang
deploy:
resources:
reservations:
devices:
- driver: nvidia
count: all
capabilities: [gpu]
command: >-
sglang serve
--trust-remote-code
--model-path /data/models/DeepSeek-V4.1-Flash
--tp 4
--ep-size 4
--mem-fraction-static 0.75
--attention-backend dsv4
--moe-runner-backend flashinfer_mxfp4
--cuda-graph-max-bs-decode 64
--reasoning-parser auto
--tool-call-parser auto
--max-running-requests 64
--host 0.0.0.0
--port 30000
--enable-cache-report
--enable-metrics
--speculative-algorithm DSPARK
--speculative-dspark-block-size 5
--json-model-override-args '{"vision_n_layers": 0}'
@@ -1,42 +0,0 @@
services:
sglang:
image: ymkymx/sglang:main-ee5fcdf0d-20260920-0142
container_name: ds41-replay-on
shm_size: "32gb"
ipc: host
privileged: true
ports:
- "30000:30000"
environment:
- CUDA_VISIBLE_DEVICES=0,1,2,3
volumes:
- /data:/data
- /data/ymk/cache/sglang:/root/.cache/sglang
deploy:
resources:
reservations:
devices:
- driver: nvidia
count: all
capabilities: [gpu]
command: >-
sglang serve
--trust-remote-code
--model-path /data/models/DeepSeek-V4.1-Flash
--tp 4
--ep-size 4
--mem-fraction-static 0.75
--attention-backend dsv4
--moe-runner-backend flashinfer_mxfp4
--cuda-graph-max-bs-decode 64
--reasoning-parser auto
--tool-call-parser auto
--max-running-requests 64
--host 0.0.0.0
--port 30000
--enable-cache-report
--enable-metrics
--speculative-algorithm DSPARK
--speculative-dspark-block-size 5
--enable-decoder-swa-bounded-replay
--json-model-override-args '{"vision_n_layers": 0}'
-92
View File
@@ -1,92 +0,0 @@
# 纯 tp2 单实例对照(2026-09-24 早):与 dockerserve-cp2x4.yml 逐参数对齐,唯一差异 = 去掉
# --enable-prefill-cp --cp-strategy interleave(即无 CP),用于测「tp2 vs cp2」的 prefill 效率差。
# GPU 4-5(a)/6-7(b),端口 30020/30030;2×tp2 聚合经 rr router 30003。
# 镜像/补丁挂载/L3/engram/tok8 全部与 cp2x4 相同。
x-common: &common
image: ymkymx/sglang:dsv41-pd-visioncp-db7d2cb7d-fix
shm_size: "32gb"
ipc: host
privileged: true
network_mode: host
volumes:
- /data:/data
- /data/ymk/cache/sglang:/root/.cache/sglang
# 与 cp2x4 保持同一份 scheduler.py(含未 commit 的 CP idle 补丁),保证唯一变量是 CP 开关
- /data/ymk/sglang/python/sglang/srt/managers/scheduler.py:/sgl-workspace/sglang/python/sglang/srt/managers/scheduler.py
environment: &env
SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE: "1"
SGLANG_RAGGED_VERIFY_MODE: static
MC_MS_AUTO_DISC: "0"
MOONCAKE_MASTER: 127.0.0.1:50051
MOONCAKE_TE_META_DATA_SERVER: P2PHANDSHAKE
MOONCAKE_PROTOCOL: tcp
MOONCAKE_GLOBAL_SEGMENT_SIZE: 300gb
deploy:
resources:
reservations:
devices:
- driver: nvidia
count: all
capabilities: [gpu]
services:
tp2:
<<: *common
container_name: ds41-tp2
environment:
<<: *env
CUDA_VISIBLE_DEVICES: "4,5"
command: >
sglang serve
--trust-remote-code
--model-path /data/models/DeepSeek-V4.1-Flash
--tp 2 --ep-size 2
--mem-fraction-static 0.75
--attention-backend dsv4
--moe-runner-backend flashinfer_mxfp4
--cuda-graph-max-bs-decode 32
--reasoning-parser auto --tool-call-parser auto
--default-chat-template-kwargs '{"thinking": true, "reasoning_effort": "high"}'
--max-running-requests 64
--tokenizer-worker-num 8
--host 0.0.0.0 --port 30020
--enable-cache-report --enable-metrics
--speculative-algorithm DSPARK --speculative-dspark-block-size 5
--enable-hierarchical-cache
--hicache-ratio 2
--hicache-mem-layout page_first_direct
--hicache-io-backend direct
--hicache-write-policy write_through
--hicache-storage-backend mooncake
--hicache-storage-prefetch-policy wait_complete
--hicache-size 0
b:
<<: *common
container_name: ds41-tp2-b
environment:
<<: *env
CUDA_VISIBLE_DEVICES: "6,7"
command: >
sglang serve
--trust-remote-code
--model-path /data/models/DeepSeek-V4.1-Flash
--tp 2 --ep-size 2
--mem-fraction-static 0.75
--attention-backend dsv4
--moe-runner-backend flashinfer_mxfp4
--cuda-graph-max-bs-decode 32
--reasoning-parser auto --tool-call-parser auto
--default-chat-template-kwargs '{"thinking": true, "reasoning_effort": "high"}'
--max-running-requests 64
--tokenizer-worker-num 8
--host 0.0.0.0 --port 30030
--enable-cache-report --enable-metrics
--speculative-algorithm DSPARK --speculative-dspark-block-size 5
--enable-hierarchical-cache
--hicache-ratio 2
--hicache-mem-layout page_first_direct
--hicache-io-backend direct
--hicache-write-policy write_through
--hicache-storage-backend mooncake
--hicache-storage-prefetch-policy wait_complete
--hicache-size 0
-43
View File
@@ -1,43 +0,0 @@
services:
sglang:
image: uhub.service.ucloud.cn/umirror/sglang:dev-dsv41
shm_size: "32gb"
ipc: host
privileged: true
ports:
- "30000:30000"
volumes:
- /data:/data
deploy:
resources:
reservations:
devices:
- driver: nvidia
count: all
capabilities: [gpu]
command: >-
sglang serve
--trust-remote-code
--model-path /data/models/DeepSeek-V4.1-Flash
--tp 4
--ep-size 4
--mem-fraction-static 0.75
--attention-backend dsv4
--moe-runner-backend flashinfer_mxfp4
--enable-decoder-swa-bounded-replay
--cuda-graph-max-bs-decode 64
--reasoning-parser auto
--tool-call-parser auto
--max-running-requests 64
--host 0.0.0.0
--port 30000
--enable-cache-report
--enable-metrics
--enable-prefill-cp
--cp-strategy interleave
--json-model-override-args '{"vision_n_layers": 0}'
-451
View File
@@ -1,451 +0,0 @@
# ds41-3p1d: DeepSeek-V4.1-Flash 3P1D PD 分离(cs32k + OTLP trace),RBG 版
# 拓扑: 3×prefill(tp2+ep2+cp2, cs32k, dspark, hicache L3) + 1×decode(tp2+ep2+dp2, cg512, dspark,
# load-balance-method=total_requests 按请求数分流) + mooncake-master(hicache L3 store) + router(mini-lb PD)
# 全部钉 b300-01(单机 8 卡,hostNetwork + hostPID——mooncake NVLink-intra 跨 Pod CUDA IPC 必需,
# 等同 docker compose 的 pid: host;参考 AGENTS.md「部署四要素」)
# 镜像已侧载 containerd: docker.io/ymkymx/sglang:dsv41-pd-visioncp-db7d2cb7d-fix(IfNotPresent)
# 参数来源: deploy/b300-ds41/dockerserve-3p1d-otlp-cs32k.yml(docker compose 版)
# 对外: 仅 ClusterIP Service ds41-3p1d:50002(router 用普通 Pod 网络,不占宿主机端口,
# 天然不对宿主机/外网暴露);new-api-gw 用集群 DNS http://ds41-3p1d:50002/v1 访问
# trace: OTLP → 127.0.0.1:4317(同节点 jaeger-trace cw,badger 6h 轮窗)
apiVersion: workloads.x-k8s.io/v1alpha2
kind: RoleBasedGroup
metadata:
name: ds41-3p1d
labels:
app: ds41-3p1d
sglang-model: ds-v41-flash
spec:
roleTemplates:
- name: prefill-base
template:
metadata:
labels:
sglang-model: ds-v41-flash
sglang-role: prefill
spec:
hostNetwork: true
hostPID: true
dnsPolicy: ClusterFirstWithHostNet
nodeSelector:
kubernetes.io/hostname: b300-01
containers:
- name: sglang
image: ymkymx/sglang:dsv41-pd-visioncp-db7d2cb7d-fix
imagePullPolicy: IfNotPresent
command:
- /bin/bash
- -c
- |
exec sglang serve \
--trust-remote-code \
--model-path /data/models/DeepSeek-V4.1-Flash \
--tp 2 --ep-size 2 \
--enable-prefill-cp --cp-strategy interleave \
--chunked-prefill-size 32768 --max-prefill-tokens 32768 \
--mem-fraction-static 0.75 \
--attention-backend dsv4 \
--moe-runner-backend flashinfer_mxfp4 \
--cuda-graph-max-bs-decode 32 \
--reasoning-parser auto --tool-call-parser auto \
--default-chat-template-kwargs '{"thinking": true, "reasoning_effort": "high"}' \
--max-running-requests 256 \
--tokenizer-worker-num 8 \
--host 0.0.0.0 --port "$SGLANG_PORT" \
--enable-cache-report --enable-metrics \
--speculative-algorithm DSPARK --speculative-dspark-block-size 5 \
--enable-hierarchical-cache \
--hicache-ratio 2 \
--hicache-mem-layout page_first_direct \
--hicache-io-backend direct \
--hicache-write-policy write_through \
--hicache-storage-backend mooncake \
--hicache-storage-prefetch-policy wait_complete \
--hicache-size 0 \
--disaggregation-mode prefill \
--disaggregation-transfer-backend mooncake \
--disaggregation-bootstrap-port "$SGLANG_BOOTSTRAP_PORT" \
--enable-trace \
--trace-modules request,mooncake \
--otlp-traces-endpoint 127.0.0.1:4317 \
--otlp-service-name "$OTLP_NAME"
env:
- name: SGLANG_TORCH_PROFILER_DIR
value: /data/ymk/prof/torch
- name: SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE
value: "1"
- name: SGLANG_RAGGED_VERIFY_MODE
value: static
- name: MC_INTRANODE_NVLINK
value: "true"
- name: MC_INTRA_NVLINK
value: "true"
- name: SGLANG_MOONCAKE_SEND_AUX_TCP
value: "1"
- name: SGLANG_HICACHE_MOONCAKE_CONFIG_PATH
value: /data/ymk/ds41/mooncake-store.json
- name: SGLANG_DISAGGREGATION_QUEUE_SIZE
value: "16"
- name: SGLANG_DISAGGREGATION_THREAD_POOL_SIZE
value: "32"
- name: TZ
value: Asia/Shanghai
readinessProbe:
httpGet:
path: /health
port: metrics
initialDelaySeconds: 60
periodSeconds: 10
timeoutSeconds: 5
failureThreshold: 90
resources:
requests:
cpu: "32"
memory: 220Gi
nvidia.com/gpu: "2"
limits:
cpu: "96"
memory: 950Gi
nvidia.com/gpu: "2"
securityContext:
privileged: true
volumeMounts:
- name: data
mountPath: /data
- name: sglang-cache
mountPath: /root/.cache/sglang
- name: dshm
mountPath: /dev/shm
volumes:
- name: data
hostPath:
path: /data
type: Directory
- name: sglang-cache
hostPath:
path: /data/ymk/cache/sglang
type: Directory
- name: dshm
emptyDir:
medium: Memory
sizeLimit: 32Gi
- name: decode-base
template:
metadata:
labels:
sglang-model: ds-v41-flash
sglang-role: decode
spec:
hostNetwork: true
hostPID: true
dnsPolicy: ClusterFirstWithHostNet
nodeSelector:
kubernetes.io/hostname: b300-01
containers:
- name: sglang
image: ymkymx/sglang:dsv41-pd-visioncp-db7d2cb7d-fix
imagePullPolicy: IfNotPresent
command:
- /bin/bash
- -c
- |
exec sglang serve \
--trust-remote-code \
--model-path /data/models/DeepSeek-V4.1-Flash \
--tp 2 --ep-size 2 --dp-size 2 \
--enable-dp-attention --enable-dp-lm-head \
--mem-fraction-static 0.80 \
--attention-backend dsv4 \
--moe-runner-backend flashinfer_mxfp4 \
--cuda-graph-max-bs-decode 512 \
--load-balance-method total_requests \
--reasoning-parser auto --tool-call-parser auto \
--default-chat-template-kwargs '{"thinking": true, "reasoning_effort": "high"}' \
--max-running-requests 256 \
--tokenizer-worker-num 8 \
--host 0.0.0.0 --port 30001 \
--enable-cache-report --enable-metrics \
--speculative-algorithm DSPARK --speculative-dspark-block-size 5 \
--disaggregation-mode decode \
--disaggregation-transfer-backend mooncake \
--disaggregation-bootstrap-port 9001 \
--enable-trace \
--trace-modules request,mooncake \
--otlp-traces-endpoint 127.0.0.1:4317 \
--otlp-service-name ds41-3p1d-d1
env:
- name: SGLANG_TORCH_PROFILER_DIR
value: /data/ymk/prof/torch
- name: SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE
value: "1"
- name: SGLANG_RAGGED_VERIFY_MODE
value: static
- name: MC_INTRANODE_NVLINK
value: "true"
- name: MC_INTRA_NVLINK
value: "true"
- name: SGLANG_MOONCAKE_SEND_AUX_TCP
value: "1"
- name: SGLANG_DISAGGREGATION_QUEUE_SIZE
value: "16"
- name: SGLANG_DISAGGREGATION_THREAD_POOL_SIZE
value: "32"
- name: TZ
value: Asia/Shanghai
ports:
- name: metrics
containerPort: 30001
readinessProbe:
httpGet:
path: /health
port: metrics
initialDelaySeconds: 60
periodSeconds: 10
timeoutSeconds: 5
failureThreshold: 90
resources:
requests:
cpu: "32"
memory: 100Gi
nvidia.com/gpu: "2"
limits:
cpu: "96"
memory: 600Gi
nvidia.com/gpu: "2"
securityContext:
privileged: true
volumeMounts:
- name: data
mountPath: /data
- name: sglang-cache
mountPath: /root/.cache/sglang
- name: dshm
mountPath: /dev/shm
volumes:
- name: data
hostPath:
path: /data
type: Directory
- name: sglang-cache
hostPath:
path: /data/ymk/cache/sglang
type: Directory
- name: dshm
emptyDir:
medium: Memory
sizeLimit: 32Gi
roles:
- name: mooncake-master
replicas: 1
standalonePattern:
template:
metadata:
labels:
app: ds41-3p1d-mooncake-master
spec:
hostNetwork: true
dnsPolicy: ClusterFirstWithHostNet
nodeSelector:
kubernetes.io/hostname: b300-01
containers:
- name: mooncake-master
image: ymkymx/sglang:dsv41-pd-visioncp-db7d2cb7d-fix
imagePullPolicy: IfNotPresent
command:
- /bin/bash
- -c
- |
exec /opt/sglang/bin/mooncake_master \
--port 50051 \
--enable_http_metadata_server=true \
--http_metadata_server_port=18080 \
--eviction_high_watermark_ratio=0.90
readinessProbe:
tcpSocket:
port: 50051
initialDelaySeconds: 5
periodSeconds: 5
failureThreshold: 60
resources:
requests:
cpu: "1"
memory: 8Gi
limits:
cpu: "4"
memory: 64Gi
- name: p1
replicas: 1
dependencies:
- mooncake-master
standalonePattern:
templateRef:
name: prefill-base
patch:
metadata:
labels:
app: ds41-3p1d-p1
spec:
containers:
- name: sglang
env:
- name: CUDA_VISIBLE_DEVICES
value: "0,1"
- name: SGLANG_PORT
value: "30000"
- name: SGLANG_BOOTSTRAP_PORT
value: "8998"
- name: OTLP_NAME
value: ds41-3p1d-p1
ports:
- name: metrics
containerPort: 30000
- name: p2
replicas: 1
dependencies:
- mooncake-master
standalonePattern:
templateRef:
name: prefill-base
patch:
metadata:
labels:
app: ds41-3p1d-p2
spec:
containers:
- name: sglang
env:
- name: CUDA_VISIBLE_DEVICES
value: "2,3"
- name: SGLANG_PORT
value: "30010"
- name: SGLANG_BOOTSTRAP_PORT
value: "8999"
- name: OTLP_NAME
value: ds41-3p1d-p2
ports:
- name: metrics
containerPort: 30010
- name: p3
replicas: 1
dependencies:
- mooncake-master
standalonePattern:
templateRef:
name: prefill-base
patch:
metadata:
labels:
app: ds41-3p1d-p3
spec:
containers:
- name: sglang
env:
- name: CUDA_VISIBLE_DEVICES
value: "4,5"
- name: SGLANG_PORT
value: "30020"
- name: SGLANG_BOOTSTRAP_PORT
value: "9000"
- name: OTLP_NAME
value: ds41-3p1d-p3
ports:
- name: metrics
containerPort: 30020
- name: d1
replicas: 1
dependencies:
- mooncake-master
standalonePattern:
templateRef:
name: decode-base
patch:
metadata:
labels:
app: ds41-3p1d-d1
spec:
containers:
- name: sglang
env:
- name: CUDA_VISIBLE_DEVICES
value: "6,7"
- name: router
replicas: 1
dependencies:
- p1
- p2
- p3
- d1
standalonePattern:
template:
metadata:
labels:
app: ds41-3p1d-router
spec:
# 入口不走 hostNetwork:普通 Pod 网络 + ClusterIP,天然不对宿主机/外网暴露。
# P/D 地址用节点内网 IP(hostNetwork 监听在 192.168.1.243 上,Calico 可达)。
nodeSelector:
kubernetes.io/hostname: b300-01
containers:
- name: router
image: ymkymx/sglang:dsv41-pd-visioncp-db7d2cb7d-fix
imagePullPolicy: IfNotPresent
command:
- /bin/bash
- -c
- |
exec sglang-router launch \
--mini-lb --pd-disaggregation \
--host 0.0.0.0 --port 50002 \
--prefill http://192.168.1.243:30000 8998 \
--prefill http://192.168.1.243:30010 8999 \
--prefill http://192.168.1.243:30020 9000 \
--decode http://192.168.1.243:30001
ports:
- name: http
containerPort: 50002
readinessProbe:
httpGet:
path: /health
port: http
initialDelaySeconds: 5
periodSeconds: 10
timeoutSeconds: 5
failureThreshold: 30
resources:
requests:
cpu: "1"
memory: 4Gi
limits:
cpu: "8"
memory: 16Gi
---
apiVersion: v1
kind: Service
metadata:
name: ds41-3p1d
labels:
app: ds41-3p1d
spec:
type: ClusterIP
selector:
rbg.workloads.x-k8s.io/group-name: ds41-3p1d
rbg.workloads.x-k8s.io/role-name: router
ports:
- name: http
port: 50002
targetPort: 50002
---
apiVersion: monitoring.coreos.com/v1
kind: PodMonitor
metadata:
name: ds41-3p1d
labels:
release: kube-prometheus-stack
spec:
selector:
matchLabels:
sglang-model: ds-v41-flash
podMetricsEndpoints:
- interval: 15s
path: /metrics
port: metrics
-6
View File
@@ -1,6 +0,0 @@
{
"master_server_address": "127.0.0.1:50051",
"metadata_server": "P2PHANDSHAKE",
"global_segment_size": "300gb",
"protocol": "tcp"
}
@@ -181,15 +181,12 @@ def validate_deepseek_v41_features(server_args: ServerArgs) -> None:
) )
cfg = resolving_view(server_args) cfg = resolving_view(server_args)
hf_config = model_config_of(server_args).hf_config if model_config_of(server_args).hf_config.model_type != "deepseek_v41":
if hf_config.model_type != "deepseek_v41":
if cfg.enable_encoder_swa_bounded_replay: if cfg.enable_encoder_swa_bounded_replay:
raise ValueError( raise ValueError(
"--enable-encoder-swa-bounded-replay requires DeepSeek-V4.1" "--enable-encoder-swa-bounded-replay requires DeepSeek-V4.1"
) )
return return
if hf_config.vision_n_layers > 0 and cfg.enable_prefill_cp:
_validate_deepseek_v41_vision_prefill_cp(server_args)
if cfg.enable_encoder_swa_bounded_replay: if cfg.enable_encoder_swa_bounded_replay:
from sglang.srt.model_executor.cuda_graph_config import Backend from sglang.srt.model_executor.cuda_graph_config import Backend
@@ -200,8 +197,7 @@ def validate_deepseek_v41_features(server_args: ServerArgs) -> None:
cfg.cuda_graph_config.prefill.backend != Backend.DISABLED, cfg.cuda_graph_config.prefill.backend != Backend.DISABLED,
), ),
("DP attention", cfg.enable_dp_attention), ("DP attention", cfg.enable_dp_attention),
# Prefill CP declares attn_cp_size and DP attention only later. ("context parallelism", cfg.attn_cp_size > 1),
("context parallelism", cfg.attn_cp_size > 1 or cfg.enable_prefill_cp),
("external cache linker", cfg.enable_unified_cache_external_linker), ("external cache linker", cfg.enable_unified_cache_external_linker),
("unified memory", cfg.enable_unified_memory), ("unified memory", cfg.enable_unified_memory),
("PD disaggregation", cfg.disaggregation_mode != "null"), ("PD disaggregation", cfg.disaggregation_mode != "null"),
@@ -253,33 +249,23 @@ def validate_deepseek_v41_features(server_args: ServerArgs) -> None:
if ( if (
read_ragged_verify_mode() is not RaggedVerifyMode.STATIC read_ragged_verify_mode() is not RaggedVerifyMode.STATIC
or cfg.disaggregation_transfer_backend != "mooncake" or cfg.disaggregation_transfer_backend != "mooncake"
or cfg.dp_size != 1
or cfg.enable_dp_attention
or cfg.attn_cp_size != 1 or cfg.attn_cp_size != 1
or cfg.dcp_size != 1 or cfg.dcp_size != 1
): ):
raise ValueError( raise ValueError(
"DeepSeek-V4.1 DSpark PD requires static verify, Mooncake, " "DeepSeek-V4.1 DSpark PD requires static verify, Mooncake, "
"and CP=1 on both servers. DP attention is supported when " "DP=1 and CP=1. Both servers must enable DSpark with the same "
"both servers use the same block size and target/draft KV layout." "block size and TP size."
) )
from sglang.srt.model_executor.cuda_graph_config import Backend, Phase, with_phase from sglang.srt.model_executor.cuda_graph_config import Backend, Phase, with_phase
prefill_graph = cfg.cuda_graph_config.prefill prefill_graph = cfg.cuda_graph_config.prefill
cp_breakable_prefill = ( if prefill_graph.backend != Backend.DISABLED and prefill_graph.max_seq_len is None:
cfg.enable_prefill_cp # The captured low-ratio indexer scores a static context width; 16k
and cfg.cp_strategy == "interleave" # keeps it inside the candidate window at under 1 ms per layer.
and cfg.tp_size > 1
and prefill_graph.backend == Backend.BREAKABLE
)
if (
prefill_graph.backend != Backend.DISABLED
and prefill_graph.max_seq_len is None
and not cp_breakable_prefill
):
# The non-CP captured low-ratio indexer scores a static context width.
# CP BCG runs these sources eagerly with live prefix metadata, so this
# default would only force long-prefix CP batches back to eager.
# Explicit max_seq_len values still constrain both paths.
declare_resolution( declare_resolution(
server_args, server_args,
"validate_deepseek_v41_features", "validate_deepseek_v41_features",
@@ -310,40 +296,3 @@ def validate_deepseek_v41_features(server_args: ServerArgs) -> None:
"--enable-decoder-swa-bounded-replay cannot be combined with " "--enable-decoder-swa-bounded-replay cannot be combined with "
f"{feature} yet; disable one of them." f"{feature} yet; disable one of them."
) )
def _validate_deepseek_v41_vision_prefill_cp(server_args: ServerArgs) -> None:
from sglang.srt.model_executor.cuda_graph_config import Backend, Phase, with_phase
cfg = resolving_view(server_args)
if cfg.cp_strategy != "interleave":
raise ValueError(
"DeepSeek-V4.1 vision with prefill CP requires --cp-strategy "
f"interleave; got {cfg.cp_strategy!r}."
)
if cfg.cuda_graph_config.prefill.backend != Backend.DISABLED:
# The CP runner merges image features eagerly; no capture path replays it.
locked = getattr(server_args, "_cuda_graph_config_locked", set())
if (Phase.PREFILL, "backend") in locked:
raise ValueError(
"DeepSeek-V4.1 vision with prefill CP runs eager prefill; remove "
"the explicit prefill CUDA graph backend."
)
declare_resolution(
server_args,
"validate_deepseek_v41_features",
cuda_graph_config=with_phase(
cfg.cuda_graph_config, Phase.PREFILL, backend=Backend.DISABLED
),
)
logger.warning(
"Disabling the prefill CUDA graph for DeepSeek-V4.1 vision with prefill CP."
)
if (
str(cfg.speculative_algorithm).upper() == "DSPARK"
and cfg.enable_decoder_swa_bounded_replay
):
raise ValueError(
"DeepSeek-V4.1 vision with prefill CP does not support DSpark together "
"with --enable-decoder-swa-bounded-replay yet."
)
+2 -4
View File
@@ -966,14 +966,12 @@ def handle_language_model_only(server_args: Any):
): ):
if flag: if flag:
raise ValueError(f"--language-model-only cannot be combined with {name}") raise ValueError(f"--language-model-only cannot be combined with {name}")
hf_config = model_config_of(server_args).hf_config if cfg.disaggregation_mode != "null":
# V4.1 text-only workers use the standard PD KV transfer path.
if cfg.disaggregation_mode != "null" and hf_config.model_type != "deepseek_v41":
raise ValueError( raise ValueError(
"--language-model-only is incompatible with --disaggregation-mode " "--language-model-only is incompatible with --disaggregation-mode "
"prefill/decode" "prefill/decode"
) )
architectures = hf_config.architectures architectures = model_config_of(server_args).hf_config.architectures
if not any( if not any(
a in server_args.LANGUAGE_MODEL_ONLY_ARCHITECTURES for a in architectures a in server_args.LANGUAGE_MODEL_ONLY_ARCHITECTURES for a in architectures
): ):
@@ -945,33 +945,9 @@ class CommonKVManager(BaseKVManager):
"enable DSpark with the same block size and target/draft KV " "enable DSpark with the same block size and target/draft KV "
"layout. Upgrade both servers together." "layout. Upgrade both servers together."
) )
same_tp_with_prefill_cp = ( if info.attn_tp_size != self.attn_tp_size:
info.attn_cp_size > 1
and (self.is_mla_backend or self.is_hybrid_mla_backend)
and self.attn_cp_size == 1
and info.attn_tp_size * info.attn_cp_size == self.attn_tp_size
)
# Combined branch (40323-series + 40177): prefill CP can also pair
# with a DP-attention decode server. MLA KV is replicated across
# prefill CP ranks, so per-rank layouts match when attn_tp matches.
dp_decode_with_prefill_cp = (
info.attn_cp_size > 1
and self.attn_cp_size == 1
and (self.is_mla_backend or self.is_hybrid_mla_backend)
and info.attn_tp_size == self.attn_tp_size
)
non_cp_mla_layout = info.attn_cp_size == self.attn_cp_size == 1 and (
self.is_mla_backend or self.is_hybrid_mla_backend
)
if info.attn_tp_size != self.attn_tp_size and not (
same_tp_with_prefill_cp
or dp_decode_with_prefill_cp
or non_cp_mla_layout
):
raise RuntimeError( raise RuntimeError(
"DeepSeek-V4.1 DSpark PD requires matching attention TP " "DeepSeek-V4.1 DSpark PD requires the same TP size on both servers"
"unless both servers use CP=1 with an MLA KV layout, "
"or prefill runs CP with an MLA KV layout"
) )
if self.dcp_size > 1: if self.dcp_size > 1:
@@ -82,115 +82,6 @@ FAILED_SESSION_RECOVERIES = Counter(
) )
# ---------------------------------------------------------------------------
# Intra-node NVLink transport helpers.
#
# Mooncake's IntraNodeNvlinkTransport can only register and reach *device*
# memory (it IPC-opens the remote cudaMalloc segments). Host-resident regions
# (aux buffers, some state components) cannot be registered: one host region
# makes the whole registerLocalMemoryBatch fail, and the engine then rolls
# back *every* region, leaving the segment descriptor empty and all KV
# transfers failing with "Requested address ... not found". When the
# intra-node NVLink transport is active we therefore
# 1. register only device-memory regions, and
# 2. route blocks whose source is host memory over the ordered zmq channel
# (same ordering guarantee the aux TCP path relies on) instead of the
# transfer engine.
# ---------------------------------------------------------------------------
import ctypes as _ctypes
_CUDA_MEMORY_TYPE_DEVICE = 2
try:
from cuda.bindings import runtime as _cudart
except ImportError: # pragma: no cover - cuda-python is always present in images
_cudart = None
def _is_device_pointer(ptr: int) -> bool:
"""Probe a *local* pointer with cudaPointerGetAttributes.
Only valid for pointers owned by this process (never probe remote
segment addresses). Returns False on any error so the caller falls back
to the safe host path.
"""
if _cudart is None:
# Cannot tell; assume device so behavior stays unchanged.
return True
err, attr = _cudart.cudaPointerGetAttributes(int(ptr))
if int(err) != 0:
# Clear the error so subsequent CUDA calls are not poisoned.
_cudart.cudaGetLastError()
return False
return int(attr.type) == _CUDA_MEMORY_TYPE_DEVICE
def _read_bytes_from_address(addr: int, length: int) -> Optional[bytes]:
if length <= 0:
return b""
if _is_device_pointer(addr):
buf = bytearray(length)
# cudaMemcpyDeviceToHost = 2; synchronous default-stream copy.
err, = _cudart.cudaMemcpy(
_ctypes.addressof((_ctypes.c_char * length).from_buffer(buf)),
int(addr),
length,
2,
)
if int(err) != 0:
logger.error(
f"cudaMemcpy D2H failed (err={err}) for addr {hex(addr)} len {length}"
)
return None
return bytes(buf)
return _ctypes.string_at(int(addr), length)
def _write_bytes_to_address(addr: int, data: bytes) -> bool:
if not data:
return True
if _is_device_pointer(addr):
buf = _ctypes.create_string_buffer(data, len(data))
# cudaMemcpyHostToDevice = 1; synchronous default-stream copy.
err, = _cudart.cudaMemcpy(
int(addr), _ctypes.addressof(buf), len(data), 1
)
if int(err) != 0:
logger.error(
f"cudaMemcpy H2D failed (err={err}) for addr {hex(addr)} "
f"len {len(data)}"
)
return False
return True
_ctypes.memmove(int(addr), data, len(data))
return True
_NVLINK_INTRA_ACTIVE = None
def _nvlink_intra_transport_active() -> bool:
"""Whether mooncake installed the intra-node NVLink transport.
Mirrors the env probing in mooncake's transfer_engine_impl.cpp: the
transport is installed iff MC_INTRANODE_NVLINK is set (any value), or an
equivalent protocol selection was made.
"""
global _NVLINK_INTRA_ACTIVE
if _NVLINK_INTRA_ACTIVE is None:
active = bool(
os.environ.get("MC_INTRANODE_NVLINK")
or os.environ.get("MC_INTRA_NVLINK")
)
if not active:
proto = (os.environ.get("MOONCAKE_PROTOCOL") or "").strip().lower()
active = proto in ("nvlink_intra", "nvlink-intra", "intra_nvlink")
_NVLINK_INTRA_ACTIVE = active
return _NVLINK_INTRA_ACTIVE
# decode # decode
@dataclasses.dataclass @dataclasses.dataclass
class TransferInfo: class TransferInfo:
@@ -323,7 +214,6 @@ class KVArgsRegisterInfo:
class MooncakeKVManager(StagingManagerMixin, CommonKVManager): class MooncakeKVManager(StagingManagerMixin, CommonKVManager):
AUX_DATA_HEADER = b"AUX_DATA" AUX_DATA_HEADER = b"AUX_DATA"
STATE_DATA_HEADER = b"STATE_DATA"
# Implements teardown() below, so runtime PD role switching is supported. # Implements teardown() below, so runtime PD role switching is supported.
supports_role_switch = True supports_role_switch = True
@@ -337,10 +227,6 @@ class MooncakeKVManager(StagingManagerMixin, CommonKVManager):
super().__init__(args, disaggregation_mode, server_args, is_mla_backend) super().__init__(args, disaggregation_mode, server_args, is_mla_backend)
self.init_engine() self.init_engine()
self.register_buffer_to_engine() self.register_buffer_to_engine()
# session_id -> (endpoint, dst_port, room), used to route host-memory
# transfer blocks over zmq when the intra-node NVLink transport is
# active (it cannot reach host memory). Populated on bootstrap.
self._session_endpoint_map = {}
self.enable_staging = envs.SGLANG_DISAGG_STAGING_BUFFER.get() self.enable_staging = envs.SGLANG_DISAGG_STAGING_BUFFER.get()
self.max_transfer_batch_indices = ( self.max_transfer_batch_indices = (
envs.SGLANG_MOONCAKE_MAX_TRANSFER_BATCH_INDICES.get() envs.SGLANG_MOONCAKE_MAX_TRANSFER_BATCH_INDICES.get()
@@ -436,13 +322,6 @@ class MooncakeKVManager(StagingManagerMixin, CommonKVManager):
Deduped because the unified memory pool reports one raw buffer as both Deduped because the unified memory pool reports one raw buffer as both
its KV and its mamba state component, and double registration fails in its KV and its mamba state component, and double registration fails in
the engine. the engine.
When the intra-node NVLink transport is active, host-memory regions
(aux buffers, some state components) are skipped: the transport only
accepts device memory, and a single host region fails the whole batch
and triggers a full engine-side rollback that would unregister the KV
pools too. Host-resident payloads are instead exchanged over the
ordered zmq channel (see _transfer_data / send_aux).
""" """
regions: List[Tuple[int, int]] = [] regions: List[Tuple[int, int]] = []
seen: Set[Tuple[int, int]] = set() seen: Set[Tuple[int, int]] = set()
@@ -459,24 +338,6 @@ class MooncakeKVManager(StagingManagerMixin, CommonKVManager):
self.kv_args.state_data_ptrs, self.kv_args.state_data_lens self.kv_args.state_data_ptrs, self.kv_args.state_data_lens
): ):
add(ptrs, lens) add(ptrs, lens)
if _nvlink_intra_transport_active():
device_regions = []
skipped = []
for ptr, length in regions:
if _is_device_pointer(ptr):
device_regions.append((ptr, length))
else:
skipped.append((ptr, length))
if skipped:
logger.info(
"Intra-node NVLink transport: skipping %d host-memory "
"regions from engine registration (they will be exchanged "
"over the zmq channel instead): %s",
len(skipped),
[(hex(p), l) for p, l in skipped[:8]],
)
regions = device_regions
return regions return regions
def register_buffer_to_engine(self): def register_buffer_to_engine(self):
@@ -887,64 +748,11 @@ class MooncakeKVManager(StagingManagerMixin, CommonKVManager):
if not transfer_blocks: if not transfer_blocks:
return 0 return 0
if not _nvlink_intra_transport_active():
src_addrs, dst_addrs, lengths = zip(*transfer_blocks) src_addrs, dst_addrs, lengths = zip(*transfer_blocks)
return self.engine.batch_transfer_sync( return self.engine.batch_transfer_sync(
mooncake_session_id, list(src_addrs), list(dst_addrs), list(lengths) mooncake_session_id, list(src_addrs), list(dst_addrs), list(lengths)
) )
# Intra-node NVLink transport can only move device memory. Partition
# blocks by the *local source* pointer (probing a local pointer is
# safe; the remote dst is never probed): device-sourced blocks go
# through the engine as usual, host-sourced blocks are shipped over
# the ordered zmq channel and written into the peer's buffer by the
# receiver (see _handle_state_data). This mirrors the aux TCP path.
device_blocks = []
host_blocks = []
for src, dst, length in transfer_blocks:
if _is_device_pointer(src):
device_blocks.append((src, dst, length))
else:
host_blocks.append((src, dst, length))
rc = 0
if device_blocks:
src_addrs, dst_addrs, lengths = zip(*device_blocks)
rc = self.engine.batch_transfer_sync(
mooncake_session_id, list(src_addrs), list(dst_addrs), list(lengths)
)
if rc == 0 and host_blocks:
rc = self._send_host_blocks_tcp(mooncake_session_id, host_blocks)
return rc
def _send_host_blocks_tcp(self, mooncake_session_id, host_blocks):
target = self._session_endpoint_map.get(mooncake_session_id)
if target is None:
logger.error(
f"No zmq endpoint known for mooncake session "
f"{mooncake_session_id}; cannot deliver {len(host_blocks)} "
"host-memory transfer blocks"
)
return -1
endpoint, dst_port, room = target
na = NetworkAddress(endpoint, dst_port)
for src, dst, length in host_blocks:
data = _read_bytes_from_address(src, length)
if data is None:
return -1
self._send_multipart_locked(
na.to_tcp(),
[
MooncakeKVManager.STATE_DATA_HEADER,
str(room).encode("ascii"),
str(int(dst)).encode("ascii"),
struct.pack(">I", len(data)),
data,
],
is_ipv6=na.is_ipv6,
)
return 0
def _send_kvcache_generic( def _send_kvcache_generic(
self, self,
mooncake_session_id: str, mooncake_session_id: str,
@@ -1674,10 +1482,8 @@ class MooncakeKVManager(StagingManagerMixin, CommonKVManager):
): ):
# TODO(shangming): Fix me when nvlink_transport of Mooncake is bug-free # TODO(shangming): Fix me when nvlink_transport of Mooncake is bug-free
if ( if (
(self.enable_custom_mem_pool and self.custom_mem_pool_type == "NVLINK") self.enable_custom_mem_pool and self.custom_mem_pool_type == "NVLINK"
or envs.SGLANG_MOONCAKE_SEND_AUX_TCP.get() ) or envs.SGLANG_MOONCAKE_SEND_AUX_TCP.get():
or _nvlink_intra_transport_active()
):
return self.send_aux_tcp(req, prefill_aux_index, dst_aux_ptrs) return self.send_aux_tcp(req, prefill_aux_index, dst_aux_ptrs)
transfer_blocks = [] transfer_blocks = []
@@ -1760,71 +1566,6 @@ class MooncakeKVManager(StagingManagerMixin, CommonKVManager):
f"Received AUX_DATA for bootstrap_room {room} with length:{len(data)}" f"Received AUX_DATA for bootstrap_room {room} with length:{len(data)}"
) )
def _host_transfer_regions(self):
"""Address ranges this process published as transfer targets.
Used to validate STATE_DATA writes. Built lazily because kv_args is
fully populated only after registration.
"""
regions = getattr(self, "_host_transfer_regions_cache", None)
if regions is None:
regions = []
for ptr, length in zip(
self.kv_args.kv_data_ptrs or [], self.kv_args.kv_data_lens or []
):
regions.append((int(ptr), int(ptr) + int(length)))
for ptr, length in zip(
self.kv_args.aux_data_ptrs or [], self.kv_args.aux_data_lens or []
):
regions.append((int(ptr), int(ptr) + int(length)))
for ptrs, lens in zip(
self.kv_args.state_data_ptrs or [], self.kv_args.state_data_lens or []
):
for ptr, length in zip(ptrs or [], lens or []):
regions.append((int(ptr), int(ptr) + int(length)))
self._host_transfer_regions_cache = regions
return regions
def _handle_state_data(self, msg: List[bytes]):
"""Handle STATE_DATA messages received by the decode thread.
Carries one host-memory transfer block that could not go through the
intra-node NVLink transport. Written directly into the local buffer at
the destination address; ordering against the final status message is
guaranteed by the shared per-endpoint zmq socket.
"""
room = int(msg[1].decode("ascii"))
dst_addr = int(msg[2].decode("ascii"))
data_length = struct.unpack(">I", msg[3])[0]
data = msg[4]
if len(data) != data_length:
logger.error(f"STATE_DATA length mismatch for bootstrap_room {room}")
return
in_region = any(
start <= dst_addr and dst_addr + len(data) <= end
for start, end in self._host_transfer_regions()
)
if not in_region:
logger.error(
f"STATE_DATA for bootstrap_room {room} targets unknown region "
f"{hex(dst_addr)}..{hex(dst_addr + len(data))}; dropping"
)
return
if not _write_bytes_to_address(dst_addr, data):
logger.error(
f"STATE_DATA write failed for bootstrap_room {room} at "
f"{hex(dst_addr)} len {len(data)}"
)
return
logger.debug(
f"Received STATE_DATA for bootstrap_room {room} at {hex(dst_addr)} "
f"with length:{len(data)}"
)
def _get_dsa_cache_transfer_skip_flags( def _get_dsa_cache_transfer_skip_flags(
self, info: Optional[KVArgsRegisterInfo] self, info: Optional[KVArgsRegisterInfo]
) -> Tuple[bool, bool]: ) -> Tuple[bool, bool]:
@@ -2671,8 +2412,6 @@ class MooncakeKVManager(StagingManagerMixin, CommonKVManager):
): ):
self._staging_outstanding.pop(kv_chunk.room, None) self._staging_outstanding.pop(kv_chunk.room, None)
if kv_chunk.room in self.transfer_infos: if kv_chunk.room in self.transfer_infos:
for sid in self.transfer_infos[kv_chunk.room]:
self._session_endpoint_map.pop(sid, None)
self.transfer_infos.pop(kv_chunk.room) self.transfer_infos.pop(kv_chunk.room)
self.req_to_decode_prefix_len.pop(kv_chunk.room, None) self.req_to_decode_prefix_len.pop(kv_chunk.room, None)
if self.enable_staging: if self.enable_staging:
@@ -2818,11 +2557,6 @@ class MooncakeKVManager(StagingManagerMixin, CommonKVManager):
self.transfer_infos[room][mooncake_session_id] = ( self.transfer_infos[room][mooncake_session_id] = (
TransferInfo.from_zmq(waiting_req_bytes) TransferInfo.from_zmq(waiting_req_bytes)
) )
self._session_endpoint_map[mooncake_session_id] = (
self.transfer_infos[room][mooncake_session_id].endpoint,
self.transfer_infos[room][mooncake_session_id].dst_port,
room,
)
# NOTE: after bootstrapping we can mark the req as waiting for input # NOTE: after bootstrapping we can mark the req as waiting for input
if len(self.transfer_infos[room]) == required_dst_info_num: if len(self.transfer_infos[room]) == required_dst_info_num:
self.resolve_kv_replica_factor(self.transfer_infos[room]) self.resolve_kv_replica_factor(self.transfer_infos[room])
@@ -2851,9 +2585,6 @@ class MooncakeKVManager(StagingManagerMixin, CommonKVManager):
if msg[0] == MooncakeKVManager.AUX_DATA_HEADER: if msg[0] == MooncakeKVManager.AUX_DATA_HEADER:
self._handle_aux_data(msg) self._handle_aux_data(msg)
continue continue
if msg[0] == MooncakeKVManager.STATE_DATA_HEADER:
self._handle_state_data(msg)
continue
# Staging: prefill notifies a chunk written to staging buffer # Staging: prefill notifies a chunk written to staging buffer
if msg[0] == b"CHUNK_READY": if msg[0] == b"CHUNK_READY":
@@ -736,14 +736,7 @@ class DSV4AttnMetadata:
if src_val is None and dst_val is None: if src_val is None and dst_val is None:
continue continue
assert dst_val is not None, f"{field_name=} {src_val=} {dst_val=}" assert dst_val is not None, f"{field_name=} {src_val=} {dst_val=}"
shape_mismatch = dst_val.shape != src_val.shape dst_val.copy_(src_val)
assert not shape_mismatch or field_name in self._CP_GLOBAL_FIELDS, (
f"Only CP-global replay metadata may use a shorter live prefix, "
f"got {field_name=} {src_val.shape=} {dst_val.shape=}"
)
_copy_tensor_allowing_storage_alias(
dst_val, src_val, pad_value=0 if shape_mismatch else None
)
# These fields are safe to replace because captured kernels only need # These fields are safe to replace because captured kernels only need
# the current per-replay objects, or the field is produced inside the # the current per-replay objects, or the field is produced inside the
@@ -995,27 +988,6 @@ def _prefill_graph_max_seq_len() -> Optional[int]:
return get_exec().graph.cuda_graph_config.prefill.max_seq_len return get_exec().graph.cuda_graph_config.prefill.max_seq_len
def _copy_tensor_allowing_storage_alias(
dst: torch.Tensor, src: torch.Tensor, *, pad_value: Optional[int] = None
) -> None:
"""Copy replay metadata while preserving capture-stable destination addresses."""
if dst is src:
return
if dst.untyped_storage().data_ptr() == src.untyped_storage().data_ptr():
src = src.clone()
if dst.shape == src.shape:
dst.copy_(src)
return
assert (
pad_value is not None
and dst.ndim == src.ndim
and dst.shape[0] >= src.shape[0]
and dst.shape[1:] == src.shape[1:]
), f"Cannot copy replay metadata from {src.shape=} to {dst.shape=}"
dst.fill_(pad_value)
dst[: src.shape[0]].copy_(src)
@dataclass @dataclass
class DSV4Metadata: class DSV4Metadata:
core_attn_metadata: DSV4AttnMetadata core_attn_metadata: DSV4AttnMetadata
@@ -1282,11 +1254,6 @@ class DeepseekV4AttnBackend(
] = None ] = None
self.online_c128_mtp = OnlineC128MTPController(self) self.online_c128_mtp = OnlineC128MTPController(self)
self.sparse_prefill_workspace = SparsePrefillWorkspace(self.device) self.sparse_prefill_workspace = SparsePrefillWorkspace(self.device)
# CP V4.1 consumers share compressed KV across layers. Separate ratio
# workspaces keep those prefixes intact while each layer refreshes SWA.
self.shared_compressed_prefill_workspaces = {
ratio: SparsePrefillWorkspace(self.device) for ratio in (1, 2)
}
spec_alg = model_runner.spec_algorithm spec_alg = model_runner.spec_algorithm
self.needs_cpu_seq_lens = not spec_alg.is_dspark() and ( self.needs_cpu_seq_lens = not spec_alg.is_dspark() and (
not _is_cuda or self.online_c128_mtp.enabled() not _is_cuda or self.online_c128_mtp.enabled()
@@ -1599,12 +1566,8 @@ class DeepseekV4AttnBackend(
@property @property
def low_ratio_prefill_graph(self) -> bool: def low_ratio_prefill_graph(self) -> bool:
"""Whether ratio-1/2 sources use captured projections and indexer metadata."""
return ( return (
bool(self.low_ratios) bool(self.low_ratios) and _has_dense_fp4_indexer() and _is_sm100_or_newer()
and _has_dense_fp4_indexer()
and _is_sm100_or_newer()
and get_parallel().attn_cp_size == 1
) )
def can_run_prefill_cuda_graph(self, forward_batch: ForwardBatch) -> bool: def can_run_prefill_cuda_graph(self, forward_batch: ForwardBatch) -> bool:
@@ -2907,7 +2870,7 @@ class DeepseekV4AttnBackend(
q_lora[:num_local], q_lora[:num_local],
positions[:num_local].to(torch.int64), positions[:num_local].to(torch.int64),
forward_batch, forward_batch,
self._move_to_device(q_lens_cpu), torch.tensor(q_lens_cpu, dtype=torch.int32, device=x.device),
q_lens_cpu, q_lens_cpu,
) )
@@ -3273,9 +3236,7 @@ class DeepseekV4AttnBackend(
continue continue
j = torch.arange(lc, device=device) j = torch.arange(lc, device=device)
slot_chunks.append( slot_chunks.append(
self.req_to_token[req_pool_indices[r : r + 1], j * ratio].to( self.req_to_token[req_pool_indices[r], j * ratio].to(torch.int64)
torch.int64
)
// ratio // ratio
) )
start += lc start += lc
@@ -3300,7 +3261,7 @@ class DeepseekV4AttnBackend(
weights = indexer.head_weights(x).float() weights = indexer.head_weights(x).float()
compress_lens = ((pos + 1) // ratio).to(torch.int32) compress_lens = ((pos + 1) // ratio).to(torch.int32)
ks = torch.repeat_interleave( ks = torch.repeat_interleave(
self._move_to_device(starts), torch.tensor(starts, dtype=torch.int32, device=device),
q_lens.to(torch.int64), q_lens.to(torch.int64),
output_size=num_tokens, output_size=num_tokens,
) )
@@ -3486,31 +3447,6 @@ class DeepseekV4AttnBackend(
self.candidate_indexer.publish_decode(inputs, page_indices, raw_indices) self.candidate_indexer.publish_decode(inputs, page_indices, raw_indices)
) )
return return
if isinstance(metadata.deep_gemm_metadata, list):
topk_plans = metadata.topk_metadata_chunks
assert not metadata.use_topk_v2 or topk_plans is not None
for chunk_idx, (rows, plan) in enumerate(metadata.row_chunks()):
logits = deep_gemm_fp4_paged_mqa_logits(
(q_fp4[rows], q_sf[rows]),
k_cache,
weights[rows],
metadata.compressed_seq_lens[rows],
metadata.page_table[rows],
plan,
metadata.max_compressed_seq_len,
)
# TODO(dark): add bf16 topk
topk_transform_paged_from_metadata(
logits,
metadata,
page_indices,
raw_indices,
rows=rows,
topk_metadata=(
topk_plans[chunk_idx] if topk_plans is not None else None
),
)
else:
logits = deep_gemm_fp4_paged_mqa_logits( logits = deep_gemm_fp4_paged_mqa_logits(
(q_fp4, q_sf), (q_fp4, q_sf),
k_cache, k_cache,
@@ -3521,9 +3457,7 @@ class DeepseekV4AttnBackend(
metadata.max_compressed_seq_len, metadata.max_compressed_seq_len,
) )
# TODO(dark): add bf16 topk # TODO(dark): add bf16 topk
topk_transform_paged_from_metadata( topk_transform_paged_from_metadata(logits, metadata, page_indices, raw_indices)
logits, metadata, page_indices, raw_indices
)
# TODO(candidate): Hopper decode still publishes / consumes masks inline (torch # TODO(candidate): Hopper decode still publishes / consumes masks inline (torch
# top-k); move into the candidate indexer with the prefill paths. # top-k); move into the candidate indexer with the prefill paths.
@@ -4011,29 +3945,13 @@ class DeepseekV4AttnBackend(
compress_ratio, core_attn_metadata, extra_page_size compress_ratio, core_attn_metadata, extra_page_size
) )
n_compressed = flat_token_ids.shape[0] n_compressed = flat_token_ids.shape[0]
reuse_compressed = compress_ratio in (1, 2) and is_cp_active(forward_batch) workspace = self.sparse_prefill_workspace.get(
workspace_pool = ( n_compressed + cache.swa_token_ids.shape[0]
self.shared_compressed_prefill_workspaces[compress_ratio]
if reuse_compressed
else self.sparse_prefill_workspace
) )
workspace = workspace_pool.get(n_compressed + cache.swa_token_ids.shape[0])
compressed_slice = workspace[:n_compressed] compressed_slice = workspace[:n_compressed]
swa_slice = workspace[n_compressed:] swa_slice = workspace[n_compressed:]
if compressed_slice is not None: if compressed_slice is not None:
source_key = None
if reuse_compressed:
source_layer = token_to_kv_pool.source_layer_of(layer_id)
source_key = (source_layer, workspace.data_ptr())
gather = cache.compressed[compress_ratio]
# A source layer may have just updated its cache in place. Consumer
# layers only reuse the compressed prefix; their top-k and SWA stay live.
if (
source_key is None
or layer_id == source_key[0]
or gather.dequantized_source != source_key
):
dequantize_k_cache_paged( dequantize_k_cache_paged(
extra_k_cache, extra_k_cache,
flat_token_ids, flat_token_ids,
@@ -4041,8 +3959,6 @@ class DeepseekV4AttnBackend(
out=compressed_slice, out=compressed_slice,
layout=token_to_kv_pool.get_extra_key_layout(layer_id), layout=token_to_kv_pool.get_extra_key_layout(layer_id),
) )
if source_key is not None:
gather.dequantized_source = source_key
dequantize_k_cache_paged( dequantize_k_cache_paged(
token_to_kv_pool.get_swa_key_buffer_radix(layer_id), token_to_kv_pool.get_swa_key_buffer_radix(layer_id),
cache.swa_token_ids, cache.swa_token_ids,
@@ -21,7 +21,6 @@ from sglang.srt.layers.attention.dsv4.candidate_indexer import (
) )
from sglang.srt.layers.attention.dsv4.indexer import ( from sglang.srt.layers.attention.dsv4.indexer import (
deep_gemm_fp4_paged_mqa_logits, deep_gemm_fp4_paged_mqa_logits,
topk_transform_paged_from_metadata,
) )
CANDIDATE_BLOCK_SIZE = 8 # positions per block; DeepGEMM accepts 8 or 16 CANDIDATE_BLOCK_SIZE = 8 # positions per block; DeepGEMM accepts 8 or 16
@@ -177,10 +176,6 @@ class DeepGemmCandidateIndexer:
metadata.""" metadata."""
metadata = inputs.metadata metadata = inputs.metadata
seq_lens = metadata.compressed_seq_lens.reshape(-1) seq_lens = metadata.compressed_seq_lens.reshape(-1)
if isinstance(metadata.deep_gemm_metadata, list):
return self._publish_decode_chunked(
inputs, page_indices, raw_indices, seq_lens
)
logits = deep_gemm_fp4_paged_mqa_logits( logits = deep_gemm_fp4_paged_mqa_logits(
(inputs.q_fp4, inputs.q_sf), (inputs.q_fp4, inputs.q_sf),
inputs.k_cache, inputs.k_cache,
@@ -236,83 +231,6 @@ class DeepGemmCandidateIndexer:
ready=ready, ready=ready,
) )
def _publish_decode_chunked(
self,
inputs: IndexerInputs,
page_indices: torch.Tensor,
raw_indices: Optional[torch.Tensor],
seq_lens: torch.Tensor,
) -> SparseBlockTable:
"""Publish an eager forward whose dense logits are bounded by row chunks.
CUDA-graph metadata always carries one tensor schedule and keeps using the
asynchronous fast path above. The exceptional eager path stays on the
current stream so each chunk's full logits can be released before the next.
"""
metadata = inputs.metadata
block_chunks = []
phys_block_chunks = []
valid_len_chunks = []
topk_plans = metadata.topk_metadata_chunks
assert not metadata.use_topk_v2 or topk_plans is not None
for chunk_idx, (rows, plan) in enumerate(metadata.row_chunks()):
logits = deep_gemm_fp4_paged_mqa_logits(
(inputs.q_fp4[rows], inputs.q_sf[rows]),
inputs.k_cache,
inputs.weights[rows],
metadata.compressed_seq_lens[rows],
metadata.page_table[rows],
plan,
metadata.max_compressed_seq_len,
)
topk_transform_paged_from_metadata(
logits,
metadata,
page_indices,
raw_indices,
rows=rows,
topk_metadata=(
topk_plans[chunk_idx] if topk_plans is not None else None
),
)
chunk_seq_lens = seq_lens[rows]
nblocks, row_valid_lens = candidate_row_lens(
chunk_seq_lens, self.topk_blocks
)
blocks = amax_topk_blocks(logits, chunk_seq_lens, nblocks, self.topk_blocks)
phys_blocks = sort_candidate_blocks(
blocks,
chunk_seq_lens,
metadata.page_table[rows],
metadata.compressed_page_size,
)
block_chunks.append(blocks)
phys_block_chunks.append(phys_blocks)
valid_len_chunks.append(row_valid_lens)
blocks = torch.cat(block_chunks)
phys_blocks = torch.cat(phys_block_chunks)
row_valid_lens = torch.cat(valid_len_chunks)
schedule = build_sparse_indexer_schedule(
blocks,
seq_lens,
metadata.page_table,
metadata.compressed_page_size,
inputs.q_fp4.dtype,
self._request_ids(inputs.request_ids, inputs.num_rows, blocks.device),
)
ready = torch.cuda.Event()
ready.record(torch.cuda.current_stream())
return SparseBlockTable(
blocks=blocks,
schedule=schedule,
phys_blocks=phys_blocks,
valid_lens=row_valid_lens,
ready=ready,
)
def _scores(self, table: SparseBlockTable, inputs: IndexerInputs) -> torch.Tensor: def _scores(self, table: SparseBlockTable, inputs: IndexerInputs) -> torch.Tensor:
return sparse_logits( return sparse_logits(
inputs.q_fp4, inputs.q_fp4,
@@ -475,40 +475,27 @@ def topk_transform_paged_from_metadata(
metadata, metadata,
page_indices: torch.Tensor, page_indices: torch.Tensor,
raw_indices: Optional[torch.Tensor] = None, raw_indices: Optional[torch.Tensor] = None,
*,
rows: Optional[slice] = None,
topk_metadata: Optional[torch.Tensor] = None,
) -> None: ) -> None:
"""Pool slots into ``page_indices`` (``-1`` past the valid count) and, when given, """Pool slots into ``page_indices`` (``-1`` past the valid count) and, when given,
positions into ``raw_indices``; ``metadata`` is a ``PagedIndexerMetadata``.""" positions into ``raw_indices``; ``metadata`` is a ``PagedIndexerMetadata``."""
if rows is None:
seq_lens = metadata.compressed_seq_lens
page_table = metadata.page_table
out_page_indices = page_indices
out_raw_indices = raw_indices
else:
seq_lens = metadata.compressed_seq_lens[rows]
page_table = metadata.page_table[rows]
out_page_indices = page_indices[rows]
out_raw_indices = raw_indices[rows] if raw_indices is not None else None
if metadata.use_topk_v2: if metadata.use_topk_v2:
topk_transform_paged_v2( topk_transform_paged_v2(
logits, logits,
seq_lens, metadata.compressed_seq_lens,
page_table, metadata.page_table,
out_page_indices, page_indices,
metadata.compressed_page_size, metadata.compressed_page_size,
metadata.topk_metadata if topk_metadata is None else topk_metadata, metadata.topk_metadata,
out_raw_indices, raw_indices,
) )
else: else:
topk_transform_paged( topk_transform_paged(
logits, logits,
seq_lens, metadata.compressed_seq_lens,
page_table, metadata.page_table,
out_page_indices, page_indices,
metadata.compressed_page_size, metadata.compressed_page_size,
out_raw_indices, raw_indices,
) )
@@ -21,7 +21,6 @@ from sglang.srt.model_executor.runner_backend_utils.breakable_cuda_graph.context
from sglang.srt.model_executor.runner_backend_utils.tc_piecewise_cuda_graph import ( from sglang.srt.model_executor.runner_backend_utils.tc_piecewise_cuda_graph import (
is_in_tc_piecewise_cuda_graph, is_in_tc_piecewise_cuda_graph,
) )
from sglang.srt.model_executor.runner_utils.capture_mode import get_is_capture_mode
from sglang.srt.utils import is_hip, is_sm120_supported, is_xpu from sglang.srt.utils import is_hip, is_sm120_supported, is_xpu
logger = logging.getLogger(__name__) logger = logging.getLogger(__name__)
@@ -307,8 +306,7 @@ class PagedIndexerMetadata:
): ):
return None return None
if ( if (
get_is_capture_mode() torch.cuda.is_current_stream_capturing()
or torch.cuda.is_current_stream_capturing()
or is_in_breakable_cuda_graph() or is_in_breakable_cuda_graph()
or is_in_tc_piecewise_cuda_graph() or is_in_tc_piecewise_cuda_graph()
): ):
@@ -327,27 +325,14 @@ class PagedIndexerMetadata:
def row_chunks(self): def row_chunks(self):
num_rows = self.compressed_seq_lens.shape[0] num_rows = self.compressed_seq_lens.shape[0]
if self.row_chunk > 0: if self.row_chunk <= 0:
rows_per_chunk = self.row_chunk
elif isinstance(self.deep_gemm_metadata, list):
assert self.rows_per_chunk is not None, (
"chunked DeepGEMM metadata requires rows_per_chunk"
)
rows_per_chunk = self.rows_per_chunk
else:
return [(slice(0, num_rows), self.deep_gemm_metadata)] return [(slice(0, num_rows), self.deep_gemm_metadata)]
return [
chunks = [ (slice(start, min(start + self.row_chunk, num_rows)), plan)
(slice(start, min(start + rows_per_chunk, num_rows)), plan)
for start, plan in zip( for start, plan in zip(
range(0, num_rows, rows_per_chunk), self.deep_gemm_metadata range(0, num_rows, self.row_chunk), self.deep_gemm_metadata
) )
] ]
assert chunks and chunks[-1][0].stop == num_rows, (
f"chunk schedules do not cover all rows: {num_rows=} {rows_per_chunk=} "
f"{len(chunks)=}"
)
return chunks
def copy_(self, other: PagedIndexerMetadata): def copy_(self, other: PagedIndexerMetadata):
# A chunked schedule list has no in-place copy; rebind it instead. # A chunked schedule list has no in-place copy; rebind it instead.
@@ -78,11 +78,10 @@ def use_dsv4_q8kv8_sparse_prefill(dsv4_prefill_backend: str = "auto") -> bool:
class SparsePrefillWorkspace: class SparsePrefillWorkspace:
"""Backend-owned scratch storage for sparse prefill KV dequantization. """Backend-owned scratch storage for sparse prefill KV dequantization.
Callers normally overwrite the entire workspace. Shared compressed-KV The workspace contents are fully overwritten before every attention call,
callers keep separate workspaces per ratio and track prefix validity in the so token buckets and compression ratios can safely share one buffer. Sparse
per-forward gather cache, including the allocation address. Sparse prefill prefill executes eagerly and serially on the supported paths, which makes it
executes eagerly and serially on the supported paths, so the allocation can safe to replace the scratch allocation when a larger extent is needed.
be replaced when a larger extent is needed.
""" """
def __init__(self, device: torch.device): def __init__(self, device: torch.device):
@@ -276,18 +275,14 @@ class CompressedGather:
# chunk-invariant per request; subsequent layers only overwrite that prefix. # chunk-invariant per request; subsequent layers only overwrite that prefix.
combined_indices: Optional[torch.Tensor] = None combined_indices: Optional[torch.Tensor] = None
combined_lens: Optional[torch.Tensor] = None combined_lens: Optional[torch.Tensor] = None
# Valid only for this forward's gather layout. Each ratio has its own
# workspace; its compressed prefix survives consumer layers' SWA writes.
dequantized_source: Optional[tuple[int, int]] = None
@dataclass @dataclass
class SparsePrefillChunkCache: class SparsePrefillChunkCache:
"""Cache prefill-chunk metadata shared across layers. """Cache prefill-chunk metadata shared across layers.
Gather layouts depend on request/token mappings and compressed page tables. Fields depend on request/token mappings and compressed page tables, not
Shared-source dequantization keys live only for this forward; per-layer per-layer k_cache; per-layer top-k combinations are recomputed into reused
top-k combinations are recomputed into reused
buffers. buffers.
""" """
+10 -97
View File
@@ -23,22 +23,17 @@ import torch
from sglang.srt.arg_groups.overrides import ( from sglang.srt.arg_groups.overrides import (
attention_backends_of, attention_backends_of,
model_config_of,
resolved_view, resolved_view,
resolving_view, resolving_view,
) )
from sglang.srt.configs.model_config import is_deepseek_v4
from sglang.srt.layers.cp.base import get_cp_strategy from sglang.srt.layers.cp.base import get_cp_strategy
from sglang.srt.layers.cp.interleave import InterleaveCPStrategy
from sglang.srt.layers.cp.padding import get_cp_padding_align_size from sglang.srt.layers.cp.padding import get_cp_padding_align_size
from sglang.srt.layers.cp.utils import ( from sglang.srt.layers.cp.utils import (
cp_gather_after_forward, cp_gather_after_forward,
cp_shard_hidden_states,
cp_split_before_forward, cp_split_before_forward,
prepare_cp_forward, prepare_cp_forward,
) )
from sglang.srt.layers.cp.zigzag import ZigzagCPStrategy from sglang.srt.layers.cp.zigzag import ZigzagCPStrategy
from sglang.srt.layers.logits_processor import LogitsMetadata
from sglang.srt.model_executor.forward_batch_info import PPProxyTensors from sglang.srt.model_executor.forward_batch_info import PPProxyTensors
if TYPE_CHECKING: if TYPE_CHECKING:
@@ -55,18 +50,12 @@ def supports_prefill_cp_bcg(server_args: ServerArgs) -> bool:
cfg = resolving_view(server_args) cfg = resolving_view(server_args)
resolved = resolved_view(server_args) resolved = resolved_view(server_args)
prefill_attention_backend, _ = attention_backends_of(resolved_view(server_args)) prefill_attention_backend, _ = attention_backends_of(resolved_view(server_args))
supports_layout = (
cfg.cp_strategy == "zigzag" and prefill_attention_backend == "trtllm_mha"
) or (
cfg.cp_strategy == "interleave"
and prefill_attention_backend == "dsv4"
and is_deepseek_v4(model_config_of(server_args).hf_config)
)
return ( return (
cfg.enable_prefill_cp cfg.enable_prefill_cp
and cfg.pp_size == 1 and cfg.pp_size == 1
and resolved.attn_cp_size == cfg.tp_size and resolved.attn_cp_size == cfg.tp_size
and supports_layout and cfg.cp_strategy == "zigzag"
and prefill_attention_backend == "trtllm_mha"
) )
@@ -78,12 +67,8 @@ def enable_cp_bcg_capture(server_args: ServerArgs) -> bool:
def filter_prefill_cp_bcg_capture_num_tokens( def filter_prefill_cp_bcg_capture_num_tokens(
capture_num_tokens: list[int], server_args: ServerArgs capture_num_tokens: list[int], server_args: ServerArgs
) -> list[int]: ) -> list[int]:
"""Keep only token buckets where the configured CP strategy can run.""" """Keep only token buckets where the zigzag CP strategy can run."""
cfg = resolving_view(server_args) min_num_tokens = resolved_view(server_args).attn_cp_size * 2
cp_segments_per_token_block = 2 if cfg.cp_strategy == "zigzag" else 1
min_num_tokens = (
resolved_view(server_args).attn_cp_size * cp_segments_per_token_block
)
filtered = [size for size in capture_num_tokens if size >= min_num_tokens] filtered = [size for size in capture_num_tokens if size >= min_num_tokens]
if not filtered: if not filtered:
raise ValueError( raise ValueError(
@@ -111,8 +96,6 @@ class PrefillCPBCGInput:
input_embeds: torch.Tensor input_embeds: torch.Tensor
positions: torch.Tensor positions: torch.Tensor
input_ids: Optional[torch.Tensor] = None
num_token_non_padded: Optional[torch.Tensor] = None
bucket_local_tokens: Dict[int, int] = field(default_factory=dict) bucket_local_tokens: Dict[int, int] = field(default_factory=dict)
live_local_tokens: int = 0 live_local_tokens: int = 0
@@ -131,22 +114,12 @@ class PrefillCPBCGInput:
(runner.max_num_tokens,), (runner.max_num_tokens,),
dtype=torch.int64, dtype=torch.int64,
), ),
input_ids=torch.zeros((runner.max_num_tokens,), dtype=torch.int64),
num_token_non_padded=torch.zeros((), dtype=torch.int32),
) )
def required_local_tokens(self, extend_seq_lens: Any) -> Optional[int]: def required_local_tokens(self, extend_seq_lens: Any) -> Optional[int]:
"""Return the aligned CP-local rows required by the active layout.""" """Return the aligned CP-local rows required by a live zigzag layout."""
strategy = get_cp_strategy() strategy = get_cp_strategy()
if extend_seq_lens is None: if not isinstance(strategy, ZigzagCPStrategy) or extend_seq_lens is None:
return None
if isinstance(strategy, InterleaveCPStrategy):
logical_tokens = (
sum(int(length) for length in extend_seq_lens) + strategy.cp_size - 1
) // strategy.cp_size
align_size = get_cp_padding_align_size()
return (logical_tokens + align_size - 1) // align_size * align_size
if not isinstance(strategy, ZigzagCPStrategy):
return None return None
cp_segment_num = strategy.cp_size * 2 cp_segment_num = strategy.cp_size * 2
@@ -246,7 +219,6 @@ class PrefillCPBCGInput:
raw_tokens = int(forward_batch.extend_num_tokens) raw_tokens = int(forward_batch.extend_num_tokens)
global_input_ids = forward_batch.input_ids[:raw_tokens] global_input_ids = forward_batch.input_ids[:raw_tokens]
global_positions = forward_batch.positions[:raw_tokens] global_positions = forward_batch.positions[:raw_tokens]
local_input_ids = cp_shard_hidden_states(global_input_ids, forward_batch)
global_input_embeds = runner.model_runner.model.get_input_embeddings()( global_input_embeds = runner.model_runner.model.get_input_embeddings()(
global_input_ids global_input_ids
) )
@@ -277,31 +249,12 @@ class PrefillCPBCGInput:
input_embeds = self.input_embeds[:captured_local_tokens] input_embeds = self.input_embeds[:captured_local_tokens]
positions = self.positions[:captured_local_tokens] positions = self.positions[:captured_local_tokens]
assert self.input_ids is not None
input_ids = self.input_ids[:captured_local_tokens]
input_embeds.zero_() input_embeds.zero_()
positions.zero_() positions.zero_()
input_ids.zero_()
input_embeds[:live_local_tokens].copy_(local_input_embeds) input_embeds[:live_local_tokens].copy_(local_input_embeds)
positions[:live_local_tokens].copy_(local_positions) positions[:live_local_tokens].copy_(local_positions)
input_ids[:live_local_tokens].copy_(local_input_ids)
forward_batch.input_embeds = input_embeds forward_batch.input_embeds = input_embeds
forward_batch._cp_positions = positions forward_batch.positions = positions
# Keep the global input_ids field intact: the runner uses its length to
# select the global capture bucket. The DSV4 body consumes this fixed,
# rank-local view for hash routing and MegaMoE.
forward_batch._cp_input_ids = input_ids
forward_batch.input_ids_global = input_ids
if forward_batch.num_token_non_padded is not None:
assert self.num_token_non_padded is not None
metadata = forward_batch.attn_cp_metadata
logical_tokens = (
metadata.per_rank_logical_token or metadata.per_rank_actual_token
)
strategy = get_cp_strategy()
assert strategy is not None
self.num_token_non_padded.fill_(logical_tokens[strategy.cp_rank])
forward_batch.num_token_non_padded = self.num_token_non_padded
self.live_local_tokens = live_local_tokens self.live_local_tokens = live_local_tokens
@@ -354,50 +307,10 @@ def execute_prefill_cp_bcg(
static_forward_batch, static_forward_batch,
torch.cuda.current_stream(), torch.cuda.current_stream(),
) )
if aux_hidden_states is not None: return model.logits_processor(
if torch.is_tensor(aux_hidden_states): forward_batch.input_ids,
aux_hidden_states = cp_gather_after_forward(
aux_hidden_states, static_forward_batch, torch.cuda.current_stream()
)
else:
aux_hidden_states = [
cp_gather_after_forward(
aux, static_forward_batch, torch.cuda.current_stream()
)
for aux in aux_hidden_states
]
hidden_states_before_norm = None
if isinstance(hidden_states, tuple):
assert len(hidden_states) == 2
hidden_states, hidden_states_before_norm = hidden_states
input_ids = forward_batch.input_ids
logits_metadata = forward_batch
tail = None
language_model = getattr(model, "model", None)
if (
capture_aux_hidden_states
and getattr(language_model, "late_layer_start", None) is not None
and forward_batch.forward_mode.is_extend_without_speculative()
):
tail_metadata = runner.model_runner.attn_backend.tail_forward_metadata
tail = tail_metadata.late_layer_tail
input_ids = tail.rows(input_ids)
logits_metadata = LogitsMetadata.from_forward_batch(forward_batch)
logits_metadata.extend_seq_lens = tail.extend_seq_lens
logits_metadata.extend_seq_lens_cpu = tail.extend_seq_lens_cpu
logits_metadata.extend_logprob_start_lens_cpu = tail.extend_seq_lens_cpu
output = model.logits_processor(
input_ids,
hidden_states, hidden_states,
model.lm_head, model.lm_head,
logits_metadata, forward_batch,
aux_hidden_states, aux_hidden_states,
hidden_states_before_norm=(
None if aux_hidden_states is not None else hidden_states_before_norm
),
) )
if tail is not None:
output.hidden_states_token_indices = tail.token_indices
return output
+5 -18
View File
@@ -216,28 +216,15 @@ def _run_mega_routed(
if num_tokens > 0: if num_tokens > 0:
router_logits = moe.gate(hidden_states, forward_batch=forward_batch) router_logits = moe.gate(hidden_states, forward_batch=forward_batch)
num_token_non_padded = (
forward_batch.num_token_non_padded if forward_batch is not None else None
)
if isinstance(
getattr(moe.gate, "e_score_correction_bias_vl", None), torch.Tensor
):
# V4.1 uses a different correction bias for image-token rows. The
# MegaMoE transport consumes the same routed ids/weights as TopK.
from sglang.srt.multimodal.dsv41.vl_routing import vision_topk
topk_output = vision_topk(
moe,
router_logits,
input_ids_global,
num_token_non_padded=num_token_non_padded,
)
else:
topk_kwargs = {"input_ids": input_ids_global} if moe.is_hash else {} topk_kwargs = {"input_ids": input_ids_global} if moe.is_hash else {}
topk_output = moe.topk( topk_output = moe.topk(
hidden_states, hidden_states,
router_logits, router_logits,
num_token_non_padded=num_token_non_padded, num_token_non_padded=(
forward_batch.num_token_non_padded
if forward_batch is not None
else None
),
expert_location_dispatch_info=ExpertLocationDispatchInfo.init_new( expert_location_dispatch_info=ExpertLocationDispatchInfo.init_new(
layer_id=moe.layer_id, layer_id=moe.layer_id,
), ),
@@ -1,486 +0,0 @@
"""One owner rank encodes each image span and broadcasts it to the ranks that
run the same prefill chunk; every agreement precedes the payload it guards."""
from __future__ import annotations
import logging
from contextlib import contextmanager
from typing import Any, Callable, Dict, Iterator, List, Optional, Sequence, Tuple
import msgspec
import torch
from sglang.srt.distributed.device_communicators.pynccl_allocator import (
disable_symmetric_memory_context,
restore_symmetric_memory_context,
)
from sglang.srt.mem_cache.multimodal_cache import EmbeddingResult, MultiModalStaticCache
logger = logging.getLogger(__name__)
SpanKey = Tuple[Optional[int], int]
SpanEncoder = Callable[[List[Any]], torch.Tensor | List[torch.Tensor]]
SpanSignature = Callable[[Any, int], Tuple[Any, ...]]
LOCAL_HIT = 0
OWNER_CACHE_BROADCAST = 1
OWNER_ENCODE_BROADCAST = 2
PHASE_PREPARE = "prepare"
PHASE_FEATURES = "features"
PHASE_FINALIZE = "finalize"
class MmOwnerProtocolError(RuntimeError):
"""Raised with identical text on every group member after a group-agreed failure."""
class ImageSpanRequest(msgspec.Struct, frozen=True):
hash: Optional[int]
span_len: int
item: Any
inside_chunk: bool
duplicates: List[Any] = []
class ImageSpanKey(msgspec.Struct, frozen=True):
hash: Optional[int]
span_len: int
geometry: Optional[Tuple[Any, ...]]
class RankManifest(msgspec.Struct, frozen=True):
rank: int
keys: List[ImageSpanKey]
cached: List[bool]
dtype: str
width: int
rids: List[str]
error: Optional[str] = None
class OwnerPlan(msgspec.Struct, frozen=True):
actions: List[int]
owners: List[int]
error: Optional[str] = None
class RankStatus(msgspec.Struct, frozen=True):
rank: int
error: Optional[str] = None
def select_owner_group(parallel) -> Optional[Any]:
"""The group whose members all execute the same requests, or None when a
single rank already encodes every image it sees."""
replication = parallel.tp_size // parallel.attn_dp_size
if replication <= 1:
return None
if parallel.attn_cp_size == 1:
group = parallel.attn_tp_group
elif parallel.attn_dp_size == 1 and parallel.attn_cp_size == parallel.tp_size:
group = parallel.attn_cp_group
else:
return None
return group if group.world_size == replication else None
def has_owner_span_work(
mm_inputs: Sequence[Any],
extend_prefix_lens: Sequence[int],
extend_seq_lens: Sequence[int],
) -> bool:
"""Host-side mirror of the per-image scheduling path: does any raw
single-span image overlap the chunk on every rank of the group."""
for mm_input, prefix_len, extend_len in zip(
mm_inputs, extend_prefix_lens, extend_seq_lens
):
if mm_input is None or extend_len <= 0:
continue
items = [item for item in mm_input.mm_items if item is not None]
if not items or any(
item.precomputed_embeddings is not None or len(item.offsets) != 1
for item in items
):
continue
for item in items:
start, end = item.offsets[0]
if end >= prefix_len and start < prefix_len + extend_len:
return True
return False
class MmOwnerSession(msgspec.Struct):
group: Any
device: Any
dtype: Any
width: int
rids: List[str]
signature: Any
engaged: bool
phase: str = PHASE_PREPARE
in_collective: bool = False
def resolve(
self,
requests: Sequence[ImageSpanRequest],
cache: MultiModalStaticCache,
encode: SpanEncoder,
) -> Dict[SpanKey, torch.Tensor]:
if not self.engaged:
raise RuntimeError(
"owner protocol reached for a chunk whose host metadata has no image span"
)
# Owners allocate different amounts than receivers, so none of these
# buffers may come out of a symmetric pool.
saved_context = disable_symmetric_memory_context()
try:
return _resolve_owner_features(self, requests, cache, encode)
finally:
restore_symmetric_memory_context(saved_context)
def features_ready(self) -> None:
self._complete()
self.phase = PHASE_FINALIZE
@contextmanager
def uncaptured(self) -> Iterator[None]:
# A failure inside a collective leaves the group in an unknown state;
# no later exchange may try to agree on it.
self.in_collective = True
yield
self.in_collective = False
@contextmanager
def fence(self) -> Iterator[None]:
try:
yield
except Exception as exc:
self._fail(exc)
raise
self._complete()
def _fail(self, exc: BaseException) -> None:
if (
not self.engaged
or self.in_collective
or isinstance(exc, MmOwnerProtocolError)
):
raise exc
text = _describe(self, self.phase, exc)
if self.phase == PHASE_PREPARE:
try:
_exchange_manifest(self, _manifest(self, [], [], error=text))
except MmOwnerProtocolError as agreed:
raise agreed from exc
_exchange_status(self, text, exc)
def _complete(self) -> None:
if not self.engaged:
return
if self.phase == PHASE_PREPARE:
raise RuntimeError(
f"owner protocol {self.phase} completed without a manifest exchange"
)
error = None
cause = None
try:
_synchronize(self.device)
except Exception as exc:
cause = exc
error = _describe(self, self.phase, exc)
_exchange_status(self, error, cause)
def _manifest(
session: MmOwnerSession,
keys: List[ImageSpanKey],
cached: List[bool],
error: Optional[str] = None,
) -> RankManifest:
return RankManifest(
rank=session.group.rank_in_group,
keys=keys,
cached=cached,
dtype=str(session.dtype),
width=session.width,
rids=list(session.rids),
error=error,
)
def _exchange_manifest(session: MmOwnerSession, manifest: RankManifest) -> OwnerPlan:
group = session.group
with session.uncaptured():
manifests = group.all_gather_object(manifest)
plan = _plan_or_error(session, manifests) if group.rank_in_group == 0 else None
plan = group.broadcast_object(plan, src=0)
session.phase = PHASE_FEATURES
if plan.error is not None:
raise MmOwnerProtocolError(plan.error)
return plan
def _plan_or_error(session: MmOwnerSession, manifests: List[RankManifest]) -> OwnerPlan:
try:
return _make_plan(manifests)
except Exception as exc:
return OwnerPlan(actions=[], owners=[], error=_describe(session, "plan", exc))
def _exchange_status(
session: MmOwnerSession, error: Optional[str], cause: Optional[BaseException]
) -> None:
with session.uncaptured():
statuses = session.group.all_gather_object(
RankStatus(rank=session.group.rank_in_group, error=error)
)
_raise_first_error(statuses, cause)
def _resolve_owner_features(
session: MmOwnerSession,
requests: Sequence[ImageSpanRequest],
cache: MultiModalStaticCache,
encode: SpanEncoder,
) -> Dict[SpanKey, torch.Tensor]:
group = session.group
features: Dict[SpanKey, torch.Tensor] = {}
keys: List[ImageSpanKey] = []
cached: List[bool] = []
error = None
try:
keys, cached = _pin_local_cache(session, requests, cache, features)
except Exception as exc:
error = _describe(session, "manifest", exc)
plan = _exchange_manifest(session, _manifest(session, keys, cached, error))
if all(action == LOCAL_HIT for action in plan.actions):
return features
buffers: Dict[int, torch.Tensor] = {}
error = None
try:
buffers = _prepare_transfers(session, requests, keys, plan, features, encode)
_synchronize(session.device)
except Exception as exc:
error = _describe(session, "encode", exc)
_exchange_status(session, error, None)
with session.uncaptured():
for index, (action, owner) in enumerate(zip(plan.actions, plan.owners)):
if action != LOCAL_HIT:
group.broadcast(buffers[index], src=owner)
for index, key in enumerate(keys):
if plan.actions[index] == LOCAL_HIT:
continue
span = buffers[index]
features[(key.hash, key.span_len)] = span
cache.set(key.hash, EmbeddingResult(embedding=span))
return features
def _pin_local_cache(
session: MmOwnerSession,
requests: Sequence[ImageSpanRequest],
cache: MultiModalStaticCache,
features: Dict[SpanKey, torch.Tensor],
) -> Tuple[List[ImageSpanKey], List[bool]]:
keys: List[ImageSpanKey] = []
cached: List[bool] = []
for request in requests:
if request.hash is None:
raise ValueError(
f"image span of {request.span_len} tokens has no content hash"
)
geometry = session.signature(request.item, request.span_len)
for duplicate in request.duplicates:
other = session.signature(duplicate, request.span_len)
if other != geometry:
raise ValueError(
f"image hash {request.hash} ({request.span_len} tokens) occurs "
f"with different geometry: {geometry} vs {other}"
)
keys.append(
ImageSpanKey(
hash=request.hash, span_len=request.span_len, geometry=geometry
)
)
span = _valid_cached_span(session, cache, request)
if span is not None:
features[(request.hash, request.span_len)] = span
cached.append(span is not None)
return keys, cached
def _valid_cached_span(
session: MmOwnerSession,
cache: MultiModalStaticCache,
request: ImageSpanRequest,
) -> Optional[torch.Tensor]:
entry = cache.get_single(request.hash)
if entry is None:
return None
span = entry.embedding
if (
span.dim() == 2
and span.shape[0] == request.span_len
and span.shape[1] == session.width
and span.dtype == session.dtype
and span.device == session.device
):
return span
logger.warning(
"Discarding cached multimodal embedding that cannot serve the current "
"image span: cache_key=%s expected=(%d, %d, %s) cached=(%s, %s).",
request.hash,
request.span_len,
session.width,
session.dtype,
tuple(span.shape),
span.dtype,
)
cache.free(request.hash, None)
return None
def _make_plan(manifests: List[RankManifest]) -> OwnerPlan:
for manifest in manifests:
if manifest.error is not None:
return OwnerPlan(actions=[], owners=[], error=manifest.error)
lead = manifests[0]
for manifest in manifests[1:]:
if (manifest.keys, manifest.dtype, manifest.width, manifest.rids) != (
lead.keys,
lead.dtype,
lead.width,
lead.rids,
):
return OwnerPlan(
actions=[],
owners=[],
error=(
"image manifest mismatch between group ranks 0 and "
f"{manifest.rank}: rids={lead.rids} vs {manifest.rids}, "
f"keys={lead.keys} vs {manifest.keys}, "
f"dtype={lead.dtype} vs {manifest.dtype}, "
f"width={lead.width} vs {manifest.width}"
),
)
replication = len(manifests)
actions: List[int] = []
owners: List[int] = []
for index, key in enumerate(lead.keys):
owner = key.hash % replication
if all(manifest.cached[index] for manifest in manifests):
action = LOCAL_HIT
elif manifests[owner].cached[index]:
action = OWNER_CACHE_BROADCAST
else:
action = OWNER_ENCODE_BROADCAST
actions.append(action)
owners.append(owner)
return OwnerPlan(actions=actions, owners=owners)
def _prepare_transfers(
session: MmOwnerSession,
requests: Sequence[ImageSpanRequest],
keys: List[ImageSpanKey],
plan: OwnerPlan,
features: Dict[SpanKey, torch.Tensor],
encode: SpanEncoder,
) -> Dict[int, torch.Tensor]:
rank = session.group.rank_in_group
buffers: Dict[int, torch.Tensor] = {}
owned: List[int] = []
for index, (action, owner) in enumerate(zip(plan.actions, plan.owners)):
if action == LOCAL_HIT:
continue
if owner != rank:
try:
buffers[index] = _new_span_buffer(session, keys[index])
except Exception as exc:
raise RuntimeError(
f"receive buffer for image hash {keys[index].hash} shape "
f"{(keys[index].span_len, session.width)} {session.dtype} "
f"failed: {type(exc).__name__}: {exc}"
) from exc
elif action == OWNER_CACHE_BROADCAST:
key = (keys[index].hash, keys[index].span_len)
buffers[index] = features[key].contiguous()
else:
owned.append(index)
if owned:
owned_hashes = [keys[index].hash for index in owned]
try:
encoded = encode([requests[index].item for index in owned])
except Exception as exc:
raise RuntimeError(
f"owner encode of image hashes {owned_hashes} failed: "
f"{type(exc).__name__}: {exc}"
) from exc
spans = _split_spans(encoded, [keys[index].span_len for index in owned])
for index, span in zip(owned, spans):
buffers[index] = _validated_span(session, keys[index], span)
return buffers
def _new_span_buffer(session: MmOwnerSession, key: ImageSpanKey) -> torch.Tensor:
return torch.empty(
(key.span_len, session.width), device=session.device, dtype=session.dtype
)
def _split_spans(
encoded: torch.Tensor | List[torch.Tensor], span_lens: List[int]
) -> List[torch.Tensor]:
if isinstance(encoded, list):
if len(encoded) != len(span_lens):
raise ValueError(
f"encoder returned {len(encoded)} spans for {len(span_lens)} images"
)
return [span.reshape(-1, span.shape[-1]) for span in encoded]
encoded = encoded.reshape(-1, encoded.shape[-1])
if encoded.shape[0] != sum(span_lens):
raise ValueError(
f"encoder returned {encoded.shape[0]} rows for spans of {span_lens}"
)
return list(torch.split(encoded, span_lens, dim=0))
def _validated_span(
session: MmOwnerSession, key: ImageSpanKey, span: torch.Tensor
) -> torch.Tensor:
expected = (key.span_len, session.width)
if tuple(span.shape) != expected or span.dtype != session.dtype:
raise ValueError(
f"encoded span for hash={key.hash} has shape {tuple(span.shape)} "
f"dtype {span.dtype}; expected {expected} {session.dtype}"
)
if span.device != session.device:
span = span.to(session.device)
return span.contiguous()
def _synchronize(device) -> None:
if device.type == "cuda":
torch.cuda.current_stream(device).synchronize()
def _describe(session: MmOwnerSession, stage: str, exc: BaseException) -> str:
return (
f"multimodal owner protocol failed during {stage} on group rank "
f"{session.group.rank_in_group} (global rank "
f"{session.group.ranks[session.group.rank_in_group]}, rids={list(session.rids)}): "
f"{type(exc).__name__}: {exc}"
)
def _raise_first_error(
statuses: List[RankStatus], cause: Optional[BaseException]
) -> None:
for status in statuses:
if status.error is not None:
raise MmOwnerProtocolError(status.error) from cause
+31 -77
View File
@@ -5,7 +5,6 @@ from typing import Callable, Dict, List, Optional, Tuple
import torch import torch
from sglang.srt.managers.mm_owner_embedding import ImageSpanRequest, MmOwnerSession
from sglang.srt.managers.schedule_batch import MultimodalDataItem from sglang.srt.managers.schedule_batch import MultimodalDataItem
from sglang.srt.mem_cache.multimodal_cache import EmbeddingResult, MultiModalStaticCache from sglang.srt.mem_cache.multimodal_cache import EmbeddingResult, MultiModalStaticCache
from sglang.srt.multimodal.evs import EVSEmbeddingResult from sglang.srt.multimodal.evs import EVSEmbeddingResult
@@ -340,24 +339,43 @@ def _batch_encode_per_image_misses(
unique_misses: Dict[Tuple[Optional[int], int], Tuple[MultimodalDataItem, int]] = {} unique_misses: Dict[Tuple[Optional[int], int], Tuple[MultimodalDataItem, int]] = {}
hash_to_embedding: Dict[Tuple[Optional[int], int], torch.Tensor] = {} hash_to_embedding: Dict[Tuple[Optional[int], int], torch.Tensor] = {}
# Phase 1a: collect cache misses over the unique overlapping spans # Phase 1a: find overlapping items per request and collect cache misses
for span in _collect_image_span_requests(per_image_requests): for req_info in per_image_requests:
cache_key = (span.hash, span.span_len) chunk_start = req_info.extend_prefix_len
cached = embedding_cache.get_single(span.hash) chunk_end = chunk_start + req_info.extend_seq_len # exclusive
overlapping = []
if req_info.extend_seq_len > 0:
for idx, (item, (start, end)) in enumerate(
zip(req_info.items, req_info.items_offset)
):
if end >= chunk_start and start < chunk_end:
overlapping.append((idx, item, start, end))
req_info.overlapping = overlapping
for _idx, item, start, end in overlapping:
expected_token_count = end - start + 1
cache_key = (item.hash, expected_token_count)
if cache_key in hash_to_embedding:
continue
cached = embedding_cache.get_single(item.hash)
if cached is not None: if cached is not None:
cached_embedding = cached.embedding cached_embedding = cached.embedding
cached_token_count = _embedding_token_count(cached_embedding) cached_token_count = _embedding_token_count(cached_embedding)
if cached_token_count == span.span_len: if cached_token_count == expected_token_count:
hash_to_embedding[cache_key] = cached_embedding hash_to_embedding[cache_key] = cached_embedding
continue else:
_discard_mismatched_cached_embedding( _discard_mismatched_cached_embedding(
span.hash, span.span_len, cached_token_count item.hash, expected_token_count, cached_token_count
) )
elif ( unique_misses[cache_key] = (item, expected_token_count)
span.inside_chunk and span.item.can_defer_cuda_ipc_feature_reconstruction() elif cache_key not in unique_misses:
if (
start >= chunk_start
and end < chunk_end
and item.can_defer_cuda_ipc_feature_reconstruction()
): ):
span.item.model_specific_data[BORROW_CUDA_IPC_FEATURE_KEY] = True item.model_specific_data[BORROW_CUDA_IPC_FEATURE_KEY] = True
unique_misses[cache_key] = (span.item, span.span_len) unique_misses[cache_key] = (item, expected_token_count)
# Phase 1b: single ViT call for all unique cache misses # Phase 1b: single ViT call for all unique cache misses
if unique_misses: if unique_misses:
@@ -394,52 +412,6 @@ def _batch_encode_per_image_misses(
return hash_to_embedding return hash_to_embedding
def _collect_image_span_requests(
per_image_requests: List[PerImageRequestInfo],
) -> List[ImageSpanRequest]:
spans: Dict[
Tuple[Optional[int], int],
Tuple[MultimodalDataItem, bool, List[MultimodalDataItem]],
] = {}
for req_info in per_image_requests:
chunk_start = req_info.extend_prefix_len
chunk_end = chunk_start + req_info.extend_seq_len # exclusive
overlapping = []
if req_info.extend_seq_len > 0:
for idx, (item, (start, end)) in enumerate(
zip(req_info.items, req_info.items_offset)
):
if end >= chunk_start and start < chunk_end:
overlapping.append((idx, item, start, end))
req_info.overlapping = overlapping
for _idx, item, start, end in overlapping:
cache_key = (item.hash, end - start + 1)
if cache_key in spans:
spans[cache_key][2].append(item)
continue
spans[cache_key] = (item, start >= chunk_start and end < chunk_end, [])
return [
ImageSpanRequest(
hash=item_hash,
span_len=span_len,
item=item,
inside_chunk=inside_chunk,
duplicates=duplicates,
)
for (item_hash, span_len), (item, inside_chunk, duplicates) in spans.items()
]
def _owner_span_encoder(data_embedding_func: DataEmbeddingFunc, device: torch.device):
def encode(items: List[MultimodalDataItem]):
if not _can_skip_pre_embed_feature_move(data_embedding_func):
_move_items_to_device(items, device)
return data_embedding_func(items)
return encode
def _get_chunked_embedding_by_item( def _get_chunked_embedding_by_item(
data_embedding_func: DataEmbeddingFunc, data_embedding_func: DataEmbeddingFunc,
embedding_items_per_req: List[MultimodalDataItem], embedding_items_per_req: List[MultimodalDataItem],
@@ -565,7 +537,6 @@ def _get_chunked_prefill_embedding(
extend_length: List[int], extend_length: List[int],
items_offset_list: List[List[Tuple[int, int]]], items_offset_list: List[List[Tuple[int, int]]],
input_ids: torch.Tensor, input_ids: torch.Tensor,
mm_owner: Optional[MmOwnerSession] = None,
) -> tuple[torch.Tensor | None, torch.Tensor]: ) -> tuple[torch.Tensor | None, torch.Tensor]:
""" """
Chunked prefill embedding: encode items across all requests and extract Chunked prefill embedding: encode items across all requests and extract
@@ -627,22 +598,7 @@ def _get_chunked_prefill_embedding(
# Phase 1: batch encode all per-image cache misses in ONE ViT call # Phase 1: batch encode all per-image cache misses in ONE ViT call
hash_to_embedding: Dict[Tuple[Optional[int], int], torch.Tensor] = {} hash_to_embedding: Dict[Tuple[Optional[int], int], torch.Tensor] = {}
if per_image_requests and mm_owner is not None: if per_image_requests:
# The owner protocol must see every overlapping span before any local
# cache filtering: a rank-local hit can never skip a group collective.
span_requests = _collect_image_span_requests(per_image_requests)
if mm_owner.engaged:
hash_to_embedding = mm_owner.resolve(
span_requests,
cache=embedding_cache,
encode=_owner_span_encoder(data_embedding_func, device),
)
elif span_requests:
raise RuntimeError(
"owner eligibility saw no image span in this chunk, but "
f"scheduling found {len(span_requests)}"
)
elif per_image_requests:
hash_to_embedding = _batch_encode_per_image_misses( hash_to_embedding = _batch_encode_per_image_misses(
data_embedding_func, per_image_requests, device data_embedding_func, per_image_requests, device
) )
@@ -745,7 +701,6 @@ def get_embedding_and_mask(
prefix_length: List[int], prefix_length: List[int],
extend_length: List[int], extend_length: List[int],
items_offset_list: List[List[Tuple[int, int]]], items_offset_list: List[List[Tuple[int, int]]],
mm_owner: Optional[MmOwnerSession] = None,
) -> Tuple[torch.Tensor | None, torch.Tensor | None, torch.Tensor]: ) -> Tuple[torch.Tensor | None, torch.Tensor | None, torch.Tensor]:
""" """
Generate multimodal embeddings and create a mask for identifying their positions in the input sequence. Generate multimodal embeddings and create a mask for identifying their positions in the input sequence.
@@ -786,7 +741,6 @@ def get_embedding_and_mask(
extend_length, extend_length,
items_offset_list, items_offset_list,
input_ids, input_ids,
mm_owner=mm_owner,
) )
if embedding is None: if embedding is None:
return None, None, input_ids return None, None, input_ids
+1 -12
View File
@@ -10,7 +10,6 @@ import pickle
import sys import sys
from abc import abstractmethod from abc import abstractmethod
from collections import defaultdict from collections import defaultdict
from contextlib import nullcontext
from multiprocessing import shared_memory from multiprocessing import shared_memory
from typing import Any, Dict, List, Optional, Tuple from typing import Any, Dict, List, Optional, Tuple
@@ -25,7 +24,6 @@ from sglang.srt.managers.io_struct import (
TokenizedEmbeddingReqInput, TokenizedEmbeddingReqInput,
TokenizedGenerateReqInput, TokenizedGenerateReqInput,
) )
from sglang.srt.managers.mm_owner_embedding import MmOwnerSession
# Preserve the existing initialization import for downstream callers. # Preserve the existing initialization import for downstream callers.
from sglang.srt.managers.mm_schedule import ( from sglang.srt.managers.mm_schedule import (
@@ -399,7 +397,6 @@ def embed_mm_inputs(
data_embedding_func_mapping: Dict[Modality, DataEmbeddingFunc] = None, data_embedding_func_mapping: Dict[Modality, DataEmbeddingFunc] = None,
placeholder_tokens: dict[Modality, List[int]] = None, placeholder_tokens: dict[Modality, List[int]] = None,
use_deepstack: Dict[Modality, bool] = {}, use_deepstack: Dict[Modality, bool] = {},
mm_owner: Optional[MmOwnerSession] = None,
) -> Optional[torch.Tensor]: ) -> Optional[torch.Tensor]:
""" """
Embed multimodal inputs and integrate them with text token embeddings. Embed multimodal inputs and integrate them with text token embeddings.
@@ -481,7 +478,6 @@ def embed_mm_inputs(
prefix_length=extend_prefix_lens, prefix_length=extend_prefix_lens,
extend_length=extend_seq_lens, extend_length=extend_seq_lens,
items_offset_list=items_offsets, items_offset_list=items_offsets,
mm_owner=mm_owner,
) )
if use_deepstack.get(modality, None) and embedding is not None: if use_deepstack.get(modality, None) and embedding is not None:
@@ -502,11 +498,6 @@ def embed_mm_inputs(
# filled with the hash values of the multimodal for the prefix matching in the radix attention. # filled with the hash values of the multimodal for the prefix matching in the radix attention.
# There values are useless because their embeddings will be replaced by vision embeddings anyway. # There values are useless because their embeddings will be replaced by vision embeddings anyway.
input_ids.clamp_(min=0, max=vocab_size - 1) input_ids.clamp_(min=0, max=vocab_size - 1)
if mm_owner is not None:
# The text embedding may all-reduce across TP; a rank-local failure in
# feature preparation has to be agreed on before any rank enters it.
mm_owner.features_ready()
with mm_owner.uncaptured() if mm_owner is not None else nullcontext():
input_embeds = input_embedding(input_ids) input_embeds = input_embedding(input_ids)
# deepstack embedding # deepstack embedding
@@ -534,9 +525,7 @@ def embed_mm_inputs(
_scatter_mm_embedding(dest=input_embeds, mask=mask, src=embedding) _scatter_mm_embedding(dest=input_embeds, mask=mask, src=embedding)
if use_deepstack.get(modality, None): if use_deepstack.get(modality, None):
_scatter_mm_embedding( _scatter_mm_embedding(
dest=input_deepstack_embeds, dest=input_deepstack_embeds, mask=mask, src=deepstack_embeddings[i]
mask=mask,
src=deepstack_embeddings[i],
) )
return input_embeds, other_info return input_embeds, other_info
@@ -1019,15 +1019,6 @@ class PrefillAdder:
else AddReqResult.CONTINUE else AddReqResult.CONTINUE
) )
def can_share_extend_batch(self, req: Req) -> bool:
# Token embedding overrides embed the batch's raw input_ids before the
# model runs, and that lookup cannot index multimodal placeholder hash IDs.
if req.positional_embed_overrides is not None:
return all(r.multimodal_inputs is None for r in self.can_run_list)
if req.multimodal_inputs is not None:
return all(r.positional_embed_overrides is None for r in self.can_run_list)
return True
def add_chunked_req(self, req: Req): def add_chunked_req(self, req: Req):
if self.dllm_config is not None: if self.dllm_config is not None:
_rem_tokens = self._get_dllm_remain_tokens() _rem_tokens = self._get_dllm_remain_tokens()
+3 -34
View File
@@ -2097,13 +2097,9 @@ class Scheduler(
vmm_errors = self._materialize_cuda_vmm_inputs(recv_req) vmm_errors = self._materialize_cuda_vmm_inputs(recv_req)
# Skip health check when server is busy — ongoing requests already carry health info. # Skip health check when server is busy — ongoing requests already carry health info.
# NOTE: the admit/skip decision must be identical on every CP/TP rank. if is_health_check_generate_req(recv_req) and not self.is_fully_idle(
# is_fully_idle() includes rank-local hicache drain queues, which diverge for_health_check=True
# across ranks right after activity; a divergent decision lets one rank ):
# dispatch the health-check generate while others piggyback-skip, breaking
# collective ordering (deadlock: one rank blocks in the hicache drain
# all_reduce while another waits in the CP request broadcast).
if is_health_check_generate_req(recv_req) and not self.is_sched_idle_cp_symmetric():
self.return_health_check_ipcs.append( self.return_health_check_ipcs.append(
getattr(recv_req, "http_worker_ipc", None) getattr(recv_req, "http_worker_ipc", None)
) )
@@ -3944,8 +3940,6 @@ class Scheduler(
for req in self.waiting_queue: for req in self.waiting_queue:
if self.enable_lora and not self.can_schedule_lora_req(req, running_loras): if self.enable_lora and not self.can_schedule_lora_req(req, running_loras):
continue continue
if not adder.can_share_extend_batch(req):
break
running_bs = len(running_batch.reqs) running_bs = len(running_batch.reqs)
candidate_beam_width = ( candidate_beam_width = (
@@ -4986,31 +4980,6 @@ class Scheduler(
else: else:
self.metrics_reporter.record_scheduler_active() self.metrics_reporter.record_scheduler_active()
def is_sched_idle_cp_symmetric(self) -> bool:
"""Idle check using only state that is identical across CP/TP ranks.
Request/batch/queue state is collectively maintained (requests arrive
via broadcast, batches are collectively scheduled), so every rank
computes the same result. Rank-local hicache drain and disagg transfer
queues are deliberately excluded: those are exactly the terms that
diverge across ranks and caused the CP health-check deadlock
(hicache drain all_reduce vs CP request broadcast cross-collective
wait, seen on cp2/cp4 + hicache L3 right after router health checks).
Used only for health-check admission; all other idle logic keeps using
is_fully_idle().
"""
return (
self.running_batch.is_empty()
and self.chunked_req is None
and not self.dllm_manager.any_staging_reqs()
and (self.last_batch is None or self.last_batch.is_empty())
and (not self.enable_overlap or len(self.result_queue) == 0)
and self._pp_microbatches_drained()
and len(self.waiting_queue) == 0
and len(self.grammar_manager.grammar_queue) == 0
)
def is_fully_idle(self, for_health_check=False) -> bool: def is_fully_idle(self, for_health_check=False) -> bool:
# Health check piggybacks on running requests in process_output. # Health check piggybacks on running requests in process_output.
# Only running_batch + waiting_queue guarantee active GPU processing; # Only running_batch + waiting_queue guarantee active GPU processing;
@@ -1279,16 +1279,6 @@ class TokenizerManager(TokenizerControlMixin, TokenizerManagerScoreMixin):
raise ValueError( raise ValueError(
"encoder SWA replay cannot return cached prompt logprobs" "encoder SWA replay cannot return cached prompt logprobs"
) )
requests_embed_overrides = obj.positional_embed_overrides is not None or (
isinstance(obj, EmbeddingReqInput)
and obj.embed_overrides is not None
and obj.embed_override_token_id is not None
)
if requests_embed_overrides and obj.contains_mm_input():
raise ValueError(
"embedding overrides cannot be combined with image, video, or audio "
"inputs"
)
_max_req_len = self.context_len _max_req_len = self.context_len
input_token_num = len(input_ids) if input_ids is not None else 0 input_token_num = len(input_ids) if input_ids is not None else 0
input_token_num += self.num_reserved_tokens input_token_num += self.num_reserved_tokens
-4
View File
@@ -155,10 +155,6 @@ def free_kv_row_segments(
def maybe_cache_unfinished_req(req: Req, tree_cache: BasePrefixCache, **kwargs): def maybe_cache_unfinished_req(req: Req, tree_cache: BasePrefixCache, **kwargs):
if getattr(req, "skip_radix_cache_insert", False): if getattr(req, "skip_radix_cache_insert", False):
kv_indices = tree_cache.req_to_token_pool.req_to_token[
req.kv.req_pool_idx, : len(req.get_fill_ids())
]
req.prefix_indices = kv_indices.to(dtype=torch.int64, copy=True)
return return
tree_cache.cache_unfinished_req(req, **kwargs) tree_cache.cache_unfinished_req(req, **kwargs)
@@ -1646,7 +1646,6 @@ class ModelRunner:
forward_batch.replace_embeds is not None forward_batch.replace_embeds is not None
and forward_batch.replace_positions is not None and forward_batch.replace_positions is not None
): ):
misc_utils.validate_replace_embeds_batch(forward_batch)
# Token embedding overrides: get base embeddings, scatter replacements # Token embedding overrides: get base embeddings, scatter replacements
if "input_embeds" not in kwargs: if "input_embeds" not in kwargs:
embed_layer = self.model.get_input_embeddings() embed_layer = self.model.get_input_embeddings()
@@ -18,7 +18,6 @@ from sglang.srt.server_args import CHUNKED_PREFIX_CACHE_SUPPORTED_ATTENTION_BACK
if TYPE_CHECKING: if TYPE_CHECKING:
from sglang.srt.configs.model_config import ModelConfig from sglang.srt.configs.model_config import ModelConfig
from sglang.srt.model_executor.forward_batch_info import ForwardBatch
logger = logging.getLogger(__name__) logger = logging.getLogger(__name__)
@@ -106,24 +105,3 @@ def resolve_pp_proxy_dspark_hidden_size(
if isinstance(model, _SupportsDSparkPPProxy): if isinstance(model, _SupportsDSparkPPProxy):
return model.get_pp_proxy_dspark_hidden_size() return model.get_pp_proxy_dspark_hidden_size()
return 0 return 0
def validate_replace_embeds_batch(forward_batch: ForwardBatch) -> None:
if forward_batch.mm_inputs is None:
return
for mm_inputs, prefix_len, extend_len in zip(
forward_batch.mm_inputs,
forward_batch.extend_prefix_lens_cpu,
forward_batch.extend_seq_lens_cpu,
):
if mm_inputs is None:
continue
chunk_end = prefix_len + extend_len
for item in mm_inputs.mm_items:
for start, end in item.offsets or ():
if start < chunk_end and end >= prefix_len:
# Placeholder rows carry hash IDs the base embedding lookup cannot index.
raise ValueError(
"Token embedding overrides cannot share an extend batch with "
"multimodal placeholders"
)
@@ -385,23 +385,14 @@ class EagerRunner(BaseRunner):
""" """
model = self.model_runner.model model = self.model_runner.model
input_ids = forward_batch.input_ids
input_embeds = kwargs.get("input_embeds") input_embeds = kwargs.get("input_embeds")
if hasattr(model, "prepare_model_inputs"):
# Multimodal offsets are request-global, so the merge and the
# placeholder-ID remap must see the full extend layout first.
input_ids, input_embeds = model.prepare_model_inputs(
input_ids=input_ids,
forward_batch=forward_batch,
input_embeds=input_embeds,
)
if input_embeds is None: if input_embeds is None:
input_embeds = model.get_input_embeddings()(input_ids) input_embeds = model.get_input_embeddings()(forward_batch.input_ids)
with cp_shard_model_inputs( with cp_shard_model_inputs(
input_embeds, input_embeds,
forward_batch.positions, forward_batch.positions,
forward_batch, forward_batch,
input_ids, forward_batch.input_ids,
) as (sharded_input_embeds, sharded_positions, model_input_ids): ) as (sharded_input_embeds, sharded_positions, model_input_ids):
model_kwargs = {"input_embeds": sharded_input_embeds} model_kwargs = {"input_embeds": sharded_input_embeds}
if (pp_proxy_tensors := kwargs.get("pp_proxy_tensors")) is not None: if (pp_proxy_tensors := kwargs.get("pp_proxy_tensors")) is not None:
@@ -446,7 +437,7 @@ class EagerRunner(BaseRunner):
if aux_hidden_states is None: if aux_hidden_states is None:
logits_kwargs["hidden_states_before_norm"] = hidden_states_before_norm logits_kwargs["hidden_states_before_norm"] = hidden_states_before_norm
return model.logits_processor( return model.logits_processor(
input_ids, forward_batch.input_ids,
hidden_states, hidden_states,
model.lm_head, model.lm_head,
forward_batch, forward_batch,
@@ -701,9 +701,6 @@ class PrefillCudaGraphRunner(BaseCudaGraphRunner):
def _get_layer_model_positions(self, forward_batch: ForwardBatch) -> torch.Tensor: def _get_layer_model_positions(self, forward_batch: ForwardBatch) -> torch.Tensor:
"""Mirror outer multimodal wrappers when BCG captures layer_model directly.""" """Mirror outer multimodal wrappers when BCG captures layer_model directly."""
cp_positions = getattr(forward_batch, "_cp_positions", None)
if cp_positions is not None:
return cp_positions
if forward_batch.mrope_positions is None: if forward_batch.mrope_positions is None:
return forward_batch.positions return forward_batch.positions
@@ -785,9 +782,7 @@ class PrefillCudaGraphRunner(BaseCudaGraphRunner):
if self._uses_eager_prefill_tail(): if self._uses_eager_prefill_tail():
# BCG / Full: capture the transformer body only. # BCG / Full: capture the transformer body only.
positions = self._get_layer_model_positions(forward_batch) positions = self._get_layer_model_positions(forward_batch)
input_ids = getattr( input_ids = forward_batch.input_ids
forward_batch, "_cp_input_ids", forward_batch.input_ids
)
kwargs = _build_layer_model_forward_kwargs( kwargs = _build_layer_model_forward_kwargs(
self.layer_model, forward_batch, pp_proxy_tensors self.layer_model, forward_batch, pp_proxy_tensors
) )
@@ -1341,9 +1336,9 @@ class PrefillCudaGraphRunner(BaseCudaGraphRunner):
batch_max_context_len=batch_max_context_len, batch_max_context_len=batch_max_context_len,
): ):
return False return False
if getattr(self, "enable_cp_bcg_capture", False): if getattr(self, "enable_cp_bcg_capture", False) and is_cp_active(
if not is_cp_active(forward_batch): forward_batch
return False ):
assert self.prefill_cp_bcg_input is not None assert self.prefill_cp_bcg_input is not None
if ( if (
self.prefill_cp_bcg_input.select_replay_bucket_for_batch( self.prefill_cp_bcg_input.select_replay_bucket_for_batch(
+30 -144
View File
@@ -74,7 +74,6 @@ from sglang.srt.layers.communicator_dsa_cp import (
dsa_cp_gather_hidden_states, dsa_cp_gather_hidden_states,
dsa_cp_reduce_scatter_hidden_states, dsa_cp_reduce_scatter_hidden_states,
) )
from sglang.srt.layers.cp.base import is_zigzag
from sglang.srt.layers.cp.cp_decode_attn_tp import get_cp_decode_attn_tp_ctx from sglang.srt.layers.cp.cp_decode_attn_tp import get_cp_decode_attn_tp_ctx
from sglang.srt.layers.cp.utils import ( from sglang.srt.layers.cp.utils import (
cp_gather_full_sequence_states, cp_gather_full_sequence_states,
@@ -121,11 +120,6 @@ from sglang.srt.layers.quantization.mxfp8_input import Mxfp8SwizzledInput
from sglang.srt.layers.rotary_embedding import get_rope_wrapper from sglang.srt.layers.rotary_embedding import get_rope_wrapper
from sglang.srt.layers.utils import PPMissingLayer, get_layer_id from sglang.srt.layers.utils import PPMissingLayer, get_layer_id
from sglang.srt.layers.vocab_parallel_embedding import VocabParallelEmbedding from sglang.srt.layers.vocab_parallel_embedding import VocabParallelEmbedding
from sglang.srt.managers.mm_owner_embedding import (
MmOwnerSession,
has_owner_span_work,
select_owner_group,
)
from sglang.srt.managers.mm_utils import ( from sglang.srt.managers.mm_utils import (
MultiModalityDataPaddingPatternMultimodalTokens, MultiModalityDataPaddingPatternMultimodalTokens,
embed_mm_inputs, embed_mm_inputs,
@@ -188,7 +182,6 @@ from sglang.srt.multimodal.deepseek_v41_image_processing import (
) )
from sglang.srt.runtime_context import ( from sglang.srt.runtime_context import (
get_device, get_device,
get_disagg,
get_exec, get_exec,
get_forward, get_forward,
get_parallel, get_parallel,
@@ -2046,10 +2039,7 @@ class MQALayer(MqaAttentionBase):
if ( if (
forward_batch.forward_mode.is_extend() forward_batch.forward_mode.is_extend()
and is_in_breakable_cuda_graph() and is_in_breakable_cuda_graph()
and ( and not getattr(attn_backend, "low_ratio_prefill_graph", False)
dsa_use_prefill_cp(forward_batch)
or not getattr(attn_backend, "low_ratio_prefill_graph", False)
)
): ):
bcg_deepseek_v4_low_ratio_sources(self, x, q_lora, positions) bcg_deepseek_v4_low_ratio_sources(self, x, q_lora, positions)
else: else:
@@ -2660,8 +2650,7 @@ class DeepseekV4DecoderLayer(nn.Module):
is_nextn=is_nextn, is_nextn=is_nextn,
is_deepseek_v4=True, is_deepseek_v4=True,
vl_correction_bias=config.model_type == "deepseek_v41" vl_correction_bias=config.model_type == "deepseek_v41"
and config.vision_n_layers > 0 and config.vision_n_layers > 0,
and not getattr(config, "language_model_only", False),
) )
self.input_layernorm = RMSNorm(config.hidden_size, eps=config.rms_norm_eps) self.input_layernorm = RMSNorm(config.hidden_size, eps=config.rms_norm_eps)
@@ -3883,14 +3872,6 @@ class DeepseekV4DecoderLayer(nn.Module):
finally: finally:
forward_batch.num_token_non_padded = saved_num_token_non_padded forward_batch.num_token_non_padded = saved_num_token_non_padded
if _use_cp and get_moe_a2a_backend().is_none(): if _use_cp and get_moe_a2a_backend().is_none():
if self.config.model_type == "deepseek_v41":
parallel = get_parallel()
hidden_states = parallel.tp_group.all_reduce(hidden_states)
parallel = get_parallel()
hidden_states = hidden_states.tensor_split(parallel.attn_cp_size)[
parallel.attn_cp_rank
].contiguous()
else:
hidden_states = dsa_cp_reduce_scatter_hidden_states(hidden_states) hidden_states = dsa_cp_reduce_scatter_hidden_states(hidden_states)
elif _use_tp_moe_gather: elif _use_tp_moe_gather:
hidden_states, global_hidden_states = ( hidden_states, global_hidden_states = (
@@ -4421,18 +4402,11 @@ class DeepseekV4Model(nn.Module):
) )
if self.engram_hasher is not None: if self.engram_hasher is not None:
if cp_extend: if cp_extend:
# N-gram hashing needs each token's predecessors, so hash the # n-gram hashing needs each token's predecessors: hash the whole prompt
# whole prompt before selecting this CP rank's interleaved rows.
# The hasher builds request-to-token indices dynamically; keep
# that work at an eager break during breakable graph capture.
total = int(forward_batch.attn_cp_metadata.total_seq_lens) total = int(forward_batch.attn_cp_metadata.total_seq_lens)
global_input_ids = forward_batch.input_ids[:total] hash_ids = self.engram_hasher(
if is_in_breakable_cuda_graph(): forward_batch.input_ids[:total], forward_batch
hash_ids = bcg_deepseek_v4_engram_hash_ids(
self.engram_hasher, global_input_ids
) )
else:
hash_ids = self.engram_hasher(global_input_ids, forward_batch)
parallel = get_parallel() parallel = get_parallel()
hash_ids = hash_ids[parallel.attn_cp_rank :: parallel.attn_cp_size] hash_ids = hash_ids[parallel.attn_cp_rank :: parallel.attn_cp_size]
pad_rows = hidden_states.shape[0] - hash_ids.shape[0] pad_rows = hidden_states.shape[0] - hash_ids.shape[0]
@@ -4865,13 +4839,6 @@ class DeepseekV4Model(nn.Module):
return hidden_states, pre_hc_head return hidden_states, pre_hc_head
def _v41_vision_a2a_supported() -> bool:
backend = get_moe_a2a_backend()
return backend.is_none() or (
backend.is_megamoe() and get_disagg().disaggregation_mode == "decode"
)
class DeepseekV4ForCausalLM(nn.Module): class DeepseekV4ForCausalLM(nn.Module):
supports_cuda_vmm_feature_transport = True supports_cuda_vmm_feature_transport = True
@@ -4897,23 +4864,14 @@ class DeepseekV4ForCausalLM(nn.Module):
self.wo_a_fp8 = wo_a_fp8_gemm_enabled(quant_config) self.wo_a_fp8 = wo_a_fp8_gemm_enabled(quant_config)
self.determine_num_fused_shared_experts() self.determine_num_fused_shared_experts()
self.vision = None self.vision = None
if config.model_type == "deepseek_v41" and config.vision_n_layers > 0:
if ( if (
config.model_type == "deepseek_v41" get_parallel().attn_cp_size != 1
and config.vision_n_layers > 0 or get_parallel().pp_group.world_size != 1
and not getattr(config, "language_model_only", False) or not get_moe_a2a_backend().is_none()
):
if (
get_parallel().pp_group.world_size != 1
or not _v41_vision_a2a_supported()
): ):
raise ValueError( raise ValueError(
"V4.1 vision supports TP/EP/DP without PP; " "V4.1 vision currently supports TP/EP/DP without CP, PP or MoE A2A"
"MoE A2A is supported only with MegaMoE on a PD decode node"
)
if get_parallel().attn_cp_size != 1 and (_is_npu or is_zigzag()):
raise ValueError(
"V4.1 vision context parallelism requires the CUDA interleave "
"strategy; NPU and zigzag CP are not supported yet"
) )
args = SimpleNamespace(**vars(config), dim=config.hidden_size) args = SimpleNamespace(**vars(config), dim=config.hidden_size)
@@ -4922,11 +4880,6 @@ class DeepseekV4ForCausalLM(nn.Module):
self.image_start = nn.Parameter(torch.empty(config.hidden_size)) self.image_start = nn.Parameter(torch.empty(config.hidden_size))
self.image_end = nn.Parameter(torch.empty(config.hidden_size)) self.image_end = nn.Parameter(torch.empty(config.hidden_size))
self.image_newline = nn.Parameter(torch.empty(config.hidden_size)) self.image_newline = nn.Parameter(torch.empty(config.hidden_size))
self.mm_owner_group = (
select_owner_group(get_parallel())
if self.vision is not None and _is_cuda
else None
)
self.model = DeepseekV4Model( self.model = DeepseekV4Model(
config, quant_config, prefix=add_prefix("model", prefix) config, quant_config, prefix=add_prefix("model", prefix)
) )
@@ -5061,42 +5014,7 @@ class DeepseekV4ForCausalLM(nn.Module):
spans.append(span) spans.append(span)
return spans return spans
def _image_span_signature(self, item, span_len: int): def _prepare_mm_embeddings(self, input_ids, forward_batch):
h, w = int(item.n_vit_h), int(item.n_vit_w)
r = self.config.vision_downsample_ratio
expected = len(image_token_types((h + r - 1) // r, (w + r - 1) // r))
if expected != span_len:
raise ValueError(
f"image grid {(h, w)} yields {expected} span tokens, "
f"placeholder has {span_len}"
)
plan = item.model_specific_data.get(GPU_PLAN_KEY)
feature = item.feature
return (
h,
w,
tuple(feature.shape) if isinstance(feature, torch.Tensor) else None,
None if plan is None else tuple(sorted(plan.items())),
)
def _mm_owner_session(self, forward_batch) -> Optional[MmOwnerSession]:
if self.mm_owner_group is None:
return None
return MmOwnerSession(
group=self.mm_owner_group,
device=self.image_start.device,
dtype=self.image_start.dtype,
width=self.config.hidden_size,
rids=list(forward_batch.rids or ()),
signature=self._image_span_signature,
engaged=has_owner_span_work(
forward_batch.mm_inputs,
forward_batch.extend_prefix_lens_cpu,
forward_batch.extend_seq_lens_cpu,
),
)
def _prepare_mm_embeddings(self, input_ids, forward_batch, mm_owner):
# Keep scheduler hash IDs intact: the shared embedder clamps its input in place. # Keep scheduler hash IDs intact: the shared embedder clamps its input in place.
input_embeds, _ = embed_mm_inputs( input_embeds, _ = embed_mm_inputs(
mm_inputs_list=[ mm_inputs_list=[
@@ -5108,7 +5026,6 @@ class DeepseekV4ForCausalLM(nn.Module):
input_ids=input_ids.clone(), input_ids=input_ids.clone(),
input_embedding=self.get_input_embeddings(), input_embedding=self.get_input_embeddings(),
multimodal_model=self, multimodal_model=self,
mm_owner=mm_owner,
) )
forward_batch.mm_input_embeds = input_embeds forward_batch.mm_input_embeds = input_embeds
return input_embeds return input_embeds
@@ -5116,41 +5033,6 @@ class DeepseekV4ForCausalLM(nn.Module):
def get_input_embeddings(self) -> nn.Module: def get_input_embeddings(self) -> nn.Module:
return self.model.get_input_embeddings() return self.model.get_input_embeddings()
def prepare_model_inputs(
self,
input_ids: torch.Tensor,
forward_batch: ForwardBatch,
input_embeds: Optional[torch.Tensor],
) -> Tuple[torch.Tensor, Optional[torch.Tensor]]:
if self.vision is None:
return input_ids, input_embeds
has_images = (
not forward_batch.forward_mode.is_decode()
and not forward_batch.forward_mode.is_target_verify()
and forward_batch.mm_inputs is not None
and any(x is not None for x in forward_batch.mm_inputs)
)
if has_images and input_embeds is not None:
raise ValueError("Cannot combine input_embeds and image inputs")
mm_owner = self._mm_owner_session(forward_batch) if has_images else None
# Peers may only enter the body or the CP shard once every rank has
# finished all of its fallible input preparation, the remap included.
with mm_owner.fence() if mm_owner is not None else nullcontext():
if has_images:
input_embeds = self._prepare_mm_embeddings(
input_ids, forward_batch, mm_owner
)
if not (
forward_batch.forward_mode.is_decode_or_idle()
or forward_batch.forward_mode.is_target_verify()
):
# Decode/verify IDs are already vocabulary IDs; remap prompt image
# hashes for Engram and routing.
input_ids = input_ids.masked_fill(
input_ids >= MM_PAD_SHIFT_VALUE, self.config.image_token_id
)
return input_ids, input_embeds
def set_dspark_layers_to_capture(self, layer_ids: List[int]) -> None: def set_dspark_layers_to_capture(self, layer_ids: List[int]) -> None:
if not self.pp_group.is_last_rank: if not self.pp_group.is_last_rank:
return return
@@ -5196,19 +5078,6 @@ class DeepseekV4ForCausalLM(nn.Module):
0 if is_shared_experts_fusion_disabled() else self.config.n_shared_experts 0 if is_shared_experts_fusion_disabled() else self.config.n_shared_experts
) )
def prepare_language_model_inputs(
self,
input_ids: torch.Tensor,
forward_batch: ForwardBatch,
input_embeds: Optional[torch.Tensor] = None,
pp_proxy_tensors: Optional[PPProxyTensors] = None,
) -> torch.Tensor:
input_ids, input_embeds = self.prepare_model_inputs(
input_ids=input_ids, forward_batch=forward_batch, input_embeds=input_embeds
)
return input_ids, input_embeds
def forward( def forward(
self, self,
input_ids: torch.Tensor, input_ids: torch.Tensor,
@@ -5217,9 +5086,26 @@ class DeepseekV4ForCausalLM(nn.Module):
input_embeds: Optional[torch.Tensor] = None, input_embeds: Optional[torch.Tensor] = None,
pp_proxy_tensors: Optional[PPProxyTensors] = None, pp_proxy_tensors: Optional[PPProxyTensors] = None,
) -> torch.Tensor: ) -> torch.Tensor:
input_ids, input_embeds = self.prepare_language_model_inputs( if (
input_ids, forward_batch, input_embeds self.vision is not None
and not forward_batch.forward_mode.is_decode()
and not forward_batch.forward_mode.is_target_verify()
and forward_batch.mm_inputs is not None
and any(x is not None for x in forward_batch.mm_inputs)
):
if input_embeds is not None:
raise ValueError("Cannot combine input_embeds and image inputs")
input_embeds = self._prepare_mm_embeddings(input_ids, forward_batch)
if self.vision is not None and not (
forward_batch.forward_mode.is_decode_or_idle()
or forward_batch.forward_mode.is_target_verify()
):
# Decode/verify IDs are already vocabulary IDs; remap prompt image
# hashes for Engram and routing.
input_ids = input_ids.masked_fill(
input_ids >= MM_PAD_SHIFT_VALUE, self.config.image_token_id
) )
with get_attn_tp_context().maybe_input_scattered(forward_batch): with get_attn_tp_context().maybe_input_scattered(forward_batch):
hidden_states = self.model.forward( hidden_states = self.model.forward(
input_ids, positions, forward_batch, input_embeds, pp_proxy_tensors input_ids, positions, forward_batch, input_embeds, pp_proxy_tensors
@@ -222,7 +222,6 @@ class DeepseekV4ForCausalLMNextN(DeepseekV4ForCausalLM):
self.quant_config = quant_config self.quant_config = quant_config
self.wo_a_fp8 = wo_a_fp8_gemm_enabled(quant_config) self.wo_a_fp8 = wo_a_fp8_gemm_enabled(quant_config)
self.determine_num_fused_shared_experts() self.determine_num_fused_shared_experts()
self.vision = None
self.model = DeepseekV4ModelNextN( self.model = DeepseekV4ModelNextN(
config, quant_config, prefix=add_prefix("model", prefix) config, quant_config, prefix=add_prefix("model", prefix)
-1
View File
@@ -376,7 +376,6 @@ class ServerArgs:
# ===== END TO BE REFACTORED ==== # ===== END TO BE REFACTORED ====
LANGUAGE_MODEL_ONLY_ARCHITECTURES = ( LANGUAGE_MODEL_ONLY_ARCHITECTURES = (
"DeepseekV4ForCausalLM",
"MuseGlimmerForConditionalGeneration", "MuseGlimmerForConditionalGeneration",
"Cosmos3ForConditionalGeneration", "Cosmos3ForConditionalGeneration",
"Cosmos3EdgeForConditionalGeneration", "Cosmos3EdgeForConditionalGeneration",
-79
View File
@@ -1,79 +0,0 @@
"""Small CPU tensors; production CP slicing/gather, mocked collective transport."""
from contextlib import ExitStack, contextmanager, nullcontext
from types import SimpleNamespace as NS
from unittest.mock import patch
import torch
from sglang.srt.layers.cp.interleave import InterleaveCPStrategy
from sglang.srt.layers.cp.padding import pad_logical_token_to_physical
from sglang.srt.model_executor.forward_batch_info import ForwardMode
CP = "sglang.srt.layers.cp"
@contextmanager
def cp_context(size, rank, lengths=(3, 6), prefix_lengths=(7, 13)):
"""Keep real interleave indexing/padding; replace only runtime context."""
strategy = InterleaveCPStrategy(size)
parallel = NS(attn_cp_size=size, attn_cp_rank=rank, attn_cp_group=None)
batch = NS(
forward_mode=ForwardMode.EXTEND,
input_ids=torch.arange(1, sum(lengths) + 1),
positions=torch.cat(
[
torch.arange(prefix, prefix + length)
for prefix, length in zip(prefix_lengths, lengths)
]
),
extend_seq_lens_cpu=list(lengths),
extend_prefix_lens_cpu=list(prefix_lengths),
mm_inputs=None,
spec_info=None,
)
batch.attn_cp_metadata = strategy.build_metadata(
sum(lengths), [p + n for p, n in zip(prefix_lengths, lengths)], list(lengths)
)
with ExitStack() as stack:
for module in ("base", "utils", "padding", "interleave"):
stack.enter_context(
patch(CP + "." + module + ".get_parallel", return_value=parallel)
)
stack.enter_context(patch(CP + ".utils.get_cp_strategy", return_value=strategy))
stack.enter_context(
patch(CP + ".padding.get_cp_padding_align_size", return_value=size)
)
stack.enter_context(
patch(
CP + ".utils.get_moe_a2a_backend", return_value=NS(is_none=lambda: True)
)
)
pad_logical_token_to_physical(batch.attn_cp_metadata)
yield strategy, batch
@contextmanager
def simulated_collective(strategy, batch, global_tensor):
"""Inject peer buffers into all-gather; retain production unpadding/reordering."""
physical = max(batch.attn_cp_metadata.per_rank_actual_token)
buffers = []
for rank in range(strategy.cp_size):
buf = global_tensor.new_zeros((physical, *global_tensor.shape[1:]))
local = global_tensor[rank :: strategy.cp_size]
buf[: len(local)] = local
buffers.append(buf)
def gather(output, local):
torch.testing.assert_close(local, buffers[strategy.cp_rank], rtol=0, atol=0)
output.copy_(torch.cat(buffers))
with (
patch(
CP + ".interleave.use_symmetric_memory",
side_effect=lambda *a, **k: nullcontext(),
),
patch(CP + ".interleave.is_allocation_symmetric", return_value=False),
patch(CP + ".interleave.attn_cp_all_gather_into_tensor", side_effect=gather),
):
yield
@@ -15,7 +15,7 @@ from sglang.srt.layers.attention import aiter_mla_gluon as mod
from sglang.test.ci.ci_register import register_cpu_ci from sglang.test.ci.ci_register import register_cpu_ci
from sglang.test.test_utils import CustomTestCase from sglang.test.test_utils import CustomTestCase
register_cpu_ci(est_time=15, suite="base-a-test-cpu") register_cpu_ci(est_time=8, suite="base-a-test-cpu")
_GLUON_FN = "sglang.srt.layers.attention.aiter_mla_gluon._gluon_fn" _GLUON_FN = "sglang.srt.layers.attention.aiter_mla_gluon._gluon_fn"
@@ -13,7 +13,7 @@ from sglang.test.ci.ci_register import (
register_xpu_ci, register_xpu_ci,
) )
register_cuda_ci(est_time=6, stage="base-b", runner_config="1-gpu-large") register_cuda_ci(est_time=7, stage="base-b", runner_config="1-gpu-large")
register_amd_ci(est_time=11, stage="stage-b", runner_config="1-gpu-large-amd") register_amd_ci(est_time=11, stage="stage-b", runner_config="1-gpu-large-amd")
register_xpu_ci(est_time=900, suite="stage-b-test-1-gpu-xpu") register_xpu_ci(est_time=900, suite="stage-b-test-1-gpu-xpu")
@@ -9,7 +9,7 @@ from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci
from sglang.test.test_utils import CustomTestCase from sglang.test.test_utils import CustomTestCase
# Triton kernel unit test for KV indices creation # Triton kernel unit test for KV indices creation
register_cuda_ci(est_time=10, stage="base-b", runner_config="1-gpu-small") register_cuda_ci(est_time=9, stage="base-b", runner_config="1-gpu-small")
register_amd_ci(est_time=10, suite="stage-b-test-1-gpu-small-amd") register_amd_ci(est_time=10, suite="stage-b-test-1-gpu-small-amd")
@@ -24,7 +24,7 @@ from sglang.test.test_utils import (
is_in_amd_ci, is_in_amd_ci,
) )
register_cuda_ci(est_time=280, stage="base-b", runner_config="1-gpu-large") register_cuda_ci(est_time=272, stage="base-b", runner_config="1-gpu-large")
register_amd_ci(est_time=278, suite="stage-b-test-1-gpu-small-amd") register_amd_ci(est_time=278, suite="stage-b-test-1-gpu-small-amd")
register_xpu_ci(est_time=207, suite="stage-b-test-1-gpu-xpu") register_xpu_ci(est_time=207, suite="stage-b-test-1-gpu-xpu")
@@ -13,7 +13,7 @@ from sglang.test.test_utils import (
) )
# FlashAttention4 integration test (requires SM 100+ / Blackwell B200) # FlashAttention4 integration test (requires SM 100+ / Blackwell B200)
register_cuda_ci(est_time=230, stage="base-b", runner_config="4-gpu-b200") register_cuda_ci(est_time=220, stage="base-b", runner_config="4-gpu-b200")
@unittest.skipIf(get_device_sm() < 100, "Test requires CUDA SM 100 or higher") @unittest.skipIf(get_device_sm() < 100, "Test requires CUDA SM 100 or higher")
@@ -11,7 +11,7 @@ from sglang.test.test_utils import (
# Hybrid attention backend tests (FA3 prefill + FlashInfer decode, requires SM 90+ / H100) # Hybrid attention backend tests (FA3 prefill + FlashInfer decode, requires SM 90+ / H100)
# Multiple test classes: base, MLA, TorchCompile, SpecDecode variants # Multiple test classes: base, MLA, TorchCompile, SpecDecode variants
register_cuda_ci(est_time=393, stage="extra-a", runner_config="1-gpu-large") register_cuda_ci(est_time=368, stage="extra-a", runner_config="1-gpu-large")
class TestHybridAttnBackendMLA(TestHybridAttnBackendBase): class TestHybridAttnBackendMLA(TestHybridAttnBackendBase):
@@ -19,7 +19,7 @@ from sglang.srt.utils.common import get_device
from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci
from sglang.test.test_utils import CustomTestCase from sglang.test.test_utils import CustomTestCase
register_cuda_ci(est_time=25, stage="base-b", runner_config="1-gpu-large") register_cuda_ci(est_time=32, stage="base-b", runner_config="1-gpu-large")
register_amd_ci(est_time=12, stage="stage-b", runner_config="1-gpu-large-amd") register_amd_ci(est_time=12, stage="stage-b", runner_config="1-gpu-large-amd")
@@ -21,7 +21,7 @@ from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci
from sglang.test.test_utils import CustomTestCase from sglang.test.test_utils import CustomTestCase
# Register this test for CUDA CI in base-b (fast attention/kernel tests) # Register this test for CUDA CI in base-b (fast attention/kernel tests)
register_cuda_ci(est_time=11, stage="base-b", runner_config="1-gpu-large") register_cuda_ci(est_time=10, stage="base-b", runner_config="1-gpu-large")
register_amd_ci(est_time=17, suite="stage-b-test-1-gpu-large-amd") register_amd_ci(est_time=17, suite="stage-b-test-1-gpu-large-amd")
@@ -12,7 +12,7 @@ from sglang.test.test_deterministic_utils import (
TestDeterministicBase, TestDeterministicBase,
) )
register_cuda_ci(est_time=119, stage="extra-b", runner_config="4-gpu-h100") register_cuda_ci(est_time=135, stage="extra-b", runner_config="4-gpu-h100")
QWEN35 = "Qwen/Qwen3.5-35B-A3B" QWEN35 = "Qwen/Qwen3.5-35B-A3B"
@@ -18,7 +18,7 @@ from sglang.test.test_utils import (
) )
# Torch native attention backend integration test with MMLU eval # Torch native attention backend integration test with MMLU eval
register_cuda_ci(est_time=310, stage="extra-a", runner_config="1-gpu-small") register_cuda_ci(est_time=312, stage="extra-a", runner_config="1-gpu-small")
register_amd_ci(est_time=150, suite="stage-b-test-1-gpu-small-amd") register_amd_ci(est_time=150, suite="stage-b-test-1-gpu-small-amd")
@@ -15,7 +15,7 @@ from sglang.test.test_utils import (
) )
# Sliding window attention with Triton backend (Gemma-3 model) # Sliding window attention with Triton backend (Gemma-3 model)
register_cuda_ci(est_time=80, stage="extra-a", runner_config="1-gpu-large") register_cuda_ci(est_time=81, stage="extra-a", runner_config="1-gpu-large")
register_amd_ci(est_time=200, suite="stage-b-test-1-gpu-small-amd") register_amd_ci(est_time=200, suite="stage-b-test-1-gpu-small-amd")
@@ -17,7 +17,7 @@ from sglang.test.ci.ci_register import register_cuda_ci
from sglang.test.test_utils import CustomTestCase from sglang.test.test_utils import CustomTestCase
# trtllm_mha kernels are sm100-only; run this kernel-unit test on Blackwell. # trtllm_mha kernels are sm100-only; run this kernel-unit test on Blackwell.
register_cuda_ci(est_time=10, stage="base-b", runner_config="4-gpu-b200") register_cuda_ci(est_time=8, stage="base-b", runner_config="4-gpu-b200")
DEVICE = "cuda" DEVICE = "cuda"
PAGE_SIZE = 32 PAGE_SIZE = 32
@@ -23,7 +23,7 @@ from sglang.srt.model_executor.forward_batch_info import ForwardMode
from sglang.test.ci.ci_register import register_cuda_ci from sglang.test.ci.ci_register import register_cuda_ci
# trtllm_mha kernels are sm100-only; run this kernel-unit test on Blackwell. # trtllm_mha kernels are sm100-only; run this kernel-unit test on Blackwell.
register_cuda_ci(est_time=16, stage="base-b", runner_config="4-gpu-b200") register_cuda_ci(est_time=12, stage="base-b", runner_config="4-gpu-b200")
DEVICE = "cuda" DEVICE = "cuda"
PAGE_SIZE = 128 PAGE_SIZE = 128
@@ -25,7 +25,7 @@ from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci
from sglang.test.test_utils import CustomTestCase from sglang.test.test_utils import CustomTestCase
# Triton kernel unit test for the trtllm_mha device-side page-table build. # Triton kernel unit test for the trtllm_mha device-side page-table build.
register_cuda_ci(est_time=12, stage="base-b", runner_config="1-gpu-small") register_cuda_ci(est_time=10, stage="base-b", runner_config="1-gpu-small")
register_amd_ci(est_time=14, stage="stage-b", runner_config="1-gpu-small-amd") register_amd_ci(est_time=14, stage="stage-b", runner_config="1-gpu-small-amd")
@@ -27,7 +27,7 @@ from sglang.test.test_utils import (
popen_launch_server, popen_launch_server,
) )
register_cuda_ci(est_time=71, stage="base-b", runner_config="1-gpu-large") register_cuda_ci(est_time=90, stage="base-b", runner_config="1-gpu-large")
KIMI_LINEAR_MODEL = "yujiepan/kimi-linear-tiny-random" KIMI_LINEAR_MODEL = "yujiepan/kimi-linear-tiny-random"
SERVER_ENV = {"SGLANG_BATCH_INVARIANT_OPS_ENABLE_MM_DEEPGEMM": "0"} SERVER_ENV = {"SGLANG_BATCH_INVARIANT_OPS_ENABLE_MM_DEEPGEMM": "0"}
@@ -36,7 +36,7 @@ from sglang.test.kits.attention_unittest.attention_methods.dense_attention impor
) )
from sglang.test.test_utils import CustomTestCase from sglang.test.test_utils import CustomTestCase
register_cuda_ci(est_time=11, stage="base-a", runner_config="1-gpu-small") register_cuda_ci(est_time=10, stage="base-a", runner_config="1-gpu-small")
_EXTEND_CASE = DenseAttentionCase( _EXTEND_CASE = DenseAttentionCase(
name="extend_no_prefix_smoke", name="extend_no_prefix_smoke",
@@ -30,7 +30,7 @@ from sglang.test.kits.attention_unittest.runner_modes.split_op_runner import (
) )
from sglang.test.test_utils import CustomTestCase from sglang.test.test_utils import CustomTestCase
register_cuda_ci(est_time=16, stage="base-b", runner_config="1-gpu-large") register_cuda_ci(est_time=18, stage="base-b", runner_config="1-gpu-large")
@unittest.skipIf(not torch.cuda.is_available(), "CUDA is required") @unittest.skipIf(not torch.cuda.is_available(), "CUDA is required")
@@ -29,8 +29,8 @@ from sglang.test.kits.attention_unittest.runner_modes.split_op_runner import (
) )
from sglang.test.test_utils import CustomTestCase from sglang.test.test_utils import CustomTestCase
register_cuda_ci(est_time=27, stage="base-b", runner_config="4-gpu-b200") register_cuda_ci(est_time=23, stage="base-b", runner_config="4-gpu-b200")
register_cuda_ci(est_time=22, stage="base-b", runner_config="1-gpu-large") register_cuda_ci(est_time=15, stage="base-b", runner_config="1-gpu-large")
@unittest.skipIf(not torch.cuda.is_available(), "CUDA is required") @unittest.skipIf(not torch.cuda.is_available(), "CUDA is required")
@@ -26,8 +26,8 @@ from sglang.test.kits.attention_unittest.runner_modes.split_op_runner import (
) )
from sglang.test.test_utils import CustomTestCase from sglang.test.test_utils import CustomTestCase
register_cuda_ci(est_time=18, stage="base-b", runner_config="4-gpu-b200") register_cuda_ci(est_time=16, stage="base-b", runner_config="4-gpu-b200")
register_cuda_ci(est_time=17, stage="base-b", runner_config="1-gpu-large") register_cuda_ci(est_time=26, stage="base-b", runner_config="1-gpu-large")
@unittest.skipIf( @unittest.skipIf(
@@ -14,8 +14,8 @@ from sglang.test.kits.attention_unittest.runner_modes.split_op_runner import (
) )
from sglang.test.test_utils import CustomTestCase from sglang.test.test_utils import CustomTestCase
register_cuda_ci(est_time=18, stage="base-b", runner_config="4-gpu-b200") register_cuda_ci(est_time=16, stage="base-b", runner_config="4-gpu-b200")
register_cuda_ci(est_time=17, stage="base-b", runner_config="1-gpu-large") register_cuda_ci(est_time=36, stage="base-b", runner_config="1-gpu-large")
@unittest.skipIf(not torch.cuda.is_available(), "CUDA is required") @unittest.skipIf(not torch.cuda.is_available(), "CUDA is required")
@@ -18,8 +18,8 @@ from sglang.test.kits.attention_unittest.attention_methods.dense_attention impor
) )
from sglang.test.test_utils import CustomTestCase from sglang.test.test_utils import CustomTestCase
register_cuda_ci(est_time=11, stage="base-b", runner_config="4-gpu-b200") register_cuda_ci(est_time=10, stage="base-b", runner_config="4-gpu-b200")
register_cuda_ci(est_time=10, stage="base-b", runner_config="1-gpu-large") register_cuda_ci(est_time=11, stage="base-b", runner_config="1-gpu-large")
@unittest.skipIf( @unittest.skipIf(
@@ -21,8 +21,8 @@ from sglang.test.kits.attention_unittest.runner_modes.speculative_target_verify_
) )
from sglang.test.test_utils import CustomTestCase from sglang.test.test_utils import CustomTestCase
register_cuda_ci(est_time=11, stage="base-b", runner_config="4-gpu-b200") register_cuda_ci(est_time=10, stage="base-b", runner_config="4-gpu-b200")
register_cuda_ci(est_time=12, stage="base-b", runner_config="1-gpu-large") register_cuda_ci(est_time=11, stage="base-b", runner_config="1-gpu-large")
@unittest.skipIf(not torch.cuda.is_available(), "CUDA is required") @unittest.skipIf(not torch.cuda.is_available(), "CUDA is required")
@@ -11,8 +11,8 @@ from sglang.test.kits.attention_unittest.attention_methods.dense_attention impor
) )
from sglang.test.test_utils import CustomTestCase from sglang.test.test_utils import CustomTestCase
register_cuda_ci(est_time=12, stage="base-b", runner_config="4-gpu-b200") register_cuda_ci(est_time=11, stage="base-b", runner_config="4-gpu-b200")
register_cuda_ci(est_time=11, stage="base-b", runner_config="1-gpu-large") register_cuda_ci(est_time=12, stage="base-b", runner_config="1-gpu-large")
register_amd_ci(est_time=20, suite="stage-b-test-1-gpu-large-amd") register_amd_ci(est_time=20, suite="stage-b-test-1-gpu-large-amd")
@@ -30,8 +30,8 @@ from sglang.test.kits.attention_unittest.runner_modes.split_op_runner import (
) )
from sglang.test.test_utils import CustomTestCase from sglang.test.test_utils import CustomTestCase
register_cuda_ci(est_time=18, stage="base-b", runner_config="4-gpu-b200") register_cuda_ci(est_time=38, stage="base-b", runner_config="4-gpu-b200")
register_cuda_ci(est_time=19, stage="base-b", runner_config="1-gpu-large") register_cuda_ci(est_time=36, stage="base-b", runner_config="1-gpu-large")
register_amd_ci(est_time=25, suite="stage-b-test-1-gpu-large-amd") register_amd_ci(est_time=25, suite="stage-b-test-1-gpu-large-amd")
@@ -28,8 +28,8 @@ from sglang.test.kits.attention_unittest.runner_modes.split_op_runner import (
) )
from sglang.test.test_utils import CustomTestCase from sglang.test.test_utils import CustomTestCase
register_cuda_ci(est_time=17, stage="base-b", runner_config="4-gpu-b200") register_cuda_ci(est_time=15, stage="base-b", runner_config="4-gpu-b200")
register_cuda_ci(est_time=16, stage="base-b", runner_config="1-gpu-large") register_cuda_ci(est_time=18, stage="base-b", runner_config="1-gpu-large")
@unittest.skipIf( @unittest.skipIf(
@@ -30,8 +30,8 @@ from sglang.test.kits.attention_unittest.runner_modes.speculative_draft_runner i
) )
from sglang.test.test_utils import CustomTestCase from sglang.test.test_utils import CustomTestCase
register_cuda_ci(est_time=14, stage="base-b", runner_config="4-gpu-b200") register_cuda_ci(est_time=22, stage="base-b", runner_config="4-gpu-b200")
register_cuda_ci(est_time=14, stage="base-b", runner_config="1-gpu-large") register_cuda_ci(est_time=26, stage="base-b", runner_config="1-gpu-large")
@unittest.skipIf(not torch.cuda.is_available(), "CUDA is required") @unittest.skipIf(not torch.cuda.is_available(), "CUDA is required")
@@ -48,7 +48,7 @@ from sglang.test.kits.attention_unittest.runner_modes.speculative_target_verify_
) )
register_cuda_ci(est_time=14, stage="base-b", runner_config="4-gpu-b200") register_cuda_ci(est_time=14, stage="base-b", runner_config="4-gpu-b200")
register_cuda_ci(est_time=13, stage="base-b", runner_config="1-gpu-large") register_cuda_ci(est_time=16, stage="base-b", runner_config="1-gpu-large")
@unittest.skipIf(not torch.cuda.is_available(), "CUDA is required") @unittest.skipIf(not torch.cuda.is_available(), "CUDA is required")
@@ -25,7 +25,7 @@ from sglang.test.kits.attention_unittest.runner_modes.split_op_runner import (
) )
from sglang.test.test_utils import CustomTestCase from sglang.test.test_utils import CustomTestCase
register_cuda_ci(est_time=15, stage="base-b", runner_config="4-gpu-b200") register_cuda_ci(est_time=14, stage="base-b", runner_config="4-gpu-b200")
register_cuda_ci(est_time=14, stage="base-b", runner_config="1-gpu-large") register_cuda_ci(est_time=14, stage="base-b", runner_config="1-gpu-large")
_cuda_major = int(torch.version.cuda.split(".")[0]) if torch.version.cuda else 0 _cuda_major = int(torch.version.cuda.split(".")[0]) if torch.version.cuda else 0
@@ -45,7 +45,7 @@ import torch
from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci
from sglang.test.test_utils import CustomTestCase from sglang.test.test_utils import CustomTestCase
register_cuda_ci(est_time=11, stage="base-b", runner_config="1-gpu-large") register_cuda_ci(est_time=10, stage="base-b", runner_config="1-gpu-large")
register_amd_ci(est_time=20, suite="stage-b-test-1-gpu-large-amd") register_amd_ci(est_time=20, suite="stage-b-test-1-gpu-large-amd")
@@ -15,7 +15,7 @@ from sglang.test.kits.attention_unittest.runner_modes.split_op_runner import (
) )
from sglang.test.test_utils import CustomTestCase from sglang.test.test_utils import CustomTestCase
register_cuda_ci(est_time=12, stage="base-b", runner_config="4-gpu-b200") register_cuda_ci(est_time=10, stage="base-b", runner_config="4-gpu-b200")
register_cuda_ci(est_time=11, stage="base-b", runner_config="1-gpu-large") register_cuda_ci(est_time=11, stage="base-b", runner_config="1-gpu-large")
register_amd_ci(est_time=20, suite="stage-b-test-1-gpu-large-amd") register_amd_ci(est_time=20, suite="stage-b-test-1-gpu-large-amd")
@@ -29,8 +29,8 @@ from sglang.test.kits.attention_unittest.runner_modes.split_op_runner import (
) )
from sglang.test.test_utils import CustomTestCase from sglang.test.test_utils import CustomTestCase
register_cuda_ci(est_time=13, stage="base-b", runner_config="4-gpu-b200") register_cuda_ci(est_time=11, stage="base-b", runner_config="4-gpu-b200")
register_cuda_ci(est_time=11, stage="base-b", runner_config="1-gpu-large") register_cuda_ci(est_time=12, stage="base-b", runner_config="1-gpu-large")
register_amd_ci(est_time=20, suite="stage-b-test-1-gpu-large-amd") register_amd_ci(est_time=20, suite="stage-b-test-1-gpu-large-amd")

Some files were not shown because too many files have changed in this diff Show More