deploy: k8s RBG 3p1d cw manifest; D cg512 + load-balance-method total_requests

This commit is contained in:
2026-09-28 16:47:16 +08:00
parent e17fbc9a2d
commit 735434fcde
+451
View File
@@ -0,0 +1,451 @@
# ds41-3p1d: DeepSeek-V4.1-Flash 3P1D PD 分离(cs32k + OTLP trace),RBG 版
# 拓扑: 3×prefill(tp2+ep2+cp2, cs32k, dspark, hicache L3) + 1×decode(tp2+ep2+dp2, cg512, dspark,
# load-balance-method=total_requests 按请求数分流) + mooncake-master(hicache L3 store) + router(mini-lb PD)
# 全部钉 b300-01(单机 8 卡,hostNetwork + hostPID——mooncake NVLink-intra 跨 Pod CUDA IPC 必需,
# 等同 docker compose 的 pid: host;参考 AGENTS.md「部署四要素」)
# 镜像已侧载 containerd: docker.io/ymkymx/sglang:dsv41-pd-visioncp-db7d2cb7d-fix(IfNotPresent)
# 参数来源: deploy/b300-ds41/dockerserve-3p1d-otlp-cs32k.yml(docker compose 版)
# 对外: 仅 ClusterIP Service ds41-3p1d:50002(router 用普通 Pod 网络,不占宿主机端口,
# 天然不对宿主机/外网暴露);new-api-gw 用集群 DNS http://ds41-3p1d:50002/v1 访问
# trace: OTLP → 127.0.0.1:4317(同节点 jaeger-trace cw,badger 6h 轮窗)
apiVersion: workloads.x-k8s.io/v1alpha2
kind: RoleBasedGroup
metadata:
name: ds41-3p1d
labels:
app: ds41-3p1d
sglang-model: ds-v41-flash
spec:
roleTemplates:
- name: prefill-base
template:
metadata:
labels:
sglang-model: ds-v41-flash
sglang-role: prefill
spec:
hostNetwork: true
hostPID: true
dnsPolicy: ClusterFirstWithHostNet
nodeSelector:
kubernetes.io/hostname: b300-01
containers:
- name: sglang
image: ymkymx/sglang:dsv41-pd-visioncp-db7d2cb7d-fix
imagePullPolicy: IfNotPresent
command:
- /bin/bash
- -c
- |
exec sglang serve \
--trust-remote-code \
--model-path /data/models/DeepSeek-V4.1-Flash \
--tp 2 --ep-size 2 \
--enable-prefill-cp --cp-strategy interleave \
--chunked-prefill-size 32768 --max-prefill-tokens 32768 \
--mem-fraction-static 0.75 \
--attention-backend dsv4 \
--moe-runner-backend flashinfer_mxfp4 \
--cuda-graph-max-bs-decode 32 \
--reasoning-parser auto --tool-call-parser auto \
--default-chat-template-kwargs '{"thinking": true, "reasoning_effort": "high"}' \
--max-running-requests 256 \
--tokenizer-worker-num 8 \
--host 0.0.0.0 --port "$SGLANG_PORT" \
--enable-cache-report --enable-metrics \
--speculative-algorithm DSPARK --speculative-dspark-block-size 5 \
--enable-hierarchical-cache \
--hicache-ratio 2 \
--hicache-mem-layout page_first_direct \
--hicache-io-backend direct \
--hicache-write-policy write_through \
--hicache-storage-backend mooncake \
--hicache-storage-prefetch-policy wait_complete \
--hicache-size 0 \
--disaggregation-mode prefill \
--disaggregation-transfer-backend mooncake \
--disaggregation-bootstrap-port "$SGLANG_BOOTSTRAP_PORT" \
--enable-trace \
--trace-modules request,mooncake \
--otlp-traces-endpoint 127.0.0.1:4317 \
--otlp-service-name "$OTLP_NAME"
env:
- name: SGLANG_TORCH_PROFILER_DIR
value: /data/ymk/prof/torch
- name: SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE
value: "1"
- name: SGLANG_RAGGED_VERIFY_MODE
value: static
- name: MC_INTRANODE_NVLINK
value: "true"
- name: MC_INTRA_NVLINK
value: "true"
- name: SGLANG_MOONCAKE_SEND_AUX_TCP
value: "1"
- name: SGLANG_HICACHE_MOONCAKE_CONFIG_PATH
value: /data/ymk/ds41/mooncake-store.json
- name: SGLANG_DISAGGREGATION_QUEUE_SIZE
value: "16"
- name: SGLANG_DISAGGREGATION_THREAD_POOL_SIZE
value: "32"
- name: TZ
value: Asia/Shanghai
readinessProbe:
httpGet:
path: /health
port: metrics
initialDelaySeconds: 60
periodSeconds: 10
timeoutSeconds: 5
failureThreshold: 90
resources:
requests:
cpu: "32"
memory: 220Gi
nvidia.com/gpu: "2"
limits:
cpu: "96"
memory: 950Gi
nvidia.com/gpu: "2"
securityContext:
privileged: true
volumeMounts:
- name: data
mountPath: /data
- name: sglang-cache
mountPath: /root/.cache/sglang
- name: dshm
mountPath: /dev/shm
volumes:
- name: data
hostPath:
path: /data
type: Directory
- name: sglang-cache
hostPath:
path: /data/ymk/cache/sglang
type: Directory
- name: dshm
emptyDir:
medium: Memory
sizeLimit: 32Gi
- name: decode-base
template:
metadata:
labels:
sglang-model: ds-v41-flash
sglang-role: decode
spec:
hostNetwork: true
hostPID: true
dnsPolicy: ClusterFirstWithHostNet
nodeSelector:
kubernetes.io/hostname: b300-01
containers:
- name: sglang
image: ymkymx/sglang:dsv41-pd-visioncp-db7d2cb7d-fix
imagePullPolicy: IfNotPresent
command:
- /bin/bash
- -c
- |
exec sglang serve \
--trust-remote-code \
--model-path /data/models/DeepSeek-V4.1-Flash \
--tp 2 --ep-size 2 --dp-size 2 \
--enable-dp-attention --enable-dp-lm-head \
--mem-fraction-static 0.80 \
--attention-backend dsv4 \
--moe-runner-backend flashinfer_mxfp4 \
--cuda-graph-max-bs-decode 512 \
--load-balance-method total_requests \
--reasoning-parser auto --tool-call-parser auto \
--default-chat-template-kwargs '{"thinking": true, "reasoning_effort": "high"}' \
--max-running-requests 256 \
--tokenizer-worker-num 8 \
--host 0.0.0.0 --port 30001 \
--enable-cache-report --enable-metrics \
--speculative-algorithm DSPARK --speculative-dspark-block-size 5 \
--disaggregation-mode decode \
--disaggregation-transfer-backend mooncake \
--disaggregation-bootstrap-port 9001 \
--enable-trace \
--trace-modules request,mooncake \
--otlp-traces-endpoint 127.0.0.1:4317 \
--otlp-service-name ds41-3p1d-d1
env:
- name: SGLANG_TORCH_PROFILER_DIR
value: /data/ymk/prof/torch
- name: SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE
value: "1"
- name: SGLANG_RAGGED_VERIFY_MODE
value: static
- name: MC_INTRANODE_NVLINK
value: "true"
- name: MC_INTRA_NVLINK
value: "true"
- name: SGLANG_MOONCAKE_SEND_AUX_TCP
value: "1"
- name: SGLANG_DISAGGREGATION_QUEUE_SIZE
value: "16"
- name: SGLANG_DISAGGREGATION_THREAD_POOL_SIZE
value: "32"
- name: TZ
value: Asia/Shanghai
ports:
- name: metrics
containerPort: 30001
readinessProbe:
httpGet:
path: /health
port: metrics
initialDelaySeconds: 60
periodSeconds: 10
timeoutSeconds: 5
failureThreshold: 90
resources:
requests:
cpu: "32"
memory: 100Gi
nvidia.com/gpu: "2"
limits:
cpu: "96"
memory: 600Gi
nvidia.com/gpu: "2"
securityContext:
privileged: true
volumeMounts:
- name: data
mountPath: /data
- name: sglang-cache
mountPath: /root/.cache/sglang
- name: dshm
mountPath: /dev/shm
volumes:
- name: data
hostPath:
path: /data
type: Directory
- name: sglang-cache
hostPath:
path: /data/ymk/cache/sglang
type: Directory
- name: dshm
emptyDir:
medium: Memory
sizeLimit: 32Gi
roles:
- name: mooncake-master
replicas: 1
standalonePattern:
template:
metadata:
labels:
app: ds41-3p1d-mooncake-master
spec:
hostNetwork: true
dnsPolicy: ClusterFirstWithHostNet
nodeSelector:
kubernetes.io/hostname: b300-01
containers:
- name: mooncake-master
image: ymkymx/sglang:dsv41-pd-visioncp-db7d2cb7d-fix
imagePullPolicy: IfNotPresent
command:
- /bin/bash
- -c
- |
exec /opt/sglang/bin/mooncake_master \
--port 50051 \
--enable_http_metadata_server=true \
--http_metadata_server_port=18080 \
--eviction_high_watermark_ratio=0.90
readinessProbe:
tcpSocket:
port: 50051
initialDelaySeconds: 5
periodSeconds: 5
failureThreshold: 60
resources:
requests:
cpu: "1"
memory: 8Gi
limits:
cpu: "4"
memory: 64Gi
- name: p1
replicas: 1
dependencies:
- mooncake-master
standalonePattern:
templateRef:
name: prefill-base
patch:
metadata:
labels:
app: ds41-3p1d-p1
spec:
containers:
- name: sglang
env:
- name: CUDA_VISIBLE_DEVICES
value: "0,1"
- name: SGLANG_PORT
value: "30000"
- name: SGLANG_BOOTSTRAP_PORT
value: "8998"
- name: OTLP_NAME
value: ds41-3p1d-p1
ports:
- name: metrics
containerPort: 30000
- name: p2
replicas: 1
dependencies:
- mooncake-master
standalonePattern:
templateRef:
name: prefill-base
patch:
metadata:
labels:
app: ds41-3p1d-p2
spec:
containers:
- name: sglang
env:
- name: CUDA_VISIBLE_DEVICES
value: "2,3"
- name: SGLANG_PORT
value: "30010"
- name: SGLANG_BOOTSTRAP_PORT
value: "8999"
- name: OTLP_NAME
value: ds41-3p1d-p2
ports:
- name: metrics
containerPort: 30010
- name: p3
replicas: 1
dependencies:
- mooncake-master
standalonePattern:
templateRef:
name: prefill-base
patch:
metadata:
labels:
app: ds41-3p1d-p3
spec:
containers:
- name: sglang
env:
- name: CUDA_VISIBLE_DEVICES
value: "4,5"
- name: SGLANG_PORT
value: "30020"
- name: SGLANG_BOOTSTRAP_PORT
value: "9000"
- name: OTLP_NAME
value: ds41-3p1d-p3
ports:
- name: metrics
containerPort: 30020
- name: d1
replicas: 1
dependencies:
- mooncake-master
standalonePattern:
templateRef:
name: decode-base
patch:
metadata:
labels:
app: ds41-3p1d-d1
spec:
containers:
- name: sglang
env:
- name: CUDA_VISIBLE_DEVICES
value: "6,7"
- name: router
replicas: 1
dependencies:
- p1
- p2
- p3
- d1
standalonePattern:
template:
metadata:
labels:
app: ds41-3p1d-router
spec:
# 入口不走 hostNetwork:普通 Pod 网络 + ClusterIP,天然不对宿主机/外网暴露。
# P/D 地址用节点内网 IP(hostNetwork 监听在 192.168.1.243 上,Calico 可达)。
nodeSelector:
kubernetes.io/hostname: b300-01
containers:
- name: router
image: ymkymx/sglang:dsv41-pd-visioncp-db7d2cb7d-fix
imagePullPolicy: IfNotPresent
command:
- /bin/bash
- -c
- |
exec sglang-router launch \
--mini-lb --pd-disaggregation \
--host 0.0.0.0 --port 50002 \
--prefill http://192.168.1.243:30000 8998 \
--prefill http://192.168.1.243:30010 8999 \
--prefill http://192.168.1.243:30020 9000 \
--decode http://192.168.1.243:30001
ports:
- name: http
containerPort: 50002
readinessProbe:
httpGet:
path: /health
port: http
initialDelaySeconds: 5
periodSeconds: 10
timeoutSeconds: 5
failureThreshold: 30
resources:
requests:
cpu: "1"
memory: 4Gi
limits:
cpu: "8"
memory: 16Gi
---
apiVersion: v1
kind: Service
metadata:
name: ds41-3p1d
labels:
app: ds41-3p1d
spec:
type: ClusterIP
selector:
rbg.workloads.x-k8s.io/group-name: ds41-3p1d
rbg.workloads.x-k8s.io/role-name: router
ports:
- name: http
port: 50002
targetPort: 50002
---
apiVersion: monitoring.coreos.com/v1
kind: PodMonitor
metadata:
name: ds41-3p1d
labels:
release: kube-prometheus-stack
spec:
selector:
matchLabels:
sglang-model: ds-v41-flash
podMetricsEndpoints:
- interval: 15s
path: /metrics
port: metrics