deploy: archive b300 ds41 compose files; skip CI for deploy/ and .gitea/ changes

This commit is contained in:
2026-09-24 11:49:37 +08:00
parent db7d2cb7db
commit 67d8368a84
34 changed files with 1858 additions and 0 deletions
+59
View File
@@ -0,0 +1,59 @@
# PD 干净对照实验(2026-09-24 上午):验证「cp2-P → dp2-D」传输路径下 dspark 是否正常。
# 用户疑问:单机六形态健康不等于 PD 链路健康,可能两边传输没对齐。
# 以已验证可用的 dockerserve-pd-{p,d}-vision.yml 为底,缩到 2+2 卡;P 加 engram host table
# (2 卡权重放不下,见 G1.9)。P=GPU4-5 端口 30020,D=GPU6-7 端口 30030,mini_lb=30004。
services:
sglang:
image: ymkymx/sglang:dsv41-pd-visioncp-db7d2cb7d-fix
container_name: ds41-pd2-prefill
shm_size: "32gb"
ipc: host
pid: host
privileged: true
network_mode: host
environment:
- CUDA_VISIBLE_DEVICES=4,5
- SGLANG_RAGGED_VERIFY_MODE=static
- MC_INTRANODE_NVLINK=true
- MC_INTRA_NVLINK=true
- SGLANG_MOONCAKE_SEND_AUX_TCP=1
- SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE=1
volumes:
- /data:/data
- /data/ymk/cache/sglang:/root/.cache/sglang
- /data/ymk/sglang/python/sglang/srt/managers/scheduler.py:/sgl-workspace/sglang/python/sglang/srt/managers/scheduler.py
deploy:
resources:
reservations:
devices:
- driver: nvidia
count: all
capabilities: [gpu]
command: >-
sglang serve
--trust-remote-code
--model-path /data/models/DeepSeek-V4.1-Flash
--tp 2
--ep-size 2
--mem-fraction-static 0.75
--attention-backend dsv4
--moe-runner-backend flashinfer_mxfp4
--cuda-graph-max-bs-decode 32
--reasoning-parser auto
--tool-call-parser auto
--default-chat-template-kwargs '{"thinking": true, "reasoning_effort": "high"}'
--tokenizer-worker-num 8
--host 0.0.0.0
--port 30020
--enable-cache-report
--enable-metrics
--speculative-algorithm DSPARK
--speculative-dspark-block-size 5
--enable-hierarchical-cache
--hicache-ratio 2.5
--hicache-write-policy write_back
--enable-prefill-cp
--cp-strategy interleave
--disaggregation-mode prefill
--disaggregation-transfer-backend mooncake