# PD2 二分 Round 1(2026-09-24):干净基线 + 嫌疑项①「P 侧 L3 hicache 块 + mooncake-store.json」 # 相对 dockerserve-pd2-p.yml 的差异: # env 加 SGLANG_HICACHE_MOONCAKE_CONFIG_PATH=/data/ymk/ds41/mooncake-store.json # hicache 参数块换成坏部署同款:page_first_direct + direct + write_through + mooncake + # wait_complete + size 0(替代 write_back L2) # D 侧不变(坏部署 D 本无 L3)。探针:bs16 low_entropy decode 看 accept len。 services: sglang: image: ymkymx/sglang:dsv41-pd-visioncp-db7d2cb7d-fix container_name: ds41-pd2-prefill shm_size: "32gb" ipc: host pid: host privileged: true network_mode: host environment: - CUDA_VISIBLE_DEVICES=4,5 - SGLANG_RAGGED_VERIFY_MODE=static - MC_INTRANODE_NVLINK=true - MC_INTRA_NVLINK=true - SGLANG_MOONCAKE_SEND_AUX_TCP=1 - SGLANG_ENABLE_DSV41_ENGRAM_HOST_TABLE=1 - SGLANG_HICACHE_MOONCAKE_CONFIG_PATH=/data/ymk/ds41/mooncake-store.json volumes: - /data:/data - /data/ymk/cache/sglang:/root/.cache/sglang - /data/ymk/sglang/python/sglang/srt/managers/scheduler.py:/sgl-workspace/sglang/python/sglang/srt/managers/scheduler.py deploy: resources: reservations: devices: - driver: nvidia count: all capabilities: [gpu] command: >- sglang serve --trust-remote-code --model-path /data/models/DeepSeek-V4.1-Flash --tp 2 --ep-size 2 --mem-fraction-static 0.75 --attention-backend dsv4 --moe-runner-backend flashinfer_mxfp4 --cuda-graph-max-bs-decode 32 --reasoning-parser auto --tool-call-parser auto --default-chat-template-kwargs '{"thinking": true, "reasoning_effort": "high"}' --tokenizer-worker-num 8 --host 0.0.0.0 --port 30020 --enable-cache-report --enable-metrics --speculative-algorithm DSPARK --speculative-dspark-block-size 5 --enable-hierarchical-cache --hicache-ratio 2 --hicache-mem-layout page_first_direct --hicache-io-backend direct --hicache-write-policy write_through --hicache-storage-backend mooncake --hicache-storage-prefetch-policy wait_complete --hicache-size 0 --enable-prefill-cp --cp-strategy interleave --disaggregation-mode prefill --disaggregation-transfer-backend mooncake