[Refactor] Refactor DeepEP dispatcher (#22822)
Co-authored-by: gemini-code-assist[bot] <176961590+gemini-code-assist[bot]@users.noreply.github.com> Co-authored-by: Cheng Wan <54331508+ch-wan@users.noreply.github.com>
This commit is contained in:
co-authored by
gemini-code-assist[bot]
Cheng Wan
parent
5147de26e4
commit
a080358cac
@@ -127,7 +127,6 @@ do
|
||||
echo "${P_IP[$i]}"
|
||||
export SGLANG_USE_AG_AFTER_QLORA=1
|
||||
export HCCL_BUFFSIZE=800
|
||||
export DEEP_NORMAL_MODE_USE_INT8_QUANT=1
|
||||
export TASK_QUEUE_ENABLE=2
|
||||
export SGLANG_NPU_FUSED_MOE_MODE=2
|
||||
export SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK=131072
|
||||
@@ -136,7 +135,7 @@ do
|
||||
export GLOO_SOCKET_IFNAME=lo
|
||||
python -m sglang.launch_server --model-path ${MODEL_PATH} --disaggregation-mode prefill --host ${P_IP[$i]} \
|
||||
--port 8000 --disaggregation-bootstrap-port $((8998+$i)) --trust-remote-code --nnodes 1 --node-rank 0 \
|
||||
--tp-size 16 --mem-fraction-static 0.778 --attention-backend ascend --device npu --quantization modelslim \
|
||||
--tp-size 16 --mem-fraction-static 0.778 --attention-backend ascend --device npu \
|
||||
--disaggregation-transfer-backend ascend --max-running-requests 16 --disable-radix-cache \
|
||||
--chunked-prefill-size -1 --max-prefill-tokens 60000 --moe-a2a-backend ascend_fuseep --deepep-mode normal \
|
||||
--speculative-algorithm NEXTN --speculative-num-steps 1 --speculative-eagle-topk 1 --speculative-num-draft-tokens 2 \
|
||||
@@ -248,7 +247,6 @@ do
|
||||
then
|
||||
echo "${P_IP[$i]}"
|
||||
export HCCL_BUFFSIZE=1600
|
||||
export DEEP_NORMAL_MODE_USE_INT8_QUANT=1
|
||||
export TASK_QUEUE_ENABLE=2
|
||||
export SGLANG_USE_AG_AFTER_QLORA=1
|
||||
export HCCL_SOCKET_IFNAME=lo
|
||||
@@ -375,7 +373,6 @@ do
|
||||
then
|
||||
echo "${P_IP[$i]}"
|
||||
export HCCL_BUFFSIZE=1536
|
||||
export DEEP_NORMAL_MODE_USE_INT8_QUANT=1
|
||||
export TASK_QUEUE_ENABLE=2
|
||||
|
||||
export HCCL_SOCKET_IFNAME=lo
|
||||
@@ -501,7 +498,6 @@ do
|
||||
then
|
||||
echo "${P_IP[$i]}"
|
||||
export HCCL_BUFFSIZE=1536
|
||||
export DEEP_NORMAL_MODE_USE_INT8_QUANT=1
|
||||
export TASK_QUEUE_ENABLE=2
|
||||
export HCCL_SOCKET_IFNAME=lo
|
||||
export GLOO_SOCKET_IFNAME=lo
|
||||
@@ -690,7 +686,6 @@ export DEEPEP_NORMAL_LONG_SEQ_PER_ROUND_TOKENS=512
|
||||
|
||||
MODEL_PATH=xxx
|
||||
|
||||
export DEEP_NORMAL_MODE_USE_INT8_QUANT=1
|
||||
export SGLANG_NPU_USE_MLAPO=1
|
||||
export SGLANG_ENABLE_SPEC_V2=1
|
||||
export SGLANG_ENABLE_OVERLAP_PLAN_STREAM=1
|
||||
@@ -781,7 +776,6 @@ do
|
||||
then
|
||||
echo "${P_IP[$i]}"
|
||||
export HCCL_BUFFSIZE=2600
|
||||
export DEEP_NORMAL_MODE_USE_INT8_QUANT=1
|
||||
export TASK_QUEUE_ENABLE=2
|
||||
|
||||
export HCCL_SOCKET_IFNAME=lo
|
||||
@@ -888,7 +882,6 @@ export SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK=56
|
||||
export HCCL_BUFFSIZE=1200
|
||||
export DEEPEP_NORMAL_LONG_SEQ_ROUND=10
|
||||
export DEEPEP_NORMAL_LONG_SEQ_PER_ROUND_TOKENS=512
|
||||
export DEEP_NORMAL_MODE_USE_INT8_QUANT=1
|
||||
export SGLANG_NPU_USE_MLAPO=1
|
||||
export SGLANG_ENABLE_SPEC_V2=1
|
||||
export SGLANG_ENABLE_OVERLAP_PLAN_STREAM=1
|
||||
@@ -981,7 +974,6 @@ do
|
||||
then
|
||||
echo "${P_IP[$i]}"
|
||||
export HCCL_BUFFSIZE=3500
|
||||
export DEEP_NORMAL_MODE_USE_INT8_QUANT=1
|
||||
export TASK_QUEUE_ENABLE=2
|
||||
|
||||
export HCCL_SOCKET_IFNAME=lo
|
||||
@@ -1099,7 +1091,6 @@ do
|
||||
then
|
||||
echo "${P_IP[$i]}"
|
||||
export HCCL_BUFFSIZE=1200
|
||||
export DEEP_NORMAL_MODE_USE_INT8_QUANT=1
|
||||
export TASK_QUEUE_ENABLE=2
|
||||
export HCCL_SOCKET_IFNAME=xxx
|
||||
export GLOO_SOCKET_IFNAME=xxx
|
||||
@@ -1266,7 +1257,6 @@ do
|
||||
export HCCL_SOCKET_IFNAME=lo
|
||||
export GLOO_SOCKET_IFNAME=lo
|
||||
export STREAMS_PER_DEVICE=32
|
||||
export DEEP_NORMAL_MODE_USE_INT8_QUANT=1
|
||||
|
||||
# P节点
|
||||
python -m sglang.launch_server --model-path ${MODEL_PATH} --disaggregation-mode prefill \
|
||||
@@ -2202,7 +2192,6 @@ do
|
||||
export TASK_QUEUE_ENABLE=2
|
||||
export HCCL_SOCKET_IFNAME=lo
|
||||
export GLOO_SOCKET_IFNAME=lo
|
||||
export DEEP_NORMAL_MODE_USE_INT8_QUANT=1
|
||||
|
||||
python -m sglang.launch_server --model-path ${MODEL_PATH} --disaggregation-mode prefill \
|
||||
--host ${P_IP[$i]} --port 8000 --disaggregation-bootstrap-port 8995 --trust-remote-code \
|
||||
@@ -2299,7 +2288,6 @@ source /usr/local/Ascend/ascend-toolkit/latest/opp/vendors/customize/bin/set_env
|
||||
export SGLANG_SET_CPU_AFFINITY=1
|
||||
export PYTORCH_NPU_ALLOC_CONF=expandable_segments:True
|
||||
export SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK=72
|
||||
export DEEP_NORMAL_MODE_USE_INT8_QUANT=1
|
||||
export SGLANG_DISAGGREGATION_BOOTSTRAP_TIMEOUT=600
|
||||
|
||||
MODEL_PATH=xxx
|
||||
@@ -3213,7 +3201,6 @@ export PATH=/usr/local/Ascend/8.5.0/compiler/bishengir/bin:$PATH
|
||||
|
||||
export PYTORCH_NPU_ALLOC_CONF=expandable_segments:True
|
||||
export STREAMS_PER_DEVICE=32
|
||||
export DEEP_NORMAL_MODE_USE_INT8_QUANT=1
|
||||
export SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK=330
|
||||
export DEEPEP_NORMAL_LONG_SEQ_ROUND=5
|
||||
export DEEPEP_NORMAL_LONG_SEQ_PER_ROUND_TOKENS=3000
|
||||
@@ -3260,7 +3247,6 @@ python3 -m sglang.launch_server --model-path ${MODEL_PATH} \
|
||||
--mamba-ssm-dtype bfloat16 \
|
||||
--base-gpu-id 0 \
|
||||
--speculative-draft-model-path /home/weights/Qwen3-Next-80B-A3B-Instruct \
|
||||
--quantization modelslim \
|
||||
--moe-a2a-backend deepep --deepep-mode auto \
|
||||
```
|
||||
|
||||
@@ -3308,7 +3294,6 @@ export PATH=/usr/local/Ascend/8.5.0/compiler/bishengir/bin:$PATH
|
||||
|
||||
export PYTORCH_NPU_ALLOC_CONF=expandable_segments:True
|
||||
export STREAMS_PER_DEVICE=32
|
||||
export DEEP_NORMAL_MODE_USE_INT8_QUANT=1
|
||||
export SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK=330
|
||||
export DEEPEP_NORMAL_LONG_SEQ_ROUND=5
|
||||
export DEEPEP_NORMAL_LONG_SEQ_PER_ROUND_TOKENS=3000
|
||||
@@ -3703,7 +3688,6 @@ export PATH=/usr/local/Ascend/8.5.0/compiler/bishengir/bin:$PATH
|
||||
export SGLANG_SET_CPU_AFFINITY=1
|
||||
export PYTORCH_NPU_ALLOC_CONF=expandable_segments:True
|
||||
export STREAMS_PER_DEVICE=32
|
||||
export DEEP_NORMAL_MODE_USE_INT8_QUANT=1
|
||||
export SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK=400
|
||||
export DEEPEP_NORMAL_LONG_SEQ_ROUND=10
|
||||
export DEEPEP_NORMAL_LONG_SEQ_PER_ROUND_TOKENS=2048
|
||||
|
||||
@@ -11,7 +11,6 @@ export PYTORCH_NPU_ALLOC_CONF=expandable_segments:True
|
||||
export STREAMS_PER_DEVICE=32
|
||||
|
||||
#Deepep communication settings
|
||||
export DEEP_NORMAL_MODE_USE_INT8_QUANT=1
|
||||
export SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK=32
|
||||
export HCCL_BUFFSIZE=1600
|
||||
|
||||
@@ -64,7 +63,6 @@ export STREAMS_PER_DEVICE=32
|
||||
export ASCEND_MF_STORE_URL="tcp://<PREFILL_HOST_IP>:<PORT>"
|
||||
|
||||
#Deepep communication settings
|
||||
export DEEP_NORMAL_MODE_USE_INT8_QUANT=1
|
||||
export HCCL_BUFFSIZE=1536
|
||||
|
||||
#npu acceleration operator
|
||||
@@ -214,7 +212,6 @@ do
|
||||
then
|
||||
echo "${P_IP[$i]}"
|
||||
export HCCL_BUFFSIZE=1536
|
||||
export DEEP_NORMAL_MODE_USE_INT8_QUANT=1
|
||||
export TASK_QUEUE_ENABLE=2
|
||||
|
||||
export HCCL_SOCKET_IFNAME=lo
|
||||
|
||||
@@ -21,7 +21,6 @@ This document provides a list of commonly used environment variables and aims to
|
||||
| `DEEPEP_NORMAL_LONG_SEQ_ROUND` | Enable ant-moving function in dispatch stage. Indicates <br/> the number of rounds transmitted on each rank. | `1` |
|
||||
| `DEEPEP_NORMAL_COMBINE_ENABLE_LONG_SEQ` | Enable ant-moving function in combine stage. <br/> The value `0` means disabled. | `0` |
|
||||
| `MOE_ENABLE_TOPK_NEG_ONE` | Needs to be enabled when the expert ID to be processed by <br/> DEEPEP contains -1. | `0` |
|
||||
| `DEEP_NORMAL_MODE_USE_INT8_QUANT` | Quantizes x to int8 and returns (tensor, scales) in dispatch operator. | `0` |
|
||||
|
||||
## Others
|
||||
|
||||
|
||||
@@ -61,7 +61,6 @@ export STREAMS_PER_DEVICE=32
|
||||
export HCCL_BUFFSIZE=1536
|
||||
export HCCL_OP_EXPANSION_MODE=AIV
|
||||
export SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK=32
|
||||
export SGLANG_DEEPEP_BF16_DISPATCH=1
|
||||
|
||||
python -m sglang.launch_server \
|
||||
--device npu \
|
||||
@@ -82,7 +81,6 @@ export PYTORCH_NPU_ALLOC_CONF=expandable_segments:True
|
||||
export STREAMS_PER_DEVICE=32
|
||||
export HCCL_BUFFSIZE=1536
|
||||
export SGLANG_DEEPEP_NUM_MAX_DISPATCH_TOKENS_PER_RANK=32
|
||||
export SGLANG_DEEPEP_BF16_DISPATCH=1
|
||||
|
||||
python -m sglang.launch_server \
|
||||
--model-path Qwen/Qwen3-235B-A22B-Instruct-2507 \
|
||||
@@ -114,7 +112,6 @@ MODEL_PATH=/root/.cache/modelscope/hub/models/zcgy26/Qwen3-235B-A22B-Instruct-25
|
||||
|
||||
```shell
|
||||
export ASCEND_LAUNCH_BLOCKING=1
|
||||
export DEEP_NORMAL_MODE_USE_INT8_QUANT=1
|
||||
export HCCL_BUFFSIZE=1500
|
||||
export DEEPEP_NORMAL_LONG_SEQ_PER_ROUND_TOKENS=1024
|
||||
export DEEPEP_NORMAL_LONG_SEQ_ROUND=128
|
||||
@@ -146,7 +143,6 @@ python3 -m sglang.launch_server \
|
||||
**Decode node:**
|
||||
|
||||
```shell
|
||||
export SGLANG_DEEPEP_BF16_DISPATCH=0
|
||||
export HCCL_BUFFSIZE=4000
|
||||
export DEEPEP_NORMAL_LONG_SEQ_PER_ROUND_TOKENS=4096
|
||||
export DEEPEP_NORMAL_LONG_SEQ_ROUND=16
|
||||
|
||||
Reference in New Issue
Block a user