Support V4.1 DP attention in DP-only DSpark PD
(cherry picked from commit 6ef0ebba730b3fffde4b59b953c5c9b58a35c4ed)
This commit is contained in:
@@ -249,15 +249,13 @@ def validate_deepseek_v41_features(server_args: ServerArgs) -> None:
|
|||||||
if (
|
if (
|
||||||
read_ragged_verify_mode() is not RaggedVerifyMode.STATIC
|
read_ragged_verify_mode() is not RaggedVerifyMode.STATIC
|
||||||
or cfg.disaggregation_transfer_backend != "mooncake"
|
or cfg.disaggregation_transfer_backend != "mooncake"
|
||||||
or cfg.dp_size != 1
|
|
||||||
or cfg.enable_dp_attention
|
|
||||||
or cfg.attn_cp_size != 1
|
or cfg.attn_cp_size != 1
|
||||||
or cfg.dcp_size != 1
|
or cfg.dcp_size != 1
|
||||||
):
|
):
|
||||||
raise ValueError(
|
raise ValueError(
|
||||||
"DeepSeek-V4.1 DSpark PD requires static verify, Mooncake, "
|
"DeepSeek-V4.1 DSpark PD requires static verify, Mooncake, "
|
||||||
"DP=1 and CP=1. Both servers must enable DSpark with the same "
|
"and CP=1 on both servers. DP attention is supported when "
|
||||||
"block size and TP size."
|
"both servers use the same block size and target/draft KV layout."
|
||||||
)
|
)
|
||||||
|
|
||||||
from sglang.srt.model_executor.cuda_graph_config import Backend, Phase, with_phase
|
from sglang.srt.model_executor.cuda_graph_config import Backend, Phase, with_phase
|
||||||
|
|||||||
@@ -940,10 +940,27 @@ class CommonKVManager(BaseKVManager):
|
|||||||
and self.attn_cp_size == 1
|
and self.attn_cp_size == 1
|
||||||
and info.attn_tp_size * info.attn_cp_size == self.attn_tp_size
|
and info.attn_tp_size * info.attn_cp_size == self.attn_tp_size
|
||||||
)
|
)
|
||||||
if info.attn_tp_size != self.attn_tp_size and not same_tp_with_prefill_cp:
|
# Combined branch (40323-series + 40177): prefill CP can also pair
|
||||||
|
# with a DP-attention decode server. MLA KV is replicated across
|
||||||
|
# prefill CP ranks, so per-rank layouts match when attn_tp matches.
|
||||||
|
dp_decode_with_prefill_cp = (
|
||||||
|
info.attn_cp_size > 1
|
||||||
|
and self.attn_cp_size == 1
|
||||||
|
and (self.is_mla_backend or self.is_hybrid_mla_backend)
|
||||||
|
and info.attn_tp_size == self.attn_tp_size
|
||||||
|
)
|
||||||
|
non_cp_mla_layout = info.attn_cp_size == self.attn_cp_size == 1 and (
|
||||||
|
self.is_mla_backend or self.is_hybrid_mla_backend
|
||||||
|
)
|
||||||
|
if info.attn_tp_size != self.attn_tp_size and not (
|
||||||
|
same_tp_with_prefill_cp
|
||||||
|
or dp_decode_with_prefill_cp
|
||||||
|
or non_cp_mla_layout
|
||||||
|
):
|
||||||
raise RuntimeError(
|
raise RuntimeError(
|
||||||
"DeepSeek-V4.1 DSpark PD requires the same TP size on both "
|
"DeepSeek-V4.1 DSpark PD requires matching attention TP "
|
||||||
"servers (including prefill CP ranks for MLA)"
|
"unless both servers use CP=1 with an MLA KV layout, "
|
||||||
|
"or prefill runs CP with an MLA KV layout"
|
||||||
)
|
)
|
||||||
|
|
||||||
if self.dcp_size > 1:
|
if self.dcp_size > 1:
|
||||||
|
|||||||
Reference in New Issue
Block a user