From 202e61889889a193e438e5072cc4e84bfa5bdbc8 Mon Sep 17 00:00:00 2001 From: Cheng Wan <54331508+ch-wan@users.noreply.github.com> Date: Tue, 2 Jun 2026 23:27:57 -0700 Subject: [PATCH] Revert "Fix hybrid linear attention misrouting plain-RadixAttention linear layers to the full backend (Ring-2.5-1T)" (#27116) --- .../attention/hybrid_linear_attn_backend.py | 21 ++++++++++++------- .../sglang/srt/models/bailing_moe_linear.py | 6 ++++++ 2 files changed, 19 insertions(+), 8 deletions(-) diff --git a/python/sglang/srt/layers/attention/hybrid_linear_attn_backend.py b/python/sglang/srt/layers/attention/hybrid_linear_attn_backend.py index ac5a5e57a..71248dcf2 100644 --- a/python/sglang/srt/layers/attention/hybrid_linear_attn_backend.py +++ b/python/sglang/srt/layers/attention/hybrid_linear_attn_backend.py @@ -782,17 +782,22 @@ class HybridLinearAttnBackend(AttentionBackend): self.req_to_token_pool = full_attn_backend.req_to_token_pool def _is_full_attn( - self, - layer: Optional[Union[RadixAttention, RadixLinearAttention]], - layer_id: Optional[int] = None, + self, layer: Optional[RadixAttention], layer_id: Optional[int] = None ) -> bool: - # RadixLinearAttention is unambiguously a linear-attention layer. - # Everything else (including plain RadixAttention) must be classified by - # layer id: models like Bailing/Ring use a plain RadixAttention for their - # linear layers, so an `isinstance(layer, RadixAttention) -> full` shortcut - # would misroute those linear layers to the full-attention backend. + # Explicit linear-attention subclass → strong linear signal (KDA, GDN, + # Qwen3-Next, Qwen3.5 main linear layers). if isinstance(layer, RadixLinearAttention): return False + # Some hybrid models (Ling-2.5/2.6) wrap their linear layers in plain + # `RadixAttention` rather than `RadixLinearAttention`. Those wrappers + # set `_is_linear_attention=True` on the attn module so we can + # distinguish them from full-attention RadixAttention instances — + # including MTP/NEXTN draft layers, which are full and must default to + # the full-attn path. + if layer is not None and getattr(layer, "_is_linear_attention", False): + return False + if isinstance(layer, RadixAttention): + return True if layer is not None: layer_id = layer.layer_id diff --git a/python/sglang/srt/models/bailing_moe_linear.py b/python/sglang/srt/models/bailing_moe_linear.py index c0c6be3ca..1c4a5d872 100644 --- a/python/sglang/srt/models/bailing_moe_linear.py +++ b/python/sglang/srt/models/bailing_moe_linear.py @@ -508,6 +508,12 @@ class BailingMoELinearAttention(nn.Module): quant_config=quant_config, prefix=f"{prefix}.attn", ) + # Marker for HybridLinearAttnBackend._is_full_attn: Bailing wraps + # linear-attention layers in a plain RadixAttention, so the + # dispatcher can't tell from the type alone that this is a linear + # layer (would otherwise default to the full-attn backend, e.g. the + # same way MTP/NEXTN draft layers are routed). + self.attn._is_linear_attention = True self.group_norm_size = getattr(config, "group_norm_size", 1) self.rms_norm_eps = float(getattr(config, "rms_norm_eps", 1e-5))