[Spec] Enable grammar overlap scheduling for STANDALONE speculative decoding (#32110)

This commit is contained in:
Liangsheng Yin
2026-07-23 01:47:27 -07:00
committed by GitHub
parent 5387e23ecd
commit f35411ee81
2 changed files with 9 additions and 3 deletions
+2 -1
View File
@@ -136,7 +136,8 @@ class SpeculativeAlgorithm(Enum):
def supports_grammar_overlap(self) -> bool:
# Whether the worker advances the grammar FSM inside verify() (via the
# scheduler's grammar barrier), letting spec + grammar decode overlap.
return self.is_eagle()
# STANDALONE inherits the EAGLE V2 worker's verify path, barrier included.
return self.is_eagle() or self.is_standalone()
def has_draft_kv(self) -> bool:
"""Whether the draft phase writes KV chains. NGRAM does not (its tree
+7 -2
View File
@@ -1,19 +1,24 @@
import unittest
from sglang.test.ci.ci_register import register_cuda_ci
from sglang.test.kits.json_constrained_kit import JSONConstrainedMixin
from sglang.test.kits.regex_constrained_kit import RegexConstrainedMixin
from sglang.test.server_fixtures.standalone_fixture import StandaloneServerBase
from sglang.test.test_utils import CustomTestCase
# V2 standalone speculative decoding tests (FA3, Triton, FlashInfer backends).
# Non-V2 backends moved to test_spec_standalone_extra.py.
register_cuda_ci(est_time=406, stage="base-b", runner_config="1-gpu-large")
register_cuda_ci(est_time=450, stage="base-b", runner_config="1-gpu-large")
class TestStandaloneV2SpeculativeDecodingBase(StandaloneServerBase, CustomTestCase):
attention_backend = "fa3"
class TestStandaloneV2SpeculativeDecodingTriton(StandaloneServerBase, CustomTestCase):
class TestStandaloneV2SpeculativeDecodingTriton(
StandaloneServerBase, CustomTestCase, RegexConstrainedMixin, JSONConstrainedMixin
):
# Constrained mixins reuse this server; overlap on -> grammar barrier path.
attention_backend = "triton"