[misc] depdencies & enviroment flag (#12113)

This commit is contained in:
Liangsheng Yin
2025-10-26 14:52:35 +08:00
committed by GitHub
parent bda3758fac
commit 8491c794ad
4 changed files with 15 additions and 12 deletions
-1
View File
@@ -81,7 +81,6 @@ modelopt = ["nvidia-modelopt"]
test = [ test = [
"accelerate", "accelerate",
"expecttest", "expecttest",
"gguf",
"jsonlines", "jsonlines",
"matplotlib", "matplotlib",
"pandas", "pandas",
+1
View File
@@ -231,6 +231,7 @@ class Envs:
SGLANG_TRITON_DECODE_SPLIT_TILE_SIZE = EnvInt(256) SGLANG_TRITON_DECODE_SPLIT_TILE_SIZE = EnvInt(256)
# Overlap Spec V2 # Overlap Spec V2
SGLANG_ENABLE_SPEC_V2 = EnvBool(False)
SGLANG_ENABLE_OVERLAP_PLAN_STREAM = EnvBool(False) SGLANG_ENABLE_OVERLAP_PLAN_STREAM = EnvBool(False)
# VLM # VLM
+6 -4
View File
@@ -27,6 +27,7 @@ from typing import Dict, List, Literal, Optional, Union
import orjson import orjson
from sglang.srt.connector import ConnectorType from sglang.srt.connector import ConnectorType
from sglang.srt.environ import envs
from sglang.srt.function_call.function_call_parser import FunctionCallParser from sglang.srt.function_call.function_call_parser import FunctionCallParser
from sglang.srt.lora.lora_registry import LoRARef from sglang.srt.lora.lora_registry import LoRARef
from sglang.srt.parser.reasoning_parser import ReasoningParser from sglang.srt.parser.reasoning_parser import ReasoningParser
@@ -342,7 +343,6 @@ class ServerArgs:
nsa_decode_backend: str = "fa3" nsa_decode_backend: str = "fa3"
# Speculative decoding # Speculative decoding
enable_beta_spec: bool = False
speculative_algorithm: Optional[str] = None speculative_algorithm: Optional[str] = None
speculative_draft_model_path: Optional[str] = None speculative_draft_model_path: Optional[str] = None
speculative_draft_model_revision: Optional[str] = None speculative_draft_model_revision: Optional[str] = None
@@ -1431,13 +1431,16 @@ class ServerArgs:
"Max running requests is reset to 48 for speculative decoding. You can override this by explicitly setting --max-running-requests." "Max running requests is reset to 48 for speculative decoding. You can override this by explicitly setting --max-running-requests."
) )
if self.speculative_algorithm == "EAGLE" and self.enable_beta_spec: if (
self.speculative_algorithm == "EAGLE"
and envs.SGLANG_ENABLE_SPEC_V2.get()
):
self.disable_overlap_schedule = False self.disable_overlap_schedule = False
logger.warning( logger.warning(
"Beta spec is enabled for eagle speculative decoding and overlap schedule is turned on." "Beta spec is enabled for eagle speculative decoding and overlap schedule is turned on."
) )
if not self.enable_beta_spec: if not envs.SGLANG_ENABLE_SPEC_V2.get():
self.disable_overlap_schedule = True self.disable_overlap_schedule = True
logger.warning( logger.warning(
"Overlap scheduler is disabled because of using eagle3 or standalone speculative decoding." "Overlap scheduler is disabled because of using eagle3 or standalone speculative decoding."
@@ -2573,7 +2576,6 @@ class ServerArgs:
) )
# Speculative decoding # Speculative decoding
parser.add_argument("--enable-beta-spec", action="store_true")
parser.add_argument( parser.add_argument(
"--speculative-algorithm", "--speculative-algorithm",
type=str, type=str,
+2 -1
View File
@@ -1,6 +1,7 @@
import unittest import unittest
from types import SimpleNamespace from types import SimpleNamespace
from sglang.srt.environ import envs
from sglang.srt.utils import kill_process_tree from sglang.srt.utils import kill_process_tree
from sglang.test.few_shot_gsm8k import run_eval from sglang.test.few_shot_gsm8k import run_eval
from sglang.test.kit_matched_stop import MatchedStopMixin from sglang.test.kit_matched_stop import MatchedStopMixin
@@ -29,7 +30,6 @@ class TestEagleServerBase(CustomTestCase, MatchedStopMixin):
def setUpClass(cls): def setUpClass(cls):
cls.base_url = DEFAULT_URL_FOR_TEST cls.base_url = DEFAULT_URL_FOR_TEST
launch_args = [ launch_args = [
"--enable-beta-spec",
"--trust-remote-code", "--trust-remote-code",
"--attention-backend", "--attention-backend",
cls.attention_backend, cls.attention_backend,
@@ -53,6 +53,7 @@ class TestEagleServerBase(CustomTestCase, MatchedStopMixin):
*[str(i) for i in range(1, cls.max_running_requests + 1)], *[str(i) for i in range(1, cls.max_running_requests + 1)],
] ]
launch_args.extend(cls.other_launch_args) launch_args.extend(cls.other_launch_args)
with envs.SGLANG_ENABLE_SPEC_V2.override(True):
cls.process = popen_launch_server( cls.process = popen_launch_server(
cls.model, cls.model,
cls.base_url, cls.base_url,