feat(reasoning): auto-detect reasoning/tool-call parser from chat template (#23952)
This commit is contained in:
+1
-1
@@ -1,3 +1,3 @@
|
|||||||
[codespell]
|
[codespell]
|
||||||
ignore-words-list = ans, als, hel, boostrap, childs, te, vas, hsa, ment, cann, thi, makro, wil, rouge, PRIS, ather, MIS, medias, allready, inout, nd, fo, visibles
|
ignore-words-list = ans, als, hel, boostrap, childs, te, vas, hsa, ment, cann, thi, makro, wil, rouge, PRIS, ather, MIS, medias, allready, inout, nd, fo, visibles, nothink
|
||||||
skip = *.json, *.jsonl, *.patch, *.txt, *.lock
|
skip = *.json, *.jsonl, *.patch, *.txt, *.lock
|
||||||
|
|||||||
@@ -82,6 +82,7 @@ from sglang.srt.managers.io_struct import (
|
|||||||
)
|
)
|
||||||
from sglang.srt.managers.multi_tokenizer_mixin import MultiTokenizerRouter
|
from sglang.srt.managers.multi_tokenizer_mixin import MultiTokenizerRouter
|
||||||
from sglang.srt.managers.scheduler import run_scheduler_process
|
from sglang.srt.managers.scheduler import run_scheduler_process
|
||||||
|
from sglang.srt.managers.template_detection import resolve_auto_parsers
|
||||||
from sglang.srt.managers.template_manager import TemplateManager
|
from sglang.srt.managers.template_manager import TemplateManager
|
||||||
from sglang.srt.managers.tokenizer_manager import TokenizerManager
|
from sglang.srt.managers.tokenizer_manager import TokenizerManager
|
||||||
from sglang.srt.observability.trace import process_tracing_init, trace_set_thread_info
|
from sglang.srt.observability.trace import process_tracing_init, trace_set_thread_info
|
||||||
@@ -140,6 +141,33 @@ def init_tokenizer_manager(
|
|||||||
completion_template=server_args.completion_template,
|
completion_template=server_args.completion_template,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
# Resolve any remaining auto parsers using template manager's detection results
|
||||||
|
for attr, suggested, label in (
|
||||||
|
(
|
||||||
|
"reasoning_parser",
|
||||||
|
template_manager.suggested_reasoning_parser,
|
||||||
|
"reasoning parser",
|
||||||
|
),
|
||||||
|
(
|
||||||
|
"tool_call_parser",
|
||||||
|
template_manager.suggested_tool_call_parser,
|
||||||
|
"tool-call parser",
|
||||||
|
),
|
||||||
|
):
|
||||||
|
if getattr(server_args, attr) != "auto":
|
||||||
|
continue
|
||||||
|
if suggested is not None:
|
||||||
|
setattr(server_args, attr, suggested)
|
||||||
|
logger.info(
|
||||||
|
f"Auto-detected --{attr.replace('_', '-')} as '{suggested}' from chat template"
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
logger.warning(
|
||||||
|
f"--{attr.replace('_', '-')}=auto specified but could not detect "
|
||||||
|
f"{label} from chat template. Disabling {label}."
|
||||||
|
)
|
||||||
|
setattr(server_args, attr, None)
|
||||||
|
|
||||||
return tokenizer_manager, template_manager
|
return tokenizer_manager, template_manager
|
||||||
|
|
||||||
|
|
||||||
@@ -695,6 +723,12 @@ class Engine(EngineScoreMixin, EngineBase):
|
|||||||
host=server_args.host, port=bootstrap_port
|
host=server_args.host, port=bootstrap_port
|
||||||
)
|
)
|
||||||
|
|
||||||
|
if (
|
||||||
|
server_args.reasoning_parser == "auto"
|
||||||
|
or server_args.tool_call_parser == "auto"
|
||||||
|
):
|
||||||
|
resolve_auto_parsers(server_args)
|
||||||
|
|
||||||
# Launch scheduler processes
|
# Launch scheduler processes
|
||||||
scheduler_init_result, scheduler_procs = cls._launch_scheduler_processes(
|
scheduler_init_result, scheduler_procs = cls._launch_scheduler_processes(
|
||||||
server_args, port_args, run_scheduler_process_func
|
server_args, port_args, run_scheduler_process_func
|
||||||
|
|||||||
@@ -193,6 +193,19 @@ class OpenAIServingChat(OpenAIServingBase):
|
|||||||
self.template_manager = template_manager
|
self.template_manager = template_manager
|
||||||
self.tool_call_parser = self.tokenizer_manager.server_args.tool_call_parser
|
self.tool_call_parser = self.tokenizer_manager.server_args.tool_call_parser
|
||||||
self.reasoning_parser = self.tokenizer_manager.server_args.reasoning_parser
|
self.reasoning_parser = self.tokenizer_manager.server_args.reasoning_parser
|
||||||
|
self._reasoning_detector = None
|
||||||
|
if self.reasoning_parser:
|
||||||
|
try:
|
||||||
|
rp = ReasoningParser(
|
||||||
|
model_type=self.reasoning_parser, stream_reasoning=True
|
||||||
|
)
|
||||||
|
self._reasoning_detector = rp.detector
|
||||||
|
except ValueError as e:
|
||||||
|
logger.warning(
|
||||||
|
"Failed to initialize reasoning detector for parser '%s': %s",
|
||||||
|
self.reasoning_parser,
|
||||||
|
e,
|
||||||
|
)
|
||||||
|
|
||||||
# Get default sampling parameters from model's generation config
|
# Get default sampling parameters from model's generation config
|
||||||
self.default_sampling_params = (
|
self.default_sampling_params = (
|
||||||
@@ -683,10 +696,11 @@ class OpenAIServingChat(OpenAIServingBase):
|
|||||||
prompt = prompt[: -len(conv.sep2)]
|
prompt = prompt[: -len(conv.sep2)]
|
||||||
else:
|
else:
|
||||||
prompt = conv.get_prompt()
|
prompt = conv.get_prompt()
|
||||||
if self._get_reasoning_from_request(
|
if self._get_reasoning_from_request(request) and (
|
||||||
request
|
self._reasoning_detector is None
|
||||||
) and self.reasoning_parser not in ["qwen3", "qwen3-thinking", "glm4"]:
|
or not self._reasoning_detector.thinks_internally
|
||||||
# qwen3 and glm4 think internally without a leading <think> token
|
):
|
||||||
|
# Models with thinks_internally=True think without a leading <think> token
|
||||||
prompt += "<think>" # Note(Xinyuan): hard code thinking token
|
prompt += "<think>" # Note(Xinyuan): hard code thinking token
|
||||||
|
|
||||||
image_data = conv.image_data if conv.image_data else None
|
image_data = conv.image_data if conv.image_data else None
|
||||||
@@ -1395,54 +1409,75 @@ class OpenAIServingChat(OpenAIServingBase):
|
|||||||
request.skip_special_tokens = False
|
request.skip_special_tokens = False
|
||||||
|
|
||||||
def _get_reasoning_from_request(self, request: ChatCompletionRequest) -> bool:
|
def _get_reasoning_from_request(self, request: ChatCompletionRequest) -> bool:
|
||||||
"""Judge whether the request needs reasoning for hybrid reasoning models
|
"""Determine whether reasoning mode should be enabled for this request.
|
||||||
|
|
||||||
NOTE: This is predefined based on model's chat template
|
NOTE: This is predefined based on model's chat template
|
||||||
"""
|
"""
|
||||||
if not self.reasoning_parser:
|
if not self.reasoning_parser:
|
||||||
return False
|
return False
|
||||||
|
|
||||||
if self.reasoning_parser == "deepseek-v3":
|
|
||||||
# Models that require explicit enable thinking (thinking=True)
|
|
||||||
return (
|
|
||||||
request.chat_template_kwargs is not None
|
|
||||||
and request.chat_template_kwargs.get("thinking") is True
|
|
||||||
)
|
|
||||||
if self.reasoning_parser == "gemma4":
|
|
||||||
return (
|
|
||||||
request.chat_template_kwargs is not None
|
|
||||||
and request.chat_template_kwargs.get("enable_thinking") is True
|
|
||||||
)
|
|
||||||
if self.reasoning_parser in ["kimi_k2"]:
|
|
||||||
# Models that thinking by default, and can be disabled by setting thinking=False
|
|
||||||
return (
|
|
||||||
not request.chat_template_kwargs
|
|
||||||
or request.chat_template_kwargs.get("thinking") is not False
|
|
||||||
)
|
|
||||||
if self.reasoning_parser in ["qwen3", "glm45", "nemotron_3", "interns1"]:
|
|
||||||
# Models that thinking by default, and can be disabled by setting enable_thinking=False
|
|
||||||
return (
|
|
||||||
not request.chat_template_kwargs
|
|
||||||
or request.chat_template_kwargs.get("enable_thinking") is not False
|
|
||||||
)
|
|
||||||
if self.reasoning_parser in ["mimo"]:
|
|
||||||
# Models that require explicit enable thinking (enable_thinking=True)
|
|
||||||
return (
|
|
||||||
request.chat_template_kwargs is not None
|
|
||||||
and request.chat_template_kwargs.get("enable_thinking") is True
|
|
||||||
)
|
|
||||||
if self.reasoning_parser == "hunyuan":
|
if self.reasoning_parser == "hunyuan":
|
||||||
# Hy3-preview template emits no <think> when reasoning_effort is
|
# Hy3-preview template emits no <think> when reasoning_effort is
|
||||||
# "no_think" / "none" / unset; forcing reasoning would route all
|
# "no_think" / "none" / unset; forcing reasoning would route all
|
||||||
# output into reasoning_content.
|
# output into reasoning_content.
|
||||||
return request.reasoning_effort not in (None, "none", "no_think")
|
return request.reasoning_effort not in (None, "none", "no_think")
|
||||||
if self.reasoning_parser == "mistral":
|
|
||||||
# Mistral only reasons when reasoning_effort is explicitly set
|
config = self.template_manager.reasoning_config
|
||||||
# to a non-"none" value (typically "high").
|
if config is None:
|
||||||
|
# Fallback to parser-level defaults when template toggle config
|
||||||
|
# cannot be inferred (e.g., parser-only <think> templates).
|
||||||
|
mode = (
|
||||||
|
self._reasoning_detector.reasoning_default
|
||||||
|
if self._reasoning_detector is not None
|
||||||
|
else None
|
||||||
|
)
|
||||||
|
if mode is None:
|
||||||
|
return False
|
||||||
|
if mode == "always":
|
||||||
|
return True
|
||||||
|
if mode == "mistral":
|
||||||
|
return (
|
||||||
|
request.reasoning_effort is not None
|
||||||
|
and request.reasoning_effort != "none"
|
||||||
|
)
|
||||||
|
if mode in ("thinking", "enable_thinking"):
|
||||||
|
return (
|
||||||
|
not request.chat_template_kwargs
|
||||||
|
or request.chat_template_kwargs.get(mode) is not False
|
||||||
|
)
|
||||||
|
if mode in ("explicit_thinking", "explicit_enable_thinking"):
|
||||||
|
toggle = mode.replace("explicit_", "")
|
||||||
|
return (
|
||||||
|
request.chat_template_kwargs is not None
|
||||||
|
and request.chat_template_kwargs.get(toggle) is True
|
||||||
|
)
|
||||||
|
logger.warning(
|
||||||
|
"Unknown reasoning_default mode '%s', defaulting to reasoning disabled",
|
||||||
|
mode,
|
||||||
|
)
|
||||||
|
return False
|
||||||
|
|
||||||
|
if config.special_case == "always":
|
||||||
|
return True
|
||||||
|
|
||||||
|
if config.special_case == "mistral":
|
||||||
return (
|
return (
|
||||||
request.reasoning_effort is not None
|
request.reasoning_effort is not None
|
||||||
and request.reasoning_effort != "none"
|
and request.reasoning_effort != "none"
|
||||||
)
|
)
|
||||||
return True # default
|
|
||||||
|
if config.toggle_param is None or config.default_enabled is None:
|
||||||
|
return False
|
||||||
|
|
||||||
|
if config.default_enabled:
|
||||||
|
return (
|
||||||
|
not request.chat_template_kwargs
|
||||||
|
or request.chat_template_kwargs.get(config.toggle_param) is not False
|
||||||
|
)
|
||||||
|
return (
|
||||||
|
request.chat_template_kwargs is not None
|
||||||
|
and request.chat_template_kwargs.get(config.toggle_param) is True
|
||||||
|
)
|
||||||
|
|
||||||
async def _process_tool_call_stream(
|
async def _process_tool_call_stream(
|
||||||
self,
|
self,
|
||||||
|
|||||||
@@ -0,0 +1,471 @@
|
|||||||
|
# Copyright 2026 SGLang Team
|
||||||
|
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||||
|
# you may not use this file except in compliance with the License.
|
||||||
|
# You may obtain a copy of the License at
|
||||||
|
#
|
||||||
|
# http://www.apache.org/licenses/LICENSE-2.0
|
||||||
|
#
|
||||||
|
# Unless required by applicable law or agreed to in writing, software
|
||||||
|
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||||
|
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||||
|
# See the License for the specific language governing permissions and
|
||||||
|
# limitations under the License.
|
||||||
|
# ==============================================================================
|
||||||
|
"""
|
||||||
|
Template detection utilities for auto-detecting reasoning and tool-call parsers.
|
||||||
|
|
||||||
|
Provides rule-based detection of reasoning mode, reasoning parser, and tool-call
|
||||||
|
parser from chat templates and tokenizer vocabularies.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import logging
|
||||||
|
import re
|
||||||
|
from dataclasses import dataclass
|
||||||
|
from typing import Callable, Optional, Tuple
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class TemplateDetectionContext:
|
||||||
|
template: str
|
||||||
|
reasoning_config: Optional["ReasoningToggleConfig"]
|
||||||
|
force_reasoning: bool
|
||||||
|
vocab: set[str]
|
||||||
|
|
||||||
|
def has_text(self, needle: str) -> bool:
|
||||||
|
return needle in self.template
|
||||||
|
|
||||||
|
def has_vocab(self, token: str) -> bool:
|
||||||
|
return token in self.vocab
|
||||||
|
|
||||||
|
def has_pattern(self, pattern: str, flags: int = 0) -> bool:
|
||||||
|
return re.search(pattern, self.template, flags) is not None
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class DetectionRule:
|
||||||
|
name: str
|
||||||
|
value: object
|
||||||
|
predicate: Callable[[TemplateDetectionContext], bool]
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class ReasoningToggleConfig:
|
||||||
|
toggle_param: Optional[str] = None
|
||||||
|
default_enabled: Optional[bool] = None
|
||||||
|
special_case: Optional[str] = None
|
||||||
|
|
||||||
|
@property
|
||||||
|
def always_on(self) -> bool:
|
||||||
|
return self.special_case == "always"
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Reasoning mode rules (detect toggle config from template)
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
REASONING_MODE_RULES = (
|
||||||
|
DetectionRule(
|
||||||
|
name="gpt_oss_channel_markers",
|
||||||
|
value=ReasoningToggleConfig(special_case="always"),
|
||||||
|
predicate=lambda ctx: ctx.has_text("<|channel|>"),
|
||||||
|
),
|
||||||
|
DetectionRule(
|
||||||
|
name="force_reasoning_pattern",
|
||||||
|
value=ReasoningToggleConfig(special_case="always"),
|
||||||
|
predicate=lambda ctx: ctx.has_pattern(r"<\|im_start\|>assistant\\n<think>\\n")
|
||||||
|
and not ctx.has_text("enable_thinking")
|
||||||
|
and not ctx.has_text("thinking"),
|
||||||
|
),
|
||||||
|
DetectionRule(
|
||||||
|
name="mistral_reasoning_effort",
|
||||||
|
value=ReasoningToggleConfig(special_case="mistral"),
|
||||||
|
predicate=lambda ctx: ctx.has_text("reasoning_effort")
|
||||||
|
and ctx.has_text("[THINK]"),
|
||||||
|
),
|
||||||
|
DetectionRule(
|
||||||
|
name="explicit_enable_thinking_default_false",
|
||||||
|
value=ReasoningToggleConfig(
|
||||||
|
toggle_param="enable_thinking", default_enabled=False
|
||||||
|
),
|
||||||
|
predicate=lambda ctx: ctx.has_pattern(
|
||||||
|
r"{%\s*if\s+not\s+enable_thinking\s+is\s+defined\s*%}.*?"
|
||||||
|
r"{%\s*set\s+enable_thinking\s*=\s*(?:false|False)\s*%}",
|
||||||
|
re.DOTALL,
|
||||||
|
),
|
||||||
|
),
|
||||||
|
DetectionRule(
|
||||||
|
name="enable_thinking_default_true",
|
||||||
|
value=ReasoningToggleConfig(
|
||||||
|
toggle_param="enable_thinking", default_enabled=True
|
||||||
|
),
|
||||||
|
predicate=lambda ctx: ctx.has_pattern(
|
||||||
|
r"{%\s*if\s+not\s+enable_thinking\s+is\s+defined\s*%}.*?"
|
||||||
|
r"{%\s*set\s+enable_thinking\s*=\s*(?:true|True)\s*%}",
|
||||||
|
re.DOTALL,
|
||||||
|
)
|
||||||
|
or ctx.has_pattern(
|
||||||
|
r"set\s+enable_thinking\s*=\s*enable_thinking\s+if\s+enable_thinking\s+is\s+defined\s+else\s+(?:true|True)"
|
||||||
|
)
|
||||||
|
or ctx.has_pattern(
|
||||||
|
r"enable_thinking\s+is\s+defined\s+and\s+(?:enable_thinking\s+is\s+false|not\s+enable_thinking)"
|
||||||
|
)
|
||||||
|
or ctx.has_pattern(
|
||||||
|
r"enable_thinking\s+is\s+not\s+defined\s+or\s+enable_thinking"
|
||||||
|
)
|
||||||
|
or ctx.has_pattern(r"namespace\([^)]*enable_thinking\s*=\s*true"),
|
||||||
|
),
|
||||||
|
DetectionRule(
|
||||||
|
name="explicit_thinking_default_false",
|
||||||
|
value=ReasoningToggleConfig(toggle_param="thinking", default_enabled=False),
|
||||||
|
predicate=lambda ctx: ctx.has_pattern(
|
||||||
|
r"{%\s*if\s+not\s+thinking\s+is\s+defined\s*%}.*?"
|
||||||
|
r"{%\s*set\s+thinking\s*=\s*(?:false|False)\s*%}",
|
||||||
|
re.DOTALL,
|
||||||
|
),
|
||||||
|
),
|
||||||
|
DetectionRule(
|
||||||
|
name="thinking_default_true",
|
||||||
|
value=ReasoningToggleConfig(toggle_param="thinking", default_enabled=True),
|
||||||
|
predicate=lambda ctx: ctx.has_pattern(
|
||||||
|
r"{%\s*if\s+not\s+thinking\s+is\s+defined\s*%}.*?"
|
||||||
|
r"{%\s*set\s+thinking\s*=\s*(?:true|True)\s*%}",
|
||||||
|
re.DOTALL,
|
||||||
|
)
|
||||||
|
or ctx.has_pattern(
|
||||||
|
r"set\s+thinking\s*=\s*thinking\s+if\s+thinking\s+is\s+defined\s+else\s+(?:true|True)"
|
||||||
|
)
|
||||||
|
or ctx.has_pattern(
|
||||||
|
r"thinking\s+is\s+defined\s+and\s+(?:thinking\s+is\s+false|not\s+thinking)"
|
||||||
|
)
|
||||||
|
or ctx.has_pattern(r"thinking\s+is\s+not\s+defined\s+or\s+thinking")
|
||||||
|
or ctx.has_pattern(r"namespace\([^)]*thinking\s*=\s*true"),
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Shared predicates for model-family detection
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
|
||||||
|
def _is_gemma4(ctx):
|
||||||
|
return ctx.has_text("<|channel>")
|
||||||
|
|
||||||
|
|
||||||
|
def _is_kimi(ctx):
|
||||||
|
return ctx.has_text("◁think▷")
|
||||||
|
|
||||||
|
|
||||||
|
def _is_interns1(ctx):
|
||||||
|
return ctx.has_text("default_thinking_sys") and ctx.reasoning_config == (
|
||||||
|
ReasoningToggleConfig(toggle_param="enable_thinking", default_enabled=True)
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _is_mistral(ctx):
|
||||||
|
return (
|
||||||
|
ctx.reasoning_config is not None
|
||||||
|
and ctx.reasoning_config.special_case == "mistral"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _is_gpt_oss(ctx):
|
||||||
|
return ctx.has_text("<|channel|>")
|
||||||
|
|
||||||
|
|
||||||
|
def _is_kimi_k2(ctx):
|
||||||
|
return ctx.has_vocab("<|tool_calls_section_begin|>")
|
||||||
|
|
||||||
|
|
||||||
|
def _is_nemotron_3(ctx):
|
||||||
|
return ctx.has_text("truncate_history_thinking") and ctx.reasoning_config == (
|
||||||
|
ReasoningToggleConfig(toggle_param="enable_thinking", default_enabled=True)
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _is_glm45(ctx):
|
||||||
|
return (
|
||||||
|
(
|
||||||
|
ctx.has_text("[gMASK]<sop>")
|
||||||
|
or ctx.has_pattern(r"(?<!<)/nothink")
|
||||||
|
or ctx.has_pattern(r"(?<!<)/think")
|
||||||
|
)
|
||||||
|
and ctx.has_vocab("<tool_call>")
|
||||||
|
and ctx.reasoning_config
|
||||||
|
== ReasoningToggleConfig(toggle_param="enable_thinking", default_enabled=True)
|
||||||
|
and (ctx.has_vocab("<|user|>") or ctx.has_vocab("<|endoftext|>"))
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _is_mimo(ctx):
|
||||||
|
return ctx.reasoning_config == ReasoningToggleConfig(
|
||||||
|
toggle_param="enable_thinking", default_enabled=False
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _is_minimax(ctx):
|
||||||
|
return ctx.has_text("<minimax:tool_call>")
|
||||||
|
|
||||||
|
|
||||||
|
def _is_qwen3(ctx):
|
||||||
|
return ctx.reasoning_config == ReasoningToggleConfig(
|
||||||
|
toggle_param="enable_thinking", default_enabled=True
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _is_deepseek_v3(ctx):
|
||||||
|
return ctx.reasoning_config == ReasoningToggleConfig(
|
||||||
|
toggle_param="thinking", default_enabled=False
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _is_deepseek_r1(ctx):
|
||||||
|
return ctx.force_reasoning
|
||||||
|
|
||||||
|
|
||||||
|
def _is_deepseek_r1_think_tags(ctx):
|
||||||
|
return ctx.has_text("<think>") or ctx.has_text("</think>")
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Reasoning parser rules
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
REASONING_PARSER_RULES = (
|
||||||
|
DetectionRule(name="gemma4", value="gemma4", predicate=_is_gemma4),
|
||||||
|
DetectionRule(name="kimi", value="kimi", predicate=_is_kimi),
|
||||||
|
DetectionRule(name="interns1", value="interns1", predicate=_is_interns1),
|
||||||
|
DetectionRule(name="mistral", value="mistral", predicate=_is_mistral),
|
||||||
|
DetectionRule(name="gpt_oss", value="gpt-oss", predicate=_is_gpt_oss),
|
||||||
|
DetectionRule(name="kimi_k2", value="kimi_k2", predicate=_is_kimi_k2),
|
||||||
|
DetectionRule(name="nemotron_3", value="nemotron_3", predicate=_is_nemotron_3),
|
||||||
|
DetectionRule(name="glm45", value="glm45", predicate=_is_glm45),
|
||||||
|
DetectionRule(name="mimo", value="mimo", predicate=_is_mimo),
|
||||||
|
DetectionRule(name="minimax", value="minimax", predicate=_is_minimax),
|
||||||
|
DetectionRule(name="qwen3", value="qwen3", predicate=_is_qwen3),
|
||||||
|
DetectionRule(name="deepseek_v3", value="deepseek-v3", predicate=_is_deepseek_v3),
|
||||||
|
DetectionRule(
|
||||||
|
name="deepseek_r1_force", value="deepseek-r1", predicate=_is_deepseek_r1
|
||||||
|
),
|
||||||
|
DetectionRule(
|
||||||
|
name="deepseek_r1_think_tags",
|
||||||
|
value="deepseek-r1",
|
||||||
|
predicate=_is_deepseek_r1_think_tags,
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Tool-call parser rules (reuse shared predicates, different values)
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
TOOL_CALL_PARSER_RULES = (
|
||||||
|
DetectionRule(name="gemma4", value="gemma4", predicate=_is_gemma4),
|
||||||
|
DetectionRule(name="gpt_oss", value="gpt-oss", predicate=_is_gpt_oss),
|
||||||
|
DetectionRule(name="kimi_k2", value="kimi_k2", predicate=_is_kimi_k2),
|
||||||
|
DetectionRule(name="minimax", value="minimax-m2", predicate=_is_minimax),
|
||||||
|
DetectionRule(name="interns1", value="interns1", predicate=_is_interns1),
|
||||||
|
DetectionRule(name="mistral", value="mistral", predicate=_is_mistral),
|
||||||
|
DetectionRule(name="glm45", value="glm45", predicate=_is_glm45),
|
||||||
|
DetectionRule(name="mimo", value="mimo", predicate=_is_mimo),
|
||||||
|
DetectionRule(name="qwen", value="qwen", predicate=_is_qwen3),
|
||||||
|
DetectionRule(name="deepseek_v3", value="deepseekv3", predicate=_is_deepseek_v3),
|
||||||
|
DetectionRule(name="deepseek_r1", value="deepseekv3", predicate=_is_deepseek_r1),
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Detection functions
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
|
||||||
|
def build_detection_context(
|
||||||
|
template: Optional[str],
|
||||||
|
tokenizer,
|
||||||
|
reasoning_config: Optional[ReasoningToggleConfig] = None,
|
||||||
|
force_reasoning: bool = False,
|
||||||
|
) -> Optional[TemplateDetectionContext]:
|
||||||
|
if template is None:
|
||||||
|
return None
|
||||||
|
vocab = set()
|
||||||
|
if tokenizer is not None:
|
||||||
|
try:
|
||||||
|
vocab = set(tokenizer.get_vocab().keys())
|
||||||
|
except Exception as e:
|
||||||
|
logger.warning(
|
||||||
|
"Failed to load tokenizer vocab for template detection: %s. "
|
||||||
|
"Vocab-dependent detection rules will be skipped.",
|
||||||
|
e,
|
||||||
|
)
|
||||||
|
return TemplateDetectionContext(
|
||||||
|
template=template,
|
||||||
|
reasoning_config=reasoning_config,
|
||||||
|
force_reasoning=force_reasoning,
|
||||||
|
vocab=vocab,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def match_rules(
|
||||||
|
ctx: TemplateDetectionContext,
|
||||||
|
rules: Tuple[DetectionRule, ...],
|
||||||
|
label: str,
|
||||||
|
) -> Optional[str]:
|
||||||
|
for rule in rules:
|
||||||
|
try:
|
||||||
|
if rule.predicate(ctx):
|
||||||
|
logger.info(
|
||||||
|
"Detected %s '%s' from template rule '%s'.",
|
||||||
|
label,
|
||||||
|
rule.value,
|
||||||
|
rule.name,
|
||||||
|
)
|
||||||
|
return rule.value
|
||||||
|
except Exception as e:
|
||||||
|
logger.warning(
|
||||||
|
"Detection rule '%s' for %s raised an exception: %s. Skipping.",
|
||||||
|
rule.name,
|
||||||
|
label,
|
||||||
|
e,
|
||||||
|
exc_info=True,
|
||||||
|
)
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def detect_reasoning_pattern(
|
||||||
|
template: Optional[str],
|
||||||
|
) -> Tuple[bool, Optional[ReasoningToggleConfig]]:
|
||||||
|
"""Detect if the chat template contains reasoning/thinking patterns."""
|
||||||
|
if template is None:
|
||||||
|
return False, None
|
||||||
|
|
||||||
|
ctx = TemplateDetectionContext(
|
||||||
|
template=template,
|
||||||
|
reasoning_config=None,
|
||||||
|
force_reasoning=False,
|
||||||
|
vocab=set(),
|
||||||
|
)
|
||||||
|
for rule in REASONING_MODE_RULES:
|
||||||
|
if rule.predicate(ctx):
|
||||||
|
logger.info(
|
||||||
|
"Detected reasoning config '%s' from template rule '%s'.",
|
||||||
|
rule.value,
|
||||||
|
rule.name,
|
||||||
|
)
|
||||||
|
return rule.value.always_on, rule.value
|
||||||
|
|
||||||
|
return False, None
|
||||||
|
|
||||||
|
|
||||||
|
def detect_reasoning_parser(
|
||||||
|
template: Optional[str],
|
||||||
|
tokenizer,
|
||||||
|
reasoning_config: Optional[ReasoningToggleConfig] = None,
|
||||||
|
force_reasoning: bool = False,
|
||||||
|
) -> Optional[str]:
|
||||||
|
"""Auto-detect which reasoning parser to use from the chat template."""
|
||||||
|
ctx = build_detection_context(
|
||||||
|
template, tokenizer, reasoning_config, force_reasoning
|
||||||
|
)
|
||||||
|
if ctx is None:
|
||||||
|
return None
|
||||||
|
return match_rules(ctx, REASONING_PARSER_RULES, "reasoning parser")
|
||||||
|
|
||||||
|
|
||||||
|
def detect_tool_call_parser(
|
||||||
|
template: Optional[str],
|
||||||
|
tokenizer,
|
||||||
|
reasoning_config: Optional[ReasoningToggleConfig] = None,
|
||||||
|
force_reasoning: bool = False,
|
||||||
|
) -> Optional[str]:
|
||||||
|
"""Auto-detect which tool-call parser to use from the chat template."""
|
||||||
|
ctx = build_detection_context(
|
||||||
|
template, tokenizer, reasoning_config, force_reasoning
|
||||||
|
)
|
||||||
|
if ctx is None:
|
||||||
|
return None
|
||||||
|
return match_rules(ctx, TOOL_CALL_PARSER_RULES, "tool-call parser")
|
||||||
|
|
||||||
|
|
||||||
|
def _resolve_auto_parser(
|
||||||
|
server_args,
|
||||||
|
attr: str,
|
||||||
|
ctx: TemplateDetectionContext,
|
||||||
|
rules: Tuple[DetectionRule, ...],
|
||||||
|
label: str,
|
||||||
|
) -> None:
|
||||||
|
"""Resolve a single auto parser, updating server_args in place."""
|
||||||
|
detected = match_rules(ctx, rules, label)
|
||||||
|
if detected:
|
||||||
|
setattr(server_args, attr, detected)
|
||||||
|
logger.info(
|
||||||
|
f"Auto-detected --{attr.replace('_', '-')} as '{detected}' from chat template"
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
logger.warning(
|
||||||
|
f"--{attr.replace('_', '-')}=auto specified but could not detect "
|
||||||
|
f"{label} from chat template. Disabling {label}."
|
||||||
|
)
|
||||||
|
setattr(server_args, attr, None)
|
||||||
|
|
||||||
|
|
||||||
|
def resolve_auto_parsers(server_args) -> None:
|
||||||
|
"""Resolve --reasoning-parser=auto and --tool-call-parser=auto before scheduler.
|
||||||
|
|
||||||
|
This performs a lightweight tokenizer load to detect parsers from the chat
|
||||||
|
template. Called early in engine init before scheduler subprocesses are spawned.
|
||||||
|
"""
|
||||||
|
needs_reasoning = server_args.reasoning_parser == "auto"
|
||||||
|
needs_tool_call = server_args.tool_call_parser == "auto"
|
||||||
|
|
||||||
|
if not needs_reasoning and not needs_tool_call:
|
||||||
|
return
|
||||||
|
|
||||||
|
from sglang.srt.utils.hf_transformers_utils import get_tokenizer
|
||||||
|
|
||||||
|
try:
|
||||||
|
tokenizer = get_tokenizer(
|
||||||
|
server_args.model_path,
|
||||||
|
trust_remote_code=server_args.trust_remote_code,
|
||||||
|
)
|
||||||
|
template = getattr(tokenizer, "chat_template", None)
|
||||||
|
except Exception as e:
|
||||||
|
logger.warning(f"Failed to load tokenizer for auto-detection: {e}")
|
||||||
|
if needs_reasoning:
|
||||||
|
logger.warning(
|
||||||
|
"--reasoning-parser=auto specified but could not detect "
|
||||||
|
"reasoning parser from chat template. Disabling reasoning parser."
|
||||||
|
)
|
||||||
|
server_args.reasoning_parser = None
|
||||||
|
if needs_tool_call:
|
||||||
|
logger.warning(
|
||||||
|
"--tool-call-parser=auto specified but could not detect "
|
||||||
|
"tool-call parser from chat template. Disabling tool-call parser."
|
||||||
|
)
|
||||||
|
server_args.tool_call_parser = None
|
||||||
|
return
|
||||||
|
|
||||||
|
force_reasoning, reasoning_config = detect_reasoning_pattern(template)
|
||||||
|
ctx = build_detection_context(
|
||||||
|
template, tokenizer, reasoning_config, force_reasoning
|
||||||
|
)
|
||||||
|
if ctx is None:
|
||||||
|
return
|
||||||
|
|
||||||
|
if needs_reasoning:
|
||||||
|
_resolve_auto_parser(
|
||||||
|
server_args,
|
||||||
|
"reasoning_parser",
|
||||||
|
ctx,
|
||||||
|
REASONING_PARSER_RULES,
|
||||||
|
"reasoning parser",
|
||||||
|
)
|
||||||
|
|
||||||
|
if needs_tool_call:
|
||||||
|
_resolve_auto_parser(
|
||||||
|
server_args,
|
||||||
|
"tool_call_parser",
|
||||||
|
ctx,
|
||||||
|
TOOL_CALL_PARSER_RULES,
|
||||||
|
"tool-call parser",
|
||||||
|
)
|
||||||
@@ -21,9 +21,16 @@ and code completion templates, eliminating global state and improving modularity
|
|||||||
import json
|
import json
|
||||||
import logging
|
import logging
|
||||||
import os
|
import os
|
||||||
import re
|
|
||||||
from typing import Dict, Optional
|
from typing import Dict, Optional
|
||||||
|
|
||||||
|
from sglang.srt.managers.template_detection import (
|
||||||
|
REASONING_PARSER_RULES,
|
||||||
|
TOOL_CALL_PARSER_RULES,
|
||||||
|
ReasoningToggleConfig,
|
||||||
|
build_detection_context,
|
||||||
|
detect_reasoning_pattern,
|
||||||
|
match_rules,
|
||||||
|
)
|
||||||
from sglang.srt.managers.tokenizer_manager import TokenizerManager
|
from sglang.srt.managers.tokenizer_manager import TokenizerManager
|
||||||
from sglang.srt.parser.code_completion_parser import (
|
from sglang.srt.parser.code_completion_parser import (
|
||||||
CompletionTemplate,
|
CompletionTemplate,
|
||||||
@@ -58,6 +65,9 @@ class TemplateManager:
|
|||||||
self._completion_template_name: Optional[str] = None
|
self._completion_template_name: Optional[str] = None
|
||||||
self._jinja_template_content_format: Optional[str] = "openai"
|
self._jinja_template_content_format: Optional[str] = "openai"
|
||||||
self._force_reasoning: bool = False
|
self._force_reasoning: bool = False
|
||||||
|
self._reasoning_config: Optional[ReasoningToggleConfig] = None
|
||||||
|
self._suggested_reasoning_parser: Optional[str] = None
|
||||||
|
self._suggested_tool_call_parser: Optional[str] = None
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def chat_template_name(self) -> Optional[str]:
|
def chat_template_name(self) -> Optional[str]:
|
||||||
@@ -84,21 +94,39 @@ class TemplateManager:
|
|||||||
"""
|
"""
|
||||||
return self._force_reasoning
|
return self._force_reasoning
|
||||||
|
|
||||||
def _detect_reasoning_pattern(self, template: str) -> bool:
|
@property
|
||||||
"""
|
def reasoning_config(self) -> Optional[ReasoningToggleConfig]:
|
||||||
Detect if the chat template contains reasoning/thinking patterns.
|
"""Get the reasoning toggle config inferred from chat template."""
|
||||||
"""
|
return self._reasoning_config
|
||||||
if template is None:
|
|
||||||
return False
|
|
||||||
|
|
||||||
# TODO: remove this hard code the reasoning pattern
|
@property
|
||||||
force_reasoning_pattern = r"<\|im_start\|>assistant\\n<think>\\n"
|
def suggested_reasoning_parser(self) -> Optional[str]:
|
||||||
has_reasoning = re.search(force_reasoning_pattern, template) is not None
|
"""Get the auto-detected reasoning parser name, or None."""
|
||||||
|
return self._suggested_reasoning_parser
|
||||||
|
|
||||||
if has_reasoning:
|
@property
|
||||||
logger.info("Detected the force reasoning pattern in chat template.")
|
def suggested_tool_call_parser(self) -> Optional[str]:
|
||||||
|
"""Get the auto-detected tool-call parser name, or None."""
|
||||||
|
return self._suggested_tool_call_parser
|
||||||
|
|
||||||
return has_reasoning
|
def _run_template_detection(self, template, tokenizer) -> None:
|
||||||
|
"""Run reasoning pattern and parser detection on a template."""
|
||||||
|
self._force_reasoning, self._reasoning_config = detect_reasoning_pattern(
|
||||||
|
template
|
||||||
|
)
|
||||||
|
# Build context once, reuse for both parser detections (avoids
|
||||||
|
# duplicate tokenizer.get_vocab() calls).
|
||||||
|
ctx = build_detection_context(
|
||||||
|
template, tokenizer, self._reasoning_config, self._force_reasoning
|
||||||
|
)
|
||||||
|
if ctx is None:
|
||||||
|
return
|
||||||
|
self._suggested_reasoning_parser = match_rules(
|
||||||
|
ctx, REASONING_PARSER_RULES, "reasoning parser"
|
||||||
|
)
|
||||||
|
self._suggested_tool_call_parser = match_rules(
|
||||||
|
ctx, TOOL_CALL_PARSER_RULES, "tool-call parser"
|
||||||
|
)
|
||||||
|
|
||||||
def load_chat_template(
|
def load_chat_template(
|
||||||
self,
|
self,
|
||||||
@@ -141,11 +169,18 @@ class TemplateManager:
|
|||||||
"No chat template found, defaulting to 'string' content format"
|
"No chat template found, defaulting to 'string' content format"
|
||||||
)
|
)
|
||||||
|
|
||||||
# Detect reasoning pattern from chat template
|
# Detect reasoning pattern and suggest parser from chat template
|
||||||
if tokenizer_manager.tokenizer:
|
if tokenizer_manager.tokenizer:
|
||||||
self._force_reasoning = self._detect_reasoning_pattern(
|
template = tokenizer_manager.tokenizer.chat_template
|
||||||
tokenizer_manager.tokenizer.chat_template
|
self._run_template_detection(template, tokenizer_manager.tokenizer)
|
||||||
)
|
if self._suggested_reasoning_parser:
|
||||||
|
logger.info(
|
||||||
|
f"Auto-detected reasoning parser: {self._suggested_reasoning_parser}"
|
||||||
|
)
|
||||||
|
if self._suggested_tool_call_parser:
|
||||||
|
logger.info(
|
||||||
|
f"Auto-detected tool-call parser: {self._suggested_tool_call_parser}"
|
||||||
|
)
|
||||||
|
|
||||||
def _load_explicit_chat_template(
|
def _load_explicit_chat_template(
|
||||||
self, tokenizer_manager: TokenizerManager, chat_template_arg: str
|
self, tokenizer_manager: TokenizerManager, chat_template_arg: str
|
||||||
|
|||||||
@@ -28,12 +28,17 @@ class BaseReasoningFormatDetector:
|
|||||||
tool_start_token: Optional[str] = None,
|
tool_start_token: Optional[str] = None,
|
||||||
continue_final_message: bool = False,
|
continue_final_message: bool = False,
|
||||||
previous_content: str = "",
|
previous_content: str = "",
|
||||||
|
thinks_internally: bool = False,
|
||||||
|
reasoning_default: str = "always",
|
||||||
):
|
):
|
||||||
self.think_start_token = think_start_token
|
self.think_start_token = think_start_token
|
||||||
self.think_end_token = think_end_token
|
self.think_end_token = think_end_token
|
||||||
self.tool_start_token = tool_start_token
|
self.tool_start_token = tool_start_token
|
||||||
|
self.force_reasoning = force_reasoning
|
||||||
self._in_reasoning = force_reasoning
|
self._in_reasoning = force_reasoning
|
||||||
self.stream_reasoning = stream_reasoning
|
self.stream_reasoning = stream_reasoning
|
||||||
|
self.thinks_internally = thinks_internally
|
||||||
|
self.reasoning_default = reasoning_default
|
||||||
|
|
||||||
self._buffer = ""
|
self._buffer = ""
|
||||||
self.stripped_think_start = False
|
self.stripped_think_start = False
|
||||||
@@ -244,6 +249,8 @@ class Qwen3Detector(BaseReasoningFormatDetector):
|
|||||||
stream_reasoning=stream_reasoning,
|
stream_reasoning=stream_reasoning,
|
||||||
continue_final_message=continue_final_message,
|
continue_final_message=continue_final_message,
|
||||||
previous_content=previous_content,
|
previous_content=previous_content,
|
||||||
|
thinks_internally=True,
|
||||||
|
reasoning_default="enable_thinking",
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
@@ -298,6 +305,7 @@ class KimiK2Detector(BaseReasoningFormatDetector):
|
|||||||
tool_start_token="<|tool_calls_section_begin|>",
|
tool_start_token="<|tool_calls_section_begin|>",
|
||||||
continue_final_message=continue_final_message,
|
continue_final_message=continue_final_message,
|
||||||
previous_content=previous_content,
|
previous_content=previous_content,
|
||||||
|
reasoning_default="thinking",
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
@@ -321,6 +329,8 @@ class Glm45Detector(BaseReasoningFormatDetector):
|
|||||||
force_reasoning=force_reasoning,
|
force_reasoning=force_reasoning,
|
||||||
stream_reasoning=stream_reasoning,
|
stream_reasoning=stream_reasoning,
|
||||||
tool_start_token="<tool_call>",
|
tool_start_token="<tool_call>",
|
||||||
|
thinks_internally=True,
|
||||||
|
reasoning_default="enable_thinking",
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
@@ -445,6 +455,7 @@ class Nemotron3Detector(BaseReasoningFormatDetector):
|
|||||||
stream_reasoning=stream_reasoning,
|
stream_reasoning=stream_reasoning,
|
||||||
continue_final_message=continue_final_message,
|
continue_final_message=continue_final_message,
|
||||||
previous_content=previous_content,
|
previous_content=previous_content,
|
||||||
|
reasoning_default="enable_thinking",
|
||||||
)
|
)
|
||||||
self._force_nonempty_content = force_nonempty_content
|
self._force_nonempty_content = force_nonempty_content
|
||||||
|
|
||||||
@@ -479,6 +490,7 @@ class MistralDetector(BaseReasoningFormatDetector):
|
|||||||
stream_reasoning=stream_reasoning,
|
stream_reasoning=stream_reasoning,
|
||||||
continue_final_message=continue_final_message,
|
continue_final_message=continue_final_message,
|
||||||
previous_content=previous_content,
|
previous_content=previous_content,
|
||||||
|
reasoning_default="mistral",
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
@@ -524,10 +536,27 @@ class Gemma4Detector(BaseReasoningFormatDetector):
|
|||||||
stream_reasoning=stream_reasoning,
|
stream_reasoning=stream_reasoning,
|
||||||
continue_final_message=continue_final_message,
|
continue_final_message=continue_final_message,
|
||||||
previous_content=previous_content,
|
previous_content=previous_content,
|
||||||
|
reasoning_default="explicit_enable_thinking",
|
||||||
)
|
)
|
||||||
self.think_start_self_label = "thought\n"
|
self.think_start_self_label = "thought\n"
|
||||||
|
|
||||||
|
|
||||||
|
class _DeepSeekV3Detector(Qwen3Detector):
|
||||||
|
"""DeepSeek-V3 reuses Qwen3 tokens but requires explicit thinking=True to enable."""
|
||||||
|
|
||||||
|
def __init__(self, **kwargs):
|
||||||
|
super().__init__(**kwargs)
|
||||||
|
self.reasoning_default = "explicit_thinking"
|
||||||
|
|
||||||
|
|
||||||
|
class _MimoDetector(Qwen3Detector):
|
||||||
|
"""MIMO reuses Qwen3 tokens but requires explicit enable_thinking=True to enable."""
|
||||||
|
|
||||||
|
def __init__(self, **kwargs):
|
||||||
|
super().__init__(**kwargs)
|
||||||
|
self.reasoning_default = "explicit_enable_thinking"
|
||||||
|
|
||||||
|
|
||||||
class ReasoningParser:
|
class ReasoningParser:
|
||||||
"""
|
"""
|
||||||
Parser that handles both streaming and non-streaming scenarios for extracting
|
Parser that handles both streaming and non-streaming scenarios for extracting
|
||||||
@@ -541,13 +570,13 @@ class ReasoningParser:
|
|||||||
|
|
||||||
DetectorMap: Dict[str, Type[BaseReasoningFormatDetector]] = {
|
DetectorMap: Dict[str, Type[BaseReasoningFormatDetector]] = {
|
||||||
"deepseek-r1": DeepSeekR1Detector,
|
"deepseek-r1": DeepSeekR1Detector,
|
||||||
"deepseek-v3": Qwen3Detector,
|
"deepseek-v3": _DeepSeekV3Detector,
|
||||||
"glm45": Glm45Detector,
|
"glm45": Glm45Detector,
|
||||||
"hunyuan": HunyuanDetector,
|
"hunyuan": HunyuanDetector,
|
||||||
"gpt-oss": GptOssDetector,
|
"gpt-oss": GptOssDetector,
|
||||||
"kimi": KimiDetector,
|
"kimi": KimiDetector,
|
||||||
"kimi_k2": KimiK2Detector,
|
"kimi_k2": KimiK2Detector,
|
||||||
"mimo": Qwen3Detector,
|
"mimo": _MimoDetector,
|
||||||
"qwen3": Qwen3Detector,
|
"qwen3": Qwen3Detector,
|
||||||
"qwen3-thinking": Qwen3Detector,
|
"qwen3-thinking": Qwen3Detector,
|
||||||
"minimax": Qwen3Detector,
|
"minimax": Qwen3Detector,
|
||||||
@@ -575,7 +604,11 @@ class ReasoningParser:
|
|||||||
raise ValueError(f"Unsupported model type: {model_type}")
|
raise ValueError(f"Unsupported model type: {model_type}")
|
||||||
|
|
||||||
# Special cases where we override force_reasoning
|
# Special cases where we override force_reasoning
|
||||||
if model_type.lower() in {"qwen3-thinking", "gpt-oss", "minimax"}:
|
if model_type.lower() in {
|
||||||
|
"qwen3-thinking",
|
||||||
|
"gpt-oss",
|
||||||
|
"minimax",
|
||||||
|
}:
|
||||||
force_reasoning = True
|
force_reasoning = True
|
||||||
|
|
||||||
# Only pass force_reasoning if explicitly set, let detectors use their defaults
|
# Only pass force_reasoning if explicitly set, let detectors use their defaults
|
||||||
|
|||||||
@@ -5110,12 +5110,15 @@ class ServerArgs:
|
|||||||
action="store_true",
|
action="store_true",
|
||||||
help="Return number of cached tokens in usage.prompt_tokens_details for each openai request.",
|
help="Return number of cached tokens in usage.prompt_tokens_details for each openai request.",
|
||||||
)
|
)
|
||||||
|
reasoning_parser_choices = list(ReasoningParser.DetectorMap.keys())
|
||||||
parser.add_argument(
|
parser.add_argument(
|
||||||
"--reasoning-parser",
|
"--reasoning-parser",
|
||||||
type=str,
|
type=str,
|
||||||
choices=list(ReasoningParser.DetectorMap.keys()),
|
choices=["auto"] + reasoning_parser_choices,
|
||||||
default=ServerArgs.reasoning_parser,
|
default=ServerArgs.reasoning_parser,
|
||||||
help=f"Specify the parser for reasoning models, supported parsers are: {list(ReasoningParser.DetectorMap.keys())}.",
|
help=f"Specify the parser for reasoning models. "
|
||||||
|
f"Use 'auto' to detect from chat template. "
|
||||||
|
f"Options include: {reasoning_parser_choices}.",
|
||||||
)
|
)
|
||||||
parser.add_argument(
|
parser.add_argument(
|
||||||
"--strip-thinking-cache",
|
"--strip-thinking-cache",
|
||||||
@@ -5128,9 +5131,11 @@ class ServerArgs:
|
|||||||
parser.add_argument(
|
parser.add_argument(
|
||||||
"--tool-call-parser",
|
"--tool-call-parser",
|
||||||
type=str,
|
type=str,
|
||||||
choices=tool_call_parser_choices,
|
choices=["auto"] + tool_call_parser_choices,
|
||||||
default=ServerArgs.tool_call_parser,
|
default=ServerArgs.tool_call_parser,
|
||||||
help=f"Specify the parser for handling tool-call interactions. Options include: {tool_call_parser_choices}.",
|
help=f"Specify the parser for handling tool-call interactions. "
|
||||||
|
f"Use 'auto' to detect from chat template. "
|
||||||
|
f"Options include: {tool_call_parser_choices}.",
|
||||||
)
|
)
|
||||||
parser.add_argument(
|
parser.add_argument(
|
||||||
"--tool-server",
|
"--tool-server",
|
||||||
|
|||||||
@@ -28,6 +28,7 @@ from sglang.srt.entrypoints.openai.serving_chat import (
|
|||||||
normalize_tool_content,
|
normalize_tool_content,
|
||||||
)
|
)
|
||||||
from sglang.srt.managers.io_struct import GenerateReqInput
|
from sglang.srt.managers.io_struct import GenerateReqInput
|
||||||
|
from sglang.srt.managers.template_detection import ReasoningToggleConfig
|
||||||
from sglang.srt.utils import get_or_create_event_loop
|
from sglang.srt.utils import get_or_create_event_loop
|
||||||
from sglang.test.ci.ci_register import register_cpu_ci
|
from sglang.test.ci.ci_register import register_cpu_ci
|
||||||
|
|
||||||
@@ -86,6 +87,8 @@ class _MockTemplateManager:
|
|||||||
self.chat_template_name: Optional[str] = "llama-3"
|
self.chat_template_name: Optional[str] = "llama-3"
|
||||||
self.jinja_template_content_format: Optional[str] = None
|
self.jinja_template_content_format: Optional[str] = None
|
||||||
self.completion_template_name: Optional[str] = None
|
self.completion_template_name: Optional[str] = None
|
||||||
|
self.reasoning_config = None
|
||||||
|
self.force_reasoning = False
|
||||||
|
|
||||||
|
|
||||||
class ServingChatTestCase(unittest.TestCase):
|
class ServingChatTestCase(unittest.TestCase):
|
||||||
@@ -1039,6 +1042,175 @@ class ServingChatTestCase(unittest.TestCase):
|
|||||||
req.reasoning_effort = effort
|
req.reasoning_effort = effort
|
||||||
self.assertEqual(chat._get_reasoning_from_request(req), expected)
|
self.assertEqual(chat._get_reasoning_from_request(req), expected)
|
||||||
|
|
||||||
|
# ------------- reasoning config tests -------------
|
||||||
|
def test_get_reasoning_from_request_default_true_toggle(self):
|
||||||
|
self.tm.server_args.reasoning_parser = "qwen3"
|
||||||
|
self.chat.reasoning_parser = "qwen3"
|
||||||
|
self.template_manager.reasoning_config = ReasoningToggleConfig(
|
||||||
|
toggle_param="enable_thinking", default_enabled=True
|
||||||
|
)
|
||||||
|
|
||||||
|
enabled_by_default = ChatCompletionRequest(
|
||||||
|
model="x", messages=[{"role": "user", "content": "Hi?"}]
|
||||||
|
)
|
||||||
|
disabled_explicitly = ChatCompletionRequest(
|
||||||
|
model="x",
|
||||||
|
messages=[{"role": "user", "content": "Hi?"}],
|
||||||
|
chat_template_kwargs={"enable_thinking": False},
|
||||||
|
)
|
||||||
|
|
||||||
|
self.assertTrue(self.chat._get_reasoning_from_request(enabled_by_default))
|
||||||
|
self.assertFalse(self.chat._get_reasoning_from_request(disabled_explicitly))
|
||||||
|
|
||||||
|
def test_get_reasoning_from_request_default_false_toggle(self):
|
||||||
|
self.tm.server_args.reasoning_parser = "deepseek-v3"
|
||||||
|
self.chat.reasoning_parser = "deepseek-v3"
|
||||||
|
self.template_manager.reasoning_config = ReasoningToggleConfig(
|
||||||
|
toggle_param="thinking", default_enabled=False
|
||||||
|
)
|
||||||
|
|
||||||
|
disabled_by_default = ChatCompletionRequest(
|
||||||
|
model="x", messages=[{"role": "user", "content": "Hi?"}]
|
||||||
|
)
|
||||||
|
enabled_explicitly = ChatCompletionRequest(
|
||||||
|
model="x",
|
||||||
|
messages=[{"role": "user", "content": "Hi?"}],
|
||||||
|
chat_template_kwargs={"thinking": True},
|
||||||
|
)
|
||||||
|
|
||||||
|
self.assertFalse(self.chat._get_reasoning_from_request(disabled_by_default))
|
||||||
|
self.assertTrue(self.chat._get_reasoning_from_request(enabled_explicitly))
|
||||||
|
|
||||||
|
def test_get_reasoning_from_request_special_cases(self):
|
||||||
|
self.tm.server_args.reasoning_parser = "mistral"
|
||||||
|
self.chat.reasoning_parser = "mistral"
|
||||||
|
req = ChatCompletionRequest(
|
||||||
|
model="x", messages=[{"role": "user", "content": "Hi?"}]
|
||||||
|
)
|
||||||
|
|
||||||
|
self.template_manager.reasoning_config = ReasoningToggleConfig(
|
||||||
|
special_case="always"
|
||||||
|
)
|
||||||
|
self.assertTrue(self.chat._get_reasoning_from_request(req))
|
||||||
|
|
||||||
|
self.template_manager.reasoning_config = ReasoningToggleConfig(
|
||||||
|
special_case="mistral"
|
||||||
|
)
|
||||||
|
self.assertFalse(self.chat._get_reasoning_from_request(req))
|
||||||
|
req.reasoning_effort = "medium"
|
||||||
|
self.assertTrue(self.chat._get_reasoning_from_request(req))
|
||||||
|
|
||||||
|
# --- fallback path tests (config=None, uses reasoning_default) ---
|
||||||
|
|
||||||
|
def _setup_fallback(self, parser_name):
|
||||||
|
"""Set up reasoning with config=None to exercise the fallback path."""
|
||||||
|
self.tm.server_args.reasoning_parser = parser_name
|
||||||
|
self.chat = OpenAIServingChat(self.tm, self.template_manager)
|
||||||
|
self.chat.reasoning_parser = parser_name
|
||||||
|
self.template_manager.reasoning_config = None
|
||||||
|
|
||||||
|
def test_fallback_always_mode(self):
|
||||||
|
self._setup_fallback("deepseek-r1")
|
||||||
|
req = ChatCompletionRequest(
|
||||||
|
model="x", messages=[{"role": "user", "content": "Hi?"}]
|
||||||
|
)
|
||||||
|
self.assertTrue(self.chat._get_reasoning_from_request(req))
|
||||||
|
|
||||||
|
def test_fallback_mistral_mode(self):
|
||||||
|
self._setup_fallback("mistral")
|
||||||
|
req_no_effort = ChatCompletionRequest(
|
||||||
|
model="x", messages=[{"role": "user", "content": "Hi?"}]
|
||||||
|
)
|
||||||
|
self.assertFalse(self.chat._get_reasoning_from_request(req_no_effort))
|
||||||
|
|
||||||
|
req_with_effort = ChatCompletionRequest(
|
||||||
|
model="x",
|
||||||
|
messages=[{"role": "user", "content": "Hi?"}],
|
||||||
|
reasoning_effort="high",
|
||||||
|
)
|
||||||
|
self.assertTrue(self.chat._get_reasoning_from_request(req_with_effort))
|
||||||
|
|
||||||
|
def test_fallback_enable_thinking_mode_default_on(self):
|
||||||
|
self._setup_fallback("qwen3")
|
||||||
|
req_default = ChatCompletionRequest(
|
||||||
|
model="x", messages=[{"role": "user", "content": "Hi?"}]
|
||||||
|
)
|
||||||
|
self.assertTrue(self.chat._get_reasoning_from_request(req_default))
|
||||||
|
|
||||||
|
req_disabled = ChatCompletionRequest(
|
||||||
|
model="x",
|
||||||
|
messages=[{"role": "user", "content": "Hi?"}],
|
||||||
|
chat_template_kwargs={"enable_thinking": False},
|
||||||
|
)
|
||||||
|
self.assertFalse(self.chat._get_reasoning_from_request(req_disabled))
|
||||||
|
|
||||||
|
def test_fallback_explicit_thinking_mode_default_off(self):
|
||||||
|
self._setup_fallback("deepseek-v3")
|
||||||
|
req_default = ChatCompletionRequest(
|
||||||
|
model="x", messages=[{"role": "user", "content": "Hi?"}]
|
||||||
|
)
|
||||||
|
self.assertFalse(self.chat._get_reasoning_from_request(req_default))
|
||||||
|
|
||||||
|
req_enabled = ChatCompletionRequest(
|
||||||
|
model="x",
|
||||||
|
messages=[{"role": "user", "content": "Hi?"}],
|
||||||
|
chat_template_kwargs={"thinking": True},
|
||||||
|
)
|
||||||
|
self.assertTrue(self.chat._get_reasoning_from_request(req_enabled))
|
||||||
|
|
||||||
|
def test_fallback_explicit_enable_thinking_mode_default_off(self):
|
||||||
|
self._setup_fallback("mimo")
|
||||||
|
req_default = ChatCompletionRequest(
|
||||||
|
model="x", messages=[{"role": "user", "content": "Hi?"}]
|
||||||
|
)
|
||||||
|
self.assertFalse(self.chat._get_reasoning_from_request(req_default))
|
||||||
|
|
||||||
|
req_enabled = ChatCompletionRequest(
|
||||||
|
model="x",
|
||||||
|
messages=[{"role": "user", "content": "Hi?"}],
|
||||||
|
chat_template_kwargs={"enable_thinking": True},
|
||||||
|
)
|
||||||
|
self.assertTrue(self.chat._get_reasoning_from_request(req_enabled))
|
||||||
|
|
||||||
|
def test_fallback_no_detector_returns_false(self):
|
||||||
|
self.chat.reasoning_parser = "qwen3"
|
||||||
|
self.chat._reasoning_detector = None
|
||||||
|
self.template_manager.reasoning_config = None
|
||||||
|
req = ChatCompletionRequest(
|
||||||
|
model="x", messages=[{"role": "user", "content": "Hi?"}]
|
||||||
|
)
|
||||||
|
self.assertFalse(self.chat._get_reasoning_from_request(req))
|
||||||
|
|
||||||
|
def test_build_chat_response_qwen3_thinking_forces_reasoning(self):
|
||||||
|
self.tm.server_args.reasoning_parser = "qwen3-thinking"
|
||||||
|
self.chat.reasoning_parser = "qwen3-thinking"
|
||||||
|
self.template_manager.reasoning_config = ReasoningToggleConfig(
|
||||||
|
toggle_param="enable_thinking", default_enabled=True
|
||||||
|
)
|
||||||
|
|
||||||
|
req = ChatCompletionRequest(
|
||||||
|
model="Qwen/Qwen3-0.6B",
|
||||||
|
messages=[{"role": "user", "content": "Hi?"}],
|
||||||
|
separate_reasoning=True,
|
||||||
|
chat_template_kwargs={"enable_thinking": False},
|
||||||
|
)
|
||||||
|
ret_item = {
|
||||||
|
"text": "42",
|
||||||
|
"meta_info": {
|
||||||
|
"id": f"chatcmpl-{uuid.uuid4()}",
|
||||||
|
"prompt_tokens": 10,
|
||||||
|
"completion_tokens": 1,
|
||||||
|
"weight_version": "default",
|
||||||
|
"finish_reason": {"type": "stop", "matched": None},
|
||||||
|
},
|
||||||
|
"index": 0,
|
||||||
|
}
|
||||||
|
|
||||||
|
response = self.chat._build_chat_response(req, [ret_item], created=0)
|
||||||
|
msg = response.choices[0].message
|
||||||
|
self.assertIsNone(msg.content)
|
||||||
|
self.assertEqual(msg.reasoning_content, "42")
|
||||||
|
|
||||||
|
|
||||||
class TestProcessToolCallsWithRequiredToolChoice(unittest.TestCase):
|
class TestProcessToolCallsWithRequiredToolChoice(unittest.TestCase):
|
||||||
"""Test _process_tool_calls with tool_choice='required' uses model-specific parser."""
|
"""Test _process_tool_calls with tool_choice='required' uses model-specific parser."""
|
||||||
|
|||||||
@@ -0,0 +1,334 @@
|
|||||||
|
import unittest
|
||||||
|
from types import SimpleNamespace
|
||||||
|
|
||||||
|
from sglang.srt.managers.template_detection import (
|
||||||
|
ReasoningToggleConfig,
|
||||||
|
detect_reasoning_parser,
|
||||||
|
detect_reasoning_pattern,
|
||||||
|
detect_tool_call_parser,
|
||||||
|
resolve_auto_parsers,
|
||||||
|
)
|
||||||
|
from sglang.test.ci.ci_register import register_cpu_ci
|
||||||
|
|
||||||
|
register_cpu_ci(2.0, "stage-a-test-cpu")
|
||||||
|
|
||||||
|
|
||||||
|
class _DummyTokenizer:
|
||||||
|
def __init__(self, vocab):
|
||||||
|
self._vocab = vocab
|
||||||
|
|
||||||
|
def get_vocab(self):
|
||||||
|
return {token: i for i, token in enumerate(self._vocab)}
|
||||||
|
|
||||||
|
|
||||||
|
class TestTemplateManagerReasoningDetection(unittest.TestCase):
|
||||||
|
|
||||||
|
def _detect(self, template, vocab):
|
||||||
|
force, config = detect_reasoning_pattern(template)
|
||||||
|
parser = detect_reasoning_parser(
|
||||||
|
template, _DummyTokenizer(vocab), config, force
|
||||||
|
)
|
||||||
|
return force, config, parser
|
||||||
|
|
||||||
|
def test_qwen3_template_not_misclassified_as_glm45(self):
|
||||||
|
template = """
|
||||||
|
{% set enable_thinking = enable_thinking if enable_thinking is defined else true %}
|
||||||
|
{% if '</think>' in content %}
|
||||||
|
<tool_call>
|
||||||
|
"""
|
||||||
|
_, config, parser = self._detect(
|
||||||
|
template, ["<tool_call>", "<|endoftext|>", "</think>"]
|
||||||
|
)
|
||||||
|
|
||||||
|
self.assertEqual(
|
||||||
|
config,
|
||||||
|
ReasoningToggleConfig(toggle_param="enable_thinking", default_enabled=True),
|
||||||
|
)
|
||||||
|
self.assertEqual(parser, "qwen3")
|
||||||
|
|
||||||
|
def test_glm45_requires_glm_specific_template_markers(self):
|
||||||
|
template = """
|
||||||
|
[gMASK]<sop>
|
||||||
|
{% set enable_thinking = enable_thinking if enable_thinking is defined else true %}
|
||||||
|
/nothink
|
||||||
|
<tool_call>
|
||||||
|
"""
|
||||||
|
_, config, parser = self._detect(
|
||||||
|
template, ["<tool_call>", "<|endoftext|>", "<|user|>"]
|
||||||
|
)
|
||||||
|
|
||||||
|
self.assertEqual(
|
||||||
|
config,
|
||||||
|
ReasoningToggleConfig(toggle_param="enable_thinking", default_enabled=True),
|
||||||
|
)
|
||||||
|
self.assertEqual(parser, "glm45")
|
||||||
|
|
||||||
|
def test_interns1_detects_enable_thinking_default_true(self):
|
||||||
|
template = """
|
||||||
|
{% set default_thinking_sys %}...<think>...</think>{% endset %}
|
||||||
|
{% if enable_thinking is not defined or enable_thinking %}
|
||||||
|
"""
|
||||||
|
_, config, parser = self._detect(template, ["<|endoftext|>"])
|
||||||
|
|
||||||
|
self.assertEqual(
|
||||||
|
config,
|
||||||
|
ReasoningToggleConfig(toggle_param="enable_thinking", default_enabled=True),
|
||||||
|
)
|
||||||
|
self.assertEqual(parser, "interns1")
|
||||||
|
|
||||||
|
def test_nemotron_detects_uppercase_true_assignment(self):
|
||||||
|
template = """
|
||||||
|
{% set enable_thinking = enable_thinking if enable_thinking is defined else True %}
|
||||||
|
{% set truncate_history_thinking = truncate_history_thinking if truncate_history_thinking is defined else True %}
|
||||||
|
"""
|
||||||
|
_, config, parser = self._detect(template, ["<|endoftext|>"])
|
||||||
|
|
||||||
|
self.assertEqual(
|
||||||
|
config,
|
||||||
|
ReasoningToggleConfig(toggle_param="enable_thinking", default_enabled=True),
|
||||||
|
)
|
||||||
|
self.assertEqual(parser, "nemotron_3")
|
||||||
|
|
||||||
|
def test_minimax_uses_template_signature_without_toggle_config(self):
|
||||||
|
template = """
|
||||||
|
{%- set toolcall_begin_token = '<minimax:tool_call>' -%}
|
||||||
|
"""
|
||||||
|
_, config, parser = self._detect(template, ["<minimax:tool_call>"])
|
||||||
|
|
||||||
|
self.assertIsNone(config)
|
||||||
|
self.assertEqual(parser, "minimax")
|
||||||
|
|
||||||
|
|
||||||
|
class TestTemplateDetectionRuleMatrix(unittest.TestCase):
|
||||||
|
"""Table-driven tests for REASONING_PARSER_RULES and REASONING_MODE_RULES."""
|
||||||
|
|
||||||
|
def _detect(self, template, vocab=None):
|
||||||
|
if vocab is None:
|
||||||
|
vocab = []
|
||||||
|
force, config = detect_reasoning_pattern(template)
|
||||||
|
parser = detect_reasoning_parser(
|
||||||
|
template, _DummyTokenizer(vocab), config, force
|
||||||
|
)
|
||||||
|
return force, config, parser
|
||||||
|
|
||||||
|
PARSER_RULES_MATRIX = [
|
||||||
|
# (name, template_snippet, vocab, expected_parser, expected_toggle_param)
|
||||||
|
(
|
||||||
|
"deepseek_r1_think_tags",
|
||||||
|
"<think>\nLet me reason about this\n</think>\nAnswer here",
|
||||||
|
[],
|
||||||
|
"deepseek-r1",
|
||||||
|
None, # matched by deepseek_r1_think_tags rule (has <think> text)
|
||||||
|
),
|
||||||
|
(
|
||||||
|
"deepseek_v3",
|
||||||
|
"{% if not thinking is defined %}{% set thinking = false %}{% endif %}\n"
|
||||||
|
"<think>",
|
||||||
|
[],
|
||||||
|
"deepseek-v3",
|
||||||
|
"thinking",
|
||||||
|
),
|
||||||
|
(
|
||||||
|
"qwen3_enable_thinking_true",
|
||||||
|
"{% set enable_thinking = enable_thinking if enable_thinking is defined else true %}\n",
|
||||||
|
[],
|
||||||
|
"qwen3",
|
||||||
|
"enable_thinking",
|
||||||
|
),
|
||||||
|
(
|
||||||
|
"kimi_unicode_markers",
|
||||||
|
"\u25c1think\u25b7some text\u25c1/think\u25b7",
|
||||||
|
[],
|
||||||
|
"kimi",
|
||||||
|
None,
|
||||||
|
),
|
||||||
|
(
|
||||||
|
"mistral_reasoning_effort",
|
||||||
|
"{% if reasoning_effort %}[THINK]{% endif %}",
|
||||||
|
[],
|
||||||
|
"mistral",
|
||||||
|
None, # special_case="mistral"
|
||||||
|
),
|
||||||
|
(
|
||||||
|
"gpt_oss_channel",
|
||||||
|
"<|channel|>analysis<|message|>",
|
||||||
|
[],
|
||||||
|
"gpt-oss",
|
||||||
|
None, # special_case="always"
|
||||||
|
),
|
||||||
|
(
|
||||||
|
"kimi_k2_with_tool_vocab",
|
||||||
|
"{% set thinking = thinking if thinking is defined else true %}\n<think>",
|
||||||
|
["<|tool_calls_section_begin|>", "<|tool_calls_section_end|>"],
|
||||||
|
"kimi_k2",
|
||||||
|
"thinking",
|
||||||
|
),
|
||||||
|
(
|
||||||
|
"mimo_enable_thinking_false",
|
||||||
|
"{% if not enable_thinking is defined %}{% set enable_thinking = false %}{% endif %}\n"
|
||||||
|
"enable_thinking",
|
||||||
|
[],
|
||||||
|
"mimo",
|
||||||
|
"enable_thinking",
|
||||||
|
),
|
||||||
|
]
|
||||||
|
|
||||||
|
def test_parser_rules_matrix(self):
|
||||||
|
for (
|
||||||
|
name,
|
||||||
|
template,
|
||||||
|
vocab,
|
||||||
|
expected_parser,
|
||||||
|
expected_toggle,
|
||||||
|
) in self.PARSER_RULES_MATRIX:
|
||||||
|
with self.subTest(name=name):
|
||||||
|
_, config, parser = self._detect(template, vocab)
|
||||||
|
self.assertEqual(
|
||||||
|
parser,
|
||||||
|
expected_parser,
|
||||||
|
f"Rule '{name}': expected parser '{expected_parser}', got '{parser}'",
|
||||||
|
)
|
||||||
|
if expected_toggle is not None:
|
||||||
|
self.assertIsNotNone(
|
||||||
|
config, f"Rule '{name}': expected config, got None"
|
||||||
|
)
|
||||||
|
self.assertEqual(
|
||||||
|
config.toggle_param,
|
||||||
|
expected_toggle,
|
||||||
|
f"Rule '{name}': expected toggle '{expected_toggle}', "
|
||||||
|
f"got '{config.toggle_param}'",
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_unrecognized_template_returns_none(self):
|
||||||
|
template = "Hello {{ user_message }}, how can I help you?"
|
||||||
|
_, config, parser = self._detect(template)
|
||||||
|
|
||||||
|
self.assertIsNone(config)
|
||||||
|
self.assertIsNone(parser)
|
||||||
|
|
||||||
|
def test_empty_template_returns_none(self):
|
||||||
|
_, config, parser = self._detect("")
|
||||||
|
|
||||||
|
self.assertIsNone(config)
|
||||||
|
self.assertIsNone(parser)
|
||||||
|
|
||||||
|
def test_qwen3_precedence_over_deepseek_r1(self):
|
||||||
|
"""Template with enable_thinking=true but no <think> tag should be qwen3, not deepseek_r1."""
|
||||||
|
template = "{% set enable_thinking = enable_thinking if enable_thinking is defined else true %}"
|
||||||
|
_, config, parser = self._detect(template)
|
||||||
|
|
||||||
|
self.assertEqual(parser, "qwen3")
|
||||||
|
self.assertEqual(config.toggle_param, "enable_thinking")
|
||||||
|
self.assertTrue(config.default_enabled)
|
||||||
|
|
||||||
|
|
||||||
|
class TestToolCallParserDetection(unittest.TestCase):
|
||||||
|
"""Tests for detect_tool_call_parser() using real model tokenizers."""
|
||||||
|
|
||||||
|
def _detect_all(self, model_name):
|
||||||
|
from transformers import AutoTokenizer
|
||||||
|
|
||||||
|
tok = AutoTokenizer.from_pretrained(model_name, trust_remote_code=True)
|
||||||
|
template = tok.chat_template
|
||||||
|
force, config = detect_reasoning_pattern(template)
|
||||||
|
rp = detect_reasoning_parser(template, tok, config, force)
|
||||||
|
tcp = detect_tool_call_parser(template, tok, config, force)
|
||||||
|
return rp, tcp
|
||||||
|
|
||||||
|
def test_qwen3_detects_qwen_tool_call_parser(self):
|
||||||
|
rp, tcp = self._detect_all("Qwen/Qwen3-0.6B")
|
||||||
|
self.assertEqual(rp, "qwen3")
|
||||||
|
self.assertEqual(tcp, "qwen")
|
||||||
|
|
||||||
|
def test_tool_call_parser_rule_values_via_snippets(self):
|
||||||
|
"""Table-driven: verify tool-call rule values differ from reasoning where expected."""
|
||||||
|
cases = [
|
||||||
|
# (name, template, vocab, expected_tool_call)
|
||||||
|
(
|
||||||
|
"qwen_maps_from_qwen3_config",
|
||||||
|
"{% set enable_thinking = enable_thinking if enable_thinking is defined else true %}",
|
||||||
|
[],
|
||||||
|
"qwen",
|
||||||
|
),
|
||||||
|
("gpt_oss", "<|channel|>analysis<|message|>", [], "gpt-oss"),
|
||||||
|
("gemma4", "<|channel>content", [], "gemma4"),
|
||||||
|
("minimax_maps_to_m2", "<minimax:tool_call>", [], "minimax-m2"),
|
||||||
|
(
|
||||||
|
"deepseekv3",
|
||||||
|
"{% if not thinking is defined %}{% set thinking = false %}{% endif %}",
|
||||||
|
[],
|
||||||
|
"deepseekv3",
|
||||||
|
),
|
||||||
|
(
|
||||||
|
"kimi_k2",
|
||||||
|
"{% set thinking = thinking if thinking is defined else true %}\n<think>",
|
||||||
|
["<|tool_calls_section_begin|>"],
|
||||||
|
"kimi_k2",
|
||||||
|
),
|
||||||
|
]
|
||||||
|
for name, template, vocab, expected in cases:
|
||||||
|
with self.subTest(name=name):
|
||||||
|
force, config = detect_reasoning_pattern(template)
|
||||||
|
result = detect_tool_call_parser(
|
||||||
|
template, _DummyTokenizer(vocab), config, force
|
||||||
|
)
|
||||||
|
self.assertEqual(result, expected)
|
||||||
|
|
||||||
|
def test_none_template_returns_none(self):
|
||||||
|
self.assertIsNone(detect_tool_call_parser(None, None))
|
||||||
|
|
||||||
|
def test_unrecognized_template_returns_none(self):
|
||||||
|
force, config = detect_reasoning_pattern("Hello {{ user }}")
|
||||||
|
result = detect_tool_call_parser("Hello {{ user }}", None, config, force)
|
||||||
|
self.assertIsNone(result)
|
||||||
|
|
||||||
|
|
||||||
|
class TestResolveAutoParsers(unittest.TestCase):
|
||||||
|
"""Tests for resolve_auto_parsers() using real model tokenizers."""
|
||||||
|
|
||||||
|
def _make_server_args(self, reasoning_parser=None, tool_call_parser=None):
|
||||||
|
return SimpleNamespace(
|
||||||
|
reasoning_parser=reasoning_parser,
|
||||||
|
tool_call_parser=tool_call_parser,
|
||||||
|
model_path="Qwen/Qwen3-0.6B",
|
||||||
|
trust_remote_code=False,
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_resolves_both_parsers_with_real_model(self):
|
||||||
|
args = self._make_server_args(reasoning_parser="auto", tool_call_parser="auto")
|
||||||
|
resolve_auto_parsers(args)
|
||||||
|
self.assertEqual(args.reasoning_parser, "qwen3")
|
||||||
|
self.assertEqual(args.tool_call_parser, "qwen")
|
||||||
|
|
||||||
|
def test_resolves_reasoning_parser_only(self):
|
||||||
|
args = self._make_server_args(reasoning_parser="auto", tool_call_parser=None)
|
||||||
|
resolve_auto_parsers(args)
|
||||||
|
self.assertEqual(args.reasoning_parser, "qwen3")
|
||||||
|
self.assertIsNone(args.tool_call_parser)
|
||||||
|
|
||||||
|
def test_resolves_tool_call_parser_only(self):
|
||||||
|
args = self._make_server_args(reasoning_parser="qwen3", tool_call_parser="auto")
|
||||||
|
resolve_auto_parsers(args)
|
||||||
|
self.assertEqual(args.reasoning_parser, "qwen3")
|
||||||
|
self.assertEqual(args.tool_call_parser, "qwen")
|
||||||
|
|
||||||
|
def test_neither_auto_is_noop(self):
|
||||||
|
args = self._make_server_args(reasoning_parser="qwen3", tool_call_parser="qwen")
|
||||||
|
resolve_auto_parsers(args)
|
||||||
|
self.assertEqual(args.reasoning_parser, "qwen3")
|
||||||
|
self.assertEqual(args.tool_call_parser, "qwen")
|
||||||
|
|
||||||
|
def test_nonexistent_model_disables_both_parsers(self):
|
||||||
|
args = SimpleNamespace(
|
||||||
|
reasoning_parser="auto",
|
||||||
|
tool_call_parser="auto",
|
||||||
|
model_path="nonexistent/model-does-not-exist-xyz",
|
||||||
|
trust_remote_code=False,
|
||||||
|
)
|
||||||
|
resolve_auto_parsers(args)
|
||||||
|
self.assertIsNone(args.reasoning_parser)
|
||||||
|
self.assertIsNone(args.tool_call_parser)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
unittest.main()
|
||||||
Reference in New Issue
Block a user