Stop losing Kimi-K3 tool calls to reasoning, constraint conflicts, and truncation (#34881)

Co-authored-by: Claude Opus 5 (1M context) <noreply@anthropic.com>
Co-authored-by: Xinyuan Tong <xinyuantong.cs@gmail.com>
Co-authored-by: Xinyuan Tong <115166877+JustinTong0323@users.noreply.github.com>
This commit is contained in:
Khoa Pham
2026-08-18 12:42:56 -07:00
committed by GitHub
co-authored by Claude Opus 5 Xinyuan Tong Xinyuan Tong
parent 8bb106cee9
commit 307a90f6d3
9 changed files with 374 additions and 12 deletions
@@ -27,13 +27,15 @@ from sglang.srt.entrypoints.openai.chat_encoding import (
from sglang.srt.entrypoints.openai.protocol import (
ChatCompletionRequest,
MessageProcessingResult,
ToolChoice,
ToolChoiceFuncName,
)
from sglang.srt.entrypoints.openai.serving_chat import (
OpenAIServingChat,
normalize_tool_content,
)
from sglang.srt.environ import envs
from sglang.srt.function_call.kimik3_format import TOOLS_CLOSE
from sglang.srt.function_call.kimik3_format import TOOLS_CLOSE, TOOLS_OPEN
from sglang.srt.managers.io_struct import GenerateReqInput
from sglang.srt.parser.template_detection import ReasoningToggleConfig
from sglang.srt.utils import get_or_create_event_loop
@@ -1569,6 +1571,191 @@ class ServingChatTestCase(unittest.TestCase):
self.assertEqual(tool_calls[1].id, "functions.get_weather:2")
self.assertEqual(tool_calls[1].function.name, "get_weather")
def test_required_tool_choice_skips_json_fallback_for_native_parser(self):
"""A structural-tag parser owns the output format, so a missing tool
call must not be pushed through the json_schema array fallback."""
self.chat.tool_call_parser = "kimi_k3"
tools = [
{
"type": "function",
"function": {
"name": "get_weather",
"parameters": {
"type": "object",
"properties": {"city": {"type": "string"}},
},
},
}
]
texts = {
"prose": "<|open|>response<|sep|>I'll check that.<|close|>response<|sep|>",
"empty": "",
"json_object": '{"name": "get_weather", "parameters": {"city": "Paris"}}',
}
for choice in (
"required",
ToolChoice(function=ToolChoiceFuncName(name="get_weather")),
):
for label, text in texts.items():
with self.subTest(tool_choice=choice, payload=label):
finish_reason = {"type": "stop", "matched": None}
with self.assertLogs(
"sglang.srt.entrypoints.openai.serving_chat", level="WARNING"
) as logs:
tool_calls, remaining, finish_reason = (
self.chat._process_tool_calls(
text=text,
tools=tools,
finish_reason=finish_reason,
tool_choice=choice,
)
)
self.assertIsNone(tool_calls)
self.assertEqual(remaining, text)
self.assertEqual(finish_reason["type"], "stop")
self.assertNotIn("Tool call parsing error", "\n".join(logs.output))
def test_truncated_native_tool_call_logs_and_drops(self):
"""A tools section cut off before its closing tag parses to zero calls
without raising; the sync path used to drop it with no log at all."""
self.chat.tool_call_parser = "kimi_k3"
tools = [
{
"type": "function",
"function": {
"name": "get_weather",
"parameters": {
"type": "object",
"properties": {"city": {"type": "string"}},
},
},
}
]
truncated = TOOLS_OPEN + '<|open|>call tool="get_weather" index="1"<|sep|>'
for choice in ("auto", "required"):
with self.subTest(tool_choice=choice):
finish_reason = {"type": "stop", "matched": None}
with self.assertLogs(
"sglang.srt.entrypoints.openai.serving_chat", level="WARNING"
) as logs:
tool_calls, remaining, finish_reason = (
self.chat._process_tool_calls(
text=truncated,
tools=tools,
finish_reason=finish_reason,
tool_choice=choice,
)
)
self.assertIsNone(tool_calls)
self.assertEqual(remaining, "")
self.assertEqual(finish_reason["type"], "stop")
self.assertIn("no complete call", "\n".join(logs.output))
def test_required_tool_choice_json_fallback_tolerates_odd_shapes(self):
"""Parsers without a structural tag keep the JSON array fallback, but a
non-array payload degrades instead of raising an opaque TypeError."""
self.chat.tool_call_parser = "glm45"
tools = [
{
"type": "function",
"function": {
"name": "get_weather",
"parameters": {
"type": "object",
"properties": {"city": {"type": "string"}},
},
},
}
]
cases = [
(
'[{"name": "get_weather", "parameters": {"city": "Paris"}}]',
'{"city": "Paris"}',
),
(
'{"name": "get_weather", "parameters": {"city": "Paris"}}',
'{"city": "Paris"}',
),
('{"name": "get_weather"}', "{}"),
]
for text, expected_args in cases:
with self.subTest(text=text):
tool_calls, _, finish_reason = self.chat._process_tool_calls(
text=text,
tools=tools,
finish_reason={"type": "stop", "matched": None},
tool_choice="required",
)
self.assertEqual(len(tool_calls), 1)
self.assertEqual(tool_calls[0].function.name, "get_weather")
self.assertEqual(tool_calls[0].function.arguments, expected_args)
self.assertEqual(finish_reason["type"], "tool_calls")
finish_reason = {"type": "stop", "matched": None}
with self.assertLogs(
"sglang.srt.entrypoints.openai.serving_chat", level="ERROR"
):
tool_calls, remaining, finish_reason = self.chat._process_tool_calls(
text='["get_weather"]',
tools=tools,
finish_reason=finish_reason,
tool_choice="required",
)
self.assertIsNone(tool_calls)
self.assertEqual(remaining, '["get_weather"]')
self.assertEqual(finish_reason["type"], "stop")
def test_required_tool_choice_rejects_conflicting_output_constraint(self):
"""response_format and a forced tool call cannot both be honored: the
tool-call constraint was dropped with only a warning, so the model was
constrained to a shape that can never contain a tool call."""
tools = [{"type": "function", "function": {"name": "get_weather"}}]
constraint = ("structural_tag", None)
conflicting = [
{"type": "json_object"},
{
"type": "json_schema",
"json_schema": {"name": "a", "schema": {"type": "object"}},
},
]
for tool_choice in (
"required",
ToolChoice(function=ToolChoiceFuncName(name="get_weather")),
):
for response_format in conflicting:
with self.subTest(tool_choice=tool_choice, rf=response_format["type"]):
request = ChatCompletionRequest(
model="x",
messages=[{"role": "user", "content": "hi"}],
tools=tools,
tool_choice=tool_choice,
response_format=response_format,
)
with self.assertRaises(ValueError) as ctx:
request.to_sampling_params(
stop=[],
model_generation_config={},
tool_call_constraint=constraint,
)
self.assertIn("cannot be combined", str(ctx.exception))
def test_auto_tool_choice_keeps_response_format_without_raising(self):
""" "auto" means the model need not call a tool, so dropping the
tool-call constraint still leaves a satisfiable request."""
request = ChatCompletionRequest(
model="x",
messages=[{"role": "user", "content": "hi"}],
tools=[{"type": "function", "function": {"name": "get_weather"}}],
tool_choice="auto",
response_format={"type": "json_object"},
)
sampling_params = request.to_sampling_params(
stop=[],
model_generation_config={},
tool_call_constraint=("structural_tag", None),
)
self.assertEqual(sampling_params["json_schema"], '{"type": "object"}')
def test_kimi_k2_streaming_tool_call_id_with_history(self):
"""Ensure streaming first chunk tool_call.id increase with tool calls history for kimi_k2 parser."""
@@ -691,6 +691,44 @@ class OutputItemsTestCase(CustomTestCase):
[],
)
def test_required_tool_choice_skips_json_fallback_for_native_parser(self):
"""muse reports parses_required_natively, so required output must not
be pushed through the orjson JSON-array fallback (mirrors chat)."""
serving = self.serving
serving.tool_call_parser = "muse"
request = ResponsesRequest(
model="x",
input="hi",
tool_choice="required",
tools=[
{
"type": "function",
"name": "get_weather",
"parameters": {"type": "object"},
}
],
store=False,
)
raw = '[{"name": "get_weather", "parameters": {"city": "Beijing"}}]'
output_items = serving._make_response_output_items(
request, raw, tokenizer=Mock(), require_reasoning=False
)
self.assertEqual(
[
item
for item in output_items
if isinstance(item, ResponseFunctionToolCall)
],
[],
)
message_items = [
item for item in output_items if isinstance(item, ResponseOutputMessage)
]
self.assertEqual(len(message_items), 1)
self.assertEqual(message_items[0].content[0].text, raw)
def test_no_tool_call_extraction_when_tool_choice_none(self):
serving = self.serving
request = ResponsesRequest(