[Fix] Make DeepSeek-V4 reasoning and tool-call streaming parsing chunk-invariant (#34458)

Co-authored-by: hao-cyber <89575785+hao-cyber@users.noreply.github.com>
Co-authored-by: Enrico Falco <enrico9034@gmail.com>
Co-authored-by: Svyatoslav <85786374+slivanovich@users.noreply.github.com>
Co-authored-by: Andreas Hassellof <andreas@ombori.com>
Co-authored-by: Leoyzen <leoyzen@gmail.com>
Co-authored-by: Chenglun Hu <chenglunhu@gmail.com>
Co-authored-by: robellliu-dev <robell.liu@huawei.com>
Co-authored-by: Gavin.Zhu <gavin.z@gmicloud.ai>
Co-authored-by: Xinyuan Tong <xinyuantong.cs@gmail.com>
Co-authored-by: tancheng33 <garrytancheng@gmail.com>
Co-authored-by: dineshx29 <dinesh.b.offl@gmail.com>
Co-authored-by: Kangyan Zhou <zky314343421@gmail.com>
This commit is contained in:
Liangsheng Yin
2026-08-11 20:03:28 -07:00
committed by GitHub
co-authored by hao-cyber Enrico Falco Svyatoslav Andreas Hassellof Leoyzen Chenglun Hu robellliu-dev Gavin.Zhu Xinyuan Tong tancheng33 dineshx29 Kangyan Zhou
parent 2be9773a21
commit 5899674504
4 changed files with 383 additions and 48 deletions
@@ -0,0 +1,129 @@
"""Unit tests for DeepSeekV4Detector DSML streaming — no server, no model loading."""
from unittest.mock import patch
from sglang.srt.entrypoints.openai.protocol import Function, Tool
from sglang.srt.function_call.deepseekv4_detector import DeepSeekV4Detector
from sglang.test.ci.ci_register import register_cpu_ci
from sglang.test.test_utils import CustomTestCase
register_cpu_ci(1.0, "base-a-test-cpu")
DSML = "|DSML|"
def _wrapped(invoke: str) -> str:
return f"<{DSML}tool_calls>\n{invoke}\n</{DSML}tool_calls>"
def _invoke(name: str, params: str = "") -> str:
return f'<{DSML}invoke name="{name}">\n{params}\n</{DSML}invoke>'
def _param(name: str, is_string: str, value: str) -> str:
return (
f'<{DSML}parameter name="{name}" string="{is_string}">{value}</{DSML}parameter>'
)
def _weather_call(city: str = "SF") -> str:
return _wrapped(_invoke("get_weather", _param("city", "true", city)))
class TestDeepSeekV4Streaming(CustomTestCase):
def setUp(self):
self.tools = [
Tool(
type="function",
function=Function(
name="get_weather",
description="Get weather information",
parameters={
"type": "object",
"properties": {"city": {"type": "string"}},
"required": ["city"],
},
),
)
]
def _feed(self, chunks):
"""Returns (normal_text, calls) accumulated over the chunks."""
detector = DeepSeekV4Detector()
normal, calls = "", []
for chunk in chunks:
result = detector.parse_streaming_increment(chunk, self.tools)
normal += result.normal_text
calls.extend(result.calls)
return normal, calls
def test_preamble_in_same_delta_as_tool_call(self):
"""Prose sharing a delta with the tool call must not be dropped, and the
streaming and one-shot paths must agree on it."""
text = "Let me check.\n" + _weather_call()
normal, calls = self._feed([text])
self.assertEqual([c.name for c in calls if c.name], ["get_weather"])
self.assertEqual(
normal, DeepSeekV4Detector().detect_and_parse(text, self.tools).normal_text
)
def test_preamble_before_bare_invoke_without_wrapper(self):
"""The bare `<|DSML|invoke …>` form has no tool_calls wrapper to walk
back to, so the preamble is computed from the invoke itself."""
text = "Checking.\n" + _invoke("get_weather", _param("city", "true", "SF"))
normal, calls = self._feed([text])
self.assertIn("Checking.", normal)
self.assertEqual([c.name for c in calls if c.name], ["get_weather"])
def test_no_dsml_markers_leak_into_normal_text(self):
text = "Prose.\n" + _weather_call()
normal, _ = self._feed([text[i : i + 4] for i in range(0, len(text), 4)])
self.assertNotIn(DSML, normal)
def test_malformed_partial_json_falls_back_to_raw_value(self):
"""A partial non-string parameter must not escape as MalformedJSON."""
detector = DeepSeekV4Detector()
result = detector.parse_streaming_increment(
f'<{DSML}tool_calls>\n<{DSML}invoke name="get_weather">\n'
f'<{DSML}parameter name="city" string="false">{{"a"',
self.tools,
)
self.assertEqual([c.name for c in result.calls if c.name], ["get_weather"])
def test_non_streaming_parses_every_tool_calls_section(self):
"""A turn with two tool_calls sections must yield both calls."""
result = DeepSeekV4Detector().detect_and_parse(
f"{_weather_call('SF')}\n{_weather_call('NY')}", self.tools
)
self.assertEqual(len(result.calls), 2)
def test_parse_error_neither_swallows_nor_duplicates(self):
"""An unexpected parse error must not empty the turn, and the dropped
buffer must not come back on the next delta."""
detector = DeepSeekV4Detector()
with patch.object(
DeepSeekV4Detector,
"_parse_parameters_from_xml",
side_effect=RuntimeError("boom"),
):
first = detector.parse_streaming_increment(_weather_call(), self.tools)
self.assertEqual(detector._buffer, "")
second = detector.parse_streaming_increment(" tail", self.tools)
self.assertIn("get_weather", first.normal_text)
self.assertNotIn("get_weather", second.normal_text)
# No half-formed call: the failure can land between a tool's name and its
# arguments, so an argument-less named call must not reach the client.
self.assertEqual(first.calls, [])
if __name__ == "__main__":
import unittest
unittest.main()
@@ -156,11 +156,14 @@ class TestBaseReasoningFormatDetector(CustomTestCase):
self.assertEqual(detector._buffer, "")
self.assertEqual(detector.finish().reasoning_text, "")
def test_finish_drops_partial_end_tag_when_streaming_reasoning(self):
"""With stream_reasoning=True the reasoning is emitted chunk by chunk, so
finish() must not re-emit. Only a partial end-tag fragment can linger in
_buffer; that fragment is an incomplete token, not content, and must be
dropped rather than surfaced as reasoning."""
def test_finish_flushes_partial_end_tag_when_streaming_reasoning(self):
"""With stream_reasoning=True everything in _buffer at the end of the
stream is a trailing slice that was held back precisely because it could
still have grown into `</think>`, so it was never emitted. Since the
stream ended it never became a token, and dropping it would lose content
whose only crime is looking like the start of one -- reasoning ending in
a literal `<` is the common case. Flushing matches stream_reasoning=False
and the non-streaming path, which both keep it."""
detector = BaseReasoningFormatDetector(
"<think>", "</think>", stream_reasoning=True
)
@@ -170,8 +173,11 @@ class TestBaseReasoningFormatDetector(CustomTestCase):
)
self.assertEqual(detector.parse_streaming_increment("</thi").reasoning_text, "")
end = detector.finish()
self.assertEqual(end.reasoning_text, "")
self.assertEqual(end.reasoning_text, "</thi")
self.assertEqual(end.normal_text, "")
# State is cleared, so a second finish() is a no-op.
self.assertEqual(detector._buffer, "")
self.assertEqual(detector.finish().reasoning_text, "")
class TestDeepSeekR1Detector(CustomTestCase):
@@ -236,6 +242,18 @@ class TestDeepSeekV4Detector(CustomTestCase):
self.assertEqual(detector.reasoning_default, "explicit_thinking")
self.assertTrue(detector.thinks_internally)
def test_dsml_block_is_routed_out_of_reasoning(self):
"""Without tool_start_token the DSML block stays in reasoning_content and
the tool call detector never sees it."""
detector = ReasoningParser(model_type="deepseek-v4").detector
self.assertEqual(detector.tool_start_token, "<|DSML|")
result = detector.parse_streaming_increment(
'<think>pick a tool<|DSML|tool_calls><|DSML|invoke name="s">'
)
self.assertEqual(result.reasoning_text, "pick a tool")
self.assertTrue(result.normal_text.startswith("<|DSML|tool_calls>"))
class TestInklingDetector(CustomTestCase):
def test_streaming_routes_blocks_across_all_string_boundaries(self):
@@ -416,13 +434,12 @@ class TestGlm45Detector(CustomTestCase):
self.assertEqual(result.reasoning_text, "")
self.assertEqual(result.normal_text, "")
# Tool interruption should still work - flushes buffered reasoning.
# Note: when stream_reasoning=False, the <think> tag is stripped from the
# local `current_text` variable but NOT from `self._buffer` (which is never
# cleared in the non-streaming path). So the flushed reasoning content
# includes the raw <think> tag.
# Tool interruption should still work - flushes buffered reasoning. The
# opening tag is stripped from `self._buffer` as well as from the local
# view, so the flush matches detect_and_parse instead of carrying the raw
# <think> tag into reasoning_content.
result = detector.parse_streaming_increment("<tool_call>tool call")
self.assertEqual(result.reasoning_text, "<think>thinking")
self.assertEqual(result.reasoning_text, "thinking")
self.assertEqual(result.normal_text, "<tool_call>tool call")
def test_streaming_empty_reasoning_with_tool(self):
@@ -961,6 +978,154 @@ class TestBufferLossBugFix(CustomTestCase):
self.assertTrue(detector.stripped_think_start)
class TestStreamingChunkSizeInvariance(CustomTestCase):
"""Accumulated (reasoning, normal) output must not depend on how the decode
steps happen to batch tokens, and must match one-shot detect_and_parse.
Speculative decoding and stream_interval > 1 deliver multiple tokens per
step, which splits multi-character tokens like `</think>` across chunk
boundaries. The two `_is_chunk_dependent` tests pin known exceptions.
"""
CHUNK_SIZES = [1, 2, 3, 5, 7, 11, 23, 1000]
DSML = "|DSML|"
def _feed(self, detector, text, chunk_size):
reasoning = normal = ""
for i in range(0, len(text), chunk_size):
result = detector.parse_streaming_increment(text[i : i + chunk_size])
reasoning += result.reasoning_text
normal += result.normal_text
result = detector.finish()
return reasoning + result.reasoning_text, normal + result.normal_text
def _assert_invariant(self, make_detector, text, expected):
for chunk_size in self.CHUNK_SIZES:
with self.subTest(chunk_size=chunk_size):
self.assertEqual(
self._feed(make_detector(), text, chunk_size), expected
)
one_shot = make_detector().detect_and_parse(text)
self.assertEqual((one_shot.reasoning_text, one_shot.normal_text), expected)
def test_think_end_split_across_chunks(self):
"""`</think>` straddling a chunk boundary must still end the block."""
self._assert_invariant(
DeepSeekR1Detector,
"<think>abc reasoning</think>normal text",
("abc reasoning", "normal text"),
)
def test_think_end_split_buffered_mode(self):
self._assert_invariant(
lambda: DeepSeekR1Detector(stream_reasoning=False),
"<think>abc reasoning</think>normal text",
("abc reasoning", "normal text"),
)
def test_literal_angle_bracket_in_reasoning_is_not_swallowed(self):
self._assert_invariant(
DeepSeekR1Detector,
"<think>a < b</think>tail",
("a < b", "tail"),
)
def test_reasoning_truncated_mid_partial_token(self):
"""Reasoning that happens to end in a `</think>` prefix must keep those
characters: the holdback exists to recombine them with the next chunk, so
a stream that ends first must flush rather than swallow them."""
for chunk_size in self.CHUNK_SIZES:
with self.subTest(chunk_size=chunk_size):
self.assertEqual(
self._feed(DeepSeekR1Detector(), "<think>compare a <", chunk_size),
("compare a <", ""),
)
def test_normal_text_ending_in_token_prefix_survives(self):
"""Content after the reasoning block that happens to end in a `</think>`
prefix is buffered by the prefix check; the stream ending must flush it."""
for text in ("<think>a</think>b<", "<think>a</think>b</thi"):
expected = (
text.split("</think>", 1)[0].removeprefix("<think>"),
text.split("</think>", 1)[1],
)
for chunk_size in self.CHUNK_SIZES:
with self.subTest(text=text, chunk_size=chunk_size):
self.assertEqual(
self._feed(DeepSeekR1Detector(), text, chunk_size), expected
)
def test_text_before_think_token_is_chunk_dependent(self):
"""Accepted divergence, inherited from main: text before `<think>` lands
in reasoning or content depending on where the chunk boundary falls."""
text = "lead<think>r</think>tail"
variants = {
self._feed(Qwen3Detector(), text, chunk_size)
for chunk_size in self.CHUNK_SIZES
}
self.assertEqual(
variants,
{("r", "leadtail"), ("", text), ("leadr", "tail")},
)
# And the non-streaming path produces yet a fourth split.
one_shot = Qwen3Detector().detect_and_parse(text)
self.assertEqual(
(one_shot.reasoning_text, one_shot.normal_text), ("lead<think>r", "tail")
)
def test_dsv4_reasoning_quoting_dsml_is_chunk_dependent(self):
"""Accepted divergence: streaming ends the block at the DSML marker, while
one-shot waits to see whether a `</think>` follows. Reachable because the
DSV4 system prompt shows that marker to the model."""
text = f"<think>format is <{self.DSML}tool_calls></think>answer"
by_output = {}
for chunk_size in self.CHUNK_SIZES:
by_output.setdefault(
self._feed(DeepSeekV4Detector(), text, chunk_size), []
).append(chunk_size)
self.assertEqual(len(by_output), 2, f"expected two variants, got {by_output}")
early_cut = ("format is ", f"<{self.DSML}tool_calls></think>answer")
whole_buffer = (f"format is <{self.DSML}tool_calls>", "answer")
self.assertIn(early_cut, by_output)
self.assertIn(whole_buffer, by_output)
one_shot = DeepSeekV4Detector().detect_and_parse(text)
self.assertEqual((one_shot.reasoning_text, one_shot.normal_text), whole_buffer)
def test_dsv4_tool_block_after_think_end(self):
tool_call = (
f"<{self.DSML}tool_calls>"
f'<{self.DSML}invoke name="s"></{self.DSML}invoke>'
f"</{self.DSML}tool_calls>"
)
self._assert_invariant(
DeepSeekV4Detector,
f"<think>my reasoning</think>{tool_call}",
("my reasoning", tool_call),
)
def test_dsv4_tool_block_without_think_end(self):
"""DSML directly after reasoning must still be routed to normal_text so
the tool call detector can see it."""
tool_call = (
f"<{self.DSML}tool_calls>"
f'<{self.DSML}invoke name="s"></{self.DSML}invoke>'
f"</{self.DSML}tool_calls>"
)
for chunk_size in self.CHUNK_SIZES:
with self.subTest(chunk_size=chunk_size):
self.assertEqual(
self._feed(
DeepSeekV4Detector(),
f"<think>my reasoning{tool_call}",
chunk_size,
),
("my reasoning", tool_call),
)
class TestGptOssDetector(CustomTestCase):
"""Test cases for GptOssDetector which delegates to HarmonyParser."""