Consolidate reasoning tests into test/registered/reasoning/ (#22139)

This commit is contained in:
Liangsheng Yin
2026-04-05 01:09:11 -07:00
committed by GitHub
parent 2b119ba388
commit bd6a585605
5 changed files with 172 additions and 409 deletions
@@ -1,5 +1,6 @@
import json
import openai
import requests
from sglang.srt.parser.reasoning_parser import ReasoningParser
@@ -112,3 +113,85 @@ class ReasoningTokenUsageMixin:
reported = data["meta_info"]["reasoning_tokens"]
actual = data["output_ids"].index(self.think_end_token_id) + 1
self.assertEqual(reported, actual)
class SeparateReasoningMixin:
"""Mixin for separate_reasoning tests.
Required attributes on the test class:
model: str
base_url: str (without /v1)
api_key: str
"""
def _openai_client(self):
return openai.Client(api_key=self.api_key, base_url=f"{self.base_url}/v1")
def _chat(self, stream=False, extra_body=None):
return self._openai_client().chat.completions.create(
model=self.model,
messages=[{"role": "user", "content": "What is 1+3?"}],
max_tokens=1024,
stream=stream,
extra_body=extra_body,
)
def _collect_stream(self, response):
reasoning_content = ""
content = ""
for chunk in response:
if chunk.choices[0].delta.content:
content += chunk.choices[0].delta.content
elif chunk.choices[0].delta.reasoning_content:
reasoning_content += chunk.choices[0].delta.reasoning_content
return reasoning_content, content
def test_streaming_separate_reasoning_false(self):
response = self._chat(stream=True, extra_body={"separate_reasoning": False})
reasoning_content, content = self._collect_stream(response)
self.assertEqual(len(reasoning_content), 0)
self.assertGreater(len(content), 0)
def test_streaming_separate_reasoning_true(self):
response = self._chat(stream=True, extra_body={"separate_reasoning": True})
reasoning_content, content = self._collect_stream(response)
self.assertGreater(len(reasoning_content), 0)
self.assertGreater(len(content), 0)
def test_streaming_separate_reasoning_true_stream_reasoning_false(self):
response = self._chat(
stream=True,
extra_body={"separate_reasoning": True, "stream_reasoning": False},
)
reasoning_content = ""
content = ""
first_chunk = False
for chunk in response:
if chunk.choices[0].delta.reasoning_content:
reasoning_content = chunk.choices[0].delta.reasoning_content
first_chunk = True
if chunk.choices[0].delta.content:
content += chunk.choices[0].delta.content
if not first_chunk:
reasoning_content = chunk.choices[0].delta.reasoning_content
first_chunk = True
if not first_chunk:
assert (
not chunk.choices[0].delta.reasoning_content
or len(chunk.choices[0].delta.reasoning_content) == 0
)
self.assertGreater(len(reasoning_content), 0)
self.assertGreater(len(content), 0)
def test_nonstreaming_separate_reasoning_false(self):
response = self._chat(extra_body={"separate_reasoning": False})
assert (
not response.choices[0].message.reasoning_content
or len(response.choices[0].message.reasoning_content) == 0
)
self.assertGreater(len(response.choices[0].message.content), 0)
def test_nonstreaming_separate_reasoning_true(self):
response = self._chat(extra_body={"separate_reasoning": True})
self.assertGreater(len(response.choices[0].message.reasoning_content), 0)
self.assertGreater(len(response.choices[0].message.content), 0)
@@ -7,7 +7,7 @@ from sglang.srt.environ import envs
from sglang.srt.utils import kill_process_tree
from sglang.test.accuracy_test_runner import AccuracyTestParams
from sglang.test.ci.ci_register import register_cuda_ci
from sglang.test.kits.reasoning_tokens_kit import ReasoningTokenUsageMixin
from sglang.test.kits.reasoning_kit import ReasoningTokenUsageMixin
# This eval harness applies the chat_template, which is critical for qwen3.5
# to get good accuracy on gsm8k
@@ -1,345 +0,0 @@
"""
Usage:
python3 -m unittest openai_server.features.test_reasoning_content.TestReasoningContentAPI.test_streaming_separate_reasoning_false
python3 -m unittest openai_server.features.test_reasoning_content.TestReasoningContentAPI.test_streaming_separate_reasoning_true
python3 -m unittest openai_server.features.test_reasoning_content.TestReasoningContentAPI.test_streaming_separate_reasoning_true_stream_reasoning_false
python3 -m unittest openai_server.features.test_reasoning_content.TestReasoningContentAPI.test_nonstreaming_separate_reasoning_false
python3 -m unittest openai_server.features.test_reasoning_content.TestReasoningContentAPI.test_nonstreaming_separate_reasoning_true
python3 -m unittest openai_server.features.test_reasoning_content.TestReasoningContentStartup.test_nonstreaming
python3 -m unittest openai_server.features.test_reasoning_content.TestReasoningContentStartup.test_streaming
"""
import unittest
import openai
from sglang.srt.utils import kill_process_tree
from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci
from sglang.test.test_utils import (
DEFAULT_REASONING_MODEL_NAME_FOR_TEST,
DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH,
DEFAULT_URL_FOR_TEST,
CustomTestCase,
popen_launch_server,
)
register_cuda_ci(est_time=89, suite="stage-b-test-1-gpu-small")
register_amd_ci(est_time=89, suite="stage-b-test-1-gpu-small-amd")
class TestReasoningContentAPI(CustomTestCase):
@classmethod
def setUpClass(cls):
cls.model = DEFAULT_REASONING_MODEL_NAME_FOR_TEST
cls.base_url = DEFAULT_URL_FOR_TEST
cls.api_key = "sk-1234"
cls.process = popen_launch_server(
cls.model,
cls.base_url,
timeout=DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH,
api_key=cls.api_key,
other_args=[
"--reasoning-parser",
"deepseek-r1",
],
)
cls.base_url += "/v1"
@classmethod
def tearDownClass(cls):
kill_process_tree(cls.process.pid)
def test_streaming_separate_reasoning_false(self):
# Test streaming with separate_reasoning=False, reasoning_content should be empty
client = openai.Client(api_key=self.api_key, base_url=self.base_url)
payload = {
"model": self.model,
"messages": [
{
"role": "user",
"content": "What is 1+3?",
}
],
"max_tokens": 100,
"stream": True,
"extra_body": {"separate_reasoning": False},
}
response = client.chat.completions.create(**payload)
reasoning_content = ""
content = ""
for chunk in response:
if chunk.choices[0].delta.content:
content += chunk.choices[0].delta.content
elif chunk.choices[0].delta.reasoning_content:
reasoning_content += chunk.choices[0].delta.reasoning_content
assert len(reasoning_content) == 0
assert len(content) > 0
def test_streaming_separate_reasoning_true(self):
# Test streaming with separate_reasoning=True, reasoning_content should not be empty
client = openai.Client(api_key=self.api_key, base_url=self.base_url)
payload = {
"model": self.model,
"messages": [
{
"role": "user",
"content": "What is 1+3?",
}
],
"max_tokens": 100,
"stream": True,
"extra_body": {"separate_reasoning": True},
}
response = client.chat.completions.create(**payload)
reasoning_content = ""
content = ""
for chunk in response:
if chunk.choices[0].delta.content:
content += chunk.choices[0].delta.content
elif chunk.choices[0].delta.reasoning_content:
reasoning_content += chunk.choices[0].delta.reasoning_content
assert len(reasoning_content) > 0
assert len(content) > 0
def test_streaming_separate_reasoning_true_stream_reasoning_false(self):
# Test streaming with separate_reasoning=True, reasoning_content should not be empty
client = openai.Client(api_key=self.api_key, base_url=self.base_url)
payload = {
"model": self.model,
"messages": [
{
"role": "user",
"content": "What is 1+3?",
}
],
"max_tokens": 100,
"stream": True,
"extra_body": {"separate_reasoning": True, "stream_reasoning": False},
}
response = client.chat.completions.create(**payload)
reasoning_content = ""
content = ""
first_chunk = False
for chunk in response:
if chunk.choices[0].delta.reasoning_content:
reasoning_content = chunk.choices[0].delta.reasoning_content
first_chunk = True
if chunk.choices[0].delta.content:
content += chunk.choices[0].delta.content
if not first_chunk:
reasoning_content = chunk.choices[0].delta.reasoning_content
first_chunk = True
if not first_chunk:
assert (
not chunk.choices[0].delta.reasoning_content
or len(chunk.choices[0].delta.reasoning_content) == 0
)
assert len(reasoning_content) > 0
assert len(content) > 0
def test_nonstreaming_separate_reasoning_false(self):
# Test non-streaming with separate_reasoning=False, reasoning_content should be empty
client = openai.Client(api_key=self.api_key, base_url=self.base_url)
payload = {
"model": self.model,
"messages": [
{
"role": "user",
"content": "What is 1+3?",
}
],
"max_tokens": 100,
"extra_body": {"separate_reasoning": False},
}
response = client.chat.completions.create(**payload)
assert (
not response.choices[0].message.reasoning_content
or len(response.choices[0].message.reasoning_content) == 0
)
assert len(response.choices[0].message.content) > 0
def test_nonstreaming_separate_reasoning_true(self):
# Test non-streaming with separate_reasoning=True, reasoning_content should not be empty
client = openai.Client(api_key=self.api_key, base_url=self.base_url)
payload = {
"model": self.model,
"messages": [
{
"role": "user",
"content": "What is 1+3?",
}
],
"max_tokens": 100,
"extra_body": {"separate_reasoning": True},
}
response = client.chat.completions.create(**payload)
assert len(response.choices[0].message.reasoning_content) > 0
assert len(response.choices[0].message.content) > 0
class TestReasoningContentWithoutParser(CustomTestCase):
@classmethod
def setUpClass(cls):
cls.model = DEFAULT_REASONING_MODEL_NAME_FOR_TEST
cls.base_url = DEFAULT_URL_FOR_TEST
cls.api_key = "sk-1234"
cls.process = popen_launch_server(
cls.model,
cls.base_url,
timeout=DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH,
api_key=cls.api_key,
other_args=[], # No reasoning parser
)
cls.base_url += "/v1"
@classmethod
def tearDownClass(cls):
kill_process_tree(cls.process.pid)
def test_streaming_separate_reasoning_false(self):
# Test streaming with separate_reasoning=False, reasoning_content should be empty
client = openai.Client(api_key=self.api_key, base_url=self.base_url)
payload = {
"model": self.model,
"messages": [
{
"role": "user",
"content": "What is 1+3?",
}
],
"max_tokens": 100,
"stream": True,
"extra_body": {"separate_reasoning": False},
}
response = client.chat.completions.create(**payload)
reasoning_content = ""
content = ""
for chunk in response:
if chunk.choices[0].delta.content:
content += chunk.choices[0].delta.content
elif chunk.choices[0].delta.reasoning_content:
reasoning_content += chunk.choices[0].delta.reasoning_content
assert len(reasoning_content) == 0
assert len(content) > 0
def test_streaming_separate_reasoning_true(self):
# Test streaming with separate_reasoning=True, reasoning_content should not be empty
client = openai.Client(api_key=self.api_key, base_url=self.base_url)
payload = {
"model": self.model,
"messages": [
{
"role": "user",
"content": "What is 1+3?",
}
],
"max_tokens": 100,
"stream": True,
"extra_body": {"separate_reasoning": True},
}
response = client.chat.completions.create(**payload)
reasoning_content = ""
content = ""
for chunk in response:
if chunk.choices[0].delta.content:
content += chunk.choices[0].delta.content
elif chunk.choices[0].delta.reasoning_content:
reasoning_content += chunk.choices[0].delta.reasoning_content
assert len(reasoning_content) == 0
assert len(content) > 0
def test_streaming_separate_reasoning_true_stream_reasoning_false(self):
# Test streaming with separate_reasoning=True, reasoning_content should not be empty
client = openai.Client(api_key=self.api_key, base_url=self.base_url)
payload = {
"model": self.model,
"messages": [
{
"role": "user",
"content": "What is 1+3?",
}
],
"max_tokens": 100,
"stream": True,
"extra_body": {"separate_reasoning": True, "stream_reasoning": False},
}
response = client.chat.completions.create(**payload)
reasoning_content = ""
content = ""
first_chunk = False
for chunk in response:
if chunk.choices[0].delta.reasoning_content:
reasoning_content = chunk.choices[0].delta.reasoning_content
first_chunk = True
if chunk.choices[0].delta.content:
content += chunk.choices[0].delta.content
if not first_chunk:
reasoning_content = chunk.choices[0].delta.reasoning_content
first_chunk = True
if not first_chunk:
assert (
not chunk.choices[0].delta.reasoning_content
or len(chunk.choices[0].delta.reasoning_content) == 0
)
assert not reasoning_content or len(reasoning_content) == 0
assert len(content) > 0
def test_nonstreaming_separate_reasoning_false(self):
# Test non-streaming with separate_reasoning=False, reasoning_content should be empty
client = openai.Client(api_key=self.api_key, base_url=self.base_url)
payload = {
"model": self.model,
"messages": [
{
"role": "user",
"content": "What is 1+3?",
}
],
"max_tokens": 100,
"extra_body": {"separate_reasoning": False},
}
response = client.chat.completions.create(**payload)
assert (
not response.choices[0].message.reasoning_content
or len(response.choices[0].message.reasoning_content) == 0
)
assert len(response.choices[0].message.content) > 0
def test_nonstreaming_separate_reasoning_true(self):
# Test non-streaming with separate_reasoning=True, reasoning_content should not be empty
client = openai.Client(api_key=self.api_key, base_url=self.base_url)
payload = {
"model": self.model,
"messages": [
{
"role": "user",
"content": "What is 1+3?",
}
],
"max_tokens": 100,
"extra_body": {"separate_reasoning": True},
}
response = client.chat.completions.create(**payload)
assert (
not response.choices[0].message.reasoning_content
or len(response.choices[0].message.reasoning_content) == 0
)
assert len(response.choices[0].message.content) > 0
if __name__ == "__main__":
unittest.main()
@@ -1,11 +1,3 @@
"""
Usage:
python3 -m unittest openai_server.features.test_enable_thinking.TestEnableThinking.test_chat_completion_with_reasoning
python3 -m unittest openai_server.features.test_enable_thinking.TestEnableThinking.test_chat_completion_without_reasoning
python3 -m unittest openai_server.features.test_enable_thinking.TestEnableThinking.test_stream_chat_completion_with_reasoning
python3 -m unittest openai_server.features.test_enable_thinking.TestEnableThinking.test_stream_chat_completion_without_reasoning
"""
import json
import unittest
@@ -13,7 +5,10 @@ import requests
from sglang.srt.utils import kill_process_tree
from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci
from sglang.test.kits.reasoning_tokens_kit import ReasoningTokenUsageMixin
from sglang.test.kits.reasoning_kit import (
ReasoningTokenUsageMixin,
SeparateReasoningMixin,
)
from sglang.test.test_utils import (
DEFAULT_ENABLE_THINKING_MODEL_NAME_FOR_TEST,
DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH,
@@ -26,7 +21,9 @@ register_cuda_ci(est_time=109, suite="stage-b-test-1-gpu-large")
register_amd_ci(est_time=200, suite="stage-b-test-1-gpu-small-amd")
class TestEnableThinking(ReasoningTokenUsageMixin, CustomTestCase):
class TestEnableThinking(
ReasoningTokenUsageMixin, SeparateReasoningMixin, CustomTestCase
):
reasoning_parser_name = "qwen3"
@classmethod
@@ -122,7 +119,6 @@ class TestEnableThinking(ReasoningTokenUsageMixin, CustomTestCase):
has_reasoning = False
has_content = False
print("\n=== Stream With Reasoning ===")
for line in response.iter_lines():
if line:
line = line.decode("utf-8")
@@ -146,7 +142,7 @@ class TestEnableThinking(ReasoningTokenUsageMixin, CustomTestCase):
)
def test_stream_chat_completion_without_reasoning(self):
# Test streaming with "enable_thinking": False, reasoning_content should be empty
# Test streaming with "enable_thinking": False, reasoning_content should be empty
response = requests.post(
f"{self.base_url}/v1/chat/completions",
headers={"Authorization": f"Bearer {self.api_key}"},
@@ -167,7 +163,6 @@ class TestEnableThinking(ReasoningTokenUsageMixin, CustomTestCase):
has_reasoning = False
has_content = False
print("\n=== Stream Without Reasoning ===")
for line in response.iter_lines():
if line:
line = line.decode("utf-8")
@@ -191,55 +186,5 @@ class TestEnableThinking(ReasoningTokenUsageMixin, CustomTestCase):
)
# Skip for ci test
# class TestGLM45EnableThinking(TestEnableThinking):
# @classmethod
# def setUpClass(cls):
# # Replace with the model name needed for testing; if not required, reuse DEFAULT_SMALL_MODEL_NAME_FOR_TEST
# cls.model = "THUDM/GLM-4.5"
# cls.base_url = DEFAULT_URL_FOR_TEST
# cls.api_key = "sk-1234"
# cls.process = popen_launch_server(
# cls.model,
# cls.base_url,
# timeout=DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH,
# api_key=cls.api_key,
# other_args=[
# "--tool-call-parser",
# "glm45",
# "--reasoning-parser",
# "glm45",
# "--tp-size",
# "8"
# ],
# )
# # Validate whether enable-thinking conflict with tool_calls
# cls.additional_chat_kwargs = {
# "tools": [
# {
# "type": "function",
# "function": {
# "name": "add",
# "description": "Compute the sum of two numbers",
# "parameters": {
# "type": "object",
# "properties": {
# "a": {
# "type": "int",
# "description": "A number",
# },
# "b": {
# "type": "int",
# "description": "A number",
# },
# },
# "required": ["a", "b"],
# },
# },
# }
# ]
# }
if __name__ == "__main__":
unittest.main()
@@ -0,0 +1,80 @@
import unittest
from sglang.srt.parser.reasoning_parser import ReasoningParser
from sglang.test.ci.ci_register import register_cpu_ci
from sglang.test.test_utils import CustomTestCase
register_cpu_ci(est_time=5, suite="stage-a-test-cpu")
# Simulated model output that contains think tags (e.g. from DeepSeek-R1)
THINK_OUTPUT = (
"<think>\nLet me think about this.\n1 + 3 = 4\n</think>\nThe answer is 4."
)
THINK_OUTPUT_QWEN3 = (
"<think>\nLet me think about this.\n1 + 3 = 4\n</think>\n\nThe answer is 4."
)
class TestReasoningContentWithoutParser(CustomTestCase):
"""Test the code path: when no reasoning parser is configured, reasoning
content should never be separated, even if the model output contains
think tags. This mirrors the guard in serving_chat.py:
if self.reasoning_parser and request.separate_reasoning:
...
When reasoning_parser is None the block is skipped entirely.
"""
def test_no_parser_text_passthrough(self):
"""Without a parser, raw text with <think> tags passes through as-is."""
reasoning_parser = None
# Simulate serving_chat.py logic
reasoning_text = None
text = THINK_OUTPUT
if reasoning_parser:
parser = ReasoningParser(reasoning_parser)
reasoning_text, text = parser.parse_non_stream(text)
self.assertIsNone(reasoning_text)
self.assertIn("<think>", text)
self.assertIn("The answer is 4.", text)
def test_with_parser_separates_reasoning(self):
"""With a parser, reasoning content is correctly separated."""
for parser_name, output in [
("deepseek-r1", THINK_OUTPUT),
("qwen3", THINK_OUTPUT_QWEN3),
]:
with self.subTest(parser=parser_name):
parser = ReasoningParser(parser_name, stream_reasoning=False)
reasoning_text, text = parser.parse_non_stream(output)
self.assertIsNotNone(reasoning_text)
self.assertGreater(len(reasoning_text), 0)
self.assertNotIn("<think>", reasoning_text)
self.assertIn("The answer is 4.", text)
def test_no_parser_streaming_passthrough(self):
"""Without a parser, streaming chunks pass through without reasoning separation."""
reasoning_parser = None
# Simulate serving_chat.py streaming logic
chunks = ["<think>\nLet me", " think.\n</think>\nThe answer", " is 4."]
all_text = ""
reasoning_text_seen = False
for chunk in chunks:
delta = chunk
if reasoning_parser:
# This block would separate reasoning in streaming
reasoning_text_seen = True
all_text += delta
self.assertFalse(reasoning_text_seen)
self.assertIn("<think>", all_text)
if __name__ == "__main__":
unittest.main()