@@ -14,11 +14,14 @@ from sglang.srt.layers.linear import LinearBase
|
|||||||
from sglang.srt.models.qwen2 import Qwen2MLP
|
from sglang.srt.models.qwen2 import Qwen2MLP
|
||||||
from sglang.srt.server_args import ServerArgs, set_global_server_args_for_scheduler
|
from sglang.srt.server_args import ServerArgs, set_global_server_args_for_scheduler
|
||||||
from sglang.srt.utils import add_prefix, get_device
|
from sglang.srt.utils import add_prefix, get_device
|
||||||
from sglang.test.ci.ci_register import register_cuda_ci
|
from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci
|
||||||
from sglang.test.layer_ut_utils import init_single_process_dist
|
from sglang.test.layer_ut_utils import init_single_process_dist
|
||||||
from sglang.test.test_utils import CustomTestCase
|
from sglang.test.test_utils import CustomTestCase
|
||||||
|
|
||||||
register_cuda_ci(est_time=9, stage="base-b", runner_config="1-gpu-small")
|
register_cuda_ci(est_time=9, stage="base-b", runner_config="1-gpu-small")
|
||||||
|
# Backend-specific: the hook resolves get_device() and the matching distributed
|
||||||
|
# backend, so only an AMD run exercises the HIP device and dump path.
|
||||||
|
register_amd_ci(est_time=15, suite="stage-b-test-1-gpu-small-amd")
|
||||||
|
|
||||||
TEST_HIDDEN_SIZE = 32
|
TEST_HIDDEN_SIZE = 32
|
||||||
|
|
||||||
|
|||||||
@@ -34,7 +34,7 @@ from sglang.srt.observability.trace import (
|
|||||||
)
|
)
|
||||||
from sglang.srt.utils import kill_process_tree
|
from sglang.srt.utils import kill_process_tree
|
||||||
from sglang.srt.utils.network import get_zmq_socket
|
from sglang.srt.utils.network import get_zmq_socket
|
||||||
from sglang.test.ci.ci_register import register_cuda_ci
|
from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci
|
||||||
from sglang.test.test_utils import (
|
from sglang.test.test_utils import (
|
||||||
DEFAULT_SMALL_MODEL_NAME_FOR_TEST,
|
DEFAULT_SMALL_MODEL_NAME_FOR_TEST,
|
||||||
DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH,
|
DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH,
|
||||||
@@ -47,6 +47,10 @@ logger = logging.getLogger(__name__)
|
|||||||
|
|
||||||
# CI registration
|
# CI registration
|
||||||
register_cuda_ci(est_time=113, stage="extra-a", runner_config="1-gpu-small")
|
register_cuda_ci(est_time=113, stage="extra-a", runner_config="1-gpu-small")
|
||||||
|
# Backend-specific: the span assertions require PREFILL_FORWARD/DECODE_FORWARD
|
||||||
|
# to be emitted from the scheduler forward path, which ROCm reaches through its
|
||||||
|
# own attention backend and graph replay.
|
||||||
|
register_amd_ci(est_time=113, suite="stage-b-test-1-gpu-small-amd")
|
||||||
|
|
||||||
|
|
||||||
# ============================================================================
|
# ============================================================================
|
||||||
|
|||||||
@@ -17,6 +17,7 @@ import requests
|
|||||||
|
|
||||||
from sglang.srt.utils import kill_process_tree
|
from sglang.srt.utils import kill_process_tree
|
||||||
from sglang.test.ci.ci_register import (
|
from sglang.test.ci.ci_register import (
|
||||||
|
register_amd_ci,
|
||||||
register_cpu_ci,
|
register_cpu_ci,
|
||||||
register_cuda_ci,
|
register_cuda_ci,
|
||||||
)
|
)
|
||||||
@@ -29,6 +30,9 @@ from sglang.test.test_utils import (
|
|||||||
)
|
)
|
||||||
|
|
||||||
register_cuda_ci(est_time=50, stage="base-b", runner_config="1-gpu-large")
|
register_cuda_ci(est_time=50, stage="base-b", runner_config="1-gpu-large")
|
||||||
|
# Backend-specific: streaming tool-argument assembly is rebuilt from
|
||||||
|
# incremental decode output, so ROCm chunk-boundary divergence surfaces here.
|
||||||
|
register_amd_ci(est_time=140, suite="stage-b-test-1-gpu-small-amd")
|
||||||
register_cpu_ci(est_time=54, suite="stage-b-test-cpu-intel")
|
register_cpu_ci(est_time=54, suite="stage-b-test-cpu-intel")
|
||||||
|
|
||||||
# System message to guide Llama3.2 to produce proper tool call format
|
# System message to guide Llama3.2 to produce proper tool call format
|
||||||
|
|||||||
@@ -5,7 +5,11 @@ import openai
|
|||||||
|
|
||||||
from sglang.srt.utils import is_npu, kill_process_tree
|
from sglang.srt.utils import is_npu, kill_process_tree
|
||||||
from sglang.srt.utils.hf_transformers_utils import get_tokenizer
|
from sglang.srt.utils.hf_transformers_utils import get_tokenizer
|
||||||
from sglang.test.ci.ci_register import register_cuda_ci, register_npu_ci
|
from sglang.test.ci.ci_register import (
|
||||||
|
register_amd_ci,
|
||||||
|
register_cuda_ci,
|
||||||
|
register_npu_ci,
|
||||||
|
)
|
||||||
from sglang.test.test_utils import (
|
from sglang.test.test_utils import (
|
||||||
DEFAULT_SMALL_MODEL_NAME_FOR_TEST,
|
DEFAULT_SMALL_MODEL_NAME_FOR_TEST,
|
||||||
DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH,
|
DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH,
|
||||||
@@ -16,6 +20,9 @@ from sglang.test.test_utils import (
|
|||||||
)
|
)
|
||||||
|
|
||||||
register_cuda_ci(est_time=100, stage="base-b", runner_config="1-gpu-large")
|
register_cuda_ci(est_time=100, stage="base-b", runner_config="1-gpu-large")
|
||||||
|
# Backend-specific: the llama3 and pythonic parsers run on real decoded text,
|
||||||
|
# so ROCm decode divergence surfaces as tool calls that no longer parse.
|
||||||
|
register_amd_ci(est_time=73, suite="stage-b-test-1-gpu-small-amd")
|
||||||
# Backend-specific: Ascend uses a local model mirror and its native
|
# Backend-specific: Ascend uses a local model mirror and its native
|
||||||
# attention backend, while sharing the protocol assertions below.
|
# attention backend, while sharing the protocol assertions below.
|
||||||
register_npu_ci(est_time=400, suite="full-1-npu-a3", nightly=True)
|
register_npu_ci(est_time=400, suite="full-1-npu-a3", nightly=True)
|
||||||
|
|||||||
@@ -4,7 +4,7 @@ import unittest
|
|||||||
import requests
|
import requests
|
||||||
|
|
||||||
from sglang.srt.utils import kill_process_tree
|
from sglang.srt.utils import kill_process_tree
|
||||||
from sglang.test.ci.ci_register import register_cuda_ci
|
from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci
|
||||||
from sglang.test.kits.reasoning_kit import (
|
from sglang.test.kits.reasoning_kit import (
|
||||||
ReasoningTokenUsageMixin,
|
ReasoningTokenUsageMixin,
|
||||||
SeparateReasoningMixin,
|
SeparateReasoningMixin,
|
||||||
@@ -18,6 +18,10 @@ from sglang.test.test_utils import (
|
|||||||
)
|
)
|
||||||
|
|
||||||
register_cuda_ci(est_time=129, stage="base-b", runner_config="1-gpu-large")
|
register_cuda_ci(est_time=129, stage="base-b", runner_config="1-gpu-large")
|
||||||
|
# Backend-specific: the thinking-token and separate-reasoning assertions read
|
||||||
|
# decoded output from a 30B MoE, so a ROCm fused-MoE or sampling regression
|
||||||
|
# shows up here as malformed reasoning_content that CUDA cannot catch.
|
||||||
|
register_amd_ci(est_time=200, suite="stage-b-test-1-gpu-small-amd")
|
||||||
|
|
||||||
|
|
||||||
class TestEnableThinking(
|
class TestEnableThinking(
|
||||||
|
|||||||
Reference in New Issue
Block a user