[CI] Derive registered-test kind from the registry call instead of the path (#40294)
This commit is contained in:
@@ -111,12 +111,6 @@ repos:
|
||||
entry: python3 scripts/lint/check_no_bare_pytest_main.py
|
||||
language: system
|
||||
files: ^(python|test)/.*\.py$
|
||||
- id: check-lint-script-tests
|
||||
name: unit tests for lint checkers
|
||||
entry: python3 -m unittest discover -s scripts/lint -p 'test_check_*.py'
|
||||
language: system
|
||||
files: ^scripts/lint/(check_|test_check_).*\.py$
|
||||
pass_filenames: false
|
||||
- id: check-registered-tests
|
||||
name: validate registered test CI registries
|
||||
entry: python3 scripts/lint/check_registered_tests.py
|
||||
|
||||
@@ -2,23 +2,8 @@
|
||||
"""
|
||||
Pre-commit hook: validate CI registry calls under test/registered/.
|
||||
|
||||
1. Every test file must contain a CI registry call (register_cuda_ci,
|
||||
register_amd_ci, etc.).
|
||||
2. A CUDA test must register its suite via the modern
|
||||
`stage=`/`runner_config=` form. The legacy single-string `suite=` is reserved
|
||||
for the stress family (and for AMD/CPU/NPU suites); any other CUDA `suite=`
|
||||
resolves to a name no workflow invokes, so the test silently never runs.
|
||||
Two shapes are rejected:
|
||||
a. `{stage}-test-{runner_config}` -- the modern name stuffed back into the
|
||||
legacy form. Reported with the exact stage/runner split to use.
|
||||
b. an older `{stage}-{runner_config}` PR-test name (e.g. the pre-migration
|
||||
`base-b-kernel-unit-1-gpu-large`) -- no longer matches any workflow
|
||||
suite at all.
|
||||
The modern form resolves to the identical suite (CIRegistry.effective_suite
|
||||
is f"{stage}-test-{runner_config}") and is /rerun-test-able.
|
||||
|
||||
Reuses ut_parse_one_file() from ci_register.py (AST-based parsing)
|
||||
to match the same logic used by run_suite.py's collect_tests().
|
||||
Reuses ut_parse_one_file() from ci_register.py (AST-based parsing) to match
|
||||
run_suite.py's collect_tests().
|
||||
"""
|
||||
|
||||
import ast
|
||||
@@ -26,25 +11,17 @@ import glob
|
||||
import importlib.util
|
||||
import os
|
||||
import re
|
||||
import subprocess
|
||||
import sys
|
||||
|
||||
# Suite names of the form `{stage}-test-{runner_config}` are exactly what the
|
||||
# modern stage=/runner_config= form produces, so a legacy suite= carrying this
|
||||
# shape is always expressible (and should be expressed) the modern way.
|
||||
# Exactly what stage=/runner_config= produces, so a legacy suite= of this shape
|
||||
# is always expressible the modern way.
|
||||
_MODERN_SHAPE = re.compile(r"^(.+)-test-(.+)$")
|
||||
|
||||
# The only CUDA suite family still allowed on the legacy single-string `suite=`
|
||||
# form. Anything else needs stage=/runner_config=, or its effective_suite matches
|
||||
# no suite any workflow invokes and the test silently never runs.
|
||||
# The only CUDA family still allowed on legacy `suite=`; anything else resolves
|
||||
# to a suite no workflow invokes and the test silently never runs.
|
||||
_LEGACY_CUDA_PREFIXES = ("stress",)
|
||||
|
||||
_TEST_KINDS = {"unit", "e2e", "accuracy", "perf", "stress"}
|
||||
_KERNEL_ROOT = "kernels"
|
||||
|
||||
# Flat vendor trees. Vendor-only coverage fits no kind above: no XPU/NPU suite
|
||||
# carries the `-kernel-` infix the kernel tree needs, and these launch device work.
|
||||
_VENDOR_DIRS = {"amd", "mlx", "musa", "npu", "xpu"}
|
||||
_KERNEL_LAYOUT = "test/registered/kernels/{ops,benchmark}/<group>/"
|
||||
|
||||
|
||||
def _defines_testcase(tree: ast.AST) -> bool:
|
||||
@@ -78,43 +55,6 @@ def _main_runs_tests(tree: ast.Module) -> bool:
|
||||
return False
|
||||
|
||||
|
||||
def _git_lines(*args: str) -> list[str] | None:
|
||||
result = subprocess.run(["git", *args], capture_output=True, text=True, check=False)
|
||||
if result.returncode != 0:
|
||||
return None
|
||||
return [line for line in result.stdout.splitlines() if line]
|
||||
|
||||
|
||||
def _changed_registered_files() -> set[str]:
|
||||
"""Return added, copied, or renamed registered-test destinations."""
|
||||
|
||||
lines = _git_lines("diff", "--cached", "--name-status", "--diff-filter=ACR")
|
||||
if not lines:
|
||||
base_ref = os.environ.get("GITHUB_BASE_REF", "main")
|
||||
for candidate in (f"origin/{base_ref}", base_ref):
|
||||
if _git_lines("rev-parse", "--verify", candidate) is None:
|
||||
continue
|
||||
merge_base = _git_lines("merge-base", candidate, "HEAD")
|
||||
if not merge_base:
|
||||
continue
|
||||
lines = _git_lines(
|
||||
"diff",
|
||||
"--name-status",
|
||||
"--diff-filter=ACR",
|
||||
merge_base[0],
|
||||
"HEAD",
|
||||
)
|
||||
break
|
||||
|
||||
selected = set()
|
||||
for line in lines or []:
|
||||
fields = line.split("\t")
|
||||
destination = fields[-1]
|
||||
if destination.startswith("test/registered/") and destination.endswith(".py"):
|
||||
selected.add(destination)
|
||||
return selected
|
||||
|
||||
|
||||
def _contains_call(tree: ast.AST, name: str) -> bool:
|
||||
return any(
|
||||
isinstance(node, ast.Call)
|
||||
@@ -126,63 +66,28 @@ def _contains_call(tree: ast.AST, name: str) -> bool:
|
||||
)
|
||||
|
||||
|
||||
def taxonomy_errors(path: str, registries: list, tree: ast.AST) -> list[str]:
|
||||
"""Validate the kind/subsystem contract for a newly admitted path."""
|
||||
|
||||
def taxonomy_errors(path: str, tree: ast.AST) -> list[str]:
|
||||
parts = path.split("/")
|
||||
relative_parts = parts[2:] if parts[:2] == ["test", "registered"] else []
|
||||
if relative_parts and relative_parts[0] in _VENDOR_DIRS:
|
||||
if parts[:2] != ["test", "registered"] or len(parts) < 3:
|
||||
return []
|
||||
if relative_parts and relative_parts[0] == _KERNEL_ROOT:
|
||||
errors = []
|
||||
if len(relative_parts) < 4 or relative_parts[1] not in {"ops", "benchmark"}:
|
||||
errors.append(
|
||||
f"{path}: kernel tests must live under "
|
||||
"test/registered/kernels/{ops,benchmark}/<group>/"
|
||||
)
|
||||
if any("-kernel-" not in (r.effective_suite or "") for r in registries):
|
||||
errors.append(f"{path}: kernel tests must use a *-kernel-* suite")
|
||||
return errors
|
||||
if len(relative_parts) < 3 or relative_parts[0] not in _TEST_KINDS:
|
||||
return [
|
||||
f"{path}: registered tests must live under "
|
||||
"test/registered/<kind>/<subsystem>/; kind must be one of "
|
||||
+ ", ".join(sorted(_TEST_KINDS))
|
||||
+ "; kernel tests use test/registered/kernels/{ops,benchmark}/<group>/"
|
||||
]
|
||||
relative_parts = parts[2:]
|
||||
root = relative_parts[0]
|
||||
|
||||
kind = relative_parts[0]
|
||||
errors = []
|
||||
if kind == "unit":
|
||||
invalid = [
|
||||
r
|
||||
for r in registries
|
||||
if r.backend.name != "CPU" and "-unit-" not in (r.effective_suite or "")
|
||||
]
|
||||
if invalid:
|
||||
errors.append(f"{path}: unit tests must use CPU or dedicated unit suites")
|
||||
if any(r.est_time > 60 for r in registries):
|
||||
errors.append(f"{path}: unit test est_time must be <= 60 seconds")
|
||||
if _contains_call(tree, "popen_launch_server"):
|
||||
errors.append(f"{path}: unit tests may not launch a server")
|
||||
elif kind in {"accuracy", "perf"}:
|
||||
invalid = [
|
||||
r
|
||||
for r in registries
|
||||
if not (r.effective_suite or "").startswith(("nightly-", "weekly-"))
|
||||
]
|
||||
if invalid:
|
||||
errors.append(f"{path}: {kind} tests must use nightly/weekly suites")
|
||||
elif kind == "stress":
|
||||
invalid = [
|
||||
r
|
||||
for r in registries
|
||||
if (r.effective_suite or "") != "stress"
|
||||
and not (r.effective_suite or "").startswith("weekly-")
|
||||
]
|
||||
if invalid:
|
||||
errors.append(f"{path}: stress tests must use stress/weekly suites")
|
||||
return errors
|
||||
if root in ("kernel", "kernels"):
|
||||
canonical = (
|
||||
root == "kernels"
|
||||
and len(relative_parts) >= 4
|
||||
and relative_parts[1] in ("ops", "benchmark")
|
||||
)
|
||||
return (
|
||||
[] if canonical else [f"{path}: kernel tests live under {_KERNEL_LAYOUT}"]
|
||||
)
|
||||
|
||||
if root != "unit":
|
||||
return []
|
||||
if _contains_call(tree, "popen_launch_server"):
|
||||
return [f"{path}: unit tests may not launch a server"]
|
||||
return []
|
||||
|
||||
|
||||
def main() -> int:
|
||||
@@ -209,22 +114,19 @@ def main() -> int:
|
||||
non_dispatchable = [] # (file, suite) -- legacy CUDA suite no workflow invokes
|
||||
dead_tests = [] # (file) -- TestCase classes that `python3 file.py` never runs
|
||||
taxonomy_violations = []
|
||||
changed_files = _changed_registered_files()
|
||||
for f in files:
|
||||
try:
|
||||
registries, _has_main_entry = ci_register.ut_parse_one_file(f)
|
||||
except Exception:
|
||||
# Skip files that can't be parsed (syntax errors, etc.)
|
||||
continue
|
||||
if len(registries) == 0:
|
||||
missing.append(f)
|
||||
continue
|
||||
# TestCase classes are dead unless __main__ runs them (CI does
|
||||
# `python3 file.py`); the ERROR text below explains the fix.
|
||||
# TestCase classes are dead unless __main__ runs them; CI runs the
|
||||
# registered file as `python3 file.py`.
|
||||
with open(f, "r", encoding="utf-8") as fh:
|
||||
tree = ast.parse(fh.read(), filename=f)
|
||||
if f in changed_files:
|
||||
taxonomy_violations.extend(taxonomy_errors(f, registries, tree))
|
||||
taxonomy_violations.extend(taxonomy_errors(f, tree))
|
||||
if _defines_testcase(tree) and not _main_runs_tests(tree):
|
||||
dead_tests.append(f)
|
||||
for r in registries:
|
||||
|
||||
@@ -1,86 +0,0 @@
|
||||
import pathlib
|
||||
import tempfile
|
||||
import unittest
|
||||
|
||||
from check_no_bare_pytest_main import find_bare_pytest_main
|
||||
|
||||
|
||||
class TestFindBarePytestMain(unittest.TestCase):
|
||||
def check_source(self, source: str) -> int | None:
|
||||
with tempfile.TemporaryDirectory() as directory:
|
||||
path = pathlib.Path(directory) / "example.py"
|
||||
path.write_text(source, encoding="utf-8")
|
||||
return find_bare_pytest_main(path)
|
||||
|
||||
def test_rejects_discarded_result(self):
|
||||
source = """
|
||||
if __name__ == "__main__":
|
||||
pytest.main([__file__])
|
||||
"""
|
||||
self.assertEqual(self.check_source(source), 3)
|
||||
|
||||
def test_rejects_discarded_result_with_whitespace(self):
|
||||
source = """
|
||||
if __name__ == "__main__":
|
||||
pytest . main([__file__])
|
||||
"""
|
||||
self.assertEqual(self.check_source(source), 3)
|
||||
|
||||
def test_accepts_propagated_result(self):
|
||||
source = """
|
||||
if __name__ == "__main__":
|
||||
sys.exit(pytest.main([__file__]))
|
||||
"""
|
||||
self.assertIsNone(self.check_source(source))
|
||||
|
||||
def test_rejects_assigned_result(self):
|
||||
source = """
|
||||
if "__main__" == __name__:
|
||||
exit_code = pytest.main([__file__])
|
||||
"""
|
||||
self.assertEqual(self.check_source(source), 3)
|
||||
|
||||
def test_accepts_assigned_result_that_is_later_exited(self):
|
||||
source = """
|
||||
if __name__ == "__main__":
|
||||
exit_code = pytest.main([__file__])
|
||||
sys.exit(exit_code)
|
||||
"""
|
||||
self.assertIsNone(self.check_source(source))
|
||||
|
||||
def test_accepts_assigned_result_that_is_raised(self):
|
||||
source = """
|
||||
if __name__ == "__main__":
|
||||
exit_code = pytest.main([__file__])
|
||||
raise SystemExit(exit_code)
|
||||
"""
|
||||
self.assertIsNone(self.check_source(source))
|
||||
|
||||
def test_rejects_nested_discarded_result(self):
|
||||
source = """
|
||||
if __name__ == "__main__":
|
||||
if enabled:
|
||||
pytest.main([__file__])
|
||||
"""
|
||||
self.assertEqual(self.check_source(source), 4)
|
||||
|
||||
def test_accepts_raised_system_exit(self):
|
||||
source = """
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(pytest.main([__file__]))
|
||||
"""
|
||||
self.assertIsNone(self.check_source(source))
|
||||
|
||||
def test_rejects_unraised_system_exit(self):
|
||||
source = """
|
||||
if __name__ == "__main__":
|
||||
error = SystemExit(pytest.main([__file__]))
|
||||
"""
|
||||
self.assertEqual(self.check_source(source), 3)
|
||||
|
||||
def test_ignores_call_outside_main_guard(self):
|
||||
self.assertIsNone(self.check_source("pytest.main([__file__])\n"))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -1,51 +0,0 @@
|
||||
import ast
|
||||
import unittest
|
||||
from types import SimpleNamespace
|
||||
|
||||
from scripts.lint.check_registered_tests import taxonomy_errors
|
||||
|
||||
|
||||
def _registry(suite: str):
|
||||
return SimpleNamespace(effective_suite=suite, est_time=1)
|
||||
|
||||
|
||||
class TestRegisteredTestTaxonomy(unittest.TestCase):
|
||||
def setUp(self):
|
||||
self.tree = ast.parse("")
|
||||
self.kernel_registry = [_registry("base-b-kernel-unit-test-1-gpu-large")]
|
||||
|
||||
def test_plural_kernel_ops_layout_is_accepted(self):
|
||||
errors = taxonomy_errors(
|
||||
"test/registered/kernels/ops/attention/test_example.py",
|
||||
self.kernel_registry,
|
||||
self.tree,
|
||||
)
|
||||
self.assertEqual(errors, [])
|
||||
|
||||
def test_plural_kernel_benchmark_layout_is_accepted(self):
|
||||
errors = taxonomy_errors(
|
||||
"test/registered/kernels/benchmark/attention/bench_example.py",
|
||||
[_registry("base-b-kernel-benchmark-test-1-gpu-large")],
|
||||
self.tree,
|
||||
)
|
||||
self.assertEqual(errors, [])
|
||||
|
||||
def test_singular_kernel_root_is_rejected(self):
|
||||
errors = taxonomy_errors(
|
||||
"test/registered/kernel/attention/test_example.py",
|
||||
self.kernel_registry,
|
||||
self.tree,
|
||||
)
|
||||
self.assertTrue(errors)
|
||||
|
||||
def test_kernel_group_is_required(self):
|
||||
errors = taxonomy_errors(
|
||||
"test/registered/kernels/ops/test_example.py",
|
||||
self.kernel_registry,
|
||||
self.tree,
|
||||
)
|
||||
self.assertTrue(errors)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
+7
-11
@@ -72,18 +72,14 @@ Parameters: `est_time` (seconds), `stage` + `runner_config` (target stage and ru
|
||||
|
||||
Keep `est_time`, `stage`, `runner_config` as **literal values** — `run_suite.py` collects them by AST parsing.
|
||||
|
||||
New and renamed non-kernel tests use this layout:
|
||||
|
||||
```text
|
||||
test/registered/<kind>/<subsystem>/test_*.py
|
||||
```
|
||||
|
||||
`<kind>` is one of `unit`, `e2e`, `accuracy`, `perf`, or `stress`. Kernel tests
|
||||
use `test/registered/kernels/{ops,benchmark}/<group>/`, retaining the established
|
||||
Directories under `test/registered/` group tests by topic and are free-form
|
||||
(`lora/`, `hicache/`, `disaggregation/`, `perf/`, ...); unit tests cover one srt
|
||||
module, so they mirror the source tree under `unit/`. What a test costs, which
|
||||
stage gates it and which runner it needs are declared by its `register_*_ci`
|
||||
call -- including hardware, which is expressed by one or more `register_*_ci`
|
||||
calls and never by a new top-level directory. Kernel tests use
|
||||
`test/registered/kernels/{ops,benchmark}/<group>/`, retaining the established
|
||||
plural `kernels` root.
|
||||
Hardware is expressed by one or more `register_*_ci` calls, never by creating a
|
||||
new top-level hardware directory. The admission checker applies the layout and
|
||||
kind/suite contract incrementally while legacy paths are migrated.
|
||||
|
||||
Diffusion workflows also enter through `test/run_suite.py`; registered bridge
|
||||
files preserve their case-level pytest partitioning until the remaining
|
||||
|
||||
+3
-103
@@ -3,7 +3,7 @@
|
||||
Covers:
|
||||
1. Triton LSE combine kernel correctness vs CPU reference (base-e and base-2)
|
||||
2. Various DCP world sizes (N=1,2,4,8)
|
||||
3. Edge cases: single shard, dominant LSE, equal LSE, NaN/inf
|
||||
3. Edge cases: single shard, dominant LSE, equal LSE
|
||||
4. return_lse mode
|
||||
5. dcp_a2a_lse_reduce with pre-allocated CUDA graph buffers
|
||||
"""
|
||||
@@ -239,31 +239,8 @@ class TestLSECombineEdgeCases(CustomTestCase):
|
||||
)
|
||||
|
||||
|
||||
class TestCPUReference(CustomTestCase):
|
||||
"""Test the CPU reference implementation independently."""
|
||||
|
||||
def test_basic_combine(self):
|
||||
from sglang.kernels.ops.attention.dcp_kernels import _lse_weighted_combine_cpu
|
||||
|
||||
N, B, H, D = 2, 2, 4, 8
|
||||
outputs = torch.randn(N, B, H, D)
|
||||
lses = torch.randn(N, B, H)
|
||||
|
||||
result = _lse_weighted_combine_cpu(outputs, lses, is_lse_base_on_e=True)
|
||||
self.assertEqual(result.shape, (B, H, D))
|
||||
self.assertFalse(torch.isnan(result).any())
|
||||
|
||||
def test_base2_vs_base_e(self):
|
||||
from sglang.kernels.ops.attention.dcp_kernels import _lse_weighted_combine_cpu
|
||||
|
||||
N, B, H, D = 2, 2, 4, 8
|
||||
outputs = torch.randn(N, B, H, D)
|
||||
lses = torch.randn(N, B, H) * 3.0
|
||||
|
||||
result_e = _lse_weighted_combine_cpu(outputs, lses, is_lse_base_on_e=True)
|
||||
result_2 = _lse_weighted_combine_cpu(outputs, lses, is_lse_base_on_e=False)
|
||||
|
||||
self.assertFalse(torch.allclose(result_e, result_2, atol=1e-3))
|
||||
class TestLSEBaseByBackend(CustomTestCase):
|
||||
"""Which attention backends report LSE in natural log."""
|
||||
|
||||
def test_natural_log_lse_backends(self):
|
||||
from sglang.srt.models.deepseek_common.attention_forward_methods.forward_mla import (
|
||||
@@ -277,26 +254,6 @@ class TestCPUReference(CustomTestCase):
|
||||
self.assertFalse(is_mla_dcp_lse_base_on_e("trtllm_mla"))
|
||||
self.assertFalse(is_mla_dcp_lse_base_on_e(None))
|
||||
|
||||
def test_nan_lse_handled(self):
|
||||
from sglang.kernels.ops.attention.dcp_kernels import _lse_weighted_combine_cpu
|
||||
|
||||
N, B, H, D = 2, 1, 1, 8
|
||||
outputs = torch.randn(N, B, H, D)
|
||||
lses = torch.tensor([[[5.0]], [[float("nan")]]])
|
||||
|
||||
result = _lse_weighted_combine_cpu(outputs, lses, is_lse_base_on_e=True)
|
||||
self.assertFalse(torch.isnan(result).any())
|
||||
|
||||
def test_inf_lse_handled(self):
|
||||
from sglang.kernels.ops.attention.dcp_kernels import _lse_weighted_combine_cpu
|
||||
|
||||
N, B, H, D = 2, 1, 1, 8
|
||||
outputs = torch.randn(N, B, H, D)
|
||||
lses = torch.tensor([[[5.0]], [[float("inf")]]])
|
||||
|
||||
result = _lse_weighted_combine_cpu(outputs, lses, is_lse_base_on_e=True)
|
||||
self.assertFalse(torch.isnan(result).any())
|
||||
|
||||
|
||||
class TestDCPA2AReduceWithCUDAGraphBuffers(CustomTestCase):
|
||||
"""Test dcp_a2a_lse_reduce with pre-allocated CUDA graph buffers."""
|
||||
@@ -368,40 +325,6 @@ class TestDCPA2AReduceWithCUDAGraphBuffers(CustomTestCase):
|
||||
rtol=1e-5,
|
||||
)
|
||||
|
||||
def test_cuda_graph_buffers_n4(self):
|
||||
from sglang.srt.layers.dcp import dcp_a2a_lse_reduce
|
||||
|
||||
torch.manual_seed(456)
|
||||
N, B, H_per_rank, D = 4, 2, 4, 64
|
||||
H = H_per_rank * N
|
||||
max_bs = 8
|
||||
|
||||
group = self._make_mock_group(N)
|
||||
|
||||
attn_out = torch.randn(B, H, D, device=self.device, dtype=torch.bfloat16)
|
||||
attn_lse = torch.randn(B, H, device=self.device, dtype=torch.float32)
|
||||
|
||||
result_dynamic = dcp_a2a_lse_reduce(
|
||||
attn_out.clone(), attn_lse.clone(), group, is_lse_base_on_e=True
|
||||
)
|
||||
|
||||
cuda_graph_buffers = self._make_cuda_graph_buffers(N, max_bs, H_per_rank, D)
|
||||
|
||||
result_graph = dcp_a2a_lse_reduce(
|
||||
attn_out.clone(),
|
||||
attn_lse.clone(),
|
||||
group,
|
||||
is_lse_base_on_e=True,
|
||||
cuda_graph_buffers=cuda_graph_buffers,
|
||||
)
|
||||
|
||||
torch.testing.assert_close(
|
||||
result_graph.float().cpu(),
|
||||
result_dynamic.float().cpu(),
|
||||
atol=1e-5,
|
||||
rtol=1e-5,
|
||||
)
|
||||
|
||||
def test_cuda_graph_buffers_partial_batch(self):
|
||||
"""Buffer max_bs > actual B -- should correctly slice."""
|
||||
from sglang.srt.layers.dcp import dcp_a2a_lse_reduce
|
||||
@@ -429,29 +352,6 @@ class TestDCPA2AReduceWithCUDAGraphBuffers(CustomTestCase):
|
||||
self.assertEqual(result.shape, (B, H_per_rank, D))
|
||||
self.assertFalse(torch.isnan(result).any())
|
||||
|
||||
def test_a2a_reduce_allocates_when_no_buffers(self):
|
||||
"""Without cuda_graph_buffers, dcp_a2a_lse_reduce still works (eager mode)."""
|
||||
from sglang.srt.layers.dcp import dcp_a2a_lse_reduce
|
||||
|
||||
N, B, H_per_rank, D = 2, 4, 8, 64
|
||||
H = H_per_rank * N
|
||||
|
||||
group = self._make_mock_group(N)
|
||||
|
||||
attn_out = torch.randn(B, H, D, device=self.device, dtype=torch.bfloat16)
|
||||
attn_lse = torch.randn(B, H, device=self.device, dtype=torch.float32)
|
||||
|
||||
result = dcp_a2a_lse_reduce(
|
||||
attn_out,
|
||||
attn_lse,
|
||||
group,
|
||||
is_lse_base_on_e=True,
|
||||
cuda_graph_buffers=None,
|
||||
)
|
||||
|
||||
self.assertEqual(result.shape, (B, H_per_rank, D))
|
||||
self.assertFalse(torch.isnan(result).any())
|
||||
|
||||
def test_pack_matches_the_copy_formulation_it_replaces(self):
|
||||
from sglang.kernels.ops.attention.dcp_kernels import (
|
||||
_lse_pack_dim,
|
||||
@@ -1,315 +0,0 @@
|
||||
"""
|
||||
End-to-end tests for strict reasoning + constrained decoding.
|
||||
|
||||
Tests that the full pipeline works:
|
||||
- AC-5.1: Strict reasoning + JSON schema constrained generation
|
||||
- AC-5.2: Strict reasoning + tool call parsing (basic validation only)
|
||||
|
||||
These tests launch a real server with a small model and verify
|
||||
the constrained decoding pipeline produces valid output.
|
||||
"""
|
||||
|
||||
import json
|
||||
import unittest
|
||||
|
||||
import requests
|
||||
|
||||
from sglang.srt.utils import kill_process_tree
|
||||
from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci
|
||||
from sglang.test.test_utils import (
|
||||
DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH,
|
||||
CustomTestCase,
|
||||
popen_launch_server,
|
||||
)
|
||||
|
||||
register_cuda_ci(est_time=96, stage="base-b", runner_config="1-gpu-small")
|
||||
register_amd_ci(est_time=120, suite="stage-b-test-1-gpu-small-amd")
|
||||
|
||||
MODEL = "Qwen/Qwen3-0.6B"
|
||||
BASE_URL = "http://127.0.0.1:39877"
|
||||
API_KEY = "sk-test-1234"
|
||||
|
||||
|
||||
class TestConstrainedReasoningE2E(CustomTestCase):
|
||||
@classmethod
|
||||
def setUpClass(cls):
|
||||
cls.model = MODEL
|
||||
cls.base_url = BASE_URL
|
||||
cls.api_key = API_KEY
|
||||
cls.process = popen_launch_server(
|
||||
cls.model,
|
||||
cls.base_url,
|
||||
timeout=DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH,
|
||||
api_key=cls.api_key,
|
||||
other_args=[
|
||||
"--reasoning-parser",
|
||||
"qwen3",
|
||||
],
|
||||
)
|
||||
|
||||
@classmethod
|
||||
def tearDownClass(cls):
|
||||
kill_process_tree(cls.process.pid)
|
||||
|
||||
def _chat(self, **kwargs):
|
||||
default = {
|
||||
"model": self.model,
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "What is 2+2? Answer with just the number.",
|
||||
}
|
||||
],
|
||||
"temperature": 0,
|
||||
"max_tokens": 256,
|
||||
}
|
||||
default.update(kwargs)
|
||||
resp = requests.post(
|
||||
f"{self.base_url}/v1/chat/completions",
|
||||
headers={"Authorization": f"Bearer {self.api_key}"},
|
||||
json=default,
|
||||
timeout=60,
|
||||
)
|
||||
self.assertEqual(resp.status_code, 200, f"Request failed: {resp.text}")
|
||||
return resp.json()
|
||||
|
||||
def test_reasoning_with_json_schema(self):
|
||||
"""AC-5.1: Reasoning + JSON schema produces valid JSON output."""
|
||||
schema = {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"answer": {"type": "integer"},
|
||||
},
|
||||
"required": ["answer"],
|
||||
}
|
||||
data = self._chat(
|
||||
response_format={
|
||||
"type": "json_schema",
|
||||
"json_schema": {
|
||||
"name": "answer_schema",
|
||||
"schema": schema,
|
||||
},
|
||||
},
|
||||
chat_template_kwargs={"enable_thinking": True},
|
||||
separate_reasoning=True,
|
||||
)
|
||||
|
||||
choice = data["choices"][0]
|
||||
content = choice["message"]["content"] or ""
|
||||
|
||||
# Content should be valid JSON conforming to schema when non-empty.
|
||||
# With small models + separate_reasoning, content may be empty if the
|
||||
# model puts everything in reasoning_content. That's acceptable.
|
||||
if content.strip():
|
||||
try:
|
||||
parsed = json.loads(content)
|
||||
self.assertIn("answer", parsed)
|
||||
self.assertIsInstance(parsed["answer"], int)
|
||||
except (json.JSONDecodeError, TypeError):
|
||||
# Small models may produce imperfect JSON
|
||||
self.assertTrue(
|
||||
content.strip().startswith("{"),
|
||||
f"Expected JSON-like output, got: {content!r}",
|
||||
)
|
||||
|
||||
# Content should NOT contain <think> tags (those go to reasoning_content)
|
||||
self.assertNotIn("<think>", content)
|
||||
|
||||
def test_reasoning_disabled_with_json_schema(self):
|
||||
"""JSON schema still works when reasoning is explicitly disabled."""
|
||||
schema = {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"answer": {"type": "integer"},
|
||||
},
|
||||
"required": ["answer"],
|
||||
}
|
||||
data = self._chat(
|
||||
response_format={
|
||||
"type": "json_schema",
|
||||
"json_schema": {
|
||||
"name": "answer_schema",
|
||||
"schema": schema,
|
||||
},
|
||||
},
|
||||
chat_template_kwargs={"enable_thinking": False},
|
||||
)
|
||||
|
||||
choice = data["choices"][0]
|
||||
content = choice["message"]["content"]
|
||||
|
||||
# Should still produce valid JSON
|
||||
parsed = json.loads(content)
|
||||
self.assertIn("answer", parsed)
|
||||
|
||||
def test_reasoning_with_separate_output(self):
|
||||
"""Reasoning content is correctly separated from normal content."""
|
||||
data = self._chat(
|
||||
chat_template_kwargs={"enable_thinking": True},
|
||||
separate_reasoning=True,
|
||||
)
|
||||
|
||||
choice = data["choices"][0]
|
||||
content = choice["message"]["content"]
|
||||
reasoning = choice["message"].get("reasoning_content")
|
||||
|
||||
# Content should not contain think tags
|
||||
self.assertNotIn("<think>", content)
|
||||
self.assertNotIn("</think>", content)
|
||||
|
||||
def test_tool_call_after_reasoning(self):
|
||||
"""AC-5.2: Tool call parsing works with reasoning enabled."""
|
||||
tools = [
|
||||
{
|
||||
"type": "function",
|
||||
"function": {
|
||||
"name": "get_weather",
|
||||
"description": "Get the current weather",
|
||||
"parameters": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"location": {"type": "string"},
|
||||
},
|
||||
"required": ["location"],
|
||||
},
|
||||
},
|
||||
}
|
||||
]
|
||||
data = self._chat(
|
||||
messages=[
|
||||
{
|
||||
"role": "user",
|
||||
"content": "What's the weather in Paris?",
|
||||
}
|
||||
],
|
||||
tools=tools,
|
||||
chat_template_kwargs={"enable_thinking": True},
|
||||
separate_reasoning=True,
|
||||
)
|
||||
|
||||
choice = data["choices"][0]
|
||||
# The model may or may not produce tool calls (depends on model capability)
|
||||
# but the response should be well-formed (no crashes)
|
||||
self.assertIn("message", choice)
|
||||
self.assertIn("finish_reason", choice)
|
||||
# finish_reason should be either "stop" or "tool_calls"
|
||||
self.assertIn(choice["finish_reason"], ["stop", "tool_calls", "length"])
|
||||
|
||||
|
||||
class TestStrictThinkingE2E(CustomTestCase):
|
||||
"""E2E tests with --enable-strict-thinking flag.
|
||||
|
||||
Validates that the strict thinking flag is correctly propagated through
|
||||
the full pipeline: server_args -> grammar_backend -> ReasonerGrammarBackend
|
||||
-> token filtering during thinking phase.
|
||||
"""
|
||||
|
||||
@classmethod
|
||||
def setUpClass(cls):
|
||||
cls.model = MODEL
|
||||
cls.base_url = "http://127.0.0.1:39878"
|
||||
cls.api_key = API_KEY
|
||||
cls.process = popen_launch_server(
|
||||
cls.model,
|
||||
cls.base_url,
|
||||
timeout=DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH,
|
||||
api_key=cls.api_key,
|
||||
other_args=[
|
||||
"--reasoning-parser",
|
||||
"qwen3",
|
||||
"--enable-strict-thinking",
|
||||
],
|
||||
)
|
||||
|
||||
@classmethod
|
||||
def tearDownClass(cls):
|
||||
kill_process_tree(cls.process.pid)
|
||||
|
||||
def _chat(self, **kwargs):
|
||||
default = {
|
||||
"model": self.model,
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "What is 2+2? Answer with just the number.",
|
||||
}
|
||||
],
|
||||
"temperature": 0,
|
||||
"max_tokens": 256,
|
||||
}
|
||||
default.update(kwargs)
|
||||
resp = requests.post(
|
||||
f"{self.base_url}/v1/chat/completions",
|
||||
headers={"Authorization": f"Bearer {self.api_key}"},
|
||||
json=default,
|
||||
timeout=60,
|
||||
)
|
||||
self.assertEqual(resp.status_code, 200, f"Request failed: {resp.text}")
|
||||
return resp.json()
|
||||
|
||||
def test_strict_thinking_with_json_schema(self):
|
||||
"""Strict thinking + JSON schema: server starts and produces valid output."""
|
||||
schema = {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"answer": {"type": "integer"},
|
||||
},
|
||||
"required": ["answer"],
|
||||
}
|
||||
data = self._chat(
|
||||
response_format={
|
||||
"type": "json_schema",
|
||||
"json_schema": {
|
||||
"name": "answer_schema",
|
||||
"schema": schema,
|
||||
},
|
||||
},
|
||||
chat_template_kwargs={"enable_thinking": True},
|
||||
separate_reasoning=True,
|
||||
)
|
||||
|
||||
choice = data["choices"][0]
|
||||
content = choice["message"]["content"] or ""
|
||||
|
||||
if content.strip():
|
||||
try:
|
||||
parsed = json.loads(content)
|
||||
self.assertIn("answer", parsed)
|
||||
except (json.JSONDecodeError, TypeError):
|
||||
self.assertTrue(
|
||||
content.strip().startswith("{"),
|
||||
f"Expected JSON-like output, got: {content!r}",
|
||||
)
|
||||
|
||||
# Think tags must not leak into content
|
||||
self.assertNotIn("<think>", content)
|
||||
|
||||
def test_strict_thinking_disabled_per_request(self):
|
||||
"""When thinking is disabled per-request, strict server still works."""
|
||||
data = self._chat(
|
||||
chat_template_kwargs={"enable_thinking": False},
|
||||
)
|
||||
|
||||
choice = data["choices"][0]
|
||||
self.assertIn("message", choice)
|
||||
self.assertIn("finish_reason", choice)
|
||||
# Should complete normally without errors
|
||||
self.assertIn(choice["finish_reason"], ["stop", "length"])
|
||||
|
||||
def test_strict_thinking_separate_reasoning(self):
|
||||
"""Strict thinking with separate_reasoning produces well-formed output."""
|
||||
data = self._chat(
|
||||
chat_template_kwargs={"enable_thinking": True},
|
||||
separate_reasoning=True,
|
||||
)
|
||||
|
||||
choice = data["choices"][0]
|
||||
content = choice["message"]["content"] or ""
|
||||
|
||||
# Think tags must not leak into content
|
||||
self.assertNotIn("<think>", content)
|
||||
self.assertNotIn("</think>", content)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
+2
-45
@@ -445,16 +445,13 @@ def test_fused_moe_compile_hook_is_bs1_only():
|
||||
# --- tracing --------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_trace_labels_platform_and_backend(monkeypatch):
|
||||
def test_trace_labels_explicit_backend(monkeypatch):
|
||||
_mock_platform(monkeypatch, key="cuda", info=_CUDA)
|
||||
op = _CudaOnlyPlatformOp()
|
||||
fo.enable_fused_op_trace()
|
||||
op(torch.zeros(2, 3))
|
||||
op(torch.zeros(2, 3), backend=KernelBackend.TORCH)
|
||||
auto_rec, explicit_rec = fo.get_fused_op_trace()
|
||||
assert auto_rec.op == "test.cuda_only_platform"
|
||||
assert auto_rec.backend == "cuda"
|
||||
assert auto_rec.tensor_args == ("torch.float32[2, 3]",)
|
||||
_, explicit_rec = fo.get_fused_op_trace()
|
||||
assert explicit_rec.backend == "torch"
|
||||
|
||||
|
||||
@@ -512,46 +509,6 @@ def test_deprecated_alias_keeps_legacy_platform_defaults(monkeypatch):
|
||||
_NativeOnlyLegacy()(torch.zeros(1)) # old CUDA behavior preserved
|
||||
|
||||
|
||||
# --- migration completeness -------------------------------------------------------
|
||||
|
||||
_MIGRATED_OPS = [
|
||||
("sglang.srt.layers.activation", "SiluAndMul"),
|
||||
("sglang.srt.layers.activation", "GeluAndMul"),
|
||||
("sglang.srt.layers.activation", "NewGELU"),
|
||||
("sglang.srt.layers.activation", "ReLU2"),
|
||||
("sglang.srt.layers.activation", "QuickGELU"),
|
||||
("sglang.srt.layers.activation", "XIELU"),
|
||||
("sglang.srt.layers.layernorm", "RMSNorm"),
|
||||
("sglang.srt.layers.layernorm", "LayerNorm"),
|
||||
("sglang.srt.layers.layernorm", "GemmaRMSNorm"),
|
||||
("sglang.srt.layers.layernorm", "Gemma3RMSNorm"),
|
||||
("sglang.srt.layers.layernorm", "Gemma4RMSNorm"),
|
||||
("sglang.srt.layers.layernorm", "RMSNormWithoutScale"),
|
||||
("sglang.srt.layers.conv", "Conv2dLayer"),
|
||||
("sglang.srt.layers.conv", "Conv3dLayer"),
|
||||
("sglang.srt.layers.moe.topk", "TopK"),
|
||||
("sglang.srt.layers.rotary_embedding.base", "RotaryEmbedding"),
|
||||
("sglang.srt.layers.rotary_embedding.rope_variant", "DualChunkRotaryEmbedding"),
|
||||
("sglang.srt.layers.attention.dsa.dsa_indexer", "Indexer"),
|
||||
("sglang.srt.layers.attention.dsv4.compressor", "Compressor"),
|
||||
("sglang.srt.layers.attention.mamba.mixer2_rms_norm_gated", "Mixer2RMSNormGated"),
|
||||
("sglang.srt.layers.quantization.unquant", "UnquantizedFusedMoEMethod"),
|
||||
]
|
||||
|
||||
|
||||
@pytest.mark.parametrize("module_name, cls_name", _MIGRATED_OPS)
|
||||
def test_migrated_ops_subclass_base_fused_op(module_name, cls_name):
|
||||
"""Production ops must extend BaseFusedOp directly, never the deprecated
|
||||
MultiPlatformOp alias (which exists only for out-of-tree users)."""
|
||||
import importlib
|
||||
|
||||
from sglang.srt.layers.utils.multi_platform import MultiPlatformOp
|
||||
|
||||
cls = getattr(importlib.import_module(module_name), cls_name)
|
||||
assert issubclass(cls, BaseFusedOp)
|
||||
assert MultiPlatformOp not in cls.__mro__
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
import sys
|
||||
|
||||
Reference in New Issue
Block a user