diff --git a/.pre-commit-config.yaml b/.pre-commit-config.yaml index 32ac18680..5584fd5dd 100644 --- a/.pre-commit-config.yaml +++ b/.pre-commit-config.yaml @@ -111,12 +111,6 @@ repos: entry: python3 scripts/lint/check_no_bare_pytest_main.py language: system files: ^(python|test)/.*\.py$ - - id: check-lint-script-tests - name: unit tests for lint checkers - entry: python3 -m unittest discover -s scripts/lint -p 'test_check_*.py' - language: system - files: ^scripts/lint/(check_|test_check_).*\.py$ - pass_filenames: false - id: check-registered-tests name: validate registered test CI registries entry: python3 scripts/lint/check_registered_tests.py diff --git a/scripts/lint/check_registered_tests.py b/scripts/lint/check_registered_tests.py index 79d753255..800b0f0e7 100755 --- a/scripts/lint/check_registered_tests.py +++ b/scripts/lint/check_registered_tests.py @@ -2,23 +2,8 @@ """ Pre-commit hook: validate CI registry calls under test/registered/. -1. Every test file must contain a CI registry call (register_cuda_ci, - register_amd_ci, etc.). -2. A CUDA test must register its suite via the modern - `stage=`/`runner_config=` form. The legacy single-string `suite=` is reserved - for the stress family (and for AMD/CPU/NPU suites); any other CUDA `suite=` - resolves to a name no workflow invokes, so the test silently never runs. - Two shapes are rejected: - a. `{stage}-test-{runner_config}` -- the modern name stuffed back into the - legacy form. Reported with the exact stage/runner split to use. - b. an older `{stage}-{runner_config}` PR-test name (e.g. the pre-migration - `base-b-kernel-unit-1-gpu-large`) -- no longer matches any workflow - suite at all. - The modern form resolves to the identical suite (CIRegistry.effective_suite - is f"{stage}-test-{runner_config}") and is /rerun-test-able. - -Reuses ut_parse_one_file() from ci_register.py (AST-based parsing) -to match the same logic used by run_suite.py's collect_tests(). +Reuses ut_parse_one_file() from ci_register.py (AST-based parsing) to match +run_suite.py's collect_tests(). """ import ast @@ -26,25 +11,17 @@ import glob import importlib.util import os import re -import subprocess import sys -# Suite names of the form `{stage}-test-{runner_config}` are exactly what the -# modern stage=/runner_config= form produces, so a legacy suite= carrying this -# shape is always expressible (and should be expressed) the modern way. +# Exactly what stage=/runner_config= produces, so a legacy suite= of this shape +# is always expressible the modern way. _MODERN_SHAPE = re.compile(r"^(.+)-test-(.+)$") -# The only CUDA suite family still allowed on the legacy single-string `suite=` -# form. Anything else needs stage=/runner_config=, or its effective_suite matches -# no suite any workflow invokes and the test silently never runs. +# The only CUDA family still allowed on legacy `suite=`; anything else resolves +# to a suite no workflow invokes and the test silently never runs. _LEGACY_CUDA_PREFIXES = ("stress",) -_TEST_KINDS = {"unit", "e2e", "accuracy", "perf", "stress"} -_KERNEL_ROOT = "kernels" - -# Flat vendor trees. Vendor-only coverage fits no kind above: no XPU/NPU suite -# carries the `-kernel-` infix the kernel tree needs, and these launch device work. -_VENDOR_DIRS = {"amd", "mlx", "musa", "npu", "xpu"} +_KERNEL_LAYOUT = "test/registered/kernels/{ops,benchmark}//" def _defines_testcase(tree: ast.AST) -> bool: @@ -78,43 +55,6 @@ def _main_runs_tests(tree: ast.Module) -> bool: return False -def _git_lines(*args: str) -> list[str] | None: - result = subprocess.run(["git", *args], capture_output=True, text=True, check=False) - if result.returncode != 0: - return None - return [line for line in result.stdout.splitlines() if line] - - -def _changed_registered_files() -> set[str]: - """Return added, copied, or renamed registered-test destinations.""" - - lines = _git_lines("diff", "--cached", "--name-status", "--diff-filter=ACR") - if not lines: - base_ref = os.environ.get("GITHUB_BASE_REF", "main") - for candidate in (f"origin/{base_ref}", base_ref): - if _git_lines("rev-parse", "--verify", candidate) is None: - continue - merge_base = _git_lines("merge-base", candidate, "HEAD") - if not merge_base: - continue - lines = _git_lines( - "diff", - "--name-status", - "--diff-filter=ACR", - merge_base[0], - "HEAD", - ) - break - - selected = set() - for line in lines or []: - fields = line.split("\t") - destination = fields[-1] - if destination.startswith("test/registered/") and destination.endswith(".py"): - selected.add(destination) - return selected - - def _contains_call(tree: ast.AST, name: str) -> bool: return any( isinstance(node, ast.Call) @@ -126,63 +66,28 @@ def _contains_call(tree: ast.AST, name: str) -> bool: ) -def taxonomy_errors(path: str, registries: list, tree: ast.AST) -> list[str]: - """Validate the kind/subsystem contract for a newly admitted path.""" - +def taxonomy_errors(path: str, tree: ast.AST) -> list[str]: parts = path.split("/") - relative_parts = parts[2:] if parts[:2] == ["test", "registered"] else [] - if relative_parts and relative_parts[0] in _VENDOR_DIRS: + if parts[:2] != ["test", "registered"] or len(parts) < 3: return [] - if relative_parts and relative_parts[0] == _KERNEL_ROOT: - errors = [] - if len(relative_parts) < 4 or relative_parts[1] not in {"ops", "benchmark"}: - errors.append( - f"{path}: kernel tests must live under " - "test/registered/kernels/{ops,benchmark}//" - ) - if any("-kernel-" not in (r.effective_suite or "") for r in registries): - errors.append(f"{path}: kernel tests must use a *-kernel-* suite") - return errors - if len(relative_parts) < 3 or relative_parts[0] not in _TEST_KINDS: - return [ - f"{path}: registered tests must live under " - "test/registered///; kind must be one of " - + ", ".join(sorted(_TEST_KINDS)) - + "; kernel tests use test/registered/kernels/{ops,benchmark}//" - ] + relative_parts = parts[2:] + root = relative_parts[0] - kind = relative_parts[0] - errors = [] - if kind == "unit": - invalid = [ - r - for r in registries - if r.backend.name != "CPU" and "-unit-" not in (r.effective_suite or "") - ] - if invalid: - errors.append(f"{path}: unit tests must use CPU or dedicated unit suites") - if any(r.est_time > 60 for r in registries): - errors.append(f"{path}: unit test est_time must be <= 60 seconds") - if _contains_call(tree, "popen_launch_server"): - errors.append(f"{path}: unit tests may not launch a server") - elif kind in {"accuracy", "perf"}: - invalid = [ - r - for r in registries - if not (r.effective_suite or "").startswith(("nightly-", "weekly-")) - ] - if invalid: - errors.append(f"{path}: {kind} tests must use nightly/weekly suites") - elif kind == "stress": - invalid = [ - r - for r in registries - if (r.effective_suite or "") != "stress" - and not (r.effective_suite or "").startswith("weekly-") - ] - if invalid: - errors.append(f"{path}: stress tests must use stress/weekly suites") - return errors + if root in ("kernel", "kernels"): + canonical = ( + root == "kernels" + and len(relative_parts) >= 4 + and relative_parts[1] in ("ops", "benchmark") + ) + return ( + [] if canonical else [f"{path}: kernel tests live under {_KERNEL_LAYOUT}"] + ) + + if root != "unit": + return [] + if _contains_call(tree, "popen_launch_server"): + return [f"{path}: unit tests may not launch a server"] + return [] def main() -> int: @@ -209,22 +114,19 @@ def main() -> int: non_dispatchable = [] # (file, suite) -- legacy CUDA suite no workflow invokes dead_tests = [] # (file) -- TestCase classes that `python3 file.py` never runs taxonomy_violations = [] - changed_files = _changed_registered_files() for f in files: try: registries, _has_main_entry = ci_register.ut_parse_one_file(f) except Exception: - # Skip files that can't be parsed (syntax errors, etc.) continue if len(registries) == 0: missing.append(f) continue - # TestCase classes are dead unless __main__ runs them (CI does - # `python3 file.py`); the ERROR text below explains the fix. + # TestCase classes are dead unless __main__ runs them; CI runs the + # registered file as `python3 file.py`. with open(f, "r", encoding="utf-8") as fh: tree = ast.parse(fh.read(), filename=f) - if f in changed_files: - taxonomy_violations.extend(taxonomy_errors(f, registries, tree)) + taxonomy_violations.extend(taxonomy_errors(f, tree)) if _defines_testcase(tree) and not _main_runs_tests(tree): dead_tests.append(f) for r in registries: diff --git a/scripts/lint/test_check_no_bare_pytest_main.py b/scripts/lint/test_check_no_bare_pytest_main.py deleted file mode 100644 index 4721021a0..000000000 --- a/scripts/lint/test_check_no_bare_pytest_main.py +++ /dev/null @@ -1,86 +0,0 @@ -import pathlib -import tempfile -import unittest - -from check_no_bare_pytest_main import find_bare_pytest_main - - -class TestFindBarePytestMain(unittest.TestCase): - def check_source(self, source: str) -> int | None: - with tempfile.TemporaryDirectory() as directory: - path = pathlib.Path(directory) / "example.py" - path.write_text(source, encoding="utf-8") - return find_bare_pytest_main(path) - - def test_rejects_discarded_result(self): - source = """ -if __name__ == "__main__": - pytest.main([__file__]) -""" - self.assertEqual(self.check_source(source), 3) - - def test_rejects_discarded_result_with_whitespace(self): - source = """ -if __name__ == "__main__": - pytest . main([__file__]) -""" - self.assertEqual(self.check_source(source), 3) - - def test_accepts_propagated_result(self): - source = """ -if __name__ == "__main__": - sys.exit(pytest.main([__file__])) -""" - self.assertIsNone(self.check_source(source)) - - def test_rejects_assigned_result(self): - source = """ -if "__main__" == __name__: - exit_code = pytest.main([__file__]) -""" - self.assertEqual(self.check_source(source), 3) - - def test_accepts_assigned_result_that_is_later_exited(self): - source = """ -if __name__ == "__main__": - exit_code = pytest.main([__file__]) - sys.exit(exit_code) -""" - self.assertIsNone(self.check_source(source)) - - def test_accepts_assigned_result_that_is_raised(self): - source = """ -if __name__ == "__main__": - exit_code = pytest.main([__file__]) - raise SystemExit(exit_code) -""" - self.assertIsNone(self.check_source(source)) - - def test_rejects_nested_discarded_result(self): - source = """ -if __name__ == "__main__": - if enabled: - pytest.main([__file__]) -""" - self.assertEqual(self.check_source(source), 4) - - def test_accepts_raised_system_exit(self): - source = """ -if __name__ == "__main__": - raise SystemExit(pytest.main([__file__])) -""" - self.assertIsNone(self.check_source(source)) - - def test_rejects_unraised_system_exit(self): - source = """ -if __name__ == "__main__": - error = SystemExit(pytest.main([__file__])) -""" - self.assertEqual(self.check_source(source), 3) - - def test_ignores_call_outside_main_guard(self): - self.assertIsNone(self.check_source("pytest.main([__file__])\n")) - - -if __name__ == "__main__": - unittest.main() diff --git a/scripts/lint/test_check_registered_tests.py b/scripts/lint/test_check_registered_tests.py deleted file mode 100644 index cbed176d1..000000000 --- a/scripts/lint/test_check_registered_tests.py +++ /dev/null @@ -1,51 +0,0 @@ -import ast -import unittest -from types import SimpleNamespace - -from scripts.lint.check_registered_tests import taxonomy_errors - - -def _registry(suite: str): - return SimpleNamespace(effective_suite=suite, est_time=1) - - -class TestRegisteredTestTaxonomy(unittest.TestCase): - def setUp(self): - self.tree = ast.parse("") - self.kernel_registry = [_registry("base-b-kernel-unit-test-1-gpu-large")] - - def test_plural_kernel_ops_layout_is_accepted(self): - errors = taxonomy_errors( - "test/registered/kernels/ops/attention/test_example.py", - self.kernel_registry, - self.tree, - ) - self.assertEqual(errors, []) - - def test_plural_kernel_benchmark_layout_is_accepted(self): - errors = taxonomy_errors( - "test/registered/kernels/benchmark/attention/bench_example.py", - [_registry("base-b-kernel-benchmark-test-1-gpu-large")], - self.tree, - ) - self.assertEqual(errors, []) - - def test_singular_kernel_root_is_rejected(self): - errors = taxonomy_errors( - "test/registered/kernel/attention/test_example.py", - self.kernel_registry, - self.tree, - ) - self.assertTrue(errors) - - def test_kernel_group_is_required(self): - errors = taxonomy_errors( - "test/registered/kernels/ops/test_example.py", - self.kernel_registry, - self.tree, - ) - self.assertTrue(errors) - - -if __name__ == "__main__": - unittest.main() diff --git a/test/README.md b/test/README.md index 9ce06d2af..4f0e95bcc 100644 --- a/test/README.md +++ b/test/README.md @@ -72,18 +72,14 @@ Parameters: `est_time` (seconds), `stage` + `runner_config` (target stage and ru Keep `est_time`, `stage`, `runner_config` as **literal values** — `run_suite.py` collects them by AST parsing. -New and renamed non-kernel tests use this layout: - -```text -test/registered///test_*.py -``` - -`` is one of `unit`, `e2e`, `accuracy`, `perf`, or `stress`. Kernel tests -use `test/registered/kernels/{ops,benchmark}//`, retaining the established +Directories under `test/registered/` group tests by topic and are free-form +(`lora/`, `hicache/`, `disaggregation/`, `perf/`, ...); unit tests cover one srt +module, so they mirror the source tree under `unit/`. What a test costs, which +stage gates it and which runner it needs are declared by its `register_*_ci` +call -- including hardware, which is expressed by one or more `register_*_ci` +calls and never by a new top-level directory. Kernel tests use +`test/registered/kernels/{ops,benchmark}//`, retaining the established plural `kernels` root. -Hardware is expressed by one or more `register_*_ci` calls, never by creating a -new top-level hardware directory. The admission checker applies the layout and -kind/suite contract incrementally while legacy paths are migrated. Diffusion workflows also enter through `test/run_suite.py`; registered bridge files preserve their case-level pytest partitioning until the remaining diff --git a/test/registered/kernels/benchmark/bench_get_block_table.py b/test/registered/kernels/benchmark/minicpm_sala/bench_get_block_table.py similarity index 100% rename from test/registered/kernels/benchmark/bench_get_block_table.py rename to test/registered/kernels/benchmark/minicpm_sala/bench_get_block_table.py diff --git a/test/registered/kernels/test_dcp_lse_combine.py b/test/registered/kernels/ops/attention/test_dcp_lse_combine.py similarity index 81% rename from test/registered/kernels/test_dcp_lse_combine.py rename to test/registered/kernels/ops/attention/test_dcp_lse_combine.py index fadf5a417..f177178c1 100644 --- a/test/registered/kernels/test_dcp_lse_combine.py +++ b/test/registered/kernels/ops/attention/test_dcp_lse_combine.py @@ -3,7 +3,7 @@ Covers: 1. Triton LSE combine kernel correctness vs CPU reference (base-e and base-2) 2. Various DCP world sizes (N=1,2,4,8) -3. Edge cases: single shard, dominant LSE, equal LSE, NaN/inf +3. Edge cases: single shard, dominant LSE, equal LSE 4. return_lse mode 5. dcp_a2a_lse_reduce with pre-allocated CUDA graph buffers """ @@ -239,31 +239,8 @@ class TestLSECombineEdgeCases(CustomTestCase): ) -class TestCPUReference(CustomTestCase): - """Test the CPU reference implementation independently.""" - - def test_basic_combine(self): - from sglang.kernels.ops.attention.dcp_kernels import _lse_weighted_combine_cpu - - N, B, H, D = 2, 2, 4, 8 - outputs = torch.randn(N, B, H, D) - lses = torch.randn(N, B, H) - - result = _lse_weighted_combine_cpu(outputs, lses, is_lse_base_on_e=True) - self.assertEqual(result.shape, (B, H, D)) - self.assertFalse(torch.isnan(result).any()) - - def test_base2_vs_base_e(self): - from sglang.kernels.ops.attention.dcp_kernels import _lse_weighted_combine_cpu - - N, B, H, D = 2, 2, 4, 8 - outputs = torch.randn(N, B, H, D) - lses = torch.randn(N, B, H) * 3.0 - - result_e = _lse_weighted_combine_cpu(outputs, lses, is_lse_base_on_e=True) - result_2 = _lse_weighted_combine_cpu(outputs, lses, is_lse_base_on_e=False) - - self.assertFalse(torch.allclose(result_e, result_2, atol=1e-3)) +class TestLSEBaseByBackend(CustomTestCase): + """Which attention backends report LSE in natural log.""" def test_natural_log_lse_backends(self): from sglang.srt.models.deepseek_common.attention_forward_methods.forward_mla import ( @@ -277,26 +254,6 @@ class TestCPUReference(CustomTestCase): self.assertFalse(is_mla_dcp_lse_base_on_e("trtllm_mla")) self.assertFalse(is_mla_dcp_lse_base_on_e(None)) - def test_nan_lse_handled(self): - from sglang.kernels.ops.attention.dcp_kernels import _lse_weighted_combine_cpu - - N, B, H, D = 2, 1, 1, 8 - outputs = torch.randn(N, B, H, D) - lses = torch.tensor([[[5.0]], [[float("nan")]]]) - - result = _lse_weighted_combine_cpu(outputs, lses, is_lse_base_on_e=True) - self.assertFalse(torch.isnan(result).any()) - - def test_inf_lse_handled(self): - from sglang.kernels.ops.attention.dcp_kernels import _lse_weighted_combine_cpu - - N, B, H, D = 2, 1, 1, 8 - outputs = torch.randn(N, B, H, D) - lses = torch.tensor([[[5.0]], [[float("inf")]]]) - - result = _lse_weighted_combine_cpu(outputs, lses, is_lse_base_on_e=True) - self.assertFalse(torch.isnan(result).any()) - class TestDCPA2AReduceWithCUDAGraphBuffers(CustomTestCase): """Test dcp_a2a_lse_reduce with pre-allocated CUDA graph buffers.""" @@ -368,40 +325,6 @@ class TestDCPA2AReduceWithCUDAGraphBuffers(CustomTestCase): rtol=1e-5, ) - def test_cuda_graph_buffers_n4(self): - from sglang.srt.layers.dcp import dcp_a2a_lse_reduce - - torch.manual_seed(456) - N, B, H_per_rank, D = 4, 2, 4, 64 - H = H_per_rank * N - max_bs = 8 - - group = self._make_mock_group(N) - - attn_out = torch.randn(B, H, D, device=self.device, dtype=torch.bfloat16) - attn_lse = torch.randn(B, H, device=self.device, dtype=torch.float32) - - result_dynamic = dcp_a2a_lse_reduce( - attn_out.clone(), attn_lse.clone(), group, is_lse_base_on_e=True - ) - - cuda_graph_buffers = self._make_cuda_graph_buffers(N, max_bs, H_per_rank, D) - - result_graph = dcp_a2a_lse_reduce( - attn_out.clone(), - attn_lse.clone(), - group, - is_lse_base_on_e=True, - cuda_graph_buffers=cuda_graph_buffers, - ) - - torch.testing.assert_close( - result_graph.float().cpu(), - result_dynamic.float().cpu(), - atol=1e-5, - rtol=1e-5, - ) - def test_cuda_graph_buffers_partial_batch(self): """Buffer max_bs > actual B -- should correctly slice.""" from sglang.srt.layers.dcp import dcp_a2a_lse_reduce @@ -429,29 +352,6 @@ class TestDCPA2AReduceWithCUDAGraphBuffers(CustomTestCase): self.assertEqual(result.shape, (B, H_per_rank, D)) self.assertFalse(torch.isnan(result).any()) - def test_a2a_reduce_allocates_when_no_buffers(self): - """Without cuda_graph_buffers, dcp_a2a_lse_reduce still works (eager mode).""" - from sglang.srt.layers.dcp import dcp_a2a_lse_reduce - - N, B, H_per_rank, D = 2, 4, 8, 64 - H = H_per_rank * N - - group = self._make_mock_group(N) - - attn_out = torch.randn(B, H, D, device=self.device, dtype=torch.bfloat16) - attn_lse = torch.randn(B, H, device=self.device, dtype=torch.float32) - - result = dcp_a2a_lse_reduce( - attn_out, - attn_lse, - group, - is_lse_base_on_e=True, - cuda_graph_buffers=None, - ) - - self.assertEqual(result.shape, (B, H_per_rank, D)) - self.assertFalse(torch.isnan(result).any()) - def test_pack_matches_the_copy_formulation_it_replaces(self): from sglang.kernels.ops.attention.dcp_kernels import ( _lse_pack_dim, diff --git a/test/registered/kernels/test_dsa_kpool_multi_pool.py b/test/registered/kernels/ops/attention/test_dsa_kpool_multi_pool.py similarity index 100% rename from test/registered/kernels/test_dsa_kpool_multi_pool.py rename to test/registered/kernels/ops/attention/test_dsa_kpool_multi_pool.py diff --git a/test/registered/kernels/test_fused_kda_conv_recurrent_verify.py b/test/registered/kernels/ops/attention/test_fused_kda_conv_recurrent_verify.py similarity index 100% rename from test/registered/kernels/test_fused_kda_conv_recurrent_verify.py rename to test/registered/kernels/ops/attention/test_fused_kda_conv_recurrent_verify.py diff --git a/test/registered/kernels/test_kda_mtp_cutedsl_replayssm_ring.py b/test/registered/kernels/ops/attention/test_kda_mtp_cutedsl_replayssm_ring.py similarity index 100% rename from test/registered/kernels/test_kda_mtp_cutedsl_replayssm_ring.py rename to test/registered/kernels/ops/attention/test_kda_mtp_cutedsl_replayssm_ring.py diff --git a/test/registered/kernels/test_kda_replayssm_fold.py b/test/registered/kernels/ops/attention/test_kda_replayssm_fold.py similarity index 100% rename from test/registered/kernels/test_kda_replayssm_fold.py rename to test/registered/kernels/ops/attention/test_kda_replayssm_fold.py diff --git a/test/registered/kernels/test_kda_replayssm_fold_batched.py b/test/registered/kernels/ops/attention/test_kda_replayssm_fold_batched.py similarity index 100% rename from test/registered/kernels/test_kda_replayssm_fold_batched.py rename to test/registered/kernels/ops/attention/test_kda_replayssm_fold_batched.py diff --git a/test/registered/kernels/test_kda_replayssm_ring_fused.py b/test/registered/kernels/ops/attention/test_kda_replayssm_ring_fused.py similarity index 100% rename from test/registered/kernels/test_kda_replayssm_ring_fused.py rename to test/registered/kernels/ops/attention/test_kda_replayssm_ring_fused.py diff --git a/test/registered/kernels/test_kda_replayssm_ring_ragged.py b/test/registered/kernels/ops/attention/test_kda_replayssm_ring_ragged.py similarity index 100% rename from test/registered/kernels/test_kda_replayssm_ring_ragged.py rename to test/registered/kernels/ops/attention/test_kda_replayssm_ring_ragged.py diff --git a/test/registered/kernels/test_lean_attention.py b/test/registered/kernels/ops/attention/test_lean_attention.py similarity index 100% rename from test/registered/kernels/test_lean_attention.py rename to test/registered/kernels/ops/attention/test_lean_attention.py diff --git a/test/registered/kernels/test_quick_allreduce_bf16_range.py b/test/registered/kernels/ops/communication/test_quick_allreduce_bf16_range.py similarity index 100% rename from test/registered/kernels/test_quick_allreduce_bf16_range.py rename to test/registered/kernels/ops/communication/test_quick_allreduce_bf16_range.py diff --git a/test/registered/kernels/ops/test_kimi_k3_prerequisite_ops.py b/test/registered/kernels/ops/kimi_k3/test_kimi_k3_prerequisite_ops.py similarity index 100% rename from test/registered/kernels/ops/test_kimi_k3_prerequisite_ops.py rename to test/registered/kernels/ops/kimi_k3/test_kimi_k3_prerequisite_ops.py diff --git a/test/registered/kernels/ops/test_get_block_table.py b/test/registered/kernels/ops/minicpm_sala/test_get_block_table.py similarity index 100% rename from test/registered/kernels/ops/test_get_block_table.py rename to test/registered/kernels/ops/minicpm_sala/test_get_block_table.py diff --git a/test/registered/kernels/ops/test_minicpm_compress_k.py b/test/registered/kernels/ops/minicpm_sala/test_minicpm_compress_k.py similarity index 100% rename from test/registered/kernels/ops/test_minicpm_compress_k.py rename to test/registered/kernels/ops/minicpm_sala/test_minicpm_compress_k.py diff --git a/test/registered/kernel/quant/test_nvfp4_native_kv_layout.py b/test/registered/kernels/ops/quantization/test_nvfp4_native_kv_layout.py similarity index 100% rename from test/registered/kernel/quant/test_nvfp4_native_kv_layout.py rename to test/registered/kernels/ops/quantization/test_nvfp4_native_kv_layout.py diff --git a/test/registered/unit/managers/test_customized_info_streaming.py b/test/registered/observability/test_customized_info_streaming.py similarity index 100% rename from test/registered/unit/managers/test_customized_info_streaming.py rename to test/registered/observability/test_customized_info_streaming.py diff --git a/test/registered/unit/constrained/test_e2e_constrained_reasoning.py b/test/registered/unit/constrained/test_e2e_constrained_reasoning.py deleted file mode 100644 index 1d64425ac..000000000 --- a/test/registered/unit/constrained/test_e2e_constrained_reasoning.py +++ /dev/null @@ -1,315 +0,0 @@ -""" -End-to-end tests for strict reasoning + constrained decoding. - -Tests that the full pipeline works: -- AC-5.1: Strict reasoning + JSON schema constrained generation -- AC-5.2: Strict reasoning + tool call parsing (basic validation only) - -These tests launch a real server with a small model and verify -the constrained decoding pipeline produces valid output. -""" - -import json -import unittest - -import requests - -from sglang.srt.utils import kill_process_tree -from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci -from sglang.test.test_utils import ( - DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH, - CustomTestCase, - popen_launch_server, -) - -register_cuda_ci(est_time=96, stage="base-b", runner_config="1-gpu-small") -register_amd_ci(est_time=120, suite="stage-b-test-1-gpu-small-amd") - -MODEL = "Qwen/Qwen3-0.6B" -BASE_URL = "http://127.0.0.1:39877" -API_KEY = "sk-test-1234" - - -class TestConstrainedReasoningE2E(CustomTestCase): - @classmethod - def setUpClass(cls): - cls.model = MODEL - cls.base_url = BASE_URL - cls.api_key = API_KEY - cls.process = popen_launch_server( - cls.model, - cls.base_url, - timeout=DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH, - api_key=cls.api_key, - other_args=[ - "--reasoning-parser", - "qwen3", - ], - ) - - @classmethod - def tearDownClass(cls): - kill_process_tree(cls.process.pid) - - def _chat(self, **kwargs): - default = { - "model": self.model, - "messages": [ - { - "role": "user", - "content": "What is 2+2? Answer with just the number.", - } - ], - "temperature": 0, - "max_tokens": 256, - } - default.update(kwargs) - resp = requests.post( - f"{self.base_url}/v1/chat/completions", - headers={"Authorization": f"Bearer {self.api_key}"}, - json=default, - timeout=60, - ) - self.assertEqual(resp.status_code, 200, f"Request failed: {resp.text}") - return resp.json() - - def test_reasoning_with_json_schema(self): - """AC-5.1: Reasoning + JSON schema produces valid JSON output.""" - schema = { - "type": "object", - "properties": { - "answer": {"type": "integer"}, - }, - "required": ["answer"], - } - data = self._chat( - response_format={ - "type": "json_schema", - "json_schema": { - "name": "answer_schema", - "schema": schema, - }, - }, - chat_template_kwargs={"enable_thinking": True}, - separate_reasoning=True, - ) - - choice = data["choices"][0] - content = choice["message"]["content"] or "" - - # Content should be valid JSON conforming to schema when non-empty. - # With small models + separate_reasoning, content may be empty if the - # model puts everything in reasoning_content. That's acceptable. - if content.strip(): - try: - parsed = json.loads(content) - self.assertIn("answer", parsed) - self.assertIsInstance(parsed["answer"], int) - except (json.JSONDecodeError, TypeError): - # Small models may produce imperfect JSON - self.assertTrue( - content.strip().startswith("{"), - f"Expected JSON-like output, got: {content!r}", - ) - - # Content should NOT contain tags (those go to reasoning_content) - self.assertNotIn("", content) - - def test_reasoning_disabled_with_json_schema(self): - """JSON schema still works when reasoning is explicitly disabled.""" - schema = { - "type": "object", - "properties": { - "answer": {"type": "integer"}, - }, - "required": ["answer"], - } - data = self._chat( - response_format={ - "type": "json_schema", - "json_schema": { - "name": "answer_schema", - "schema": schema, - }, - }, - chat_template_kwargs={"enable_thinking": False}, - ) - - choice = data["choices"][0] - content = choice["message"]["content"] - - # Should still produce valid JSON - parsed = json.loads(content) - self.assertIn("answer", parsed) - - def test_reasoning_with_separate_output(self): - """Reasoning content is correctly separated from normal content.""" - data = self._chat( - chat_template_kwargs={"enable_thinking": True}, - separate_reasoning=True, - ) - - choice = data["choices"][0] - content = choice["message"]["content"] - reasoning = choice["message"].get("reasoning_content") - - # Content should not contain think tags - self.assertNotIn("", content) - self.assertNotIn("", content) - - def test_tool_call_after_reasoning(self): - """AC-5.2: Tool call parsing works with reasoning enabled.""" - tools = [ - { - "type": "function", - "function": { - "name": "get_weather", - "description": "Get the current weather", - "parameters": { - "type": "object", - "properties": { - "location": {"type": "string"}, - }, - "required": ["location"], - }, - }, - } - ] - data = self._chat( - messages=[ - { - "role": "user", - "content": "What's the weather in Paris?", - } - ], - tools=tools, - chat_template_kwargs={"enable_thinking": True}, - separate_reasoning=True, - ) - - choice = data["choices"][0] - # The model may or may not produce tool calls (depends on model capability) - # but the response should be well-formed (no crashes) - self.assertIn("message", choice) - self.assertIn("finish_reason", choice) - # finish_reason should be either "stop" or "tool_calls" - self.assertIn(choice["finish_reason"], ["stop", "tool_calls", "length"]) - - -class TestStrictThinkingE2E(CustomTestCase): - """E2E tests with --enable-strict-thinking flag. - - Validates that the strict thinking flag is correctly propagated through - the full pipeline: server_args -> grammar_backend -> ReasonerGrammarBackend - -> token filtering during thinking phase. - """ - - @classmethod - def setUpClass(cls): - cls.model = MODEL - cls.base_url = "http://127.0.0.1:39878" - cls.api_key = API_KEY - cls.process = popen_launch_server( - cls.model, - cls.base_url, - timeout=DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH, - api_key=cls.api_key, - other_args=[ - "--reasoning-parser", - "qwen3", - "--enable-strict-thinking", - ], - ) - - @classmethod - def tearDownClass(cls): - kill_process_tree(cls.process.pid) - - def _chat(self, **kwargs): - default = { - "model": self.model, - "messages": [ - { - "role": "user", - "content": "What is 2+2? Answer with just the number.", - } - ], - "temperature": 0, - "max_tokens": 256, - } - default.update(kwargs) - resp = requests.post( - f"{self.base_url}/v1/chat/completions", - headers={"Authorization": f"Bearer {self.api_key}"}, - json=default, - timeout=60, - ) - self.assertEqual(resp.status_code, 200, f"Request failed: {resp.text}") - return resp.json() - - def test_strict_thinking_with_json_schema(self): - """Strict thinking + JSON schema: server starts and produces valid output.""" - schema = { - "type": "object", - "properties": { - "answer": {"type": "integer"}, - }, - "required": ["answer"], - } - data = self._chat( - response_format={ - "type": "json_schema", - "json_schema": { - "name": "answer_schema", - "schema": schema, - }, - }, - chat_template_kwargs={"enable_thinking": True}, - separate_reasoning=True, - ) - - choice = data["choices"][0] - content = choice["message"]["content"] or "" - - if content.strip(): - try: - parsed = json.loads(content) - self.assertIn("answer", parsed) - except (json.JSONDecodeError, TypeError): - self.assertTrue( - content.strip().startswith("{"), - f"Expected JSON-like output, got: {content!r}", - ) - - # Think tags must not leak into content - self.assertNotIn("", content) - - def test_strict_thinking_disabled_per_request(self): - """When thinking is disabled per-request, strict server still works.""" - data = self._chat( - chat_template_kwargs={"enable_thinking": False}, - ) - - choice = data["choices"][0] - self.assertIn("message", choice) - self.assertIn("finish_reason", choice) - # Should complete normally without errors - self.assertIn(choice["finish_reason"], ["stop", "length"]) - - def test_strict_thinking_separate_reasoning(self): - """Strict thinking with separate_reasoning produces well-formed output.""" - data = self._chat( - chat_template_kwargs={"enable_thinking": True}, - separate_reasoning=True, - ) - - choice = data["choices"][0] - content = choice["message"]["content"] or "" - - # Think tags must not leak into content - self.assertNotIn("", content) - self.assertNotIn("", content) - - -if __name__ == "__main__": - unittest.main() diff --git a/test/registered/kernels/test_fused_op_dispatch.py b/test/registered/unit/kernels/test_fused_op_dispatch.py similarity index 88% rename from test/registered/kernels/test_fused_op_dispatch.py rename to test/registered/unit/kernels/test_fused_op_dispatch.py index 4d4aa16b1..fd18fc951 100644 --- a/test/registered/kernels/test_fused_op_dispatch.py +++ b/test/registered/unit/kernels/test_fused_op_dispatch.py @@ -445,16 +445,13 @@ def test_fused_moe_compile_hook_is_bs1_only(): # --- tracing -------------------------------------------------------------------- -def test_trace_labels_platform_and_backend(monkeypatch): +def test_trace_labels_explicit_backend(monkeypatch): _mock_platform(monkeypatch, key="cuda", info=_CUDA) op = _CudaOnlyPlatformOp() fo.enable_fused_op_trace() op(torch.zeros(2, 3)) op(torch.zeros(2, 3), backend=KernelBackend.TORCH) - auto_rec, explicit_rec = fo.get_fused_op_trace() - assert auto_rec.op == "test.cuda_only_platform" - assert auto_rec.backend == "cuda" - assert auto_rec.tensor_args == ("torch.float32[2, 3]",) + _, explicit_rec = fo.get_fused_op_trace() assert explicit_rec.backend == "torch" @@ -512,46 +509,6 @@ def test_deprecated_alias_keeps_legacy_platform_defaults(monkeypatch): _NativeOnlyLegacy()(torch.zeros(1)) # old CUDA behavior preserved -# --- migration completeness ------------------------------------------------------- - -_MIGRATED_OPS = [ - ("sglang.srt.layers.activation", "SiluAndMul"), - ("sglang.srt.layers.activation", "GeluAndMul"), - ("sglang.srt.layers.activation", "NewGELU"), - ("sglang.srt.layers.activation", "ReLU2"), - ("sglang.srt.layers.activation", "QuickGELU"), - ("sglang.srt.layers.activation", "XIELU"), - ("sglang.srt.layers.layernorm", "RMSNorm"), - ("sglang.srt.layers.layernorm", "LayerNorm"), - ("sglang.srt.layers.layernorm", "GemmaRMSNorm"), - ("sglang.srt.layers.layernorm", "Gemma3RMSNorm"), - ("sglang.srt.layers.layernorm", "Gemma4RMSNorm"), - ("sglang.srt.layers.layernorm", "RMSNormWithoutScale"), - ("sglang.srt.layers.conv", "Conv2dLayer"), - ("sglang.srt.layers.conv", "Conv3dLayer"), - ("sglang.srt.layers.moe.topk", "TopK"), - ("sglang.srt.layers.rotary_embedding.base", "RotaryEmbedding"), - ("sglang.srt.layers.rotary_embedding.rope_variant", "DualChunkRotaryEmbedding"), - ("sglang.srt.layers.attention.dsa.dsa_indexer", "Indexer"), - ("sglang.srt.layers.attention.dsv4.compressor", "Compressor"), - ("sglang.srt.layers.attention.mamba.mixer2_rms_norm_gated", "Mixer2RMSNormGated"), - ("sglang.srt.layers.quantization.unquant", "UnquantizedFusedMoEMethod"), -] - - -@pytest.mark.parametrize("module_name, cls_name", _MIGRATED_OPS) -def test_migrated_ops_subclass_base_fused_op(module_name, cls_name): - """Production ops must extend BaseFusedOp directly, never the deprecated - MultiPlatformOp alias (which exists only for out-of-tree users).""" - import importlib - - from sglang.srt.layers.utils.multi_platform import MultiPlatformOp - - cls = getattr(importlib.import_module(module_name), cls_name) - assert issubclass(cls, BaseFusedOp) - assert MultiPlatformOp not in cls.__mro__ - - if __name__ == "__main__": import sys diff --git a/test/registered/kernels/test_jit_cache.py b/test/registered/unit/kernels/test_jit_cache.py similarity index 100% rename from test/registered/kernels/test_jit_cache.py rename to test/registered/unit/kernels/test_jit_cache.py