Co-authored-by: Claude Opus 4.8 (1M context) <noreply@anthropic.com> Co-authored-by: Mohammad Miadh Angkad <176301910+mmangkad@users.noreply.github.com>
This commit is contained in:
co-authored by
Claude Opus 4.8
Mohammad Miadh Angkad
parent
dad0120c57
commit
dbb51c46ac
@@ -4,13 +4,18 @@ Pre-commit hook: validate CI registry calls under test/registered/.
|
|||||||
|
|
||||||
1. Every test file must contain a CI registry call (register_cuda_ci,
|
1. Every test file must contain a CI registry call (register_cuda_ci,
|
||||||
register_amd_ci, etc.).
|
register_amd_ci, etc.).
|
||||||
2. A CUDA test must not register a `{stage}-test-{runner_config}`-shaped suite
|
2. A CUDA test must register its PR-test suite via the modern
|
||||||
via the legacy single-string `suite=`: that form is not dispatchable via
|
`stage=`/`runner_config=` form. The legacy single-string `suite=` is reserved
|
||||||
/rerun-test. Use the modern `stage=`/`runner_config=` form instead -- it
|
for the nightly/stress/weekly families (and for AMD/CPU/NPU suites); any other
|
||||||
resolves to the identical suite (CIRegistry.effective_suite is
|
CUDA `suite=` resolves to a name no PR-test workflow invokes, so the test
|
||||||
f"{stage}-test-{runner_config}"), so same stage and same runner, just
|
silently never runs. Two shapes are rejected:
|
||||||
/rerun-test-able. Legacy `suite=` stays valid for nightly/stress/weekly +
|
a. `{stage}-test-{runner_config}` -- the modern name stuffed back into the
|
||||||
AMD/CPU/NPU suites whose names don't follow that shape.
|
legacy form. Reported with the exact stage/runner split to use.
|
||||||
|
b. an older `{stage}-{runner_config}` PR-test name (e.g. the pre-migration
|
||||||
|
`base-b-kernel-unit-1-gpu-large`) -- no longer matches any workflow
|
||||||
|
suite at all.
|
||||||
|
The modern form resolves to the identical suite (CIRegistry.effective_suite
|
||||||
|
is f"{stage}-test-{runner_config}") and is /rerun-test-able.
|
||||||
|
|
||||||
Reuses ut_parse_one_file() from ci_register.py (AST-based parsing)
|
Reuses ut_parse_one_file() from ci_register.py (AST-based parsing)
|
||||||
to match the same logic used by run_suite.py's collect_tests().
|
to match the same logic used by run_suite.py's collect_tests().
|
||||||
@@ -27,6 +32,12 @@ import sys
|
|||||||
# shape is always expressible (and should be expressed) the modern way.
|
# shape is always expressible (and should be expressed) the modern way.
|
||||||
_MODERN_SHAPE = re.compile(r"^(.+)-test-(.+)$")
|
_MODERN_SHAPE = re.compile(r"^(.+)-test-(.+)$")
|
||||||
|
|
||||||
|
# The only suite families a CUDA registry may keep on the legacy single-string
|
||||||
|
# `suite=` form. Everything else is a PR-test/base stage that must use the
|
||||||
|
# modern stage=/runner_config= form (otherwise its effective_suite matches no
|
||||||
|
# suite the PR-test workflows invoke, and the test silently never runs).
|
||||||
|
_LEGACY_CUDA_PREFIXES = ("nightly", "stress", "weekly")
|
||||||
|
|
||||||
|
|
||||||
def main() -> int:
|
def main() -> int:
|
||||||
# Import ci_register directly to avoid pulling in all of sglang
|
# Import ci_register directly to avoid pulling in all of sglang
|
||||||
@@ -48,7 +59,8 @@ def main() -> int:
|
|||||||
return 0
|
return 0
|
||||||
|
|
||||||
missing = []
|
missing = []
|
||||||
legacy_shape = [] # (file, suite, stage, runner_config)
|
legacy_shape = [] # (file, suite, stage, runner_config) -- has a -test- split
|
||||||
|
non_dispatchable = [] # (file, suite) -- legacy CUDA suite no workflow invokes
|
||||||
for f in files:
|
for f in files:
|
||||||
try:
|
try:
|
||||||
registries, _has_main_entry = ci_register.ut_parse_one_file(f)
|
registries, _has_main_entry = ci_register.ut_parse_one_file(f)
|
||||||
@@ -67,9 +79,15 @@ def main() -> int:
|
|||||||
and r.runner_config is None
|
and r.runner_config is None
|
||||||
):
|
):
|
||||||
continue
|
continue
|
||||||
|
# nightly/stress/weekly are the only CUDA suites allowed to stay on
|
||||||
|
# the legacy single-string form.
|
||||||
|
if r.suite.split("-", 1)[0] in _LEGACY_CUDA_PREFIXES:
|
||||||
|
continue
|
||||||
m = _MODERN_SHAPE.match(r.suite)
|
m = _MODERN_SHAPE.match(r.suite)
|
||||||
if m:
|
if m:
|
||||||
legacy_shape.append((f, r.suite, m.group(1), m.group(2)))
|
legacy_shape.append((f, r.suite, m.group(1), m.group(2)))
|
||||||
|
else:
|
||||||
|
non_dispatchable.append((f, r.suite))
|
||||||
|
|
||||||
exit_code = 0
|
exit_code = 0
|
||||||
if missing:
|
if missing:
|
||||||
@@ -94,6 +112,21 @@ def main() -> int:
|
|||||||
)
|
)
|
||||||
print()
|
print()
|
||||||
exit_code = 1
|
exit_code = 1
|
||||||
|
if non_dispatchable:
|
||||||
|
print(
|
||||||
|
'ERROR: CUDA test(s) register a legacy `suite="..."` that is neither a '
|
||||||
|
"nightly/stress/weekly suite nor the modern `stage=`/`runner_config=` "
|
||||||
|
"form. This name matches no suite the PR-test workflows invoke, so the "
|
||||||
|
"test silently never runs. Switch to the modern form:\n"
|
||||||
|
)
|
||||||
|
for f, suite in non_dispatchable:
|
||||||
|
print(
|
||||||
|
f" {f}\n"
|
||||||
|
f' suite="{suite}"'
|
||||||
|
f' -> stage="...", runner_config="..."'
|
||||||
|
)
|
||||||
|
print()
|
||||||
|
exit_code = 1
|
||||||
|
|
||||||
return exit_code
|
return exit_code
|
||||||
|
|
||||||
|
|||||||
@@ -29,7 +29,8 @@ from sglang.test.ci.ci_register import register_cuda_ci
|
|||||||
|
|
||||||
register_cuda_ci(
|
register_cuda_ci(
|
||||||
est_time=120,
|
est_time=120,
|
||||||
suite="base-b-kernel-unit-8-gpu-h200",
|
stage="extra-b",
|
||||||
|
runner_config="8-gpu-h200",
|
||||||
)
|
)
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
|
|||||||
@@ -20,7 +20,9 @@ from sglang.srt.utils.common import is_sm120_supported
|
|||||||
from sglang.test.ci.ci_register import register_cuda_ci
|
from sglang.test.ci.ci_register import register_cuda_ci
|
||||||
from sglang.utils import is_in_ci
|
from sglang.utils import is_in_ci
|
||||||
|
|
||||||
register_cuda_ci(est_time=12, suite="base-b-kernel-benchmark-1-gpu-large")
|
register_cuda_ci(
|
||||||
|
est_time=12, stage="base-b-kernel-benchmark", runner_config="1-gpu-large"
|
||||||
|
)
|
||||||
|
|
||||||
IS_CI = is_in_ci()
|
IS_CI = is_in_ci()
|
||||||
|
|
||||||
|
|||||||
@@ -18,7 +18,9 @@ from sglang.jit_kernel.dsv3_router_gemm import dsv3_router_gemm
|
|||||||
from sglang.jit_kernel.utils import get_jit_cuda_arch, is_hip_runtime
|
from sglang.jit_kernel.utils import get_jit_cuda_arch, is_hip_runtime
|
||||||
from sglang.test.ci.ci_register import register_cuda_ci
|
from sglang.test.ci.ci_register import register_cuda_ci
|
||||||
|
|
||||||
register_cuda_ci(est_time=5, suite="base-b-kernel-benchmark-1-gpu-large")
|
register_cuda_ci(
|
||||||
|
est_time=5, stage="base-b-kernel-benchmark", runner_config="1-gpu-large"
|
||||||
|
)
|
||||||
|
|
||||||
# sgl_kernel AOT kernel is specialized for hidden_dim=7168 only.
|
# sgl_kernel AOT kernel is specialized for hidden_dim=7168 only.
|
||||||
SGL_KERNEL_HIDDEN_DIM = 7168
|
SGL_KERNEL_HIDDEN_DIM = 7168
|
||||||
|
|||||||
@@ -20,7 +20,9 @@ except ImportError:
|
|||||||
flash_mla_sparse_fwd = None
|
flash_mla_sparse_fwd = None
|
||||||
HAS_Q16_FLASHMLA = False
|
HAS_Q16_FLASHMLA = False
|
||||||
|
|
||||||
register_cuda_ci(est_time=120, suite="base-b-kernel-benchmark-1-gpu-large")
|
register_cuda_ci(
|
||||||
|
est_time=120, stage="base-b-kernel-benchmark", runner_config="1-gpu-large"
|
||||||
|
)
|
||||||
|
|
||||||
IS_CI = is_in_ci()
|
IS_CI = is_in_ci()
|
||||||
DTYPE_FP8 = torch.float8_e4m3fn
|
DTYPE_FP8 = torch.float8_e4m3fn
|
||||||
|
|||||||
@@ -37,7 +37,8 @@ from sglang.test.ci.ci_register import register_cuda_ci
|
|||||||
|
|
||||||
register_cuda_ci(
|
register_cuda_ci(
|
||||||
est_time=120,
|
est_time=120,
|
||||||
suite="base-b-kernel-benchmark-1-gpu-large",
|
stage="base-b-kernel-benchmark",
|
||||||
|
runner_config="1-gpu-large",
|
||||||
disabled="requires multi-GPU, self-skips in CI",
|
disabled="requires multi-GPU, self-skips in CI",
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|||||||
@@ -13,7 +13,8 @@ from sglang.test.ci.ci_register import register_cuda_ci
|
|||||||
|
|
||||||
register_cuda_ci(
|
register_cuda_ci(
|
||||||
est_time=20,
|
est_time=20,
|
||||||
suite="base-b-kernel-benchmark-1-gpu-large",
|
stage="base-b-kernel-benchmark",
|
||||||
|
runner_config="1-gpu-large",
|
||||||
disabled="standalone benchmark",
|
disabled="standalone benchmark",
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|||||||
@@ -9,7 +9,9 @@ from sglang.jit_kernel.diffusion.triton.scale_shift import fuse_scale_shift_kern
|
|||||||
from sglang.test.ci.ci_register import register_cuda_ci
|
from sglang.test.ci.ci_register import register_cuda_ci
|
||||||
from sglang.utils import is_in_ci
|
from sglang.utils import is_in_ci
|
||||||
|
|
||||||
register_cuda_ci(est_time=30, suite="base-b-kernel-benchmark-1-gpu-large")
|
register_cuda_ci(
|
||||||
|
est_time=30, stage="base-b-kernel-benchmark", runner_config="1-gpu-large"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
@dataclass(frozen=True)
|
@dataclass(frozen=True)
|
||||||
|
|||||||
@@ -12,8 +12,8 @@ from sglang.jit_kernel.diffusion.triton.causal_conv3d_pad import (
|
|||||||
from sglang.jit_kernel.utils import get_ci_test_range
|
from sglang.jit_kernel.utils import get_ci_test_range
|
||||||
from sglang.test.ci.ci_register import register_cuda_ci
|
from sglang.test.ci.ci_register import register_cuda_ci
|
||||||
|
|
||||||
register_cuda_ci(est_time=45, suite="base-b-kernel-unit-1-gpu-large")
|
register_cuda_ci(est_time=45, stage="base-b-kernel-unit", runner_config="1-gpu-large")
|
||||||
register_cuda_ci(est_time=45, suite="base-b-kernel-unit-1-gpu-b200")
|
register_cuda_ci(est_time=45, stage="base-b-kernel-unit", runner_config="4-gpu-b200")
|
||||||
|
|
||||||
DEVICE = "cuda"
|
DEVICE = "cuda"
|
||||||
DTYPE = torch.bfloat16
|
DTYPE = torch.bfloat16
|
||||||
|
|||||||
@@ -6,7 +6,7 @@ import torch
|
|||||||
from sglang.jit_kernel.diffusion.triton.ltx2_ada_values import ltx2_ada_values9
|
from sglang.jit_kernel.diffusion.triton.ltx2_ada_values import ltx2_ada_values9
|
||||||
from sglang.test.ci.ci_register import register_cuda_ci
|
from sglang.test.ci.ci_register import register_cuda_ci
|
||||||
|
|
||||||
register_cuda_ci(est_time=8, suite="base-b-kernel-unit-1-gpu-large")
|
register_cuda_ci(est_time=8, stage="base-b-kernel-unit", runner_config="1-gpu-large")
|
||||||
|
|
||||||
DEVICE = "cuda"
|
DEVICE = "cuda"
|
||||||
|
|
||||||
|
|||||||
@@ -9,8 +9,8 @@ from sglang.jit_kernel.diffusion.residual_gate_add import (
|
|||||||
)
|
)
|
||||||
from sglang.test.ci.ci_register import register_cuda_ci
|
from sglang.test.ci.ci_register import register_cuda_ci
|
||||||
|
|
||||||
register_cuda_ci(est_time=30, suite="base-b-kernel-unit-1-gpu-large")
|
register_cuda_ci(est_time=30, stage="base-b-kernel-unit", runner_config="1-gpu-large")
|
||||||
register_cuda_ci(est_time=30, suite="base-b-kernel-unit-1-gpu-b200")
|
register_cuda_ci(est_time=30, stage="base-b-kernel-unit", runner_config="4-gpu-b200")
|
||||||
|
|
||||||
|
|
||||||
CASES = [
|
CASES = [
|
||||||
|
|||||||
@@ -9,7 +9,7 @@ from sglang.jit_kernel.cutedsl_dsv3_fused_a_gemm import dsv3_fused_a_gemm
|
|||||||
from sglang.jit_kernel.utils import get_jit_cuda_arch, is_hip_runtime
|
from sglang.jit_kernel.utils import get_jit_cuda_arch, is_hip_runtime
|
||||||
from sglang.test.ci.ci_register import register_cuda_ci
|
from sglang.test.ci.ci_register import register_cuda_ci
|
||||||
|
|
||||||
register_cuda_ci(est_time=30, suite="base-b-kernel-unit-1-gpu-large")
|
register_cuda_ci(est_time=30, stage="base-b-kernel-unit", runner_config="1-gpu-large")
|
||||||
|
|
||||||
# hd_in must be a multiple of 256; 6144/7168 cover the real fused-A shapes.
|
# hd_in must be a multiple of 256; 6144/7168 cover the real fused-A shapes.
|
||||||
HD_INS = [6144, 7168]
|
HD_INS = [6144, 7168]
|
||||||
|
|||||||
@@ -31,7 +31,7 @@ from sglang.test.ci.ci_register import register_cuda_ci
|
|||||||
|
|
||||||
_is_hip = is_hip()
|
_is_hip = is_hip()
|
||||||
|
|
||||||
register_cuda_ci(est_time=45, suite="base-b-kernel-unit-1-gpu-large")
|
register_cuda_ci(est_time=45, stage="base-b-kernel-unit", runner_config="1-gpu-large")
|
||||||
register_cuda_ci(est_time=90, suite="nightly-kernel-1-gpu", nightly=True)
|
register_cuda_ci(est_time=90, suite="nightly-kernel-1-gpu", nightly=True)
|
||||||
|
|
||||||
HEAD_DIM = 128
|
HEAD_DIM = 128
|
||||||
|
|||||||
@@ -10,7 +10,7 @@ from sglang.jit_kernel.dsv3_fused_a_gemm import dsv3_fused_a_gemm
|
|||||||
from sglang.jit_kernel.utils import get_jit_cuda_arch, is_hip_runtime
|
from sglang.jit_kernel.utils import get_jit_cuda_arch, is_hip_runtime
|
||||||
from sglang.test.ci.ci_register import register_cuda_ci
|
from sglang.test.ci.ci_register import register_cuda_ci
|
||||||
|
|
||||||
register_cuda_ci(est_time=30, suite="base-b-kernel-unit-1-gpu-large")
|
register_cuda_ci(est_time=30, stage="base-b-kernel-unit", runner_config="1-gpu-large")
|
||||||
|
|
||||||
# hd_in must be a multiple of 256; 6144/7168 cover the real fused-A shapes.
|
# hd_in must be a multiple of 256; 6144/7168 cover the real fused-A shapes.
|
||||||
HD_INS = [6144, 7168]
|
HD_INS = [6144, 7168]
|
||||||
|
|||||||
@@ -9,7 +9,7 @@ from sglang.jit_kernel.dsv3_router_gemm import dsv3_router_gemm
|
|||||||
from sglang.jit_kernel.utils import get_jit_cuda_arch, is_hip_runtime
|
from sglang.jit_kernel.utils import get_jit_cuda_arch, is_hip_runtime
|
||||||
from sglang.test.ci.ci_register import register_cuda_ci
|
from sglang.test.ci.ci_register import register_cuda_ci
|
||||||
|
|
||||||
register_cuda_ci(est_time=37, suite="base-b-kernel-unit-1-gpu-large")
|
register_cuda_ci(est_time=37, stage="base-b-kernel-unit", runner_config="1-gpu-large")
|
||||||
register_cuda_ci(est_time=148, suite="nightly-kernel-1-gpu", nightly=True)
|
register_cuda_ci(est_time=148, suite="nightly-kernel-1-gpu", nightly=True)
|
||||||
|
|
||||||
HIDDEN_DIMS = [1024, 4096, 5120, 6144, 7168]
|
HIDDEN_DIMS = [1024, 4096, 5120, 6144, 7168]
|
||||||
|
|||||||
@@ -47,7 +47,7 @@ from sglang.test.ci.ci_register import register_cuda_ci
|
|||||||
|
|
||||||
# The NVFP4 expert-quant kernels are Blackwell-only (sm100a), so this runs on
|
# The NVFP4 expert-quant kernels are Blackwell-only (sm100a), so this runs on
|
||||||
# the B200 unit suite.
|
# the B200 unit suite.
|
||||||
register_cuda_ci(est_time=20, suite="base-b-kernel-unit-1-gpu-b200")
|
register_cuda_ci(est_time=20, stage="base-b-kernel-unit", runner_config="4-gpu-b200")
|
||||||
|
|
||||||
FLOAT8_E4M3_MAX = 448.0
|
FLOAT8_E4M3_MAX = 448.0
|
||||||
FLOAT4_E2M1_MAX = 6.0
|
FLOAT4_E2M1_MAX = 6.0
|
||||||
|
|||||||
@@ -9,7 +9,7 @@ import torch
|
|||||||
from sglang.srt.utils import is_sm90_supported
|
from sglang.srt.utils import is_sm90_supported
|
||||||
from sglang.test.ci.ci_register import register_cuda_ci
|
from sglang.test.ci.ci_register import register_cuda_ci
|
||||||
|
|
||||||
register_cuda_ci(est_time=120, suite="base-b-kernel-unit-1-gpu-large")
|
register_cuda_ci(est_time=120, stage="base-b-kernel-unit", runner_config="1-gpu-large")
|
||||||
register_cuda_ci(est_time=300, suite="nightly-kernel-1-gpu", nightly=True)
|
register_cuda_ci(est_time=300, suite="nightly-kernel-1-gpu", nightly=True)
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -34,7 +34,7 @@ from sglang.srt.distributed.device_communicators.triton_symm_mem_ag import (
|
|||||||
)
|
)
|
||||||
from sglang.test.ci.ci_register import register_cuda_ci
|
from sglang.test.ci.ci_register import register_cuda_ci
|
||||||
|
|
||||||
register_cuda_ci(est_time=240, suite="base-b-kernel-unit-8-gpu-h200")
|
register_cuda_ci(est_time=240, stage="base-b-kernel-unit", runner_config="8-gpu-h200")
|
||||||
register_cuda_ci(est_time=240, suite="nightly-kernel-8-gpu-h200", nightly=True)
|
register_cuda_ci(est_time=240, suite="nightly-kernel-8-gpu-h200", nightly=True)
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
|
|||||||
Reference in New Issue
Block a user