[NVIDIA] Enable CuTe DSL BF16 GEMM on SM107 (#33617)

Co-authored-by: Lee Nau <lnau@nvidia.com>
This commit is contained in:
YAMY
2026-08-06 02:06:08 -07:00
committed by GitHub
co-authored by Lee Nau
parent dfe53232d7
commit 8b29c90218
4 changed files with 20 additions and 19 deletions
@@ -23,32 +23,33 @@ from sglang.kernels.ops.gemm.cutedsl_bf16_gemm import ( # noqa: E402
use_cutedsl_bf16_gemm,
)
N_VALUES = [1024, 2624, 6144]
K_VALUES = [2048, 6144]
SHAPES = [(n, k) for n in [1024, 2624, 6144] for k in [2048, 6144]] + [(2048, 4096)]
NUM_TOKENS = get_ci_test_range(list(range(1, 33)), [1, 15, 16, 32])
@pytest.mark.skipif(not torch.cuda.is_available(), reason="CUDA required")
@pytest.mark.parametrize("has_bias", [False, True])
@pytest.mark.parametrize("n", N_VALUES)
@pytest.mark.parametrize("k", K_VALUES)
@pytest.mark.parametrize("n,k", SHAPES)
@pytest.mark.parametrize("num_tokens", NUM_TOKENS)
def test_cutedsl_bf16_gemm(num_tokens, k, n, has_bias):
if is_hip_runtime() or get_jit_cuda_arch().major != 10:
pytest.skip("SM100/SM103 required")
pytest.skip("SM10x required")
torch.manual_seed(num_tokens)
x = torch.randn(num_tokens, k, dtype=torch.bfloat16, device="cuda")
weight = torch.randn(n, k, dtype=torch.bfloat16, device="cuda")
bias = torch.randn(n, dtype=torch.bfloat16, device="cuda") if has_bias else None
if bias is not None:
bias.requires_grad_(True)
out = cutedsl_bf16_gemm(x, weight, bias)
with torch.no_grad():
out = cutedsl_bf16_gemm(x, weight, bias)
assert out.shape == (num_tokens, n)
assert out.dtype == torch.bfloat16
ref = x.float() @ weight.float().T
if bias is not None:
ref = ref + bias.float()
ref = ref + bias.detach().float()
torch.testing.assert_close(out, ref.bfloat16(), rtol=2e-2, atol=2.5)
@@ -65,7 +66,7 @@ def test_cutedsl_bf16_gemm_empty_batch(has_bias):
"""Empty input must yield the empty [0, N] output, mirroring F.linear —
launching TGV with a 0-CTA grid fails with CUDA_ERROR_INVALID_VALUE."""
if is_hip_runtime() or get_jit_cuda_arch().major != 10:
pytest.skip("SM100/SM103 required")
pytest.skip("SM10x required")
n, k = 6144, 2048
x = torch.empty(0, k, dtype=torch.bfloat16, device="cuda")