Enable GPT-OSS TinyGEMM on CUDA 13 (#31649)

This commit is contained in:
Mohammad Miadh Angkad
2026-07-19 19:16:39 -07:00
committed by GitHub
parent bab1dd0d12
commit 8bf2ab9be9
+1 -2
View File
@@ -72,7 +72,6 @@ from sglang.srt.runtime_context import get_forward, get_parallel, get_server_arg
from sglang.srt.utils import ( from sglang.srt.utils import (
LazyValue, LazyValue,
add_prefix, add_prefix,
get_cuda_version,
is_blackwell_supported, is_blackwell_supported,
is_cpu, is_cpu,
is_cuda, is_cuda,
@@ -94,7 +93,7 @@ _is_tinygemm_supported = (
and (is_sm90_supported() or is_blackwell_supported()) and (is_sm90_supported() or is_blackwell_supported())
) )
if _is_tinygemm_supported and get_cuda_version()[0] < 13: if _is_tinygemm_supported:
try: try:
from flashinfer.gemm import tinygemm_bf16 from flashinfer.gemm import tinygemm_bf16
except ImportError: except ImportError: