[Ascend] Add Ascend NPU support for sglang.check_env & rework proposal (#11052)

Co-authored-by: ronnie_zheng <zl19940307@163.com>
This commit is contained in:
Zhihao Lyu
2025-11-01 19:26:45 -07:00
committed by GitHub
co-authored by ronnie_zheng
parent 086f0b79fc
commit c550ab9125
+251 -131
View File
@@ -5,11 +5,12 @@ import os
import resource import resource
import subprocess import subprocess
import sys import sys
from abc import abstractmethod
from collections import OrderedDict, defaultdict from collections import OrderedDict, defaultdict
import torch import torch
from sglang.srt.utils import is_hip from sglang.srt.utils import is_hip, is_npu
def is_cuda_v2(): def is_cuda_v2():
@@ -51,12 +52,29 @@ PACKAGE_LIST = [
] ]
def get_package_versions(packages): class BaseEnv:
"""Base class for environment check"""
def __init__(self):
self.package_list = PACKAGE_LIST
@abstractmethod
def get_info(self) -> dict:
"""
Get CUDA-related information if available.
"""
raise NotImplementedError
@abstractmethod
def get_topology(self) -> dict:
raise NotImplementedError
def get_package_versions(self) -> dict:
""" """
Get versions of specified packages. Get versions of specified packages.
""" """
versions = {} versions = {}
for package in packages: for package in self.package_list:
package_name = package.split("==")[0].split(">=")[0].split("<=")[0] package_name = package.split("==")[0].split(">=")[0].split("<=")[0]
try: try:
version = importlib.metadata.version(package_name) version = importlib.metadata.version(package_name)
@@ -65,32 +83,9 @@ def get_package_versions(packages):
versions[package_name] = "Module Not Found" versions[package_name] = "Module Not Found"
return versions return versions
def get_device_info(self):
def get_cuda_info():
""" """
Get CUDA-related information if available. Get information about available GPU devices.
"""
if is_cuda_v2():
cuda_info = {"CUDA available": torch.cuda.is_available()}
if cuda_info["CUDA available"]:
cuda_info.update(_get_gpu_info())
cuda_info.update(_get_cuda_version_info())
return cuda_info
elif is_hip():
cuda_info = {"ROCM available": torch.cuda.is_available()}
if cuda_info["ROCM available"]:
cuda_info.update(_get_gpu_info())
cuda_info.update(_get_cuda_version_info())
return cuda_info
def _get_gpu_info():
"""
Get information about available GPUs.
""" """
devices = defaultdict(list) devices = defaultdict(list)
capabilities = defaultdict(list) capabilities = defaultdict(list)
@@ -114,41 +109,67 @@ def _get_gpu_info():
return gpu_info return gpu_info
def get_hypervisor_vendor(self) -> dict:
try:
output = subprocess.check_output(["lscpu"], text=True)
for line in output.split("\n"):
if "Hypervisor vendor:" in line:
return {"Hypervisor vendor:": line.split(":")[1].strip()}
return {}
except:
return {}
def _get_cuda_version_info(): def get_ulimit_soft(self) -> dict:
ulimit_soft, _ = resource.getrlimit(resource.RLIMIT_NOFILE)
return {"ulimit soft": ulimit_soft}
def check_env(self):
"""
Check and print environment information.
"""
env_info = OrderedDict()
env_info["Python"] = sys.version.replace("\n", "")
env_info.update(self.get_info())
env_info["PyTorch"] = torch.__version__
env_info.update(self.get_package_versions())
env_info.update(self.get_topology())
env_info.update(self.get_hypervisor_vendor())
env_info.update(self.get_ulimit_soft())
for k, v in env_info.items():
print(f"{k}: {v}")
class GPUEnv(BaseEnv):
"""Environment checker for Nvidia GPU"""
def get_info(self):
cuda_info = {"CUDA available": torch.cuda.is_available()}
if cuda_info["CUDA available"]:
cuda_info.update(self.get_device_info())
cuda_info.update(self._get_cuda_version_info())
return cuda_info
def _get_cuda_version_info(self):
""" """
Get CUDA version information. Get CUDA version information.
""" """
if is_cuda_v2():
from torch.utils.cpp_extension import CUDA_HOME from torch.utils.cpp_extension import CUDA_HOME
cuda_info = {"CUDA_HOME": CUDA_HOME} cuda_info = {"CUDA_HOME": CUDA_HOME}
if CUDA_HOME and os.path.isdir(CUDA_HOME): if CUDA_HOME and os.path.isdir(CUDA_HOME):
cuda_info.update(_get_nvcc_info()) cuda_info.update(self._get_nvcc_info())
cuda_info.update(_get_cuda_driver_version()) cuda_info.update(self._get_cuda_driver_version())
return cuda_info return cuda_info
elif is_hip():
from torch.utils.cpp_extension import ROCM_HOME as ROCM_HOME
cuda_info = {"ROCM_HOME": ROCM_HOME} def _get_nvcc_info(self):
if ROCM_HOME and os.path.isdir(ROCM_HOME):
cuda_info.update(_get_nvcc_info())
cuda_info.update(_get_cuda_driver_version())
return cuda_info
else:
cuda_info = {"CUDA_HOME": ""}
return cuda_info
def _get_nvcc_info():
""" """
Get NVCC version information. Get NVCC version information.
""" """
if is_cuda_v2():
from torch.utils.cpp_extension import CUDA_HOME from torch.utils.cpp_extension import CUDA_HOME
try: try:
@@ -167,7 +188,73 @@ def _get_nvcc_info():
} }
except subprocess.SubprocessError: except subprocess.SubprocessError:
return {"NVCC": "Not Available"} return {"NVCC": "Not Available"}
elif is_hip():
def _get_cuda_driver_version(self):
"""
Get CUDA driver version.
"""
versions = set()
try:
output = subprocess.check_output(
[
"nvidia-smi",
"--query-gpu=driver_version",
"--format=csv,noheader,nounits",
]
)
versions = set(output.decode().strip().split("\n"))
if len(versions) == 1:
return {"CUDA Driver Version": versions.pop()}
else:
return {"CUDA Driver Versions": ", ".join(sorted(versions))}
except subprocess.SubprocessError:
return {"CUDA Driver Version": "Not Available"}
def get_topology(self):
"""
Get GPU topology information.
"""
try:
result = subprocess.run(
["nvidia-smi", "topo", "-m"],
stdout=subprocess.PIPE,
stderr=subprocess.PIPE,
text=True,
check=True,
)
return {
"NVIDIA Topology": (
"\n" + result.stdout if result.returncode == 0 else None
)
}
except subprocess.SubprocessError:
return {}
class HIPEnv(BaseEnv):
"""Environment checker for ROCm/HIP"""
def get_info(self):
cuda_info = {"ROCM available": torch.cuda.is_available()}
if cuda_info["ROCM available"]:
cuda_info.update(self.get_device_info())
cuda_info.update(self._get_cuda_version_info())
return cuda_info
def _get_cuda_version_info(self):
from torch.utils.cpp_extension import ROCM_HOME as ROCM_HOME
cuda_info = {"ROCM_HOME": ROCM_HOME}
if ROCM_HOME and os.path.isdir(ROCM_HOME):
cuda_info.update(self._get_hipcc_info())
cuda_info.update(self._get_rocm_driver_version())
return cuda_info
def _get_hipcc_info(self):
from torch.utils.cpp_extension import ROCM_HOME from torch.utils.cpp_extension import ROCM_HOME
try: try:
@@ -184,32 +271,8 @@ def _get_nvcc_info():
} }
except subprocess.SubprocessError: except subprocess.SubprocessError:
return {"HIPCC": "Not Available"} return {"HIPCC": "Not Available"}
else:
return {"NVCC": "Not Available"}
def _get_rocm_driver_version(self):
def _get_cuda_driver_version():
"""
Get CUDA driver version.
"""
versions = set()
if is_cuda_v2():
try:
output = subprocess.check_output(
[
"nvidia-smi",
"--query-gpu=driver_version",
"--format=csv,noheader,nounits",
]
)
versions = set(output.decode().strip().split("\n"))
if len(versions) == 1:
return {"CUDA Driver Version": versions.pop()}
else:
return {"CUDA Driver Versions": ", ".join(sorted(versions))}
except subprocess.SubprocessError:
return {"CUDA Driver Version": "Not Available"}
elif is_hip():
try: try:
output = subprocess.check_output( output = subprocess.check_output(
[ [
@@ -226,27 +289,8 @@ def _get_cuda_driver_version():
return {"ROCM Driver Version": ver} return {"ROCM Driver Version": ver}
except subprocess.SubprocessError: except subprocess.SubprocessError:
return {"ROCM Driver Version": "Not Available"} return {"ROCM Driver Version": "Not Available"}
else:
return {"CUDA Driver Version": "Not Available"}
def get_topology(self):
def get_gpu_topology():
"""
Get GPU topology information.
"""
if is_cuda_v2():
try:
result = subprocess.run(
["nvidia-smi", "topo", "-m"],
stdout=subprocess.PIPE,
stderr=subprocess.PIPE,
text=True,
check=True,
)
return "\n" + result.stdout if result.returncode == 0 else None
except subprocess.SubprocessError:
return None
elif is_hip():
try: try:
result = subprocess.run( result = subprocess.run(
["rocm-smi", "--showtopotype"], ["rocm-smi", "--showtopotype"],
@@ -255,51 +299,127 @@ def get_gpu_topology():
text=True, text=True,
check=True, check=True,
) )
return "\n" + result.stdout if result.returncode == 0 else None return {
"AMD Topology": "\n" + result.stdout if result.returncode == 0 else None
}
except subprocess.SubprocessError: except subprocess.SubprocessError:
return None return {}
class NPUEnv(BaseEnv):
"""Environment checker for Ascend NPU"""
def __init__(self):
super().__init__()
self.package_list = ["torch_npu", "sgl-kernel-npu"] + self.package_list
def get_info(self):
cuda_info = {"NPU available": torch.npu.is_available()}
if cuda_info["NPU available"]:
cuda_info.update(self.get_device_info())
cuda_info.update(self._get_cann_version_info())
return cuda_info
def get_device_info(self):
"""
Get information about available NPUs.
Need to override due to torch_npu interface differences.
"""
devices = defaultdict(list)
for k in range(torch.npu.device_count()):
devices[torch.npu.get_device_name(k)].append(str(k))
npu_info = {}
for name, device_ids in devices.items():
npu_info[f"NPU {','.join(device_ids)}"] = name
return npu_info
def _get_cann_version_info(self):
cann_envs = ["ASCEND_TOOLKIT_HOME", "ASCEND_INSTALL_PATH"]
for var in cann_envs:
path = os.environ.get(var)
if path and os.path.exists(path):
CANN_HOME = path
break
else: else:
return None default_path = "/usr/local/Ascend/ascend-toolkit/latest"
CANN_HOME = default_path if os.path.exists(default_path) else None
if CANN_HOME:
npu_info = {"CANN_HOME": CANN_HOME}
npu_info.update(self._get_cann_info(CANN_HOME))
npu_info.update(self._get_ascend_driver_version())
return npu_info
else:
return {"CANN_HOME": "Not found"}
def get_hypervisor_vendor(): def _get_cann_info(self, CANN_HOME: str):
cann_info = {}
cann_version_file = os.path.join(CANN_HOME, "version.cfg")
if os.path.exists(cann_version_file):
with open(cann_version_file, "r", encoding="utf-8") as f:
f.readline() # discard first line comment in version.cfg
cann_info["CANN"] = f.readline().split("[")[1].split("]")[0]
else:
cann_info["CANN"] = "Not Available"
try: try:
output = subprocess.check_output(["lscpu"], text=True) bisheng = os.path.join(CANN_HOME, "compiler/ccec_compiler/bin/bisheng")
for line in output.split("\n"): bisheng_output = (
if "Hypervisor vendor:" in line: subprocess.check_output([bisheng, "--version"]).decode("utf-8").strip()
return line.split(":")[1].strip() )
return None cann_info["BiSheng"] = bisheng_output.split("\n")[0].strip()
except: except subprocess.SubprocessError:
return None cann_info["BiSheng"] = "Not Available"
return cann_info
def _get_ascend_driver_version(self):
try:
output = subprocess.check_output(
[
"npu-smi",
"info",
"-t",
"board",
"-i",
"0",
]
)
for line in output.decode().strip().split("\n"):
if "Software Version" in line:
version = line.split(":")[-1].strip()
break
else:
version = "Not Available"
def check_env(): return {"Ascend Driver Version": version}
""" except subprocess.SubprocessError:
Check and print environment information. return {"Ascend Driver Version": "Not Available"}
"""
env_info = OrderedDict()
env_info["Python"] = sys.version.replace("\n", "")
env_info.update(get_cuda_info())
env_info["PyTorch"] = torch.__version__
env_info.update(get_package_versions(PACKAGE_LIST))
gpu_topo = get_gpu_topology() def get_topology(self):
if gpu_topo: try:
if is_cuda_v2(): result = subprocess.run(
env_info["NVIDIA Topology"] = gpu_topo ["npu-smi", "info", "-t", "topo"],
elif is_hip(): stdout=subprocess.PIPE,
env_info["AMD Topology"] = gpu_topo stderr=subprocess.PIPE,
text=True,
hypervisor_vendor = get_hypervisor_vendor() check=True,
if hypervisor_vendor: )
env_info["Hypervisor vendor"] = hypervisor_vendor return {
"Ascend Topology": (
ulimit_soft, _ = resource.getrlimit(resource.RLIMIT_NOFILE) "\n" + result.stdout if result.returncode == 0 else None
env_info["ulimit soft"] = ulimit_soft )
}
for k, v in env_info.items(): except subprocess.SubprocessError:
print(f"{k}: {v}") return {}
if __name__ == "__main__": if __name__ == "__main__":
check_env() if is_cuda_v2():
env = GPUEnv()
elif is_hip():
env = HIPEnv()
elif is_npu():
env = NPUEnv()
env.check_env()