Use SGLANG_CACHE_DIR env for gpu_p2p_access_cache path (#25686)
Co-authored-by: Ian O'Connell <ianoc@meta.com> Co-authored-by: ianoc <ianoc@fb.com>
This commit is contained in:
co-authored by
Ian O'Connell
ianoc
parent
745abd6cc0
commit
314dedf7c6
@@ -21,6 +21,7 @@ from typing_extensions import ParamSpec
|
|||||||
|
|
||||||
from sglang.srt.distributed.device_communicators.cuda_wrapper import CudaRTLibrary
|
from sglang.srt.distributed.device_communicators.cuda_wrapper import CudaRTLibrary
|
||||||
from sglang.srt.distributed.parallel_state import in_the_same_node_as
|
from sglang.srt.distributed.parallel_state import in_the_same_node_as
|
||||||
|
from sglang.srt.environ import envs as sglang_envs
|
||||||
from sglang.srt.utils import is_cuda, is_hip, is_musa
|
from sglang.srt.utils import is_cuda, is_hip, is_musa
|
||||||
|
|
||||||
logger = logging.getLogger(__name__)
|
logger = logging.getLogger(__name__)
|
||||||
@@ -261,8 +262,8 @@ def gpu_p2p_access_check(src: int, tgt: int) -> bool:
|
|||||||
cuda_visible_devices = ",".join(str(i) for i in range(num_dev))
|
cuda_visible_devices = ",".join(str(i) for i in range(num_dev))
|
||||||
|
|
||||||
# VLLM_CACHE_ROOT -> SGLANG_CACHE_ROOT
|
# VLLM_CACHE_ROOT -> SGLANG_CACHE_ROOT
|
||||||
# "~/.cache/vllm" -> "~/.cache/sglang"
|
# "~/.cache/vllm" -> envs.SGLANG_CACHE_DIR
|
||||||
SGLANG_CACHE_ROOT = os.path.expanduser("~/.cache/sglang")
|
SGLANG_CACHE_ROOT = os.path.expanduser(sglang_envs.SGLANG_CACHE_DIR.get())
|
||||||
path = os.path.join(
|
path = os.path.join(
|
||||||
SGLANG_CACHE_ROOT, f"gpu_p2p_access_cache_for_{cuda_visible_devices}.json"
|
SGLANG_CACHE_ROOT, f"gpu_p2p_access_cache_for_{cuda_visible_devices}.json"
|
||||||
)
|
)
|
||||||
|
|||||||
Reference in New Issue
Block a user