Install DeepEP from release wheels (#33932)
This commit is contained in:
@@ -1,177 +0,0 @@
|
||||
#!/bin/bash
|
||||
# Install the dependency in CI.
|
||||
set -euxo pipefail
|
||||
|
||||
# Source (not bash) so that venv activation, $PIP_CMD, $CU_VERSION, $NVCC_VER, and
|
||||
# $PIP_INSTALL_SUFFIX all propagate into this shell. Without sourcing, the subshell
|
||||
# exits and this script would fall back to system Python.
|
||||
#
|
||||
# Note: any `exit N` or `set -e` trip inside the sourced script terminates *this*
|
||||
# script too (bash runs sourced commands in the current shell, so `exit` is not
|
||||
# caught by `if`/`||`). The real error message appears upstream in the log.
|
||||
# shellcheck disable=SC1091
|
||||
source scripts/ci/cuda/ci_install_dependency.sh
|
||||
|
||||
# In venv mode, PIP_CMD must be set by the sourced script. If it isn't, the
|
||||
# source chain is broken and we'd silently fall back to system `pip` below —
|
||||
# exactly the split-install bug the migration is meant to prevent.
|
||||
if [ -z "${PIP_CMD:-}" ]; then
|
||||
echo "FATAL:PIP_CMD is unset after sourcing ci_install_dependency.sh"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
export GDRCOPY_HOME=/usr/src/gdrdrv-2.5.1/
|
||||
export CUDA_HOME=/usr/local/cuda
|
||||
|
||||
GRACE_BLACKWELL=${GRACE_BLACKWELL:-0}
|
||||
# Detect architecture
|
||||
ARCH=$(uname -m)
|
||||
if [ "$ARCH" != "x86_64" ] && [ "$ARCH" != "aarch64" ]; then
|
||||
echo "Unsupported architecture: $ARCH"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
if [ "${FORCE_REBUILD_DEEPEP:-0}" = "1" ]; then
|
||||
echo "FORCE_REBUILD_DEEPEP=1; uninstalling any cached deep_ep before rebuild."
|
||||
${PIP_UNINSTALL_CMD:-pip uninstall -y} deep_ep ${PIP_UNINSTALL_SUFFIX:-} || true
|
||||
elif python3 -c "import deep_ep" >/dev/null 2>&1; then
|
||||
echo "deep_ep is already installed or importable. Skipping installation."
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# Install system dependencies
|
||||
# Use fallback logic in case apt fails due to unrelated broken packages on the runner
|
||||
DEEPEP_SYSTEM_DEPS="curl wget git sudo rdma-core infiniband-diags openssh-server perftest libibumad3 libibverbs-dev libibverbs1 ibverbs-providers ibverbs-utils libnl-3-200 libnl-route-3-200 librdmacm1 build-essential cmake"
|
||||
apt-get install -y --no-install-recommends $DEEPEP_SYSTEM_DEPS || {
|
||||
echo "Warning: apt-get install failed, checking if required packages are available..."
|
||||
for pkg in $DEEPEP_SYSTEM_DEPS; do
|
||||
if ! dpkg -l "$pkg" 2>/dev/null | grep -q "^ii"; then
|
||||
echo "ERROR: Required package $pkg is not installed and apt-get failed"
|
||||
exit 1
|
||||
fi
|
||||
done
|
||||
echo "All required packages are already installed, continuing..."
|
||||
}
|
||||
|
||||
# Install GDRCopy
|
||||
rm -rf /opt/gdrcopy && mkdir -p /opt/gdrcopy
|
||||
cd /opt/gdrcopy
|
||||
git clone https://github.com/NVIDIA/gdrcopy.git .
|
||||
git checkout v2.5.1
|
||||
apt-get update || true # May fail due to unrelated broken packages
|
||||
GDRCOPY_DEPS_1="nvidia-dkms-580"
|
||||
GDRCOPY_DEPS_2="build-essential devscripts debhelper fakeroot pkg-config dkms"
|
||||
GDRCOPY_DEPS_3="check libsubunit0 libsubunit-dev python3-venv"
|
||||
for deps_group in "$GDRCOPY_DEPS_1" "$GDRCOPY_DEPS_2" "$GDRCOPY_DEPS_3"; do
|
||||
apt-get install -y --no-install-recommends $deps_group || {
|
||||
echo "Warning: apt-get install failed for '$deps_group', checking if packages are available..."
|
||||
for pkg in $deps_group; do
|
||||
if ! dpkg -l "$pkg" 2>/dev/null | grep -q "^ii"; then
|
||||
echo "ERROR: Required package $pkg is not installed and apt-get failed"
|
||||
exit 1
|
||||
fi
|
||||
done
|
||||
echo "All required packages from '$deps_group' are already installed, continuing..."
|
||||
}
|
||||
done
|
||||
cd packages
|
||||
CUDA=/usr/local/cuda ./build-deb-packages.sh
|
||||
dpkg -i gdrdrv-dkms_*.deb
|
||||
dpkg -i libgdrapi_*.deb
|
||||
dpkg -i gdrcopy-tests_*.deb
|
||||
dpkg -i gdrcopy_*.deb
|
||||
|
||||
# Set up library paths based on architecture
|
||||
LIB_PATH="/usr/lib/$ARCH-linux-gnu"
|
||||
if [ ! -e "$LIB_PATH/libmlx5.so" ]; then
|
||||
ln -s $LIB_PATH/libmlx5.so.1 $LIB_PATH/libmlx5.so
|
||||
fi
|
||||
apt-get update || true
|
||||
apt-get install -y --no-install-recommends libfabric-dev || {
|
||||
if ! dpkg -l libfabric-dev 2>/dev/null | grep -q "^ii"; then
|
||||
echo "ERROR: Required package libfabric-dev is not installed and apt-get failed"
|
||||
exit 1
|
||||
fi
|
||||
echo "libfabric-dev is already installed, continuing..."
|
||||
}
|
||||
|
||||
# Install DeepEP
|
||||
DEEPEP_DIR=/root/.cache/deepep
|
||||
rm -rf ${DEEPEP_DIR}
|
||||
if [ "$GRACE_BLACKWELL" = "1" ]; then
|
||||
GRACE_BLACKWELL_DEEPEP_BRANCH=hybrid-ep
|
||||
git clone https://github.com/deepseek-ai/DeepEP.git -b ${GRACE_BLACKWELL_DEEPEP_BRANCH} ${DEEPEP_DIR} && \
|
||||
pushd ${DEEPEP_DIR} && \
|
||||
git checkout d28bd676c2120573c9f1425f0c16c39faa4117e6 && \
|
||||
sed -i 's/#define NUM_CPU_TIMEOUT_SECS 100/#define NUM_CPU_TIMEOUT_SECS 1000/' csrc/kernels/configs.cuh && \
|
||||
popd
|
||||
else
|
||||
git clone https://github.com/deepseek-ai/DeepEP.git ${DEEPEP_DIR} && \
|
||||
pushd ${DEEPEP_DIR} && \
|
||||
git checkout 9af0e0d0e74f3577af1979c9b9e1ac2cad0104ee && \
|
||||
popd
|
||||
fi
|
||||
|
||||
cd ${DEEPEP_DIR}
|
||||
if [ "$GRACE_BLACKWELL" = "1" ]; then
|
||||
# Resolve the toolkit CUDA version. Preference order:
|
||||
# 1. $NVCC_VER inherited from the sourced ci_install_dependency.sh
|
||||
# (both scripts agree on the detected value, no re-detection cost).
|
||||
# 2. Local `nvcc --version` (authoritative — container toolkit).
|
||||
# 3. `nvidia-smi` (host driver; last resort).
|
||||
if [ -n "${NVCC_VER:-}" ]; then
|
||||
CUDA_VERSION="$NVCC_VER"
|
||||
elif command -v nvcc >/dev/null 2>&1; then
|
||||
CUDA_VERSION=$(nvcc --version | grep -oP 'release \K[0-9]+\.[0-9]+')
|
||||
else
|
||||
CUDA_VERSION=$(nvidia-smi | grep "CUDA Version" | head -n1 | awk '{print $9}' || true)
|
||||
fi
|
||||
if [ -z "${CUDA_VERSION:-}" ]; then
|
||||
echo "FATAL: could not determine CUDA toolkit version (NVCC_VER unset, nvcc missing, nvidia-smi empty)"
|
||||
exit 1
|
||||
fi
|
||||
if [ "$CUDA_VERSION" = "12.8" ]; then
|
||||
CHOSEN_TORCH_CUDA_ARCH_LIST='10.0'
|
||||
elif awk -v ver="$CUDA_VERSION" 'BEGIN {exit !(ver > 12.8)}'; then
|
||||
# CUDA > 12.8 supports sm_103 (Blackwell)
|
||||
CHOSEN_TORCH_CUDA_ARCH_LIST='10.0;10.3'
|
||||
else
|
||||
echo "Unsupported CUDA version for Grace Blackwell: $CUDA_VERSION" && exit 1
|
||||
fi && \
|
||||
if [ "${CUDA_VERSION%%.*}" = "13" ]; then \
|
||||
sed -i "/^ include_dirs = \['csrc\/'\]/a\ include_dirs.append('${CUDA_HOME}/include/cccl')" setup.py; \
|
||||
fi
|
||||
TORCH_CUDA_ARCH_LIST="${CHOSEN_TORCH_CUDA_ARCH_LIST}" ${PIP_CMD:-pip} install --no-build-isolation . ${PIP_INSTALL_SUFFIX:-}
|
||||
else
|
||||
# CUDA 13.0 puts CCCL headers in /usr/local/cuda/include/cccl/ but nvshmem
|
||||
# includes them as <cuda/__cccl_config> expecting /usr/local/cuda/include/cuda/.
|
||||
# Add the cccl path to setup.py include_dirs so the compiler finds them.
|
||||
NVCC_MAJOR=$(nvcc --version 2>/dev/null | grep -oP 'release \K[0-9]+' || echo "0")
|
||||
if [ "$NVCC_MAJOR" = "13" ]; then
|
||||
sed -i "/^ include_dirs = \['csrc\/'\]/a\ include_dirs.append('${CUDA_HOME:-/usr/local/cuda}/include/cccl')" setup.py
|
||||
fi
|
||||
|
||||
# Build for both Hopper (sm_90) and Blackwell (sm_100) so the same wheel
|
||||
# runs on H200 and B200 runners. Mirrors the CUDA-version-keyed list in
|
||||
# docker/Dockerfile's DeepEP build stage.
|
||||
if [ -n "${NVCC_VER:-}" ]; then
|
||||
CUDA_VERSION="$NVCC_VER"
|
||||
elif command -v nvcc >/dev/null 2>&1; then
|
||||
CUDA_VERSION=$(nvcc --version | grep -oP 'release \K[0-9]+\.[0-9]+')
|
||||
else
|
||||
CUDA_VERSION=$(nvidia-smi | grep "CUDA Version" | head -n1 | awk '{print $9}' || true)
|
||||
fi
|
||||
if [ -z "${CUDA_VERSION:-}" ]; then
|
||||
echo "FATAL: could not determine CUDA toolkit version (NVCC_VER unset, nvcc missing, nvidia-smi empty)"
|
||||
exit 1
|
||||
fi
|
||||
if [ "$CUDA_VERSION" = "12.8" ]; then
|
||||
CHOSEN_TORCH_CUDA_ARCH_LIST='9.0;10.0'
|
||||
elif awk -v ver="$CUDA_VERSION" 'BEGIN {exit !(ver > 12.8)}'; then
|
||||
# CUDA > 12.8 supports sm_103 (Blackwell)
|
||||
CHOSEN_TORCH_CUDA_ARCH_LIST='9.0;10.0;10.3'
|
||||
else
|
||||
CHOSEN_TORCH_CUDA_ARCH_LIST='9.0'
|
||||
fi
|
||||
TORCH_CUDA_ARCH_LIST="${CHOSEN_TORCH_CUDA_ARCH_LIST}" python3 setup.py install
|
||||
fi
|
||||
@@ -136,7 +136,9 @@ cleanup_stale_shm() {
|
||||
install_apt_packages() {
|
||||
CI_APT_PACKAGES=(
|
||||
python3 python3-pip python3-venv python3-dev git libnuma-dev libssl-dev pkg-config
|
||||
build-essential cmake rdma-core infiniband-diags perftest libibumad3
|
||||
libibverbs-dev libibverbs1 ibverbs-providers ibverbs-utils
|
||||
libfabric-dev libnl-3-200 libnl-route-3-200 librdmacm1
|
||||
ffmpeg libavcodec-dev libavformat-dev libavutil-dev libswscale-dev
|
||||
)
|
||||
|
||||
@@ -164,6 +166,68 @@ install_apt_packages() {
|
||||
mark_step_done "${FUNCNAME[0]}"
|
||||
}
|
||||
|
||||
install_gdrcopy() {
|
||||
# DeepEP tests only run on 4+ GPU hosts. Keep GDRCopy in the shared CUDA
|
||||
# bootstrap while avoiding a DKMS/package build on the 1- and 2-GPU jobs.
|
||||
local gpu_count=0
|
||||
if command -v nvidia-smi >/dev/null 2>&1; then
|
||||
gpu_count=$(
|
||||
(nvidia-smi --query-gpu=name --format=csv,noheader 2>/dev/null || true) |
|
||||
awk 'NF {count++} END {print count + 0}'
|
||||
)
|
||||
fi
|
||||
if [ "${gpu_count}" -lt 4 ]; then
|
||||
echo "Skipping GDRCopy install on ${gpu_count}-GPU runner"
|
||||
mark_step_done "${FUNCNAME[0]}"
|
||||
return
|
||||
fi
|
||||
|
||||
if ldconfig -p 2>/dev/null | grep 'libgdrapi\.so' >/dev/null; then
|
||||
echo "GDRCopy userspace library is already installed"
|
||||
mark_step_done "${FUNCNAME[0]}"
|
||||
return
|
||||
fi
|
||||
|
||||
local gdrcopy_root=/opt/gdrcopy
|
||||
local gdrcopy_version=2.5.1
|
||||
local -a gdrcopy_packages=(
|
||||
nvidia-dkms-580 devscripts debhelper fakeroot dkms
|
||||
check libsubunit0 libsubunit-dev python3-venv
|
||||
)
|
||||
|
||||
apt-get update || true
|
||||
apt-get install -y --no-install-recommends "${gdrcopy_packages[@]}" || {
|
||||
echo "Warning: apt-get failed while installing GDRCopy build dependencies; checking installed packages"
|
||||
local package
|
||||
for package in "${gdrcopy_packages[@]}"; do
|
||||
if ! dpkg -l "${package}" 2>/dev/null | grep -q '^ii'; then
|
||||
echo "ERROR: Required GDRCopy package ${package} is unavailable"
|
||||
exit 1
|
||||
fi
|
||||
done
|
||||
}
|
||||
|
||||
rm -rf "${gdrcopy_root}"
|
||||
git clone --branch "v${gdrcopy_version}" --depth 1 \
|
||||
https://github.com/NVIDIA/gdrcopy.git "${gdrcopy_root}"
|
||||
(
|
||||
cd "${gdrcopy_root}/packages"
|
||||
CUDA=/usr/local/cuda ./build-deb-packages.sh
|
||||
dpkg -i gdrdrv-dkms_*.deb
|
||||
dpkg -i libgdrapi_*.deb
|
||||
dpkg -i gdrcopy-tests_*.deb
|
||||
dpkg -i gdrcopy_*.deb
|
||||
)
|
||||
|
||||
local lib_path="/usr/lib/${ARCH}-linux-gnu"
|
||||
if [ ! -e "${lib_path}/libmlx5.so" ] && [ -e "${lib_path}/libmlx5.so.1" ]; then
|
||||
ln -s "${lib_path}/libmlx5.so.1" "${lib_path}/libmlx5.so"
|
||||
fi
|
||||
ldconfig
|
||||
|
||||
mark_step_done "${FUNCNAME[0]}"
|
||||
}
|
||||
|
||||
clean_site_packages() {
|
||||
# Clear torch compilation cache from every location it can be in; sglang
|
||||
# is not installed yet, so it cannot be asked which one is in use.
|
||||
@@ -260,6 +324,11 @@ setup_pip_toolchain() {
|
||||
PIP_UNINSTALL_CMD="uv pip uninstall"
|
||||
PIP_UNINSTALL_SUFFIX=""
|
||||
|
||||
# Remove both the legacy source distribution and the SGLang wheel before
|
||||
# resolving the pyproject pin. They own the same deep_ep module files, so
|
||||
# leaving either installed can make pip preserve a mixed installation.
|
||||
$PIP_UNINSTALL_CMD deep-ep sgl-deep-ep $PIP_UNINSTALL_SUFFIX || true
|
||||
|
||||
# sglang-kernel stays: install_sglang_kernel version-gates and reinstalls it.
|
||||
$PIP_UNINSTALL_CMD sgl-kernel sglang sgl-fa4 flash-attn-4 $PIP_UNINSTALL_SUFFIX || true
|
||||
|
||||
@@ -368,6 +437,30 @@ install_pytorch_stack() {
|
||||
mark_step_done "${FUNCNAME[0]}"
|
||||
}
|
||||
|
||||
install_cuda12_deepep_wheel() {
|
||||
if [ "$CU_MAJOR" = "13" ]; then
|
||||
echo "CUDA 13 uses the public sgl-deep-ep wheel declared in python/pyproject.toml"
|
||||
mark_step_done "${FUNCNAME[0]}"
|
||||
return
|
||||
fi
|
||||
|
||||
local version
|
||||
version=$(grep -Po -m1 '"sgl-deep-ep==\K[^"]+' python/pyproject.toml || true)
|
||||
if [ -z "$version" ]; then
|
||||
echo "ERROR: python/pyproject.toml must pin sgl-deep-ep"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# CUDA 12 wheels intentionally live only on the SGLang wheel index. Their
|
||||
# local version satisfies the public-version pyproject pin, so the later
|
||||
# editable SGLang install keeps this CUDA-matched wheel.
|
||||
$PIP_CMD install "sgl-deep-ep==${version}+${CU_VERSION}" \
|
||||
--index-url "https://docs.sglang.ai/whl/${CU_VERSION}/" \
|
||||
--force-reinstall --no-deps $PIP_INSTALL_SUFFIX
|
||||
|
||||
mark_step_done "${FUNCNAME[0]}"
|
||||
}
|
||||
|
||||
require_prebuilt_rust_exts() {
|
||||
# Stages whose download succeeded set this to none. Runs before
|
||||
# setup_pip_toolchain uninstalls sglang, so clearing it here still reaches
|
||||
@@ -708,6 +801,8 @@ verify_imports() {
|
||||
SGLANG_EXPECTED_INIT="${REPO_ROOT}/python/sglang/__init__.py" python3 -c '
|
||||
import torch
|
||||
print(torch.version.cuda)
|
||||
import deep_ep
|
||||
print(f"deep_ep loads from {deep_ep.__file__}")
|
||||
import cutlass
|
||||
import cutlass.cute
|
||||
|
||||
@@ -752,6 +847,7 @@ main() {
|
||||
kill_existing_processes
|
||||
cleanup_stale_shm
|
||||
install_apt_packages
|
||||
install_gdrcopy
|
||||
clean_site_packages
|
||||
setup_cargo_cache
|
||||
require_prebuilt_rust_exts
|
||||
@@ -759,6 +855,7 @@ main() {
|
||||
remove_stale_cuda12_nvidia_wheels
|
||||
uninstall_stale_flashinfer
|
||||
install_pytorch_stack
|
||||
install_cuda12_deepep_wheel
|
||||
install_sglang
|
||||
# Diffusion B200 CI imports torch inside install_sglang_kernel after removing
|
||||
# stale CUDA 12 NVIDIA wheels, so opt into one early LD_LIBRARY_PATH refresh.
|
||||
|
||||
Reference in New Issue
Block a user