diff --git a/.github/workflows/_pr-test-sgl-kernel-build.yml b/.github/workflows/_pr-test-sgl-kernel-build.yml index 5a72c5514..13190a8d9 100644 --- a/.github/workflows/_pr-test-sgl-kernel-build.yml +++ b/.github/workflows/_pr-test-sgl-kernel-build.yml @@ -40,7 +40,6 @@ env: SGLANG_CUDA_COREDUMP: "1" SGLANG_JIT_DEEPGEMM_FAST_WARMUP: true SKIP_PR_TEST_HEALTH_CHECK: ${{ inputs.skip_pr_test_health_check && 'true' || 'false' }} - FORCE_REBUILD_DEEPEP: '1' PR_TEST_BYPASS_MAINTENANCE_ON_MAIN: ${{ github.ref == 'refs/heads/main' && 'true' || 'false' }} USE_VENV: false diff --git a/.github/workflows/_pr-test-stage.yml b/.github/workflows/_pr-test-stage.yml index fd06921a3..7005f1f2a 100644 --- a/.github/workflows/_pr-test-stage.yml +++ b/.github/workflows/_pr-test-stage.yml @@ -75,7 +75,6 @@ env: SGLANG_CUDA_COREDUMP: "1" SGLANG_JIT_DEEPGEMM_FAST_WARMUP: true SKIP_PR_TEST_HEALTH_CHECK: ${{ (fromJson(inputs.caller_inputs).skip_pr_test_health_check || fromJson(inputs.caller_inputs).test_parallel_dispatch || fromJson(inputs.caller_inputs).run_all_tests) && 'true' || 'false' }} - FORCE_REBUILD_DEEPEP: '1' PR_TEST_BYPASS_MAINTENANCE_ON_MAIN: ${{ github.ref == 'refs/heads/main' && 'true' || 'false' }} USE_VENV: false @@ -136,8 +135,6 @@ jobs: - name: Install dependencies timeout-minutes: ${{ fromJson(steps.rc.outputs.install_timeout) }} - env: - GRACE_BLACKWELL: ${{ steps.rc.outputs.grace_blackwell || '0' }} run: | CUSTOM_BUILD_SGL_KERNEL=${{ fromJson(inputs.check_changes).sgl_kernel }} bash ${{ steps.rc.outputs.install }} diff --git a/.github/workflows/nightly-test-nvidia.yml b/.github/workflows/nightly-test-nvidia.yml index 82f9f44dc..3901d0d14 100644 --- a/.github/workflows/nightly-test-nvidia.yml +++ b/.github/workflows/nightly-test-nvidia.yml @@ -486,10 +486,8 @@ jobs: - uses: ./.github/actions/check-maintenance - name: Install dependencies - env: - GRACE_BLACKWELL: "1" run: | - bash scripts/ci/cuda/ci_install_deepep.sh + bash scripts/ci/cuda/ci_install_dependency.sh - name: Run test timeout-minutes: 600 diff --git a/.github/workflows/pr-test-extra.yml b/.github/workflows/pr-test-extra.yml index 3155c77e3..5b2349fc6 100644 --- a/.github/workflows/pr-test-extra.yml +++ b/.github/workflows/pr-test-extra.yml @@ -64,7 +64,6 @@ env: SGLANG_CUDA_COREDUMP: "1" SGLANG_JIT_DEEPGEMM_FAST_WARMUP: true SKIP_PR_TEST_HEALTH_CHECK: ${{ (inputs.skip_pr_test_health_check == true || inputs.run_all_tests == true) && 'true' || 'false' }} - FORCE_REBUILD_DEEPEP: '1' PR_TEST_BYPASS_MAINTENANCE_ON_MAIN: ${{ github.ref == 'refs/heads/main' && 'true' || 'false' }} USE_VENV: false @@ -238,49 +237,6 @@ jobs: rust_ext_artifact: ${{ needs.rust-ext-build.outputs.artifact_name }} secrets: inherit - extra-b-test-deepep-4-gpu-h100: - needs: [check-changes, call-gate, sgl-kernel-build-wheels, rust-ext-build] - if: ${{ !failure() && !cancelled() && needs.check-changes.result == 'success' && (needs.call-gate.result == 'success' || needs.call-gate.result == 'skipped') }} - uses: ./.github/workflows/_pr-test-stage.yml - with: - self_name: extra-b-test-deepep-4-gpu-h100 - runner_config: deepep-4-gpu-h100 - check_changes: ${{ toJson(needs.check-changes.outputs) }} - caller_inputs: ${{ toJson(inputs) }} - partitions: ${{ needs.check-changes.outputs.partitions }} - run_timeout_minutes: '60' - rust_ext_artifact: ${{ needs.rust-ext-build.outputs.artifact_name }} - secrets: inherit - - extra-b-test-deepep-4-gpu-b200: - needs: [check-changes, call-gate, sgl-kernel-build-wheels, rust-ext-build] - if: ${{ !failure() && !cancelled() && needs.check-changes.result == 'success' && (needs.call-gate.result == 'success' || needs.call-gate.result == 'skipped') }} - uses: ./.github/workflows/_pr-test-stage.yml - with: - self_name: extra-b-test-deepep-4-gpu-b200 - runner_config: deepep-4-gpu-b200 - check_changes: ${{ toJson(needs.check-changes.outputs) }} - caller_inputs: ${{ toJson(inputs) }} - partitions: ${{ needs.check-changes.outputs.partitions }} - run_timeout_minutes: '60' - timeout_per_file: '1800' - rust_ext_artifact: ${{ needs.rust-ext-build.outputs.artifact_name }} - secrets: inherit - - extra-b-test-deepep-8-gpu-h200: - needs: [check-changes, call-gate, sgl-kernel-build-wheels, rust-ext-build] - if: ${{ !failure() && !cancelled() && needs.check-changes.result == 'success' && (needs.call-gate.result == 'success' || needs.call-gate.result == 'skipped') }} - uses: ./.github/workflows/_pr-test-stage.yml - with: - self_name: extra-b-test-deepep-8-gpu-h200 - runner_config: deepep-8-gpu-h200 - check_changes: ${{ toJson(needs.check-changes.outputs) }} - caller_inputs: ${{ toJson(inputs) }} - partitions: ${{ needs.check-changes.outputs.partitions }} - run_timeout_minutes: '60' - rust_ext_artifact: ${{ needs.rust-ext-build.outputs.artifact_name }} - secrets: inherit - # =============================================== aggregator ==================================================== # Mirrors pr-test.yml's `pr-test-finish` so notify-pr-states below only # depends on one job rather than re-listing every stage. Fails if any @@ -298,9 +254,6 @@ jobs: extra-b-test-4-gpu-h100, extra-b-test-4-gpu-b200, extra-b-test-8-gpu-h200, - extra-b-test-deepep-4-gpu-h100, - extra-b-test-deepep-4-gpu-b200, - extra-b-test-deepep-8-gpu-h200, ] if: always() runs-on: ubuntu-latest diff --git a/.github/workflows/pr-test.yml b/.github/workflows/pr-test.yml index b7e652f1e..5ebae988b 100644 --- a/.github/workflows/pr-test.yml +++ b/.github/workflows/pr-test.yml @@ -59,8 +59,6 @@ env: SGLANG_CUDA_COREDUMP: "1" SGLANG_JIT_DEEPGEMM_FAST_WARMUP: true SKIP_PR_TEST_HEALTH_CHECK: ${{ (inputs.skip_pr_test_health_check == true || inputs.test_parallel_dispatch == true || inputs.run_all_tests == true) && 'true' || 'false' }} - # TEMP: force DeepEP rebuilds during the initial Torch 2.13 rollout so every CI runner replaces its Torch 2.11 build; remove after the runner fleet has cycled. - FORCE_REBUILD_DEEPEP: '1' # Schedule / main-branch dispatch / workflow_call from main use refs/heads/main; PR events use refs/pull/*/merge PR_TEST_BYPASS_MAINTENANCE_ON_MAIN: ${{ github.ref == 'refs/heads/main' && 'true' || 'false' }} USE_VENV: false @@ -400,6 +398,8 @@ jobs: caller_inputs: ${{ toJson(inputs) }} partitions: ${{ needs.check-changes.outputs.partitions }} run_timeout_minutes: '30' + warmup_deep_gemm_models: 'lmsys/sglang-ci-dsv3-test:4' + warmup_server_models: 'lmsys/sglang-ci-dsv3-test:4' rust_ext_artifact: ${{ needs.rust-ext-build.outputs.artifact_name }} secrets: inherit @@ -414,6 +414,7 @@ jobs: caller_inputs: ${{ toJson(inputs) }} partitions: ${{ needs.check-changes.outputs.partitions }} run_timeout_minutes: '30' + timeout_per_file: '1800' # Per-model TP must match the test's launch in test/registered/ -- see # FALLBACK_ARGS in scripts/ci/cuda/warmup_deep_gemm.py for extra dp/ep # flags. Only models that actually invoke DeepGEMM kernels at runtime @@ -438,52 +439,6 @@ jobs: rust_ext_artifact: ${{ needs.rust-ext-build.outputs.artifact_name }} secrets: inherit - base-c-test-deepep-4-gpu-h100: - needs: [check-changes, call-gate, wait-for-base-b, sgl-kernel-build-wheels, rust-ext-build] - if: ${{ !failure() && !cancelled() }} - uses: ./.github/workflows/_pr-test-stage.yml - with: - self_name: base-c-test-deepep-4-gpu-h100 - runner_config: deepep-4-gpu-h100 - check_changes: ${{ toJson(needs.check-changes.outputs) }} - caller_inputs: ${{ toJson(inputs) }} - partitions: ${{ needs.check-changes.outputs.partitions }} - run_timeout_minutes: '30' - warmup_deep_gemm_models: 'lmsys/sglang-ci-dsv3-test:4' - warmup_server_models: 'lmsys/sglang-ci-dsv3-test:4' - rust_ext_artifact: ${{ needs.rust-ext-build.outputs.artifact_name }} - secrets: inherit - - base-c-test-deepep-4-gpu-b200: - needs: [check-changes, call-gate, wait-for-base-b, sgl-kernel-build-wheels, rust-ext-build] - if: ${{ !failure() && !cancelled() }} - uses: ./.github/workflows/_pr-test-stage.yml - with: - self_name: base-c-test-deepep-4-gpu-b200 - runner_config: deepep-4-gpu-b200 - check_changes: ${{ toJson(needs.check-changes.outputs) }} - caller_inputs: ${{ toJson(inputs) }} - partitions: ${{ needs.check-changes.outputs.partitions }} - run_timeout_minutes: '30' - timeout_per_file: '1800' - rust_ext_artifact: ${{ needs.rust-ext-build.outputs.artifact_name }} - secrets: inherit - - base-c-test-deepep-8-gpu-h200: - needs: [check-changes, call-gate, wait-for-base-b, sgl-kernel-build-wheels, rust-ext-build] - if: ${{ !failure() && !cancelled() }} - uses: ./.github/workflows/_pr-test-stage.yml - with: - self_name: base-c-test-deepep-8-gpu-h200 - runner_config: deepep-8-gpu-h200 - check_changes: ${{ toJson(needs.check-changes.outputs) }} - caller_inputs: ${{ toJson(inputs) }} - partitions: ${{ needs.check-changes.outputs.partitions }} - run_timeout_minutes: '30' - timeout_per_file: '1800' - rust_ext_artifact: ${{ needs.rust-ext-build.outputs.artifact_name }} - secrets: inherit - base-c-test-4-gpu-b200: needs: [check-changes, call-gate, wait-for-base-b, sgl-kernel-build-wheels, rust-ext-build] if: ${{ !failure() && !cancelled() }} @@ -561,9 +516,6 @@ jobs: base-c-test-4-gpu-h100, base-c-test-8-gpu-h20, base-c-test-8-gpu-h200, - base-c-test-deepep-4-gpu-h100, - base-c-test-deepep-4-gpu-b200, - base-c-test-deepep-8-gpu-h200, base-c-test-4-gpu-b200, base-c-test-4-gpu-gb300, base-c-test-8-gpu-b300, diff --git a/.github/workflows/rerun-test.yml b/.github/workflows/rerun-test.yml index e25f2aaa1..c10883220 100644 --- a/.github/workflows/rerun-test.yml +++ b/.github/workflows/rerun-test.yml @@ -31,11 +31,6 @@ on: required: false type: string default: "20" - grace_blackwell: - description: "Set GRACE_BLACKWELL for the install step (cuda only)" - required: false - type: string - default: "0" rdma_devices: description: "SGLANG_CI_RDMA_ALL_DEVICES csv (cuda only; empty = unset)" required: false @@ -62,8 +57,6 @@ env: SGLANG_ENABLE_ASYNC_ASSERT: true SGLANG_CUDA_COREDUMP: "1" SGLANG_JIT_DEEPGEMM_FAST_WARMUP: true - # TEMP: force DeepEP rebuilds during the initial Torch 2.13 rollout so every CI runner replaces its Torch 2.11 build; remove after the runner fleet has cycled. - FORCE_REBUILD_DEEPEP: '1' permissions: actions: write @@ -109,8 +102,6 @@ jobs: - name: Install dependencies timeout-minutes: ${{ fromJson(inputs.install_timeout) }} - env: - GRACE_BLACKWELL: ${{ inputs.grace_blackwell }} run: | if [[ "${{ inputs.runs_on }}" == "1-gpu-5090" ]]; then source /etc/profile.d/sglang-ci.sh @@ -120,7 +111,7 @@ jobs: # then dies with `ModuleNotFoundError: No module named 'sglang'`. # /rerun-test resolves install_script from runner_configs.yml; a manual # workflow_dispatch must pass it explicitly (it can't be derived from - # runs_on, which is shared across configs, e.g. default vs deepep). + # runs_on, which can be shared across runner configs). if [ -z "${{ inputs.install_script }}" ]; then echo "::error::install_script is required for cuda mode (empty would silently skip installing sglang). Pass e.g. scripts/ci/cuda/ci_install_dependency.sh" exit 1 diff --git a/python/pyproject.toml b/python/pyproject.toml index 89d6defc0..a5e83c6de 100755 --- a/python/pyproject.toml +++ b/python/pyproject.toml @@ -67,6 +67,7 @@ dependencies = [ "scipy", "sentencepiece", "setproctitle", + "sgl-deep-ep==0.1.0", "sgl-deep-gemm==0.1.5.post2", "sglang-kernel==0.4.6.post1", "smg-grpc-servicer>=0.5.0", diff --git a/scripts/ci/cuda/ci_install_deepep.sh b/scripts/ci/cuda/ci_install_deepep.sh deleted file mode 100755 index c78ee649b..000000000 --- a/scripts/ci/cuda/ci_install_deepep.sh +++ /dev/null @@ -1,177 +0,0 @@ -#!/bin/bash -# Install the dependency in CI. -set -euxo pipefail - -# Source (not bash) so that venv activation, $PIP_CMD, $CU_VERSION, $NVCC_VER, and -# $PIP_INSTALL_SUFFIX all propagate into this shell. Without sourcing, the subshell -# exits and this script would fall back to system Python. -# -# Note: any `exit N` or `set -e` trip inside the sourced script terminates *this* -# script too (bash runs sourced commands in the current shell, so `exit` is not -# caught by `if`/`||`). The real error message appears upstream in the log. -# shellcheck disable=SC1091 -source scripts/ci/cuda/ci_install_dependency.sh - -# In venv mode, PIP_CMD must be set by the sourced script. If it isn't, the -# source chain is broken and we'd silently fall back to system `pip` below — -# exactly the split-install bug the migration is meant to prevent. -if [ -z "${PIP_CMD:-}" ]; then - echo "FATAL:PIP_CMD is unset after sourcing ci_install_dependency.sh" - exit 1 -fi - -export GDRCOPY_HOME=/usr/src/gdrdrv-2.5.1/ -export CUDA_HOME=/usr/local/cuda - -GRACE_BLACKWELL=${GRACE_BLACKWELL:-0} -# Detect architecture -ARCH=$(uname -m) -if [ "$ARCH" != "x86_64" ] && [ "$ARCH" != "aarch64" ]; then - echo "Unsupported architecture: $ARCH" - exit 1 -fi - -if [ "${FORCE_REBUILD_DEEPEP:-0}" = "1" ]; then - echo "FORCE_REBUILD_DEEPEP=1; uninstalling any cached deep_ep before rebuild." - ${PIP_UNINSTALL_CMD:-pip uninstall -y} deep_ep ${PIP_UNINSTALL_SUFFIX:-} || true -elif python3 -c "import deep_ep" >/dev/null 2>&1; then - echo "deep_ep is already installed or importable. Skipping installation." - exit 0 -fi - -# Install system dependencies -# Use fallback logic in case apt fails due to unrelated broken packages on the runner -DEEPEP_SYSTEM_DEPS="curl wget git sudo rdma-core infiniband-diags openssh-server perftest libibumad3 libibverbs-dev libibverbs1 ibverbs-providers ibverbs-utils libnl-3-200 libnl-route-3-200 librdmacm1 build-essential cmake" -apt-get install -y --no-install-recommends $DEEPEP_SYSTEM_DEPS || { - echo "Warning: apt-get install failed, checking if required packages are available..." - for pkg in $DEEPEP_SYSTEM_DEPS; do - if ! dpkg -l "$pkg" 2>/dev/null | grep -q "^ii"; then - echo "ERROR: Required package $pkg is not installed and apt-get failed" - exit 1 - fi - done - echo "All required packages are already installed, continuing..." -} - -# Install GDRCopy -rm -rf /opt/gdrcopy && mkdir -p /opt/gdrcopy -cd /opt/gdrcopy -git clone https://github.com/NVIDIA/gdrcopy.git . -git checkout v2.5.1 -apt-get update || true # May fail due to unrelated broken packages -GDRCOPY_DEPS_1="nvidia-dkms-580" -GDRCOPY_DEPS_2="build-essential devscripts debhelper fakeroot pkg-config dkms" -GDRCOPY_DEPS_3="check libsubunit0 libsubunit-dev python3-venv" -for deps_group in "$GDRCOPY_DEPS_1" "$GDRCOPY_DEPS_2" "$GDRCOPY_DEPS_3"; do - apt-get install -y --no-install-recommends $deps_group || { - echo "Warning: apt-get install failed for '$deps_group', checking if packages are available..." - for pkg in $deps_group; do - if ! dpkg -l "$pkg" 2>/dev/null | grep -q "^ii"; then - echo "ERROR: Required package $pkg is not installed and apt-get failed" - exit 1 - fi - done - echo "All required packages from '$deps_group' are already installed, continuing..." - } -done -cd packages -CUDA=/usr/local/cuda ./build-deb-packages.sh -dpkg -i gdrdrv-dkms_*.deb -dpkg -i libgdrapi_*.deb -dpkg -i gdrcopy-tests_*.deb -dpkg -i gdrcopy_*.deb - -# Set up library paths based on architecture -LIB_PATH="/usr/lib/$ARCH-linux-gnu" -if [ ! -e "$LIB_PATH/libmlx5.so" ]; then - ln -s $LIB_PATH/libmlx5.so.1 $LIB_PATH/libmlx5.so -fi -apt-get update || true -apt-get install -y --no-install-recommends libfabric-dev || { - if ! dpkg -l libfabric-dev 2>/dev/null | grep -q "^ii"; then - echo "ERROR: Required package libfabric-dev is not installed and apt-get failed" - exit 1 - fi - echo "libfabric-dev is already installed, continuing..." -} - -# Install DeepEP -DEEPEP_DIR=/root/.cache/deepep -rm -rf ${DEEPEP_DIR} -if [ "$GRACE_BLACKWELL" = "1" ]; then - GRACE_BLACKWELL_DEEPEP_BRANCH=hybrid-ep - git clone https://github.com/deepseek-ai/DeepEP.git -b ${GRACE_BLACKWELL_DEEPEP_BRANCH} ${DEEPEP_DIR} && \ - pushd ${DEEPEP_DIR} && \ - git checkout d28bd676c2120573c9f1425f0c16c39faa4117e6 && \ - sed -i 's/#define NUM_CPU_TIMEOUT_SECS 100/#define NUM_CPU_TIMEOUT_SECS 1000/' csrc/kernels/configs.cuh && \ - popd -else - git clone https://github.com/deepseek-ai/DeepEP.git ${DEEPEP_DIR} && \ - pushd ${DEEPEP_DIR} && \ - git checkout 9af0e0d0e74f3577af1979c9b9e1ac2cad0104ee && \ - popd -fi - -cd ${DEEPEP_DIR} -if [ "$GRACE_BLACKWELL" = "1" ]; then - # Resolve the toolkit CUDA version. Preference order: - # 1. $NVCC_VER inherited from the sourced ci_install_dependency.sh - # (both scripts agree on the detected value, no re-detection cost). - # 2. Local `nvcc --version` (authoritative — container toolkit). - # 3. `nvidia-smi` (host driver; last resort). - if [ -n "${NVCC_VER:-}" ]; then - CUDA_VERSION="$NVCC_VER" - elif command -v nvcc >/dev/null 2>&1; then - CUDA_VERSION=$(nvcc --version | grep -oP 'release \K[0-9]+\.[0-9]+') - else - CUDA_VERSION=$(nvidia-smi | grep "CUDA Version" | head -n1 | awk '{print $9}' || true) - fi - if [ -z "${CUDA_VERSION:-}" ]; then - echo "FATAL: could not determine CUDA toolkit version (NVCC_VER unset, nvcc missing, nvidia-smi empty)" - exit 1 - fi - if [ "$CUDA_VERSION" = "12.8" ]; then - CHOSEN_TORCH_CUDA_ARCH_LIST='10.0' - elif awk -v ver="$CUDA_VERSION" 'BEGIN {exit !(ver > 12.8)}'; then - # CUDA > 12.8 supports sm_103 (Blackwell) - CHOSEN_TORCH_CUDA_ARCH_LIST='10.0;10.3' - else - echo "Unsupported CUDA version for Grace Blackwell: $CUDA_VERSION" && exit 1 - fi && \ - if [ "${CUDA_VERSION%%.*}" = "13" ]; then \ - sed -i "/^ include_dirs = \['csrc\/'\]/a\ include_dirs.append('${CUDA_HOME}/include/cccl')" setup.py; \ - fi - TORCH_CUDA_ARCH_LIST="${CHOSEN_TORCH_CUDA_ARCH_LIST}" ${PIP_CMD:-pip} install --no-build-isolation . ${PIP_INSTALL_SUFFIX:-} -else - # CUDA 13.0 puts CCCL headers in /usr/local/cuda/include/cccl/ but nvshmem - # includes them as expecting /usr/local/cuda/include/cuda/. - # Add the cccl path to setup.py include_dirs so the compiler finds them. - NVCC_MAJOR=$(nvcc --version 2>/dev/null | grep -oP 'release \K[0-9]+' || echo "0") - if [ "$NVCC_MAJOR" = "13" ]; then - sed -i "/^ include_dirs = \['csrc\/'\]/a\ include_dirs.append('${CUDA_HOME:-/usr/local/cuda}/include/cccl')" setup.py - fi - - # Build for both Hopper (sm_90) and Blackwell (sm_100) so the same wheel - # runs on H200 and B200 runners. Mirrors the CUDA-version-keyed list in - # docker/Dockerfile's DeepEP build stage. - if [ -n "${NVCC_VER:-}" ]; then - CUDA_VERSION="$NVCC_VER" - elif command -v nvcc >/dev/null 2>&1; then - CUDA_VERSION=$(nvcc --version | grep -oP 'release \K[0-9]+\.[0-9]+') - else - CUDA_VERSION=$(nvidia-smi | grep "CUDA Version" | head -n1 | awk '{print $9}' || true) - fi - if [ -z "${CUDA_VERSION:-}" ]; then - echo "FATAL: could not determine CUDA toolkit version (NVCC_VER unset, nvcc missing, nvidia-smi empty)" - exit 1 - fi - if [ "$CUDA_VERSION" = "12.8" ]; then - CHOSEN_TORCH_CUDA_ARCH_LIST='9.0;10.0' - elif awk -v ver="$CUDA_VERSION" 'BEGIN {exit !(ver > 12.8)}'; then - # CUDA > 12.8 supports sm_103 (Blackwell) - CHOSEN_TORCH_CUDA_ARCH_LIST='9.0;10.0;10.3' - else - CHOSEN_TORCH_CUDA_ARCH_LIST='9.0' - fi - TORCH_CUDA_ARCH_LIST="${CHOSEN_TORCH_CUDA_ARCH_LIST}" python3 setup.py install -fi diff --git a/scripts/ci/cuda/ci_install_dependency.sh b/scripts/ci/cuda/ci_install_dependency.sh index 7227ac6ee..8f2d19b7f 100755 --- a/scripts/ci/cuda/ci_install_dependency.sh +++ b/scripts/ci/cuda/ci_install_dependency.sh @@ -136,7 +136,9 @@ cleanup_stale_shm() { install_apt_packages() { CI_APT_PACKAGES=( python3 python3-pip python3-venv python3-dev git libnuma-dev libssl-dev pkg-config + build-essential cmake rdma-core infiniband-diags perftest libibumad3 libibverbs-dev libibverbs1 ibverbs-providers ibverbs-utils + libfabric-dev libnl-3-200 libnl-route-3-200 librdmacm1 ffmpeg libavcodec-dev libavformat-dev libavutil-dev libswscale-dev ) @@ -164,6 +166,68 @@ install_apt_packages() { mark_step_done "${FUNCNAME[0]}" } +install_gdrcopy() { + # DeepEP tests only run on 4+ GPU hosts. Keep GDRCopy in the shared CUDA + # bootstrap while avoiding a DKMS/package build on the 1- and 2-GPU jobs. + local gpu_count=0 + if command -v nvidia-smi >/dev/null 2>&1; then + gpu_count=$( + (nvidia-smi --query-gpu=name --format=csv,noheader 2>/dev/null || true) | + awk 'NF {count++} END {print count + 0}' + ) + fi + if [ "${gpu_count}" -lt 4 ]; then + echo "Skipping GDRCopy install on ${gpu_count}-GPU runner" + mark_step_done "${FUNCNAME[0]}" + return + fi + + if ldconfig -p 2>/dev/null | grep 'libgdrapi\.so' >/dev/null; then + echo "GDRCopy userspace library is already installed" + mark_step_done "${FUNCNAME[0]}" + return + fi + + local gdrcopy_root=/opt/gdrcopy + local gdrcopy_version=2.5.1 + local -a gdrcopy_packages=( + nvidia-dkms-580 devscripts debhelper fakeroot dkms + check libsubunit0 libsubunit-dev python3-venv + ) + + apt-get update || true + apt-get install -y --no-install-recommends "${gdrcopy_packages[@]}" || { + echo "Warning: apt-get failed while installing GDRCopy build dependencies; checking installed packages" + local package + for package in "${gdrcopy_packages[@]}"; do + if ! dpkg -l "${package}" 2>/dev/null | grep -q '^ii'; then + echo "ERROR: Required GDRCopy package ${package} is unavailable" + exit 1 + fi + done + } + + rm -rf "${gdrcopy_root}" + git clone --branch "v${gdrcopy_version}" --depth 1 \ + https://github.com/NVIDIA/gdrcopy.git "${gdrcopy_root}" + ( + cd "${gdrcopy_root}/packages" + CUDA=/usr/local/cuda ./build-deb-packages.sh + dpkg -i gdrdrv-dkms_*.deb + dpkg -i libgdrapi_*.deb + dpkg -i gdrcopy-tests_*.deb + dpkg -i gdrcopy_*.deb + ) + + local lib_path="/usr/lib/${ARCH}-linux-gnu" + if [ ! -e "${lib_path}/libmlx5.so" ] && [ -e "${lib_path}/libmlx5.so.1" ]; then + ln -s "${lib_path}/libmlx5.so.1" "${lib_path}/libmlx5.so" + fi + ldconfig + + mark_step_done "${FUNCNAME[0]}" +} + clean_site_packages() { # Clear torch compilation cache from every location it can be in; sglang # is not installed yet, so it cannot be asked which one is in use. @@ -260,6 +324,11 @@ setup_pip_toolchain() { PIP_UNINSTALL_CMD="uv pip uninstall" PIP_UNINSTALL_SUFFIX="" + # Remove both the legacy source distribution and the SGLang wheel before + # resolving the pyproject pin. They own the same deep_ep module files, so + # leaving either installed can make pip preserve a mixed installation. + $PIP_UNINSTALL_CMD deep-ep sgl-deep-ep $PIP_UNINSTALL_SUFFIX || true + # sglang-kernel stays: install_sglang_kernel version-gates and reinstalls it. $PIP_UNINSTALL_CMD sgl-kernel sglang sgl-fa4 flash-attn-4 $PIP_UNINSTALL_SUFFIX || true @@ -368,6 +437,30 @@ install_pytorch_stack() { mark_step_done "${FUNCNAME[0]}" } +install_cuda12_deepep_wheel() { + if [ "$CU_MAJOR" = "13" ]; then + echo "CUDA 13 uses the public sgl-deep-ep wheel declared in python/pyproject.toml" + mark_step_done "${FUNCNAME[0]}" + return + fi + + local version + version=$(grep -Po -m1 '"sgl-deep-ep==\K[^"]+' python/pyproject.toml || true) + if [ -z "$version" ]; then + echo "ERROR: python/pyproject.toml must pin sgl-deep-ep" + exit 1 + fi + + # CUDA 12 wheels intentionally live only on the SGLang wheel index. Their + # local version satisfies the public-version pyproject pin, so the later + # editable SGLang install keeps this CUDA-matched wheel. + $PIP_CMD install "sgl-deep-ep==${version}+${CU_VERSION}" \ + --index-url "https://docs.sglang.ai/whl/${CU_VERSION}/" \ + --force-reinstall --no-deps $PIP_INSTALL_SUFFIX + + mark_step_done "${FUNCNAME[0]}" +} + require_prebuilt_rust_exts() { # Stages whose download succeeded set this to none. Runs before # setup_pip_toolchain uninstalls sglang, so clearing it here still reaches @@ -708,6 +801,8 @@ verify_imports() { SGLANG_EXPECTED_INIT="${REPO_ROOT}/python/sglang/__init__.py" python3 -c ' import torch print(torch.version.cuda) +import deep_ep +print(f"deep_ep loads from {deep_ep.__file__}") import cutlass import cutlass.cute @@ -752,6 +847,7 @@ main() { kill_existing_processes cleanup_stale_shm install_apt_packages + install_gdrcopy clean_site_packages setup_cargo_cache require_prebuilt_rust_exts @@ -759,6 +855,7 @@ main() { remove_stale_cuda12_nvidia_wheels uninstall_stale_flashinfer install_pytorch_stack + install_cuda12_deepep_wheel install_sglang # Diffusion B200 CI imports torch inside install_sglang_kernel after removing # stale CUDA 12 NVIDIA wheels, so opt into one early LD_LIBRARY_PATH refresh. diff --git a/scripts/ci/list_stage_models.py b/scripts/ci/list_stage_models.py index c00a38f9e..8a15cc3ce 100755 --- a/scripts/ci/list_stage_models.py +++ b/scripts/ci/list_stage_models.py @@ -28,8 +28,8 @@ How `file -> models` is resolved (best effort, recall-favoring) How `runner label -> models` is aggregated Registration/prewarm decisions are made per GH runner *label* (a runner's `runs-on` tag), not per suite. Each suite's runner_config maps to a label - via scripts/ci/runner_configs.yml (several configs can share one label, - e.g. `4-gpu-h100` and `deepep-4-gpu-h100`), so `runner_labels` carries the + via scripts/ci/runner_configs.yml (runner configs may share a label), so + `runner_labels` carries the per-label UNION -- the set a runner registered under that label must have cached before it takes jobs. Suites without a mappable runner_config are listed in `unmapped_suites`. @@ -367,9 +367,8 @@ def load_runner_labels(path: str) -> Dict[str, str]: """Parse ``{runner_config: runs_on label}`` out of runner_configs.yml. The mapping is what turns per-suite model sets into per-runner-LABEL sets: - a runner is registered under a `runs_on` label (several runner_configs can - share one, e.g. `4-gpu-h100` and `deepep-4-gpu-h100` both run on - `4-gpu-h100`), so a runner's cache must cover the union of every suite + a runner is registered under a `runs_on` label (runner configs may share a + label), so a runner's cache must cover the union of every suite that can land on its label. Raises ValueError on an entry without `runs_on` or a file with no entries at all -- a format drift must fail the workflow loudly, not silently empty the label aggregation. diff --git a/scripts/ci/runner_configs.yml b/scripts/ci/runner_configs.yml index dfac4efd9..70e3045bb 100644 --- a/scripts/ci/runner_configs.yml +++ b/scripts/ci/runner_configs.yml @@ -8,8 +8,6 @@ # - artifact_version: actions/download-artifact major version # - install_timeout: install-step wall-clock cap (minutes), enforced via # `timeout-minutes:` on the install step in _pr-test-stage.yml -# - grace_blackwell (optional): exported as GRACE_BLACKWELL for the install -# step. Used by GB300 DeePEP setup. # - runs_on: GHA runner label for the stage's `runs-on:`. The literal # `$b200_runner` is substituted at workflow-load time with the dynamic # b200 runner tag from check-changes (see runner_configs.py --map). @@ -18,7 +16,6 @@ _anchors: default_install: &default scripts/ci/cuda/ci_install_dependency.sh - deepep_install: &deepep scripts/ci/cuda/ci_install_deepep.sh kimi_k3_install: &kimi_k3 scripts/ci/cuda/ci_install_kimi_k3.sh runner_configs: @@ -26,12 +23,9 @@ runner_configs: 1-gpu-large: { install: *default, artifact_version: v4, install_timeout: "20", runs_on: 1-gpu-h100 } 2-gpu-large: { install: *default, artifact_version: v4, install_timeout: "20", runs_on: 2-gpu-h100 } 4-gpu-b200: { install: *default, artifact_version: v6, install_timeout: "20", runs_on: $b200_runner } - 4-gpu-gb300: { install: *deepep, artifact_version: v6, install_timeout: "20", grace_blackwell: "1", runs_on: 4-gpu-gb300 } + 4-gpu-gb300: { install: *default, artifact_version: v6, install_timeout: "20", runs_on: 4-gpu-gb300 } 4-gpu-h100: { install: *default, artifact_version: v4, install_timeout: "20", runs_on: 4-gpu-h100 } 8-gpu-h200: { install: *default, artifact_version: v4, install_timeout: "20", runs_on: 8-gpu-h200 } 8-gpu-b200: { install: *default, artifact_version: v6, install_timeout: "20", runs_on: 8-gpu-b200 } 8-gpu-b300: { install: *kimi_k3, artifact_version: v6, install_timeout: "20", runs_on: 8-gpu-b300 } - 8-gpu-h20: { install: *deepep, artifact_version: v4, install_timeout: "20", runs_on: 8-gpu-h20, rdma_devices: "mlx5_1,mlx5_2,mlx5_3,mlx5_4" } - deepep-4-gpu-h100: { install: *deepep, artifact_version: v4, install_timeout: "20", runs_on: 4-gpu-h100 } - deepep-4-gpu-b200: { install: *deepep, artifact_version: v6, install_timeout: "20", runs_on: $b200_runner } - deepep-8-gpu-h200: { install: *deepep, artifact_version: v4, install_timeout: "20", runs_on: 8-gpu-h200 } + 8-gpu-h20: { install: *default, artifact_version: v4, install_timeout: "20", runs_on: 8-gpu-h20, rdma_devices: "mlx5_1,mlx5_2,mlx5_3,mlx5_4" } diff --git a/scripts/ci/test_list_stage_models.py b/scripts/ci/test_list_stage_models.py index 82003b883..b4287800a 100644 --- a/scripts/ci/test_list_stage_models.py +++ b/scripts/ci/test_list_stage_models.py @@ -412,7 +412,7 @@ _anchors: runner_configs: 1-gpu: { install: *default, artifact_version: v4, runs_on: 1-gpu-h100 } - deepep-1-gpu: { install: *default, artifact_version: v4, runs_on: 1-gpu-h100 } + alternate-1-gpu: { install: *default, artifact_version: v4, runs_on: 1-gpu-h100 } 4-gpu-b200: { install: *default, artifact_version: v6, runs_on: $b200_runner } """ @@ -431,7 +431,7 @@ class LoadRunnerLabels(unittest.TestCase): labels, { "1-gpu": "1-gpu-h100", - "deepep-1-gpu": "1-gpu-h100", + "alternate-1-gpu": "1-gpu-h100", "4-gpu-b200": lsm.B200_SENTINEL, }, ) @@ -451,9 +451,7 @@ class LoadRunnerLabels(unittest.TestCase): labels = lsm.load_runner_labels( os.path.join(_REPO_ROOT, "scripts", "ci", "runner_configs.yml") ) - # Two configs sharing a label is the reason the aggregation exists. self.assertEqual(labels["4-gpu-h100"], "4-gpu-h100") - self.assertEqual(labels["deepep-4-gpu-h100"], "4-gpu-h100") self.assertEqual(labels["4-gpu-b200"], lsm.B200_SENTINEL) self.assertGreaterEqual(len(labels), 10) @@ -482,7 +480,7 @@ class RunnerLabelAggregation(unittest.TestCase): "b/test_b.py": self.REG.format( calls=( 'register_cuda_ci(est_time=1, stage="base-y", ' - 'runner_config="deepep-1-gpu")' + 'runner_config="alternate-1-gpu")' ), model="Qwen/Qwen3-8B", ), @@ -520,7 +518,7 @@ class RunnerLabelAggregation(unittest.TestCase): ) self.assertEqual( shared["suites"], - ["base-x-test-1-gpu", "base-y-test-deepep-1-gpu"], + ["base-x-test-1-gpu", "base-y-test-alternate-1-gpu"], ) # Sentinel stays literal without --b200-runner. self.assertIn(lsm.B200_SENTINEL, inv["runner_labels"]) diff --git a/scripts/ci/utils/slash_command_handler.py b/scripts/ci/utils/slash_command_handler.py index 1be998bf4..fb9a8fb64 100644 --- a/scripts/ci/utils/slash_command_handler.py +++ b/scripts/ci/utils/slash_command_handler.py @@ -723,7 +723,7 @@ def _extract_legacy_suites(content): # matches the runner the nightly/weekly pipeline actually uses (see # .github/workflows/{nightly,weekly}-test-nvidia.yml), so /rerun-test can still # dispatch a single nightly/weekly test. The runner label, install script, -# timeout, grace_blackwell, and rdma_devices are then resolved from +# timeout and rdma_devices are then resolved from # runner_configs.yml as usual, keeping that file the single source of truth for # runner details. # @@ -756,7 +756,6 @@ def _dispatch_err(suite, msg): "runner_label": None, "install_script": "", "install_timeout": "", - "grace_blackwell": "0", "rdma_devices": "", "is_cpu": False, "error": msg, @@ -797,7 +796,6 @@ def _resolve_runner_config(rc, full_path, suite): "runner_label": runs_on, "install_script": install_script, "install_timeout": str(cfg["install_timeout"]), - "grace_blackwell": str(cfg.get("grace_blackwell", "0")), "rdma_devices": cfg.get("rdma_devices", ""), "is_cpu": False, "error": None, @@ -811,8 +809,8 @@ def detect_suite(file_path_from_test): A CUDA file can carry multiple `register_cuda_ci(...)` calls — one per pool it should run on — so this returns a *list* of dispatch dicts, one - per registration. Runner label, install script, timeout, grace_blackwell, - and rdma_devices are all resolved from scripts/ci/runner_configs.yml — the + per registration. Runner label, install script, timeout, and rdma_devices + are all resolved from scripts/ci/runner_configs.yml — the same single source of truth that drives the main PR test pipeline. Legacy nightly/weekly CUDA suites (single-string `suite=`) are dispatchable @@ -824,7 +822,7 @@ def detect_suite(file_path_from_test): `error` set. Each dict has keys: suite, runner_label, install_script, - install_timeout, grace_blackwell, rdma_devices, is_cpu, error. + install_timeout, rdma_devices, is_cpu, error. """ full_path = f"test/{file_path_from_test}" with open(full_path, "r") as f: @@ -860,7 +858,6 @@ def detect_suite(file_path_from_test): "runner_label": "ubuntu-latest", "install_script": "", "install_timeout": "", - "grace_blackwell": "0", "rdma_devices": "", "is_cpu": True, "error": None, @@ -935,7 +932,6 @@ def _resolve_test_spec(test_spec): "runs_on": runner_label, "install_script": "", "install_timeout": "", - "grace_blackwell": "0", "rdma_devices": "", "error": None, } @@ -954,7 +950,7 @@ def _resolve_test_spec(test_spec): print( f"Resolved: file={resolved_path}, selector={test_selector}, " f"suite={info['suite']}, mode={mode}, runs_on={info['runner_label']}, " - f"install={info['install_script']}, grace_blackwell={info['grace_blackwell']}, " + f"install={info['install_script']}, " f"rdma={info['rdma_devices']}, " f"command='{test_command}'" ) @@ -966,7 +962,6 @@ def _resolve_test_spec(test_spec): "runs_on": info["runner_label"], "install_script": info["install_script"], "install_timeout": info["install_timeout"], - "grace_blackwell": info["grace_blackwell"], "rdma_devices": info["rdma_devices"], "error": None, } @@ -978,7 +973,7 @@ def _dispatch_batch(gh_repo, pr, batch, token, reply_comment_id="", reply_marker """ Dispatch a single workflow run for a batch of resolved test specs that share the same dispatch shape (mode + runs_on + install_script + - install_timeout + grace_blackwell + rdma_devices). + install_timeout + rdma_devices). Returns a dict with keys: specs, success, test_commands, runner_label, run_url, error. """ @@ -987,7 +982,6 @@ def _dispatch_batch(gh_repo, pr, batch, token, reply_comment_id="", reply_marker runs_on = batch[0]["runs_on"] install_script = batch[0]["install_script"] install_timeout = batch[0]["install_timeout"] - grace_blackwell = batch[0]["grace_blackwell"] rdma_devices = batch[0]["rdma_devices"] # Join multiple commands with newlines for the workflow to iterate over @@ -1020,7 +1014,6 @@ def _dispatch_batch(gh_repo, pr, batch, token, reply_comment_id="", reply_marker "runs_on": runs_on or "", "install_script": install_script, "install_timeout": install_timeout or "20", - "grace_blackwell": grace_blackwell or "0", "rdma_devices": rdma_devices, "reply_comment_id": str(reply_comment_id) if reply_comment_id else "", "reply_marker": reply_marker, @@ -1128,7 +1121,7 @@ def handle_rerun_test( """ Handles the /rerun-test command. Resolves all test specs, groups them by dispatch shape (mode + runs_on + install_script + install_timeout + - grace_blackwell + rdma_devices), and dispatches one workflow per group. + rdma_devices), and dispatches one workflow per group. """ if not skip_permission_check and not _check_rerun_test_permissions( gh_repo, pr, comment, user_perms, "rerun-test" @@ -1223,7 +1216,6 @@ def handle_rerun_test( r["runs_on"], r["install_script"], r["install_timeout"], - r["grace_blackwell"], r["rdma_devices"], ) groups.setdefault(key, []).append(r) diff --git a/test/registered/attention/unittests/dense/test_tbo.py b/test/registered/attention/unittests/dense/test_tbo.py index 222ff4e6f..1b98fe0e3 100644 --- a/test/registered/attention/unittests/dense/test_tbo.py +++ b/test/registered/attention/unittests/dense/test_tbo.py @@ -102,7 +102,7 @@ class TestTboAttnDenseAttentionBackendCorrectness(CustomTestCase): only by the in_capture=True branch (via ``_bind_metadata_buffers``). If TBO short-circuits its capture to its own replay path, those dicts are empty and replay raises ``KeyError: bs``. Reproduces the - deepep-4-gpu-h100 failure where + 4-gpu-h100 failure where ``flashattention_backend.target_verify_metadata[bs]`` lookup blew up during ``init_device_graphs``. diff --git a/test/registered/cp/test_deepseek_v3_cp_single_node.py b/test/registered/cp/test_deepseek_v3_cp_single_node.py index 1ced1238a..0c68500fd 100644 --- a/test/registered/cp/test_deepseek_v3_cp_single_node.py +++ b/test/registered/cp/test_deepseek_v3_cp_single_node.py @@ -13,7 +13,7 @@ from sglang.test.test_utils import ( write_github_step_summary, ) -register_cuda_ci(est_time=500, stage="extra-b", runner_config="deepep-8-gpu-h200") +register_cuda_ci(est_time=500, stage="extra-b", runner_config="8-gpu-h200") DEEPSEEK_V3_MODEL_PATH = "deepseek-ai/DeepSeek-V3-0324" diff --git a/test/registered/cp/test_deepseek_v4_flash_fp4_b200_cp.py b/test/registered/cp/test_deepseek_v4_flash_fp4_b200_cp.py index 74b7ced61..e4cd6bbe6 100644 --- a/test/registered/cp/test_deepseek_v4_flash_fp4_b200_cp.py +++ b/test/registered/cp/test_deepseek_v4_flash_fp4_b200_cp.py @@ -22,7 +22,7 @@ from sglang.test.test_utils import ( try_cached_model, ) -register_cuda_ci(est_time=235, stage="extra-b", runner_config="deepep-4-gpu-b200") +register_cuda_ci(est_time=235, stage="extra-b", runner_config="4-gpu-b200") MODEL = "deepseek-ai/DeepSeek-V4-Flash" SERVER_LAUNCH_TIMEOUT = 3600 diff --git a/test/registered/cp/test_dsa_prefill_cp.py b/test/registered/cp/test_dsa_prefill_cp.py index 2bd657549..3d82f33d7 100644 --- a/test/registered/cp/test_dsa_prefill_cp.py +++ b/test/registered/cp/test_dsa_prefill_cp.py @@ -13,7 +13,7 @@ from sglang.test.test_utils import ( write_github_step_summary, ) -register_cuda_ci(est_time=320, stage="extra-b", runner_config="deepep-8-gpu-h200") +register_cuda_ci(est_time=320, stage="extra-b", runner_config="8-gpu-h200") GLM52_MODEL_PATH = "zai-org/GLM-5.2-FP8" SERVER_LAUNCH_TIMEOUT = max(DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH, 1800) diff --git a/test/registered/cp/test_dsa_prefill_cp_legacy.py b/test/registered/cp/test_dsa_prefill_cp_legacy.py index 664e0b85b..34053f798 100644 --- a/test/registered/cp/test_dsa_prefill_cp_legacy.py +++ b/test/registered/cp/test_dsa_prefill_cp_legacy.py @@ -13,7 +13,7 @@ from sglang.test.test_utils import ( write_github_step_summary, ) -register_cuda_ci(est_time=616, stage="extra-b", runner_config="deepep-8-gpu-h200") +register_cuda_ci(est_time=616, stage="extra-b", runner_config="8-gpu-h200") DEEPSEEK_V32_MODEL_PATH = "deepseek-ai/DeepSeek-V3.2" diff --git a/test/registered/cp/test_gqa_prefill_cp.py b/test/registered/cp/test_gqa_prefill_cp.py index 4f49c143d..f4a939d85 100644 --- a/test/registered/cp/test_gqa_prefill_cp.py +++ b/test/registered/cp/test_gqa_prefill_cp.py @@ -11,7 +11,7 @@ from sglang.test.test_utils import ( popen_launch_server, ) -register_cuda_ci(est_time=500, stage="extra-b", runner_config="deepep-4-gpu-h100") +register_cuda_ci(est_time=500, stage="extra-b", runner_config="4-gpu-h100") GQA_MODEL_PATH = "Qwen/Qwen3-30B-A3B-FP8" diff --git a/test/registered/disaggregation/test_disaggregation_dsv4.py b/test/registered/disaggregation/test_disaggregation_dsv4.py index deb430331..4c3a3fab7 100644 --- a/test/registered/disaggregation/test_disaggregation_dsv4.py +++ b/test/registered/disaggregation/test_disaggregation_dsv4.py @@ -12,7 +12,7 @@ from sglang.test.test_utils import ( try_cached_model, ) -register_cuda_ci(est_time=500, stage="base-c", runner_config="deepep-8-gpu-h200") +register_cuda_ci(est_time=500, stage="base-c", runner_config="8-gpu-h200") DSV4_FLASH_MODEL = "sgl-project/DeepSeek-V4-Flash-FP8" diff --git a/test/registered/disaggregation/test_disaggregation_hisparse.py b/test/registered/disaggregation/test_disaggregation_hisparse.py index be391f632..4560de7dd 100644 --- a/test/registered/disaggregation/test_disaggregation_hisparse.py +++ b/test/registered/disaggregation/test_disaggregation_hisparse.py @@ -12,7 +12,7 @@ from sglang.test.test_utils import ( try_cached_model, ) -register_cuda_ci(est_time=1000, stage="extra-b", runner_config="deepep-8-gpu-h200") +register_cuda_ci(est_time=1000, stage="extra-b", runner_config="8-gpu-h200") DSV4_FLASH_MODEL = "sgl-project/DeepSeek-V4-Flash-FP8" DSV4_FLASH_LOADER_CONFIG = '{"enable_multithread_load": true, "num_threads": 64}' diff --git a/test/registered/ep/test_deepep_large.py b/test/registered/ep/test_deepep_large.py index fed39170c..c4bf5e959 100644 --- a/test/registered/ep/test_deepep_large.py +++ b/test/registered/ep/test_deepep_large.py @@ -15,7 +15,7 @@ from sglang.test.test_utils import ( popen_launch_server, ) -register_cuda_ci(est_time=528, stage="extra-b", runner_config="deepep-8-gpu-h200") +register_cuda_ci(est_time=528, stage="extra-b", runner_config="8-gpu-h200") DEEPSEEK_V32_MODEL_PATH = "deepseek-ai/DeepSeek-V3.2" diff --git a/test/registered/ep/test_deepep_small.py b/test/registered/ep/test_deepep_small.py index 01edfc238..56df75dc9 100644 --- a/test/registered/ep/test_deepep_small.py +++ b/test/registered/ep/test_deepep_small.py @@ -16,7 +16,7 @@ from sglang.test.test_utils import ( popen_launch_server, ) -register_cuda_ci(est_time=478, stage="base-c", runner_config="deepep-4-gpu-h100") +register_cuda_ci(est_time=478, stage="base-c", runner_config="4-gpu-h100") class TestPureDP(CustomTestCase): diff --git a/test/registered/ep/test_mooncake_ep_small.py b/test/registered/ep/test_mooncake_ep_small.py index 1db6b8ad8..40a65b97c 100644 --- a/test/registered/ep/test_mooncake_ep_small.py +++ b/test/registered/ep/test_mooncake_ep_small.py @@ -21,7 +21,7 @@ from sglang.test.test_utils import ( register_cuda_ci( est_time=189, stage="base-c", - runner_config="deepep-4-gpu-h100", + runner_config="4-gpu-h100", disabled="Temporarily disabled until the next Mooncake release includes the PyTorch 2.13 collective forwarding fix.", ) diff --git a/test/registered/ep/test_tbo_shared_experts_fusion.py b/test/registered/ep/test_tbo_shared_experts_fusion.py index 265b436bf..088106976 100644 --- a/test/registered/ep/test_tbo_shared_experts_fusion.py +++ b/test/registered/ep/test_tbo_shared_experts_fusion.py @@ -12,7 +12,7 @@ from sglang.test.test_utils import ( popen_launch_server, ) -register_cuda_ci(est_time=900, stage="extra-b", runner_config="deepep-8-gpu-h200") +register_cuda_ci(est_time=900, stage="extra-b", runner_config="8-gpu-h200") DEEPSEEK_V3_MODEL_PATH = "deepseek-ai/DeepSeek-V3-0324" diff --git a/test/registered/models_e2e/test_deepseek_v4_flash_fp4_b200.py b/test/registered/models_e2e/test_deepseek_v4_flash_fp4_b200.py index af7f02902..ed9cc83ba 100644 --- a/test/registered/models_e2e/test_deepseek_v4_flash_fp4_b200.py +++ b/test/registered/models_e2e/test_deepseek_v4_flash_fp4_b200.py @@ -4,7 +4,7 @@ Launches TP=4 with flashinfer_mxfp4 MoE runner + EAGLE speculative decoding. Runs 12 ServerSanity probes (correctness, streaming, concurrency, determinism) plus a GSM8K accuracy gate. -Registry: base-c-test-deepep-4-gpu-b200 (per-commit, 4x B200) +Registry: base-c-test-4-gpu-b200 (per-commit, 4x B200) """ import unittest @@ -21,7 +21,7 @@ from sglang.test.test_utils import ( try_cached_model, ) -register_cuda_ci(est_time=465, stage="base-c", runner_config="deepep-4-gpu-b200") +register_cuda_ci(est_time=465, stage="base-c", runner_config="4-gpu-b200") MODEL = "deepseek-ai/DeepSeek-V4-Flash" SERVER_LAUNCH_TIMEOUT = 3600 diff --git a/test/registered/models_e2e/test_deepseek_v4_flash_fp4_h200.py b/test/registered/models_e2e/test_deepseek_v4_flash_fp4_h200.py index 84ce32184..865852452 100644 --- a/test/registered/models_e2e/test_deepseek_v4_flash_fp4_h200.py +++ b/test/registered/models_e2e/test_deepseek_v4_flash_fp4_h200.py @@ -7,7 +7,7 @@ plus a GSM8K accuracy gate. Also covers SGLANG_DSV4_FP4_DEQUANT=1 (TP=8): FP4 experts dequantized to FP8 during loading and served through the plain FP8 MoE path. -Registry: base-c-test-deepep-8-gpu-h200 (per-commit, 8x H200) +Registry: base-c-test-8-gpu-h200 (per-commit, 8x H200) """ import unittest @@ -24,7 +24,7 @@ from sglang.test.test_utils import ( try_cached_model, ) -register_cuda_ci(est_time=600, stage="base-c", runner_config="deepep-8-gpu-h200") +register_cuda_ci(est_time=600, stage="base-c", runner_config="8-gpu-h200") def _flashinfer_has_sm90_cutlass_mxfp4() -> bool: diff --git a/test/registered/models_e2e/test_deepseek_v4_flash_fp4_megamoe_b200.py b/test/registered/models_e2e/test_deepseek_v4_flash_fp4_megamoe_b200.py index f24611048..8ffa029ca 100644 --- a/test/registered/models_e2e/test_deepseek_v4_flash_fp4_megamoe_b200.py +++ b/test/registered/models_e2e/test_deepseek_v4_flash_fp4_megamoe_b200.py @@ -4,7 +4,7 @@ Launches TP=4 with flashinfer_mxfp4 MoE runner + EAGLE speculative decoding. Runs 12 ServerSanity probes (correctness, streaming, concurrency, determinism) plus a GSM8K accuracy gate. -Registry: extra-b-test-deepep-4-gpu-b200 (label-gated, 4x B200) +Registry: extra-b-test-4-gpu-b200 (label-gated, 4x B200) """ import unittest @@ -21,7 +21,7 @@ from sglang.test.test_utils import ( try_cached_model, ) -register_cuda_ci(est_time=900, stage="extra-b", runner_config="deepep-4-gpu-b200") +register_cuda_ci(est_time=900, stage="extra-b", runner_config="4-gpu-b200") MODEL = "deepseek-ai/DeepSeek-V4-Flash" SERVER_LAUNCH_TIMEOUT = 3600 diff --git a/test/registered/models_e2e/test_deepseek_v4_flash_fp8_h200.py b/test/registered/models_e2e/test_deepseek_v4_flash_fp8_h200.py index f5a6e0cd7..0d945d38f 100644 --- a/test/registered/models_e2e/test_deepseek_v4_flash_fp8_h200.py +++ b/test/registered/models_e2e/test_deepseek_v4_flash_fp8_h200.py @@ -5,7 +5,7 @@ with FP4 experts disabled via SGLANG_DSV4_FP4_EXPERTS=0. Runs 12 ServerSanity probes (correctness, streaming, concurrency, determinism) plus a GSM8K accuracy gate. -Registry: extra-b-test-deepep-8-gpu-h200 (label-gated, 8x H200 — only 4 used by TP=4) +Registry: extra-b-test-8-gpu-h200 (label-gated, 8x H200 — only 4 used by TP=4) """ import unittest @@ -22,7 +22,7 @@ from sglang.test.test_utils import ( try_cached_model, ) -register_cuda_ci(est_time=560, stage="extra-b", runner_config="deepep-8-gpu-h200") +register_cuda_ci(est_time=560, stage="extra-b", runner_config="8-gpu-h200") MODEL_FP8 = "sgl-project/DeepSeek-V4-Flash-FP8" SERVER_LAUNCH_TIMEOUT = 3600 diff --git a/test/registered/rl/test_return_routed_experts.py b/test/registered/rl/test_return_routed_experts.py index 723679ad6..d8ae2be83 100644 --- a/test/registered/rl/test_return_routed_experts.py +++ b/test/registered/rl/test_return_routed_experts.py @@ -24,7 +24,7 @@ from sglang.test.test_utils import ( popen_launch_server, ) -register_cuda_ci(est_time=400, stage="extra-b", runner_config="deepep-4-gpu-h100") +register_cuda_ci(est_time=400, stage="extra-b", runner_config="4-gpu-h100") # FP8 variant of Qwen3-30B-A3B: required because DeepEP normal/LL fast paths in # ep_moe/layer.py only run for {Fp8Config (via deep_gemm), W4AFp8Config, aiter, diff --git a/test/run_suite.py b/test/run_suite.py index cab88424f..1af0d5b20 100644 --- a/test/run_suite.py +++ b/test/run_suite.py @@ -83,9 +83,6 @@ PER_COMMIT_SUITES = { "base-c-test-8-gpu-h200", "base-c-test-8-gpu-b200", "base-c-test-8-gpu-b300", - "base-c-test-deepep-4-gpu-h100", - "base-c-test-deepep-4-gpu-b200", - "base-c-test-deepep-8-gpu-h200", # extra-a / extra-b: label-gated PR opt-in suites in pr-test-extra.yml # (tests still tagged per-commit but skipped on default PR runs). "extra-a-test-1-gpu-small", @@ -94,9 +91,6 @@ PER_COMMIT_SUITES = { "extra-b-test-4-gpu-h100", "extra-b-test-4-gpu-b200", "extra-b-test-8-gpu-h200", - "extra-b-test-deepep-4-gpu-h100", - "extra-b-test-deepep-4-gpu-b200", - "extra-b-test-deepep-8-gpu-h200", ], HWBackend.NPU: [ "base-a-test-1-gpu-small",