From 340efca2447090665b769b0bda55fb5a054db696 Mon Sep 17 00:00:00 2001 From: Kangyan-Zhou Date: Thu, 30 Apr 2026 14:54:39 -0700 Subject: [PATCH] [sgl-kernel] Prep for torch 2.11 upgrade and switch PyPI default to cu130 (#24162) Co-authored-by: Claude Opus 4.7 (1M context) Co-authored-by: Baizhou Zhang --- .github/workflows/release-whl-kernel.yml | 72 +++++++++++------------- scripts/update_kernel_whl_index.py | 2 +- sgl-kernel/Dockerfile | 8 +-- sgl-kernel/README.md | 2 +- 4 files changed, 40 insertions(+), 44 deletions(-) diff --git a/.github/workflows/release-whl-kernel.yml b/.github/workflows/release-whl-kernel.yml index 98f6f24ea..775fafacc 100644 --- a/.github/workflows/release-whl-kernel.yml +++ b/.github/workflows/release-whl-kernel.yml @@ -34,6 +34,9 @@ concurrency: cancel-in-progress: true jobs: + # cu130 is the PyPI-released variant; cu129 wheels are published only to the + # sgl-project/whl index (consumed via `pip install ...+cu129` for the legacy + # cuda 12.9 path), not to PyPI. build-cu129-matrix: if: | github.repository == 'sgl-project/sglang' && @@ -78,43 +81,6 @@ jobs: BUILD_JOBS: 64 NVCC_THREADS: 8 - # rename_wheels.sh tags the cu129 wheel's METADATA Version with +cu129 - # (needed elsewhere — e.g. local install discrimination — see PR #23587), - # but PyPI rejects PEP 440 local version labels with HTTP 400. Repack a - # PyPI-clean copy with the +cu129 segment stripped into dist-pypi/, and - # upload that copy. dist/ is left untouched so the +cu129 wheel still - # flows through the upload-artifact -> sgl-project/whl index path below. - - name: Strip +cu129 local version for PyPI upload - working-directory: sgl-kernel - run: | - set -eux - pip install wheel - mkdir -p dist-pypi - for w in dist/*.whl; do - tmp=$(mktemp -d) - python3 -m wheel unpack "$w" --dest "$tmp" - unpacked=$(find "$tmp" -mindepth 1 -maxdepth 1 -type d | head -1) - info=$(find "$unpacked" -maxdepth 1 -type d -name "*.dist-info" | head -1) - meta="$info/METADATA" - orig=$(grep '^Version:' "$meta" | head -1 | sed 's/^Version:[[:space:]]*//') - new=$(echo "$orig" | sed 's/+cu[0-9]\+$//') - if [ "$orig" != "$new" ]; then - sed -i "s/^Version:.*/Version: ${new}/" "$meta" - old_base=$(basename "$info") - new_base="${old_base/${orig}/${new}}" - mv "$info" "$(dirname "$info")/${new_base}" - fi - python3 -m wheel pack "$unpacked" --dest-dir dist-pypi - rm -rf "$tmp" - done - ls -lh dist-pypi/ - - - name: Upload to PyPI - working-directory: sgl-kernel - run: | - pip install twine - python3 -m twine upload --skip-existing dist-pypi/* -u __token__ -p ${{ secrets.PYPI_TOKEN_SGLANG_KERNEL }} - - name: Upload artifacts uses: actions/upload-artifact@v4 with: @@ -172,7 +138,6 @@ jobs: git commit -m "update whl index" git push - # for now we do not release CUDA 13.0 wheels to pypi build-cu130-matrix: if: | github.repository == 'sgl-project/sglang' && @@ -217,6 +182,37 @@ jobs: BUILD_JOBS: 64 NVCC_THREADS: 8 + - name: Strip +cu130 local version for PyPI upload + working-directory: sgl-kernel + run: | + set -eux + pip install wheel + mkdir -p dist-pypi + for w in dist/*.whl; do + tmp=$(mktemp -d) + python3 -m wheel unpack "$w" --dest "$tmp" + unpacked=$(find "$tmp" -mindepth 1 -maxdepth 1 -type d | head -1) + info=$(find "$unpacked" -maxdepth 1 -type d -name "*.dist-info" | head -1) + meta="$info/METADATA" + orig=$(grep '^Version:' "$meta" | head -1 | sed 's/^Version:[[:space:]]*//') + new=$(echo "$orig" | sed 's/+cu[0-9]\+$//') + if [ "$orig" != "$new" ]; then + sed -i "s/^Version:.*/Version: ${new}/" "$meta" + old_base=$(basename "$info") + new_base="${old_base/${orig}/${new}}" + mv "$info" "$(dirname "$info")/${new_base}" + fi + python3 -m wheel pack "$unpacked" --dest-dir dist-pypi + rm -rf "$tmp" + done + ls -lh dist-pypi/ + + - name: Upload to PyPI + working-directory: sgl-kernel + run: | + pip install twine + python3 -m twine upload --skip-existing dist-pypi/* -u __token__ -p ${{ secrets.PYPI_TOKEN_SGLANG_KERNEL }} + - name: Upload artifacts uses: actions/upload-artifact@v4 with: diff --git a/scripts/update_kernel_whl_index.py b/scripts/update_kernel_whl_index.py index 4c1a57a25..6e2fccbee 100644 --- a/scripts/update_kernel_whl_index.py +++ b/scripts/update_kernel_whl_index.py @@ -7,7 +7,7 @@ import re # All the CUDA versions that the wheels will cover SUPPORTED_CUDA_VERSIONS = ["129", "130"] -DEFAULT_CUDA_VERSION = "129" +DEFAULT_CUDA_VERSION = "130" def check_wheel_cuda_version(path_name, target_cuda_version): diff --git a/sgl-kernel/Dockerfile b/sgl-kernel/Dockerfile index 9300568c8..67c63fb05 100644 --- a/sgl-kernel/Dockerfile +++ b/sgl-kernel/Dockerfile @@ -79,10 +79,10 @@ RUN set -eux; \ RUN --mount=type=cache,id=sgl-kernel-pip,target=/root/.cache/pip \ set -eux; \ case "${CUDA_VERSION}" in \ - 13.0) TORCH_VER=2.9.1; CU_TAG=cu130 ;; \ - 12.9) TORCH_VER=2.9.1; CU_TAG=cu128 ;; \ - 12.8) TORCH_VER=2.9.1; CU_TAG=cu128 ;; \ - *) TORCH_VER=2.9.1; CU_TAG=cu126 ;; \ + 13.0) TORCH_VER=2.11.0; CU_TAG=cu130 ;; \ + 12.9) TORCH_VER=2.11.0; CU_TAG=cu129 ;; \ + 12.8) TORCH_VER=2.11.0; CU_TAG=cu128 ;; \ + *) TORCH_VER=2.11.0; CU_TAG=cu126 ;; \ esac; \ ${PYTHON_ROOT_PATH}/bin/pip install torch==${TORCH_VER} --index-url https://${PYTORCH_MIRROR}/whl/${CU_TAG}; \ ${PYTHON_ROOT_PATH}/bin/pip install ninja setuptools==75.0.0 wheel==0.41.0 numpy uv scikit-build-core --index-url ${PIP_DEFAULT_INDEX} diff --git a/sgl-kernel/README.md b/sgl-kernel/README.md index a32000d39..de3bdf05d 100644 --- a/sgl-kernel/README.md +++ b/sgl-kernel/README.md @@ -12,7 +12,7 @@ `sglang-kernel` provides optimized compute primitives for LLM inference engines, enabling efficient inference for large language models and vision-language models through custom kernel operations. The source tree remains under the `sgl-kernel/` directory and the Python import path remains `sgl_kernel`. ## Installation -Requires torch == 2.9.1 +Requires torch == 2.11.0 ```bash # Latest version