Files
sglang/.github/workflows/pr-test-xpu.yml
T

309 lines
15 KiB
YAML

name: PR Test (XPU)
on:
push:
branches: [ main ]
pull_request:
branches: [ main ]
workflow_dispatch:
workflow_call:
inputs:
ref:
description: 'Git ref (branch, tag, or SHA) to test. If not provided, uses the default branch.'
required: false
type: string
default: ''
run_all_tests:
description: "Run all tests (for releasing or testing purpose)"
required: false
type: boolean
default: false
concurrency:
group: pr-test-xpu-${{ inputs.ref || github.ref }}
cancel-in-progress: ${{ github.event_name != 'workflow_call' }}
jobs:
# ==================== Check Changes ==================== #
check-changes:
runs-on: ubuntu-latest
outputs:
changes_exist: ${{ steps.filter.outputs.main_package == 'true' || steps.filter.outputs.multimodal_gen == 'true' || steps.run-mode.outputs.run_all_tests == 'true' }}
main_package: ${{ steps.filter.outputs.main_package == 'true' || steps.run-mode.outputs.run_all_tests == 'true' }}
multimodal_gen: ${{ steps.filter.outputs.multimodal_gen == 'true' || steps.run-mode.outputs.run_all_tests == 'true' }}
steps:
- name: Checkout code
uses: actions/checkout@v4
with:
ref: ${{ inputs.ref || github.ref }}
- name: Determine run mode
id: run-mode
run: |
# Run all tests for workflow_call (when ref input is provided)
# Note: github.event_name is inherited from caller, so we detect workflow_call by checking inputs.ref
if [[ "${{ inputs.run_all_tests }}" == "true" ]]; then
echo "run_all_tests=true" >> $GITHUB_OUTPUT
echo "Run mode: ALL TESTS (run_all_tests=${{ inputs.run_all_tests }})"
else
echo "run_all_tests=false" >> $GITHUB_OUTPUT
echo "Run mode: FILTERED (triggered by ${{ github.event_name }})"
fi
- name: Detect file changes
id: filter
uses: dorny/paths-filter@v3
if: steps.run-mode.outputs.run_all_tests != 'true'
with:
filters: |
main_package:
# Extend test/registered/ entries when adding a non-nightly register_xpu_ci.
- "python/sglang/!(multimodal_gen|kernels|cli|test)/**/!(*.md)"
- "python/sglang/test/*.py"
- "python/sglang/test/!(ascend|observability|mock_model|manual|external_models|kernels)/**/!(*.md)"
- "python/pyproject_xpu.toml"
- "test/registered/xpu/**/!(*.md)"
- "test/registered/attention/test_chunk_gated_delta_rule.py"
- "test/registered/attention/test_deterministic.py"
- "test/registered/lora/test_moe_lora_info.py"
- "test/registered/lora/test_virtual_experts_kernels.py"
- "test/registered/unit/sampling/test_sampling_params.py"
- "test/registered/unit/spec/test_adaptive_spec_params.py"
- "test/run_suite.py"
- "test/pytest.ini"
- ".github/workflows/pr-test-xpu.yml"
- "docker/xpu.Dockerfile"
- "scripts/ci/xpu/**"
multimodal_gen:
# Only paths imported by the 1-gpu-xpu suite; other-vendor and off-suite files are excluded.
- "python/sglang/multimodal_gen/*.py"
- "python/sglang/multimodal_gen/runtime/!(pipelines)/**/!(*.md|*.ipynb)"
- "python/sglang/multimodal_gen/runtime/*.py"
- "python/sglang/multimodal_gen/runtime/pipelines/__init__.py"
- "python/sglang/multimodal_gen/runtime/pipelines/@(zimage_pipeline|flux_2_klein|flux_2|wan_pipeline|diffusers_pipeline).py"
- "python/sglang/multimodal_gen/configs/**/!(*.md|*.ipynb)"
- "python/sglang/multimodal_gen/apps/webui/**/!(*.md|*.ipynb)"
- "python/sglang/multimodal_gen/benchmarks/compare_perf.py"
- "python/sglang/multimodal_gen/test/*.py"
- "python/sglang/multimodal_gen/test/runner/**/!(*.md|*.ipynb)"
- "python/sglang/multimodal_gen/test/server/!(test_server_1_gpu_5090|test_server_4_gpu_h100|test_server_b200|test_server_2_gpu|test_request_logger).py"
- "python/sglang/multimodal_gen/test/server/common/**/!(*.md|*.ipynb)"
- "python/sglang/multimodal_gen/test/server/perf_baselines/xpu_b60.json"
- "python/pyproject_xpu.toml"
- ".github/workflows/pr-test-xpu.yml"
- "docker/xpu.Dockerfile"
- "scripts/ci/xpu/**"
# ==================== PR Gate ==================== #
pr-gate:
needs: check-changes
if: needs.check-changes.outputs.changes_exist == 'true'
uses: ./.github/workflows/pr-gate.yml
secrets: inherit
# Runs the entire per-commit XPU test suite in one job (single install, single queue).
stage-a-test-1-gpu-xpu:
needs: [check-changes, pr-gate]
if: needs.check-changes.outputs.main_package == 'true'
runs-on: intel-bmg
env:
DOCKERHUB_INTEL_USERNAME: ${{ secrets.DOCKERHUB_INTEL_USERNAME }}
DOCKERHUB_INTEL_TOKEN: ${{ secrets.DOCKERHUB_INTEL_TOKEN }}
steps:
- name: Reset workspace ownership
run: |
# Prior runs install sglang as root inside the CI container against
# the bind-mounted workspace, leaving root-owned egg-info on the host.
# actions/checkout's pre-clean step can't unlink those as the runner
# user, so reset ownership before checkout runs.
docker run --rm -v "${{ github.workspace }}:/w" busybox:latest \
chown -R "$(id -u):$(id -g)" /w || true
- name: Checkout code
uses: actions/checkout@v4
with:
fetch-depth: 0
ref: ${{ inputs.ref || github.ref }}
- name: Start CI container
run: |
export HF_TOKEN="$(cat ~/huggingface_token.txt)"
bash scripts/ci/xpu/xpu_ci_start_container.sh
env:
GITHUB_WORKSPACE: ${{ github.workspace }}
- name: Install Dependency
timeout-minutes: 60
run: |
# Bridge fix: install libssl-dev in the running container so the
# JIT build of hicache_hash_cpp (needs <openssl/sha.h>) succeeds.
# Can be dropped once intel/sglang-dev:latest is rebuilt from a
# docker/xpu.Dockerfile that already ships libssl-dev.
docker exec -e DEBIAN_FRONTEND=noninteractive ci_sglang_xpu apt-get update
docker exec -e DEBIAN_FRONTEND=noninteractive ci_sglang_xpu apt-get install -y libssl-dev
docker exec ci_sglang_xpu /opt/venv/bin/python3 -m pip install --upgrade pip
docker exec ci_sglang_xpu /opt/venv/bin/python3 -m pip install pytest expecttest ray huggingface_hub tabulate "lmcache>=0.3.9" accelerate
docker exec ci_sglang_xpu /opt/venv/bin/python3 -m pip uninstall -y flashinfer-python sgl-kernel sglang
docker exec ci_sglang_xpu cp /sglang-checkout/python/pyproject_xpu.toml /sglang-checkout/python/pyproject.toml
# Fetch tags so setuptools_scm resolves a real version instead of
# falling back to 0.0.0 on a shallow/tag-less checkout.
docker exec -w /sglang-checkout ci_sglang_xpu git fetch origin '+refs/tags/*:refs/tags/*' --force
docker exec -w /sglang-checkout/python ci_sglang_xpu /opt/venv/bin/python3 -m pip install --no-cache-dir . --extra-index-url https://download.pytorch.org/whl/xpu
docker exec ci_sglang_xpu /opt/venv/bin/python3 -m pip install --no-cache-dir --no-deps xgrammar==0.1.33
docker exec ci_sglang_xpu /bin/bash -c '/opt/venv/bin/hf auth login --token ${HF_TOKEN}'
- name: Install checkpoint-engine extra (optional; tests skip if unavailable)
continue-on-error: true
run: |
docker exec -w /sglang-checkout/python ci_sglang_xpu /opt/venv/bin/python3 -m pip install --no-cache-dir ".[checkpoint-engine]" --extra-index-url https://download.pytorch.org/whl/xpu
- name: Run tests
# Old stage-a (30) + stage-b (120) = 150; +20% headroom for --enable-retry.
timeout-minutes: 180
run: |
docker exec ci_sglang_xpu bash -c "source /opt/venv/bin/activate && cd /sglang-checkout/test && python3 run_suite.py --hw xpu --suite stage-a-test-1-gpu-xpu,stage-b-test-1-gpu-xpu --enable-retry"
- name: Cleanup container
if: always()
run: |
# pip install ran as root inside the container against the
# bind-mounted workspace, so build artifacts are root-owned on
# the host. Chown them back before rm to avoid Permission denied.
docker run --rm -v "${{ github.workspace }}:/w" busybox:latest \
chown -R "$(id -u):$(id -g)" /w || true
# Wipe everything the run wrote into the workspace so the next
# job starts from a clean tree.
rm -rf \
python/build \
python/dist \
python/sglang.egg-info \
python/sglang/*.egg-info \
test/result.jsonl \
test/results \
test/.pytest_cache \
.pytest_cache || true
find . -type d -name "__pycache__" -prune -exec rm -rf {} + || true
find . -type f -name "*.pyc" -delete || true
# SIGTERM sglang and drain GPU context before `docker rm -f`;
# SIGKILL leaves the xe/GuC exec queue registered and triggers a
# GT reset (+ devcoredump) on B580.
if docker ps --format '{{.Names}}' | grep -qx ci_sglang_xpu; then
docker exec ci_sglang_xpu bash -c '
pkill -TERM -f "sglang|run_suite|python3.*test_" 2>/dev/null || true
for _ in $(seq 1 30); do
pgrep -f "sglang::|sglang.launch_server" >/dev/null || break
sleep 1
done
pkill -KILL -f "sglang|run_suite" 2>/dev/null || true
' || true
fi
docker rm -f ci_sglang_xpu || true
if [[ -n "${CI_SGLANG_XPU_IMAGE:-}" ]]; then
docker rmi -f "${CI_SGLANG_XPU_IMAGE}" || true
fi
# ==================== Multimodal Gen ==================== #
multimodal-gen-test-1-gpu-xpu:
needs: [check-changes, pr-gate]
if: needs.check-changes.outputs.multimodal_gen == 'true'
runs-on: bmg-multigen-models
env:
DOCKERHUB_INTEL_USERNAME: ${{ secrets.DOCKERHUB_INTEL_USERNAME }}
DOCKERHUB_INTEL_TOKEN: ${{ secrets.DOCKERHUB_INTEL_TOKEN }}
steps:
- name: Reset workspace ownership
run: |
docker run --rm -v "${{ github.workspace }}:/w" busybox:latest \
chown -R "$(id -u):$(id -g)" /w || true
- name: Checkout code
uses: actions/checkout@v4
with:
fetch-depth: 0
ref: ${{ inputs.ref || github.ref }}
- name: Start CI container
run: |
export HF_TOKEN="$(cat ~/huggingface_token.txt)"
bash scripts/ci/xpu/xpu_ci_start_container.sh
env:
GITHUB_WORKSPACE: ${{ github.workspace }}
- name: Install Dependency
timeout-minutes: 60
run: |
# Bridge fix: install libssl-dev in the running container so the
# JIT build of hicache_hash_cpp (needs <openssl/sha.h>) succeeds.
# Can be dropped once intel/sglang-dev:latest is rebuilt from a
# docker/xpu.Dockerfile that already ships libssl-dev.
docker exec -e DEBIAN_FRONTEND=noninteractive ci_sglang_xpu apt-get update
docker exec -e DEBIAN_FRONTEND=noninteractive ci_sglang_xpu apt-get install -y libssl-dev
docker exec ci_sglang_xpu /opt/venv/bin/python3 -m pip install --upgrade pip
docker exec ci_sglang_xpu /opt/venv/bin/python3 -m pip install pytest expecttest ray huggingface_hub tabulate "lmcache>=0.3.9"
docker exec ci_sglang_xpu /opt/venv/bin/python3 -m pip uninstall -y flashinfer-python sgl-kernel sglang
docker exec ci_sglang_xpu cp /sglang-checkout/python/pyproject_xpu.toml /sglang-checkout/python/pyproject.toml
# Fetch tags so setuptools_scm resolves a real version instead of
# falling back to 0.0.0 on a shallow/tag-less checkout.
docker exec -w /sglang-checkout ci_sglang_xpu git fetch origin '+refs/tags/*:refs/tags/*' --force
docker exec -w /sglang-checkout/python ci_sglang_xpu /opt/venv/bin/python3 -m pip install --no-cache-dir . --extra-index-url https://download.pytorch.org/whl/xpu
docker exec ci_sglang_xpu /opt/venv/bin/python3 -m pip install --no-cache-dir --no-deps xgrammar==0.1.33
docker exec ci_sglang_xpu /bin/bash -c '/opt/venv/bin/hf auth login --token ${HF_TOKEN}'
- name: Run diffusion server tests (1-GPU)
timeout-minutes: 60
run: |
docker exec ci_sglang_xpu bash -c "source /opt/venv/bin/activate && cd /sglang-checkout/python && python3 sglang/multimodal_gen/test/run_suite.py --suite 1-gpu-xpu"
- name: Cleanup container
if: always()
run: |
docker run --rm -v "${{ github.workspace }}:/w" busybox:latest \
chown -R "$(id -u):$(id -g)" /w || true
rm -rf \
python/build \
python/dist \
python/sglang.egg-info \
python/sglang/*.egg-info \
test/result.jsonl \
test/results \
test/.pytest_cache \
.pytest_cache || true
find . -type d -name "__pycache__" -prune -exec rm -rf {} + || true
find . -type f -name "*.pyc" -delete || true
# SIGTERM sglang and drain GPU context before `docker rm -f`;
# SIGKILL leaves the xe/GuC exec queue registered and triggers a
# GT reset (+ devcoredump) on B580.
if docker ps --format '{{.Names}}' | grep -qx ci_sglang_xpu; then
docker exec ci_sglang_xpu bash -c '
pkill -TERM -f "sglang|run_suite|python3.*test_" 2>/dev/null || true
for _ in $(seq 1 30); do
pgrep -f "sglang::|sglang.launch_server" >/dev/null || break
sleep 1
done
pkill -KILL -f "sglang|run_suite" 2>/dev/null || true
' || true
fi
docker rm -f ci_sglang_xpu || true
if [[ -n "${CI_SGLANG_XPU_IMAGE:-}" ]]; then
docker rmi -f "${CI_SGLANG_XPU_IMAGE}" || true
fi
finish:
if: always()
needs: [stage-a-test-1-gpu-xpu, multimodal-gen-test-1-gpu-xpu, pr-gate]
runs-on: ubuntu-latest
steps:
- name: Check job status
run: |
stage_a="${{ needs.stage-a-test-1-gpu-xpu.result }}"
multimodal_gen="${{ needs.multimodal-gen-test-1-gpu-xpu.result }}"
if [ "$stage_a" != "success" ] && [ "$stage_a" != "skipped" ]; then
echo "stage-a failed with result: $stage_a"
exit 1
fi
if [ "$multimodal_gen" != "success" ] && [ "$multimodal_gen" != "skipped" ]; then
echo "multimodal-gen failed with result: $multimodal_gen"
exit 1
fi
echo "All jobs completed successfully"
exit 0