312 lines
15 KiB
YAML
312 lines
15 KiB
YAML
name: PR Test (XPU)
|
|
|
|
on:
|
|
push:
|
|
branches: [ main ]
|
|
pull_request:
|
|
branches: [ main ]
|
|
workflow_dispatch:
|
|
workflow_call:
|
|
inputs:
|
|
ref:
|
|
description: 'Git ref (branch, tag, or SHA) to test. If not provided, uses the default branch.'
|
|
required: false
|
|
type: string
|
|
default: ''
|
|
run_all_tests:
|
|
description: "Run all tests (for releasing or testing purpose)"
|
|
required: false
|
|
type: boolean
|
|
default: false
|
|
|
|
concurrency:
|
|
group: pr-test-xpu-${{ inputs.ref || github.ref }}
|
|
cancel-in-progress: ${{ github.event_name != 'workflow_call' }}
|
|
|
|
jobs:
|
|
# ==================== Check Changes ==================== #
|
|
check-changes:
|
|
runs-on: ubuntu-latest
|
|
outputs:
|
|
changes_exist: ${{ steps.filter.outputs.main_package == 'true' || steps.filter.outputs.multimodal_gen == 'true' || steps.run-mode.outputs.run_all_tests == 'true' }}
|
|
main_package: ${{ steps.filter.outputs.main_package == 'true' || steps.run-mode.outputs.run_all_tests == 'true' }}
|
|
multimodal_gen: ${{ steps.filter.outputs.multimodal_gen == 'true' || steps.run-mode.outputs.run_all_tests == 'true' }}
|
|
steps:
|
|
- name: Checkout code
|
|
uses: actions/checkout@v4
|
|
with:
|
|
ref: ${{ inputs.ref || github.ref }}
|
|
|
|
- name: Determine run mode
|
|
id: run-mode
|
|
run: |
|
|
# Run all tests for workflow_call (when ref input is provided)
|
|
# Note: github.event_name is inherited from caller, so we detect workflow_call by checking inputs.ref
|
|
if [[ "${{ inputs.run_all_tests }}" == "true" ]]; then
|
|
echo "run_all_tests=true" >> $GITHUB_OUTPUT
|
|
echo "Run mode: ALL TESTS (run_all_tests=${{ inputs.run_all_tests }})"
|
|
else
|
|
echo "run_all_tests=false" >> $GITHUB_OUTPUT
|
|
echo "Run mode: FILTERED (triggered by ${{ github.event_name }})"
|
|
fi
|
|
- name: Detect file changes
|
|
id: filter
|
|
uses: dorny/paths-filter@v3
|
|
if: steps.run-mode.outputs.run_all_tests != 'true'
|
|
with:
|
|
filters: |
|
|
main_package:
|
|
# Extend test/registered/ entries when adding a non-nightly register_xpu_ci.
|
|
- "python/sglang/!(multimodal_gen|kernels|cli|test)/**/!(*.md)"
|
|
- "python/sglang/test/*.py"
|
|
- "python/sglang/test/!(ascend|observability|mock_model|manual|external_models|kernels)/**/!(*.md)"
|
|
- "python/pyproject_xpu.toml"
|
|
- "test/registered/xpu/**/!(*.md)"
|
|
- "test/registered/disaggregation/test_disaggregation_xpu.py"
|
|
- "test/registered/attention/test_chunk_gated_delta_rule.py"
|
|
- "test/registered/attention/test_deterministic.py"
|
|
- "test/registered/lora/test_moe_lora_info.py"
|
|
- "test/registered/lora/test_virtual_experts_kernels.py"
|
|
- "test/registered/unit/sampling/test_sampling_params.py"
|
|
- "test/registered/unit/spec/test_adaptive_spec_params.py"
|
|
- "test/run_suite.py"
|
|
- "test/pytest.ini"
|
|
- ".github/workflows/pr-test-xpu.yml"
|
|
- "docker/xpu.Dockerfile"
|
|
- "scripts/ci/xpu/**"
|
|
multimodal_gen:
|
|
# Only paths imported by the 1-gpu-xpu suite; other-vendor and off-suite files are excluded.
|
|
- "python/sglang/multimodal_gen/*.py"
|
|
- "python/sglang/multimodal_gen/runtime/!(pipelines)/**/!(*.md|*.ipynb)"
|
|
- "python/sglang/multimodal_gen/runtime/*.py"
|
|
- "python/sglang/multimodal_gen/runtime/pipelines/__init__.py"
|
|
- "python/sglang/multimodal_gen/runtime/pipelines/@(zimage_pipeline|flux_2_klein|flux_2|wan_pipeline|diffusers_pipeline).py"
|
|
- "python/sglang/multimodal_gen/configs/**/!(*.md|*.ipynb)"
|
|
- "python/sglang/multimodal_gen/apps/webui/**/!(*.md|*.ipynb)"
|
|
- "python/sglang/multimodal_gen/benchmarks/compare_perf.py"
|
|
- "python/sglang/multimodal_gen/test/*.py"
|
|
- "python/sglang/multimodal_gen/test/runner/**/!(*.md|*.ipynb)"
|
|
- "python/sglang/multimodal_gen/test/server/!(test_server_1_gpu_5090|test_server_4_gpu_h100|test_server_b200|test_server_2_gpu|test_request_logger).py"
|
|
- "python/sglang/multimodal_gen/test/server/common/**/!(*.md|*.ipynb)"
|
|
- "python/sglang/multimodal_gen/test/server/perf_baselines/xpu_b60.json"
|
|
- "python/pyproject_xpu.toml"
|
|
- ".github/workflows/pr-test-xpu.yml"
|
|
- "docker/xpu.Dockerfile"
|
|
- "scripts/ci/xpu/**"
|
|
|
|
# ==================== PR Gate ==================== #
|
|
pr-gate:
|
|
needs: check-changes
|
|
if: needs.check-changes.outputs.changes_exist == 'true'
|
|
uses: ./.github/workflows/pr-gate.yml
|
|
secrets: inherit
|
|
|
|
# Runs the entire per-commit XPU test suite in one job (single install, single queue).
|
|
stage-a-test-1-gpu-xpu:
|
|
needs: [check-changes, pr-gate]
|
|
if: needs.check-changes.outputs.main_package == 'true'
|
|
runs-on: intel-bmg
|
|
env:
|
|
DOCKERHUB_INTEL_USERNAME: ${{ secrets.DOCKERHUB_INTEL_USERNAME }}
|
|
DOCKERHUB_INTEL_TOKEN: ${{ secrets.DOCKERHUB_INTEL_TOKEN }}
|
|
steps:
|
|
- name: Reset workspace ownership
|
|
run: |
|
|
# Prior runs install sglang as root inside the CI container against
|
|
# the bind-mounted workspace, leaving root-owned egg-info on the host.
|
|
# actions/checkout's pre-clean step can't unlink those as the runner
|
|
# user, so reset ownership before checkout runs.
|
|
docker run --rm -v "${{ github.workspace }}:/w" busybox:latest \
|
|
chown -R "$(id -u):$(id -g)" /w || true
|
|
|
|
- name: Checkout code
|
|
uses: actions/checkout@v4
|
|
with:
|
|
fetch-depth: 0
|
|
ref: ${{ inputs.ref || github.ref }}
|
|
|
|
- name: Start CI container
|
|
run: |
|
|
export HF_TOKEN="$(cat ~/huggingface_token.txt)"
|
|
bash scripts/ci/xpu/xpu_ci_start_container.sh
|
|
env:
|
|
GITHUB_WORKSPACE: ${{ github.workspace }}
|
|
|
|
- name: Install Dependency
|
|
timeout-minutes: 60
|
|
run: |
|
|
# Bridge fix: install libssl-dev in the running container so the
|
|
# JIT build of hicache_hash_cpp (needs <openssl/sha.h>) succeeds.
|
|
# Can be dropped once intel/sglang-dev:latest is rebuilt from a
|
|
# docker/xpu.Dockerfile that already ships libssl-dev.
|
|
docker exec -e DEBIAN_FRONTEND=noninteractive ci_sglang_xpu apt-get update
|
|
docker exec -e DEBIAN_FRONTEND=noninteractive ci_sglang_xpu apt-get install -y libssl-dev
|
|
docker exec ci_sglang_xpu /opt/venv/bin/python3 -m pip install --upgrade pip
|
|
docker exec ci_sglang_xpu /opt/venv/bin/python3 -m pip install pytest expecttest ray huggingface_hub tabulate "lmcache>=0.3.9" accelerate
|
|
docker exec ci_sglang_xpu /opt/venv/bin/python3 -m pip uninstall -y flashinfer-python sgl-kernel sglang
|
|
docker exec ci_sglang_xpu cp /sglang-checkout/python/pyproject_xpu.toml /sglang-checkout/python/pyproject.toml
|
|
# Fetch tags so setuptools_scm resolves a real version instead of
|
|
# falling back to 0.0.0 on a shallow/tag-less checkout.
|
|
docker exec -w /sglang-checkout ci_sglang_xpu git fetch origin '+refs/tags/*:refs/tags/*' --force
|
|
docker exec -w /sglang-checkout/python ci_sglang_xpu /opt/venv/bin/python3 -m pip install --no-cache-dir . --extra-index-url https://download.pytorch.org/whl/xpu
|
|
docker exec ci_sglang_xpu /opt/venv/bin/python3 -m pip install --no-cache-dir --no-deps xgrammar==0.1.33
|
|
docker exec ci_sglang_xpu /bin/bash -c '/opt/venv/bin/hf auth login --token ${HF_TOKEN}'
|
|
|
|
- name: Install checkpoint-engine extra (optional; tests skip if unavailable)
|
|
continue-on-error: true
|
|
run: |
|
|
docker exec -w /sglang-checkout/python ci_sglang_xpu /opt/venv/bin/python3 -m pip install --no-cache-dir ".[checkpoint-engine]" --extra-index-url https://download.pytorch.org/whl/xpu
|
|
|
|
- name: Run tests
|
|
# Old stage-a (30) + stage-b (120) = 150; +20% headroom for --enable-retry.
|
|
timeout-minutes: 180
|
|
run: |
|
|
docker exec ci_sglang_xpu bash -c "source /opt/venv/bin/activate && cd /sglang-checkout/test && python3 run_suite.py --hw xpu --suite stage-a-test-1-gpu-xpu,stage-b-test-1-gpu-xpu --enable-retry"
|
|
|
|
- name: Cleanup container
|
|
if: always()
|
|
run: |
|
|
# pip install ran as root inside the container against the
|
|
# bind-mounted workspace, so build artifacts are root-owned on
|
|
# the host. Chown them back before rm to avoid Permission denied.
|
|
docker run --rm -v "${{ github.workspace }}:/w" busybox:latest \
|
|
chown -R "$(id -u):$(id -g)" /w || true
|
|
# Wipe everything the run wrote into the workspace so the next
|
|
# job starts from a clean tree.
|
|
rm -rf \
|
|
python/build \
|
|
python/dist \
|
|
python/sglang.egg-info \
|
|
python/sglang/*.egg-info \
|
|
test/result.jsonl \
|
|
test/results \
|
|
test/.pytest_cache \
|
|
.pytest_cache || true
|
|
find . -type d -name "__pycache__" -prune -exec rm -rf {} + || true
|
|
find . -type f -name "*.pyc" -delete || true
|
|
# SIGTERM sglang and drain GPU context before `docker rm -f`;
|
|
# SIGKILL leaves the xe/GuC exec queue registered and triggers a
|
|
# GT reset (+ devcoredump) on B580.
|
|
if docker ps --format '{{.Names}}' | grep -qx ci_sglang_xpu; then
|
|
docker exec ci_sglang_xpu bash -c '
|
|
pkill -TERM -f "sglang|run_suite|python3.*test_" 2>/dev/null || true
|
|
for _ in $(seq 1 30); do
|
|
pgrep -f "sglang::|sglang.launch_server" >/dev/null || break
|
|
sleep 1
|
|
done
|
|
pkill -KILL -f "sglang|run_suite" 2>/dev/null || true
|
|
' || true
|
|
fi
|
|
docker rm -f ci_sglang_xpu || true
|
|
if [[ -n "${CI_SGLANG_XPU_IMAGE:-}" ]]; then
|
|
docker rmi -f "${CI_SGLANG_XPU_IMAGE}" || true
|
|
fi
|
|
|
|
# ==================== Multimodal Gen ==================== #
|
|
multimodal-gen-test-1-gpu-xpu:
|
|
needs: [check-changes, pr-gate]
|
|
if: needs.check-changes.outputs.multimodal_gen == 'true'
|
|
runs-on: bmg-multigen-models
|
|
env:
|
|
DOCKERHUB_INTEL_USERNAME: ${{ secrets.DOCKERHUB_INTEL_USERNAME }}
|
|
DOCKERHUB_INTEL_TOKEN: ${{ secrets.DOCKERHUB_INTEL_TOKEN }}
|
|
steps:
|
|
- name: Reset workspace ownership
|
|
run: |
|
|
docker run --rm -v "${{ github.workspace }}:/w" busybox:latest \
|
|
chown -R "$(id -u):$(id -g)" /w || true
|
|
|
|
- name: Checkout code
|
|
uses: actions/checkout@v4
|
|
with:
|
|
fetch-depth: 0
|
|
ref: ${{ inputs.ref || github.ref }}
|
|
|
|
- name: Start CI container
|
|
run: |
|
|
export HF_TOKEN="$(cat ~/huggingface_token.txt)"
|
|
bash scripts/ci/xpu/xpu_ci_start_container.sh
|
|
env:
|
|
GITHUB_WORKSPACE: ${{ github.workspace }}
|
|
|
|
- name: Install Dependency
|
|
timeout-minutes: 60
|
|
run: |
|
|
# Bridge fix: install libssl-dev in the running container so the
|
|
# JIT build of hicache_hash_cpp (needs <openssl/sha.h>) succeeds.
|
|
# Can be dropped once intel/sglang-dev:latest is rebuilt from a
|
|
# docker/xpu.Dockerfile that already ships libssl-dev.
|
|
docker exec -e DEBIAN_FRONTEND=noninteractive ci_sglang_xpu apt-get update
|
|
docker exec -e DEBIAN_FRONTEND=noninteractive ci_sglang_xpu apt-get install -y libssl-dev
|
|
docker exec ci_sglang_xpu /opt/venv/bin/python3 -m pip install --upgrade pip
|
|
docker exec ci_sglang_xpu /opt/venv/bin/python3 -m pip install pytest expecttest ray huggingface_hub tabulate "lmcache>=0.3.9"
|
|
docker exec ci_sglang_xpu /opt/venv/bin/python3 -m pip uninstall -y flashinfer-python sgl-kernel sglang
|
|
docker exec ci_sglang_xpu cp /sglang-checkout/python/pyproject_xpu.toml /sglang-checkout/python/pyproject.toml
|
|
# Fetch tags so setuptools_scm resolves a real version instead of
|
|
# falling back to 0.0.0 on a shallow/tag-less checkout.
|
|
docker exec -w /sglang-checkout ci_sglang_xpu git fetch origin '+refs/tags/*:refs/tags/*' --force
|
|
docker exec -w /sglang-checkout/python ci_sglang_xpu /opt/venv/bin/python3 -m pip install --no-cache-dir . --extra-index-url https://download.pytorch.org/whl/xpu
|
|
docker exec ci_sglang_xpu /opt/venv/bin/python3 -m pip install --no-cache-dir --no-deps xgrammar==0.1.33
|
|
docker exec ci_sglang_xpu /bin/bash -c '/opt/venv/bin/hf auth login --token ${HF_TOKEN}'
|
|
|
|
- name: Run diffusion server tests (1-GPU)
|
|
timeout-minutes: 60
|
|
run: |
|
|
# xpu_b60.json is seeded with this flag on; without it the SYCL queue
|
|
# is deep enough that step timings record host enqueue, not device time.
|
|
docker exec -e SGLANG_DIFFUSION_SYNC_STAGE_PROFILING=1 ci_sglang_xpu bash -c "source /opt/venv/bin/activate && cd /sglang-checkout/python && python3 sglang/multimodal_gen/test/run_suite.py --suite 1-gpu-xpu"
|
|
|
|
- name: Cleanup container
|
|
if: always()
|
|
run: |
|
|
docker run --rm -v "${{ github.workspace }}:/w" busybox:latest \
|
|
chown -R "$(id -u):$(id -g)" /w || true
|
|
rm -rf \
|
|
python/build \
|
|
python/dist \
|
|
python/sglang.egg-info \
|
|
python/sglang/*.egg-info \
|
|
test/result.jsonl \
|
|
test/results \
|
|
test/.pytest_cache \
|
|
.pytest_cache || true
|
|
find . -type d -name "__pycache__" -prune -exec rm -rf {} + || true
|
|
find . -type f -name "*.pyc" -delete || true
|
|
# SIGTERM sglang and drain GPU context before `docker rm -f`;
|
|
# SIGKILL leaves the xe/GuC exec queue registered and triggers a
|
|
# GT reset (+ devcoredump) on B580.
|
|
if docker ps --format '{{.Names}}' | grep -qx ci_sglang_xpu; then
|
|
docker exec ci_sglang_xpu bash -c '
|
|
pkill -TERM -f "sglang|run_suite|python3.*test_" 2>/dev/null || true
|
|
for _ in $(seq 1 30); do
|
|
pgrep -f "sglang::|sglang.launch_server" >/dev/null || break
|
|
sleep 1
|
|
done
|
|
pkill -KILL -f "sglang|run_suite" 2>/dev/null || true
|
|
' || true
|
|
fi
|
|
docker rm -f ci_sglang_xpu || true
|
|
if [[ -n "${CI_SGLANG_XPU_IMAGE:-}" ]]; then
|
|
docker rmi -f "${CI_SGLANG_XPU_IMAGE}" || true
|
|
fi
|
|
|
|
finish:
|
|
if: always()
|
|
needs: [stage-a-test-1-gpu-xpu, multimodal-gen-test-1-gpu-xpu, pr-gate]
|
|
runs-on: ubuntu-latest
|
|
steps:
|
|
- name: Check job status
|
|
run: |
|
|
stage_a="${{ needs.stage-a-test-1-gpu-xpu.result }}"
|
|
multimodal_gen="${{ needs.multimodal-gen-test-1-gpu-xpu.result }}"
|
|
if [ "$stage_a" != "success" ] && [ "$stage_a" != "skipped" ]; then
|
|
echo "stage-a failed with result: $stage_a"
|
|
exit 1
|
|
fi
|
|
if [ "$multimodal_gen" != "success" ] && [ "$multimodal_gen" != "skipped" ]; then
|
|
echo "multimodal-gen failed with result: $multimodal_gen"
|
|
exit 1
|
|
fi
|
|
echo "All jobs completed successfully"
|
|
exit 0
|