name: PR Test (XPU) on: push: branches: [ main ] pull_request: branches: [ main ] workflow_dispatch: workflow_call: inputs: ref: description: 'Git ref (branch, tag, or SHA) to test. If not provided, uses the default branch.' required: false type: string default: '' run_all_tests: description: "Run all tests (for releasing or testing purpose)" required: false type: boolean default: false concurrency: group: pr-test-xpu-${{ inputs.ref || github.ref }} cancel-in-progress: ${{ github.event_name != 'workflow_call' }} jobs: # ==================== Check Changes ==================== # check-changes: runs-on: ubuntu-latest outputs: changes_exist: ${{ steps.filter.outputs.main_package == 'true' || steps.filter.outputs.multimodal_gen == 'true' || steps.run-mode.outputs.run_all_tests == 'true' }} main_package: ${{ steps.filter.outputs.main_package == 'true' || steps.run-mode.outputs.run_all_tests == 'true' }} multimodal_gen: ${{ steps.filter.outputs.multimodal_gen == 'true' || steps.run-mode.outputs.run_all_tests == 'true' }} steps: - name: Checkout code uses: actions/checkout@v4 with: ref: ${{ inputs.ref || github.ref }} - name: Determine run mode id: run-mode run: | # Run all tests for workflow_call (when ref input is provided) # Note: github.event_name is inherited from caller, so we detect workflow_call by checking inputs.ref if [[ "${{ inputs.run_all_tests }}" == "true" ]]; then echo "run_all_tests=true" >> $GITHUB_OUTPUT echo "Run mode: ALL TESTS (run_all_tests=${{ inputs.run_all_tests }})" else echo "run_all_tests=false" >> $GITHUB_OUTPUT echo "Run mode: FILTERED (triggered by ${{ github.event_name }})" fi - name: Detect file changes id: filter uses: dorny/paths-filter@v3 if: steps.run-mode.outputs.run_all_tests != 'true' with: filters: | main_package: # Extend test/registered/ entries when adding a non-nightly register_xpu_ci. - "python/sglang/!(multimodal_gen|kernels|cli|test)/**/!(*.md)" - "python/sglang/test/*.py" - "python/sglang/test/!(ascend|observability|mock_model|manual|external_models|kernels)/**/!(*.md)" - "python/pyproject_xpu.toml" - "test/registered/xpu/**/!(*.md)" - "test/registered/disaggregation/test_disaggregation_xpu.py" - "test/registered/attention/test_chunk_gated_delta_rule.py" - "test/registered/attention/test_deterministic.py" - "test/registered/lora/test_moe_lora_info.py" - "test/registered/lora/test_virtual_experts_kernels.py" - "test/registered/unit/sampling/test_sampling_params.py" - "test/registered/unit/spec/test_adaptive_spec_params.py" - "test/run_suite.py" - "test/pytest.ini" - ".github/workflows/pr-test-xpu.yml" - "docker/xpu.Dockerfile" - "scripts/ci/xpu/**" multimodal_gen: # Only paths imported by the 1-gpu-xpu suite; other-vendor and off-suite files are excluded. - "python/sglang/multimodal_gen/*.py" - "python/sglang/multimodal_gen/runtime/!(pipelines)/**/!(*.md|*.ipynb)" - "python/sglang/multimodal_gen/runtime/*.py" - "python/sglang/multimodal_gen/runtime/pipelines/__init__.py" - "python/sglang/multimodal_gen/runtime/pipelines/@(zimage_pipeline|flux_2_klein|flux_2|wan_pipeline|diffusers_pipeline).py" - "python/sglang/multimodal_gen/configs/**/!(*.md|*.ipynb)" - "python/sglang/multimodal_gen/apps/webui/**/!(*.md|*.ipynb)" - "python/sglang/multimodal_gen/benchmarks/compare_perf.py" - "python/sglang/multimodal_gen/test/*.py" - "python/sglang/multimodal_gen/test/runner/**/!(*.md|*.ipynb)" - "python/sglang/multimodal_gen/test/server/!(test_server_1_gpu_5090|test_server_4_gpu_h100|test_server_b200|test_server_2_gpu|test_request_logger).py" - "python/sglang/multimodal_gen/test/server/common/**/!(*.md|*.ipynb)" - "python/sglang/multimodal_gen/test/server/perf_baselines/xpu_b60.json" - "python/pyproject_xpu.toml" - ".github/workflows/pr-test-xpu.yml" - "docker/xpu.Dockerfile" - "scripts/ci/xpu/**" # ==================== PR Gate ==================== # pr-gate: needs: check-changes if: needs.check-changes.outputs.changes_exist == 'true' uses: ./.github/workflows/pr-gate.yml secrets: inherit # Runs the entire per-commit XPU test suite in one job (single install, single queue). stage-a-test-1-gpu-xpu: needs: [check-changes, pr-gate] if: needs.check-changes.outputs.main_package == 'true' runs-on: intel-bmg env: DOCKERHUB_INTEL_USERNAME: ${{ secrets.DOCKERHUB_INTEL_USERNAME }} DOCKERHUB_INTEL_TOKEN: ${{ secrets.DOCKERHUB_INTEL_TOKEN }} steps: - name: Reset workspace ownership run: | # Prior runs install sglang as root inside the CI container against # the bind-mounted workspace, leaving root-owned egg-info on the host. # actions/checkout's pre-clean step can't unlink those as the runner # user, so reset ownership before checkout runs. docker run --rm -v "${{ github.workspace }}:/w" busybox:latest \ chown -R "$(id -u):$(id -g)" /w || true - name: Checkout code uses: actions/checkout@v4 with: fetch-depth: 0 ref: ${{ inputs.ref || github.ref }} - name: Start CI container run: | export HF_TOKEN="$(cat ~/huggingface_token.txt)" bash scripts/ci/xpu/xpu_ci_start_container.sh env: GITHUB_WORKSPACE: ${{ github.workspace }} - name: Install Dependency timeout-minutes: 60 run: | # Bridge fix: install libssl-dev in the running container so the # JIT build of hicache_hash_cpp (needs ) succeeds. # Can be dropped once intel/sglang-dev:latest is rebuilt from a # docker/xpu.Dockerfile that already ships libssl-dev. docker exec -e DEBIAN_FRONTEND=noninteractive ci_sglang_xpu apt-get update docker exec -e DEBIAN_FRONTEND=noninteractive ci_sglang_xpu apt-get install -y libssl-dev docker exec ci_sglang_xpu /opt/venv/bin/python3 -m pip install --upgrade pip docker exec ci_sglang_xpu /opt/venv/bin/python3 -m pip install pytest expecttest ray huggingface_hub tabulate "lmcache>=0.3.9" accelerate docker exec ci_sglang_xpu /opt/venv/bin/python3 -m pip uninstall -y flashinfer-python sgl-kernel sglang docker exec ci_sglang_xpu cp /sglang-checkout/python/pyproject_xpu.toml /sglang-checkout/python/pyproject.toml # Fetch tags so setuptools_scm resolves a real version instead of # falling back to 0.0.0 on a shallow/tag-less checkout. docker exec -w /sglang-checkout ci_sglang_xpu git fetch origin '+refs/tags/*:refs/tags/*' --force docker exec -w /sglang-checkout/python ci_sglang_xpu /opt/venv/bin/python3 -m pip install --no-cache-dir . --extra-index-url https://download.pytorch.org/whl/xpu docker exec ci_sglang_xpu /opt/venv/bin/python3 -m pip install --no-cache-dir --no-deps xgrammar==0.1.33 docker exec ci_sglang_xpu /bin/bash -c '/opt/venv/bin/hf auth login --token ${HF_TOKEN}' - name: Install checkpoint-engine extra (optional; tests skip if unavailable) continue-on-error: true run: | docker exec -w /sglang-checkout/python ci_sglang_xpu /opt/venv/bin/python3 -m pip install --no-cache-dir ".[checkpoint-engine]" --extra-index-url https://download.pytorch.org/whl/xpu - name: Run tests # Old stage-a (30) + stage-b (120) = 150; +20% headroom for --enable-retry. timeout-minutes: 180 run: | docker exec ci_sglang_xpu bash -c "source /opt/venv/bin/activate && cd /sglang-checkout/test && python3 run_suite.py --hw xpu --suite stage-a-test-1-gpu-xpu,stage-b-test-1-gpu-xpu --enable-retry" - name: Cleanup container if: always() run: | # pip install ran as root inside the container against the # bind-mounted workspace, so build artifacts are root-owned on # the host. Chown them back before rm to avoid Permission denied. docker run --rm -v "${{ github.workspace }}:/w" busybox:latest \ chown -R "$(id -u):$(id -g)" /w || true # Wipe everything the run wrote into the workspace so the next # job starts from a clean tree. rm -rf \ python/build \ python/dist \ python/sglang.egg-info \ python/sglang/*.egg-info \ test/result.jsonl \ test/results \ test/.pytest_cache \ .pytest_cache || true find . -type d -name "__pycache__" -prune -exec rm -rf {} + || true find . -type f -name "*.pyc" -delete || true # SIGTERM sglang and drain GPU context before `docker rm -f`; # SIGKILL leaves the xe/GuC exec queue registered and triggers a # GT reset (+ devcoredump) on B580. if docker ps --format '{{.Names}}' | grep -qx ci_sglang_xpu; then docker exec ci_sglang_xpu bash -c ' pkill -TERM -f "sglang|run_suite|python3.*test_" 2>/dev/null || true for _ in $(seq 1 30); do pgrep -f "sglang::|sglang.launch_server" >/dev/null || break sleep 1 done pkill -KILL -f "sglang|run_suite" 2>/dev/null || true ' || true fi docker rm -f ci_sglang_xpu || true if [[ -n "${CI_SGLANG_XPU_IMAGE:-}" ]]; then docker rmi -f "${CI_SGLANG_XPU_IMAGE}" || true fi # ==================== Multimodal Gen ==================== # multimodal-gen-test-1-gpu-xpu: needs: [check-changes, pr-gate] if: needs.check-changes.outputs.multimodal_gen == 'true' runs-on: bmg-multigen-models env: DOCKERHUB_INTEL_USERNAME: ${{ secrets.DOCKERHUB_INTEL_USERNAME }} DOCKERHUB_INTEL_TOKEN: ${{ secrets.DOCKERHUB_INTEL_TOKEN }} steps: - name: Reset workspace ownership run: | docker run --rm -v "${{ github.workspace }}:/w" busybox:latest \ chown -R "$(id -u):$(id -g)" /w || true - name: Checkout code uses: actions/checkout@v4 with: fetch-depth: 0 ref: ${{ inputs.ref || github.ref }} - name: Start CI container run: | export HF_TOKEN="$(cat ~/huggingface_token.txt)" bash scripts/ci/xpu/xpu_ci_start_container.sh env: GITHUB_WORKSPACE: ${{ github.workspace }} - name: Install Dependency timeout-minutes: 60 run: | # Bridge fix: install libssl-dev in the running container so the # JIT build of hicache_hash_cpp (needs ) succeeds. # Can be dropped once intel/sglang-dev:latest is rebuilt from a # docker/xpu.Dockerfile that already ships libssl-dev. docker exec -e DEBIAN_FRONTEND=noninteractive ci_sglang_xpu apt-get update docker exec -e DEBIAN_FRONTEND=noninteractive ci_sglang_xpu apt-get install -y libssl-dev docker exec ci_sglang_xpu /opt/venv/bin/python3 -m pip install --upgrade pip docker exec ci_sglang_xpu /opt/venv/bin/python3 -m pip install pytest expecttest ray huggingface_hub tabulate "lmcache>=0.3.9" docker exec ci_sglang_xpu /opt/venv/bin/python3 -m pip uninstall -y flashinfer-python sgl-kernel sglang docker exec ci_sglang_xpu cp /sglang-checkout/python/pyproject_xpu.toml /sglang-checkout/python/pyproject.toml # Fetch tags so setuptools_scm resolves a real version instead of # falling back to 0.0.0 on a shallow/tag-less checkout. docker exec -w /sglang-checkout ci_sglang_xpu git fetch origin '+refs/tags/*:refs/tags/*' --force docker exec -w /sglang-checkout/python ci_sglang_xpu /opt/venv/bin/python3 -m pip install --no-cache-dir . --extra-index-url https://download.pytorch.org/whl/xpu docker exec ci_sglang_xpu /opt/venv/bin/python3 -m pip install --no-cache-dir --no-deps xgrammar==0.1.33 docker exec ci_sglang_xpu /bin/bash -c '/opt/venv/bin/hf auth login --token ${HF_TOKEN}' - name: Run diffusion server tests (1-GPU) timeout-minutes: 60 run: | # xpu_b60.json is seeded with this flag on; without it the SYCL queue # is deep enough that step timings record host enqueue, not device time. docker exec -e SGLANG_DIFFUSION_SYNC_STAGE_PROFILING=1 ci_sglang_xpu bash -c "source /opt/venv/bin/activate && cd /sglang-checkout/python && python3 sglang/multimodal_gen/test/run_suite.py --suite 1-gpu-xpu" - name: Cleanup container if: always() run: | docker run --rm -v "${{ github.workspace }}:/w" busybox:latest \ chown -R "$(id -u):$(id -g)" /w || true rm -rf \ python/build \ python/dist \ python/sglang.egg-info \ python/sglang/*.egg-info \ test/result.jsonl \ test/results \ test/.pytest_cache \ .pytest_cache || true find . -type d -name "__pycache__" -prune -exec rm -rf {} + || true find . -type f -name "*.pyc" -delete || true # SIGTERM sglang and drain GPU context before `docker rm -f`; # SIGKILL leaves the xe/GuC exec queue registered and triggers a # GT reset (+ devcoredump) on B580. if docker ps --format '{{.Names}}' | grep -qx ci_sglang_xpu; then docker exec ci_sglang_xpu bash -c ' pkill -TERM -f "sglang|run_suite|python3.*test_" 2>/dev/null || true for _ in $(seq 1 30); do pgrep -f "sglang::|sglang.launch_server" >/dev/null || break sleep 1 done pkill -KILL -f "sglang|run_suite" 2>/dev/null || true ' || true fi docker rm -f ci_sglang_xpu || true if [[ -n "${CI_SGLANG_XPU_IMAGE:-}" ]]; then docker rmi -f "${CI_SGLANG_XPU_IMAGE}" || true fi finish: if: always() needs: [stage-a-test-1-gpu-xpu, multimodal-gen-test-1-gpu-xpu, pr-gate] runs-on: ubuntu-latest steps: - name: Check job status run: | stage_a="${{ needs.stage-a-test-1-gpu-xpu.result }}" multimodal_gen="${{ needs.multimodal-gen-test-1-gpu-xpu.result }}" if [ "$stage_a" != "success" ] && [ "$stage_a" != "skipped" ]; then echo "stage-a failed with result: $stage_a" exit 1 fi if [ "$multimodal_gen" != "success" ] && [ "$multimodal_gen" != "skipped" ]; then echo "multimodal-gen failed with result: $multimodal_gen" exit 1 fi echo "All jobs completed successfully" exit 0