From 07c7b2674d4321d6ca4e7db1fb0f78779250b2b9 Mon Sep 17 00:00:00 2001 From: ashwini rathi Date: Wed, 9 Sep 2026 13:56:54 +0530 Subject: [PATCH] ci(xpu): merge stage-a+b into one job and trim main_package scope (#38014) Co-authored-by: arathi-hlab --- .github/workflows/pr-test-xpu.yml | 141 ++++++----------------- .github/workflows/xpu-ci-job-monitor.yml | 4 +- 2 files changed, 37 insertions(+), 108 deletions(-) diff --git a/.github/workflows/pr-test-xpu.yml b/.github/workflows/pr-test-xpu.yml index 2ab203de7..b035d5ae7 100644 --- a/.github/workflows/pr-test-xpu.yml +++ b/.github/workflows/pr-test-xpu.yml @@ -56,17 +56,42 @@ jobs: with: filters: | main_package: - - "python/sglang/!(multimodal_gen)/**/!(*.md)" + # Extend test/registered/ entries when adding a non-nightly register_xpu_ci. + - "python/sglang/!(multimodal_gen|kernels|cli|test)/**/!(*.md)" + - "python/sglang/test/*.py" + - "python/sglang/test/!(ascend|observability|mock_model|manual|external_models|kernels)/**/!(*.md)" - "python/pyproject_xpu.toml" - - "test/**/!(*.md)" - - "python/sglang/kernels/aot/**/!(*.md|THIRDPARTYNOTICES.txt|LICENSE)" + - "test/registered/xpu/**/!(*.md)" + - "test/registered/attention/test_chunk_gated_delta_rule.py" + - "test/registered/attention/test_deterministic.py" + - "test/registered/lora/test_moe_lora_info.py" + - "test/registered/lora/test_virtual_experts_kernels.py" + - "test/registered/unit/sampling/test_sampling_params.py" + - "test/registered/unit/spec/test_adaptive_spec_params.py" + - "test/run_suite.py" + - "test/pytest.ini" - ".github/workflows/pr-test-xpu.yml" - "docker/xpu.Dockerfile" + - "scripts/ci/xpu/**" multimodal_gen: - - "python/sglang/multimodal_gen/**/!(*.md|*.ipynb)" + # Only paths imported by the 1-gpu-xpu suite; other-vendor and off-suite files are excluded. + - "python/sglang/multimodal_gen/*.py" + - "python/sglang/multimodal_gen/runtime/!(pipelines)/**/!(*.md|*.ipynb)" + - "python/sglang/multimodal_gen/runtime/*.py" + - "python/sglang/multimodal_gen/runtime/pipelines/__init__.py" + - "python/sglang/multimodal_gen/runtime/pipelines/@(zimage_pipeline|flux_2_klein|flux_2|wan_pipeline|diffusers_pipeline).py" + - "python/sglang/multimodal_gen/configs/**/!(*.md|*.ipynb)" + - "python/sglang/multimodal_gen/apps/webui/**/!(*.md|*.ipynb)" + - "python/sglang/multimodal_gen/benchmarks/compare_perf.py" + - "python/sglang/multimodal_gen/test/*.py" + - "python/sglang/multimodal_gen/test/runner/**/!(*.md|*.ipynb)" + - "python/sglang/multimodal_gen/test/server/!(test_server_1_gpu_5090|test_server_4_gpu_h100|test_server_b200|test_server_2_gpu|test_request_logger).py" + - "python/sglang/multimodal_gen/test/server/common/**/!(*.md|*.ipynb)" + - "python/sglang/multimodal_gen/test/server/perf_baselines/xpu_b60.json" - "python/pyproject_xpu.toml" - ".github/workflows/pr-test-xpu.yml" - "docker/xpu.Dockerfile" + - "scripts/ci/xpu/**" # ==================== PR Gate ==================== # pr-gate: @@ -75,7 +100,7 @@ jobs: uses: ./.github/workflows/pr-gate.yml secrets: inherit - # ==================== Stage A ==================== # + # Runs the entire per-commit XPU test suite in one job (single install, single queue). stage-a-test-1-gpu-xpu: needs: [check-changes, pr-gate] if: needs.check-changes.outputs.main_package == 'true' @@ -120,102 +145,11 @@ jobs: docker exec ci_sglang_xpu /opt/venv/bin/python3 -m pip install --no-cache-dir --no-deps xgrammar==0.1.33 docker exec ci_sglang_xpu /bin/bash -c '/opt/venv/bin/hf auth login --token ${HF_TOKEN}' - - name: Run stage-a tests - timeout-minutes: 30 + - name: Run tests + # Old stage-a (30) + stage-b (120) = 150; +20% headroom for --enable-retry. + timeout-minutes: 180 run: | - docker exec ci_sglang_xpu bash -c "source /opt/venv/bin/activate && cd /sglang-checkout/test && python3 run_suite.py --hw xpu --suite stage-a-test-1-gpu-xpu --enable-retry" - - - name: Cleanup container - if: always() - run: | - # pip install ran as root inside the container against the - # bind-mounted workspace, so build artifacts are root-owned on - # the host. Chown them back before rm to avoid Permission denied. - docker run --rm -v "${{ github.workspace }}:/w" busybox:latest \ - chown -R "$(id -u):$(id -g)" /w || true - # Wipe everything the run wrote into the workspace so the next - # job starts from a clean tree. - rm -rf \ - python/build \ - python/dist \ - python/sglang.egg-info \ - python/sglang/*.egg-info \ - test/result.jsonl \ - test/results \ - test/.pytest_cache \ - .pytest_cache || true - find . -type d -name "__pycache__" -prune -exec rm -rf {} + || true - find . -type f -name "*.pyc" -delete || true - # SIGTERM sglang and drain GPU context before `docker rm -f`; - # SIGKILL leaves the xe/GuC exec queue registered and triggers a - # GT reset (+ devcoredump) on B580. - if docker ps --format '{{.Names}}' | grep -qx ci_sglang_xpu; then - docker exec ci_sglang_xpu bash -c ' - pkill -TERM -f "sglang|run_suite|python3.*test_" 2>/dev/null || true - for _ in $(seq 1 30); do - pgrep -f "sglang::|sglang.launch_server" >/dev/null || break - sleep 1 - done - pkill -KILL -f "sglang|run_suite" 2>/dev/null || true - ' || true - fi - docker rm -f ci_sglang_xpu || true - if [[ -n "${CI_SGLANG_XPU_IMAGE:-}" ]]; then - docker rmi -f "${CI_SGLANG_XPU_IMAGE}" || true - fi - - # ==================== Wait for Stage A ==================== # - wait-for-stage-a: - needs: [stage-a-test-1-gpu-xpu] - runs-on: ubuntu-latest - steps: - - run: echo "stage-a passed" - - # ==================== Stage B ==================== # - stage-b-test-1-gpu-xpu: - needs: [check-changes, pr-gate, wait-for-stage-a] - if: needs.check-changes.outputs.main_package == 'true' - runs-on: intel-bmg - env: - DOCKERHUB_INTEL_USERNAME: ${{ secrets.DOCKERHUB_INTEL_USERNAME }} - DOCKERHUB_INTEL_TOKEN: ${{ secrets.DOCKERHUB_INTEL_TOKEN }} - steps: - - name: Reset workspace ownership - run: | - docker run --rm -v "${{ github.workspace }}:/w" busybox:latest \ - chown -R "$(id -u):$(id -g)" /w || true - - - name: Checkout code - uses: actions/checkout@v4 - with: - fetch-depth: 0 - ref: ${{ inputs.ref || github.ref }} - - - name: Start CI container - run: | - export HF_TOKEN="$(cat ~/huggingface_token.txt)" - bash scripts/ci/xpu/xpu_ci_start_container.sh - env: - GITHUB_WORKSPACE: ${{ github.workspace }} - - - name: Install Dependency - timeout-minutes: 60 - run: | - docker exec ci_sglang_xpu /opt/venv/bin/python3 -m pip install --upgrade pip - docker exec ci_sglang_xpu /opt/venv/bin/python3 -m pip install pytest expecttest ray huggingface_hub tabulate "lmcache>=0.3.9" accelerate - docker exec ci_sglang_xpu /opt/venv/bin/python3 -m pip uninstall -y flashinfer-python sgl-kernel sglang - docker exec ci_sglang_xpu cp /sglang-checkout/python/pyproject_xpu.toml /sglang-checkout/python/pyproject.toml - # Fetch tags so setuptools_scm resolves a real version instead of - # falling back to 0.0.0 on a shallow/tag-less checkout. - docker exec -w /sglang-checkout ci_sglang_xpu git fetch origin '+refs/tags/*:refs/tags/*' --force - docker exec -w /sglang-checkout/python ci_sglang_xpu /opt/venv/bin/python3 -m pip install --no-cache-dir . --extra-index-url https://download.pytorch.org/whl/xpu - docker exec ci_sglang_xpu /opt/venv/bin/python3 -m pip install --no-cache-dir --no-deps xgrammar==0.1.33 - docker exec ci_sglang_xpu /bin/bash -c '/opt/venv/bin/hf auth login --token ${HF_TOKEN}' - - - name: Run stage-b tests - timeout-minutes: 120 - run: | - docker exec ci_sglang_xpu bash -c "source /opt/venv/bin/activate && cd /sglang-checkout/test && python3 run_suite.py --hw xpu --suite stage-b-test-1-gpu-xpu --enable-retry" + docker exec ci_sglang_xpu bash -c "source /opt/venv/bin/activate && cd /sglang-checkout/test && python3 run_suite.py --hw xpu --suite stage-a-test-1-gpu-xpu,stage-b-test-1-gpu-xpu --enable-retry" - name: Cleanup container if: always() @@ -338,22 +272,17 @@ jobs: finish: if: always() - needs: [stage-a-test-1-gpu-xpu, stage-b-test-1-gpu-xpu, multimodal-gen-test-1-gpu-xpu, pr-gate] + needs: [stage-a-test-1-gpu-xpu, multimodal-gen-test-1-gpu-xpu, pr-gate] runs-on: ubuntu-latest steps: - name: Check job status run: | stage_a="${{ needs.stage-a-test-1-gpu-xpu.result }}" - stage_b="${{ needs.stage-b-test-1-gpu-xpu.result }}" multimodal_gen="${{ needs.multimodal-gen-test-1-gpu-xpu.result }}" if [ "$stage_a" != "success" ] && [ "$stage_a" != "skipped" ]; then echo "stage-a failed with result: $stage_a" exit 1 fi - if [ "$stage_b" != "success" ] && [ "$stage_b" != "skipped" ]; then - echo "stage-b failed with result: $stage_b" - exit 1 - fi if [ "$multimodal_gen" != "success" ] && [ "$multimodal_gen" != "skipped" ]; then echo "multimodal-gen failed with result: $multimodal_gen" exit 1 diff --git a/.github/workflows/xpu-ci-job-monitor.yml b/.github/workflows/xpu-ci-job-monitor.yml index 7303a74e7..e51883ac8 100644 --- a/.github/workflows/xpu-ci-job-monitor.yml +++ b/.github/workflows/xpu-ci-job-monitor.yml @@ -127,9 +127,9 @@ jobs: id: parse run: | # Parse pr-test-xpu.yml and extract job names (exclude utility jobs) - # Excluded: check-changes, pr-gate, wait-for-stage-a, finish + # Excluded: check-changes, pr-gate, finish pr_jobs=$(yq -r '.jobs | keys | .[]' .github/workflows/pr-test-xpu.yml | \ - grep -v -E '^(check-changes|pr-gate|wait-for-stage-a|finish)$' | \ + grep -v -E '^(check-changes|pr-gate|finish)$' | \ jq -R -s -c 'split("\n") | map(select(length > 0))') echo "pr_jobs=$pr_jobs" >> $GITHUB_OUTPUT echo "PR jobs: $pr_jobs"