From 5d1949152da296c456e929f2dc23ababe269e44c Mon Sep 17 00:00:00 2001 From: Michael <13900043+michaelzhang-ai@users.noreply.github.com> Date: Wed, 17 Jun 2026 23:31:32 -0700 Subject: [PATCH] [AMD] ci: add extra-a 1-gpu-large tier (fp8kv-triton, streaming-session, spec-standalone) (#28458) --- .github/workflows/pr-test-amd-extra.yml | 45 ++++++++++++++++--- test/registered/quant/test_fp8kv_triton.py | 3 +- .../sessions/test_streaming_session_extra.py | 3 +- .../spec/test_spec_standalone_extra.py | 11 ++++- test/run_suite.py | 16 ++++--- 5 files changed, 63 insertions(+), 15 deletions(-) diff --git a/.github/workflows/pr-test-amd-extra.yml b/.github/workflows/pr-test-amd-extra.yml index ff86d6dd4..8c4044f24 100644 --- a/.github/workflows/pr-test-amd-extra.yml +++ b/.github/workflows/pr-test-amd-extra.yml @@ -10,12 +10,13 @@ name: PR Test Extra (AMD) # runs unconditionally on workflow_dispatch / workflow_call so it can be # triggered manually or chained from the AMD scheduler. # -# Stage: extra-a (1-gpu-small-amd). The job mirrors the container bring-up of -# pr-test-amd.yml and dispatches `run_suite.py --hw amd --suite -# extra-a-test-1-gpu-small-amd`. Only the mock-model / kv_canary *unit* tests -# are onboarded so far; the canary *e2e* tests (which would land in -# 1-/2-gpu-large) need the canary JIT kernel ported to ROCm first, so those -# suites are intentionally not registered for AMD yet. +# Stage: extra-a. Each job mirrors the container bring-up of pr-test-amd.yml +# and dispatches `run_suite.py --hw amd --suite extra-a-test-{config}-amd`: +# - 1-gpu-small: mock-model / kv_canary *unit* tests +# - 1-gpu-large: single-GPU model e2e tests (quant fp8kv-triton, +# sessions streaming-session, spec standalone triton) +# The canary *e2e* tests still need the canary JIT kernel ported to ROCm +# first, so they remain CUDA-only for now. on: pull_request: @@ -145,6 +146,37 @@ jobs: run: | bash scripts/ci/amd/amd_ci_exec.sh -w "/sglang-checkout/test" python3 run_suite.py --hw amd --suite extra-a-test-1-gpu-small-amd --timeout-per-file 1800 ${{ inputs.continue_on_error == true && '--continue-on-error' || '' }} + # =============================================== extra-a (1-gpu-large) =============================================== + # Single-GPU lora / spec / quant model e2e tests. Runs on the same 1-gpu + # pool as small (AMD GPUs carry enough VRAM that "large" here is a CUDA + # memory-tier label, not a separate AMD runner pool). + extra-a-test-1-gpu-large-amd: + name: ${{ format('extra-a-test-1-gpu-large-amd{0} (linux-{1}-1gpu-sglang)', inputs.rocm_version && format('-{0}', inputs.rocm_version) || '', inputs.runner_arch || 'mi325') }} + needs: [call-gate] + if: ${{ !cancelled() && needs.call-gate.result == 'success' }} + runs-on: ${{ format('linux-{0}-1gpu-sglang', inputs.runner_arch || 'mi325') }} + steps: + - name: Checkout code + uses: actions/checkout@v4 + with: + ref: ${{ inputs.ref || github.sha }} + + - name: Ensure VRAM is clear + run: bash scripts/ci/amd/ensure_vram_clear.sh rocm + + - name: Start CI container + run: bash scripts/ci/amd/amd_ci_start_container.sh ${{ inputs.rocm_version && format('--rocm-version {0}', inputs.rocm_version) || '' }} + env: + GITHUB_WORKSPACE: ${{ github.workspace }} + + - name: Install dependencies + run: bash scripts/ci/amd/amd_ci_install_dependency.sh + + - name: Run test + timeout-minutes: 60 + run: | + bash scripts/ci/amd/amd_ci_exec.sh -w "/sglang-checkout/test" python3 run_suite.py --hw amd --suite extra-a-test-1-gpu-large-amd --timeout-per-file 1800 ${{ inputs.continue_on_error == true && '--continue-on-error' || '' }} + # =============================================== aggregator ==================================================== # Single fan-in job so branch protection / notifications depend on one job # rather than every matrix leg. Fails if any dependent failed or was @@ -154,6 +186,7 @@ jobs: [ call-gate, extra-a-test-1-gpu-small-amd, + extra-a-test-1-gpu-large-amd, ] if: always() runs-on: ubuntu-latest diff --git a/test/registered/quant/test_fp8kv_triton.py b/test/registered/quant/test_fp8kv_triton.py index e9b867538..48b3843c3 100644 --- a/test/registered/quant/test_fp8kv_triton.py +++ b/test/registered/quant/test_fp8kv_triton.py @@ -3,7 +3,7 @@ from types import SimpleNamespace from urllib.parse import urlparse from sglang.srt.utils import kill_process_tree -from sglang.test.ci.ci_register import register_cuda_ci +from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci from sglang.test.run_eval import run_eval from sglang.test.test_utils import ( DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH, @@ -13,6 +13,7 @@ from sglang.test.test_utils import ( ) register_cuda_ci(est_time=73, stage="extra-a", runner_config="1-gpu-large") +register_amd_ci(est_time=94, suite="extra-a-test-1-gpu-large-amd") class TestFP8KVCacheTritonBackend(CustomTestCase): diff --git a/test/registered/sessions/test_streaming_session_extra.py b/test/registered/sessions/test_streaming_session_extra.py index af118a990..2a059b307 100644 --- a/test/registered/sessions/test_streaming_session_extra.py +++ b/test/registered/sessions/test_streaming_session_extra.py @@ -1,6 +1,6 @@ import unittest -from sglang.test.ci.ci_register import register_cuda_ci +from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci from sglang.test.kits.streaming_session_kit import StreamingSessionKitMixin from sglang.test.server_fixtures.streaming_session_fixture import ( StreamingSessionServerBase, @@ -11,6 +11,7 @@ from sglang.test.test_utils import ( ) register_cuda_ci(est_time=691, stage="extra-a", runner_config="1-gpu-large") +register_amd_ci(est_time=562, suite="extra-a-test-1-gpu-large-amd") class TestStreamingSessionRetractMixedChunk( diff --git a/test/registered/spec/test_spec_standalone_extra.py b/test/registered/spec/test_spec_standalone_extra.py index ddd37cba3..4a5c2da7a 100644 --- a/test/registered/spec/test_spec_standalone_extra.py +++ b/test/registered/spec/test_spec_standalone_extra.py @@ -1,14 +1,22 @@ import unittest -from sglang.test.ci.ci_register import register_cuda_ci +from sglang.srt.utils import is_hip +from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci from sglang.test.server_fixtures.standalone_fixture import StandaloneServerBase from sglang.test.test_utils import CustomTestCase # Non-V2 standalone speculative decoding tests (FA3, Triton, FlashInfer # backends). Sibling V2 classes stay per-commit in test_spec_standalone.py. register_cuda_ci(est_time=406, stage="extra-a", runner_config="1-gpu-large") +# AMD: fa3 / flashinfer attention backends are not built in the ROCm +# sgl_kernel, so only the triton-backend class runs on ROCm (the fa3 and +# flashinfer classes are skipped on ROCm below). +register_amd_ci(est_time=103, suite="extra-a-test-1-gpu-large-amd") + +_AMD_SKIP_BACKEND = "fa3 / flashinfer attention backends are CUDA-only (not in the ROCm sgl_kernel build)" +@unittest.skipIf(is_hip(), _AMD_SKIP_BACKEND) class TestStandaloneSpeculativeDecodingBase(StandaloneServerBase, CustomTestCase): attention_backend = "fa3" speculative_eagle_topk = 2 @@ -24,6 +32,7 @@ class TestStandaloneSpeculativeDecodingTriton(StandaloneServerBase, CustomTestCa enable_deterministic_inference = True +@unittest.skipIf(is_hip(), _AMD_SKIP_BACKEND) class TestStandaloneSpeculativeDecodingFlashinfer(StandaloneServerBase, CustomTestCase): attention_backend = "flashinfer" speculative_eagle_topk = 2 diff --git a/test/run_suite.py b/test/run_suite.py index 7ea611f1a..42688cf6e 100644 --- a/test/run_suite.py +++ b/test/run_suite.py @@ -43,14 +43,18 @@ PER_COMMIT_SUITES = { "stage-c-test-4-gpu-amd", "stage-c-test-large-8-gpu-amd", "stage-c-test-large-8-gpu-amd-mi35x", - # extra-a: label-gated PR opt-in suite in pr-test-amd-extra.yml + # extra-a: label-gated PR opt-in suites in pr-test-amd-extra.yml # (mirror of CUDA extra-a; tests stay tagged per-commit but only - # dispatch when the PR carries the `run-ci-extra` label). Only the - # 1-gpu-small mock-model / kv_canary *unit* tests are onboarded so - # far; the canary *e2e* tests (1-/2-gpu-large) need the canary JIT - # kernel ported to ROCm first, so those suites are intentionally - # not yet registered for AMD. + # dispatch when the PR carries the `run-ci-extra` label). 1-gpu-small + # carries the mock-model / kv_canary *unit* tests; 1-gpu-large carries + # the subset of model e2e tests validated to pass on mi325 (quant + # fp8kv-triton, sessions streaming-session EAGLE3, spec standalone + # triton-backend variant). The rest of CUDA + # extra-a tests fail on ROCm (missing flash_attn.cute/flash_ops + # kernels, OOM, or accuracy regressions — e.g. gemma4-mtp-31b dips + # below the gsm8k floor on the topk=3 leg) and stay CUDA-only for now. "extra-a-test-1-gpu-small-amd", + "extra-a-test-1-gpu-large-amd", ], HWBackend.MUSA: [], HWBackend.CUDA: [