[AMD] ci: add extra-a 1-gpu-large tier (fp8kv-triton, streaming-session, spec-standalone) (#28458)

This commit is contained in:
Michael
2026-06-17 23:31:32 -07:00
committed by GitHub
parent 0e5a66dca4
commit 5d1949152d
5 changed files with 63 additions and 15 deletions
+39 -6
View File
@@ -10,12 +10,13 @@ name: PR Test Extra (AMD)
# runs unconditionally on workflow_dispatch / workflow_call so it can be # runs unconditionally on workflow_dispatch / workflow_call so it can be
# triggered manually or chained from the AMD scheduler. # triggered manually or chained from the AMD scheduler.
# #
# Stage: extra-a (1-gpu-small-amd). The job mirrors the container bring-up of # Stage: extra-a. Each job mirrors the container bring-up of pr-test-amd.yml
# pr-test-amd.yml and dispatches `run_suite.py --hw amd --suite # and dispatches `run_suite.py --hw amd --suite extra-a-test-{config}-amd`:
# extra-a-test-1-gpu-small-amd`. Only the mock-model / kv_canary *unit* tests # - 1-gpu-small: mock-model / kv_canary *unit* tests
# are onboarded so far; the canary *e2e* tests (which would land in # - 1-gpu-large: single-GPU model e2e tests (quant fp8kv-triton,
# 1-/2-gpu-large) need the canary JIT kernel ported to ROCm first, so those # sessions streaming-session, spec standalone triton)
# suites are intentionally not registered for AMD yet. # The canary *e2e* tests still need the canary JIT kernel ported to ROCm
# first, so they remain CUDA-only for now.
on: on:
pull_request: pull_request:
@@ -145,6 +146,37 @@ jobs:
run: | run: |
bash scripts/ci/amd/amd_ci_exec.sh -w "/sglang-checkout/test" python3 run_suite.py --hw amd --suite extra-a-test-1-gpu-small-amd --timeout-per-file 1800 ${{ inputs.continue_on_error == true && '--continue-on-error' || '' }} bash scripts/ci/amd/amd_ci_exec.sh -w "/sglang-checkout/test" python3 run_suite.py --hw amd --suite extra-a-test-1-gpu-small-amd --timeout-per-file 1800 ${{ inputs.continue_on_error == true && '--continue-on-error' || '' }}
# =============================================== extra-a (1-gpu-large) ===============================================
# Single-GPU lora / spec / quant model e2e tests. Runs on the same 1-gpu
# pool as small (AMD GPUs carry enough VRAM that "large" here is a CUDA
# memory-tier label, not a separate AMD runner pool).
extra-a-test-1-gpu-large-amd:
name: ${{ format('extra-a-test-1-gpu-large-amd{0} (linux-{1}-1gpu-sglang)', inputs.rocm_version && format('-{0}', inputs.rocm_version) || '', inputs.runner_arch || 'mi325') }}
needs: [call-gate]
if: ${{ !cancelled() && needs.call-gate.result == 'success' }}
runs-on: ${{ format('linux-{0}-1gpu-sglang', inputs.runner_arch || 'mi325') }}
steps:
- name: Checkout code
uses: actions/checkout@v4
with:
ref: ${{ inputs.ref || github.sha }}
- name: Ensure VRAM is clear
run: bash scripts/ci/amd/ensure_vram_clear.sh rocm
- name: Start CI container
run: bash scripts/ci/amd/amd_ci_start_container.sh ${{ inputs.rocm_version && format('--rocm-version {0}', inputs.rocm_version) || '' }}
env:
GITHUB_WORKSPACE: ${{ github.workspace }}
- name: Install dependencies
run: bash scripts/ci/amd/amd_ci_install_dependency.sh
- name: Run test
timeout-minutes: 60
run: |
bash scripts/ci/amd/amd_ci_exec.sh -w "/sglang-checkout/test" python3 run_suite.py --hw amd --suite extra-a-test-1-gpu-large-amd --timeout-per-file 1800 ${{ inputs.continue_on_error == true && '--continue-on-error' || '' }}
# =============================================== aggregator ==================================================== # =============================================== aggregator ====================================================
# Single fan-in job so branch protection / notifications depend on one job # Single fan-in job so branch protection / notifications depend on one job
# rather than every matrix leg. Fails if any dependent failed or was # rather than every matrix leg. Fails if any dependent failed or was
@@ -154,6 +186,7 @@ jobs:
[ [
call-gate, call-gate,
extra-a-test-1-gpu-small-amd, extra-a-test-1-gpu-small-amd,
extra-a-test-1-gpu-large-amd,
] ]
if: always() if: always()
runs-on: ubuntu-latest runs-on: ubuntu-latest
+2 -1
View File
@@ -3,7 +3,7 @@ from types import SimpleNamespace
from urllib.parse import urlparse from urllib.parse import urlparse
from sglang.srt.utils import kill_process_tree from sglang.srt.utils import kill_process_tree
from sglang.test.ci.ci_register import register_cuda_ci from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci
from sglang.test.run_eval import run_eval from sglang.test.run_eval import run_eval
from sglang.test.test_utils import ( from sglang.test.test_utils import (
DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH, DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH,
@@ -13,6 +13,7 @@ from sglang.test.test_utils import (
) )
register_cuda_ci(est_time=73, stage="extra-a", runner_config="1-gpu-large") register_cuda_ci(est_time=73, stage="extra-a", runner_config="1-gpu-large")
register_amd_ci(est_time=94, suite="extra-a-test-1-gpu-large-amd")
class TestFP8KVCacheTritonBackend(CustomTestCase): class TestFP8KVCacheTritonBackend(CustomTestCase):
@@ -1,6 +1,6 @@
import unittest import unittest
from sglang.test.ci.ci_register import register_cuda_ci from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci
from sglang.test.kits.streaming_session_kit import StreamingSessionKitMixin from sglang.test.kits.streaming_session_kit import StreamingSessionKitMixin
from sglang.test.server_fixtures.streaming_session_fixture import ( from sglang.test.server_fixtures.streaming_session_fixture import (
StreamingSessionServerBase, StreamingSessionServerBase,
@@ -11,6 +11,7 @@ from sglang.test.test_utils import (
) )
register_cuda_ci(est_time=691, stage="extra-a", runner_config="1-gpu-large") register_cuda_ci(est_time=691, stage="extra-a", runner_config="1-gpu-large")
register_amd_ci(est_time=562, suite="extra-a-test-1-gpu-large-amd")
class TestStreamingSessionRetractMixedChunk( class TestStreamingSessionRetractMixedChunk(
@@ -1,14 +1,22 @@
import unittest import unittest
from sglang.test.ci.ci_register import register_cuda_ci from sglang.srt.utils import is_hip
from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci
from sglang.test.server_fixtures.standalone_fixture import StandaloneServerBase from sglang.test.server_fixtures.standalone_fixture import StandaloneServerBase
from sglang.test.test_utils import CustomTestCase from sglang.test.test_utils import CustomTestCase
# Non-V2 standalone speculative decoding tests (FA3, Triton, FlashInfer # Non-V2 standalone speculative decoding tests (FA3, Triton, FlashInfer
# backends). Sibling V2 classes stay per-commit in test_spec_standalone.py. # backends). Sibling V2 classes stay per-commit in test_spec_standalone.py.
register_cuda_ci(est_time=406, stage="extra-a", runner_config="1-gpu-large") register_cuda_ci(est_time=406, stage="extra-a", runner_config="1-gpu-large")
# AMD: fa3 / flashinfer attention backends are not built in the ROCm
# sgl_kernel, so only the triton-backend class runs on ROCm (the fa3 and
# flashinfer classes are skipped on ROCm below).
register_amd_ci(est_time=103, suite="extra-a-test-1-gpu-large-amd")
_AMD_SKIP_BACKEND = "fa3 / flashinfer attention backends are CUDA-only (not in the ROCm sgl_kernel build)"
@unittest.skipIf(is_hip(), _AMD_SKIP_BACKEND)
class TestStandaloneSpeculativeDecodingBase(StandaloneServerBase, CustomTestCase): class TestStandaloneSpeculativeDecodingBase(StandaloneServerBase, CustomTestCase):
attention_backend = "fa3" attention_backend = "fa3"
speculative_eagle_topk = 2 speculative_eagle_topk = 2
@@ -24,6 +32,7 @@ class TestStandaloneSpeculativeDecodingTriton(StandaloneServerBase, CustomTestCa
enable_deterministic_inference = True enable_deterministic_inference = True
@unittest.skipIf(is_hip(), _AMD_SKIP_BACKEND)
class TestStandaloneSpeculativeDecodingFlashinfer(StandaloneServerBase, CustomTestCase): class TestStandaloneSpeculativeDecodingFlashinfer(StandaloneServerBase, CustomTestCase):
attention_backend = "flashinfer" attention_backend = "flashinfer"
speculative_eagle_topk = 2 speculative_eagle_topk = 2
+10 -6
View File
@@ -43,14 +43,18 @@ PER_COMMIT_SUITES = {
"stage-c-test-4-gpu-amd", "stage-c-test-4-gpu-amd",
"stage-c-test-large-8-gpu-amd", "stage-c-test-large-8-gpu-amd",
"stage-c-test-large-8-gpu-amd-mi35x", "stage-c-test-large-8-gpu-amd-mi35x",
# extra-a: label-gated PR opt-in suite in pr-test-amd-extra.yml # extra-a: label-gated PR opt-in suites in pr-test-amd-extra.yml
# (mirror of CUDA extra-a; tests stay tagged per-commit but only # (mirror of CUDA extra-a; tests stay tagged per-commit but only
# dispatch when the PR carries the `run-ci-extra` label). Only the # dispatch when the PR carries the `run-ci-extra` label). 1-gpu-small
# 1-gpu-small mock-model / kv_canary *unit* tests are onboarded so # carries the mock-model / kv_canary *unit* tests; 1-gpu-large carries
# far; the canary *e2e* tests (1-/2-gpu-large) need the canary JIT # the subset of model e2e tests validated to pass on mi325 (quant
# kernel ported to ROCm first, so those suites are intentionally # fp8kv-triton, sessions streaming-session EAGLE3, spec standalone
# not yet registered for AMD. # triton-backend variant). The rest of CUDA
# extra-a tests fail on ROCm (missing flash_attn.cute/flash_ops
# kernels, OOM, or accuracy regressions — e.g. gemma4-mtp-31b dips
# below the gsm8k floor on the topk=3 leg) and stay CUDA-only for now.
"extra-a-test-1-gpu-small-amd", "extra-a-test-1-gpu-small-amd",
"extra-a-test-1-gpu-large-amd",
], ],
HWBackend.MUSA: [], HWBackend.MUSA: [],
HWBackend.CUDA: [ HWBackend.CUDA: [