[AMD] ci: add label-gated extra-a tier (kv_canary + mock_model unit tests) (#27822)
This commit is contained in:
@@ -0,0 +1,174 @@
|
|||||||
|
name: PR Test Extra (AMD)
|
||||||
|
# Label-gated AMD extra CI workflow — the AMD mirror of pr-test-extra.yml.
|
||||||
|
#
|
||||||
|
# Adds AMD runtime to a PR only when the author opts in: the PR must carry
|
||||||
|
# BOTH `run-ci` (basic-CI prerequisite) and `run-ci-extra` (explicit opt-in).
|
||||||
|
# The label check happens at runtime in pr-gate.yml via a live
|
||||||
|
# `gh pr view`-style fetch, so reruns after adding the labels (e.g. via a
|
||||||
|
# slash command) pick up the new label set — a workflow-level `if` would read
|
||||||
|
# the frozen event payload, which never updates on rerun. The job graph also
|
||||||
|
# runs unconditionally on workflow_dispatch / workflow_call so it can be
|
||||||
|
# triggered manually or chained from the AMD scheduler.
|
||||||
|
#
|
||||||
|
# Stage: extra-a (1-gpu-small-amd). The job mirrors the container bring-up of
|
||||||
|
# pr-test-amd.yml and dispatches `run_suite.py --hw amd --suite
|
||||||
|
# extra-a-test-1-gpu-small-amd`. Only the mock-model / kv_canary *unit* tests
|
||||||
|
# are onboarded so far; the canary *e2e* tests (which would land in
|
||||||
|
# 1-/2-gpu-large) need the canary JIT kernel ported to ROCm first, so those
|
||||||
|
# suites are intentionally not registered for AMD yet.
|
||||||
|
|
||||||
|
on:
|
||||||
|
pull_request:
|
||||||
|
# `labeled` lets the workflow re-fire when `run-ci-extra` (or `run-ci`)
|
||||||
|
# is added after the latest push. See call-gate.if for the matching guard
|
||||||
|
# that prevents unrelated label additions from dispatching a full run.
|
||||||
|
types: [opened, synchronize, reopened, labeled]
|
||||||
|
workflow_dispatch:
|
||||||
|
inputs:
|
||||||
|
runner_arch:
|
||||||
|
description: 'AMD runner pool to dispatch GPU jobs to'
|
||||||
|
required: false
|
||||||
|
type: choice
|
||||||
|
default: mi325
|
||||||
|
options:
|
||||||
|
- mi300
|
||||||
|
- mi325
|
||||||
|
rocm_version:
|
||||||
|
description: 'ROCm container variant (empty = Dockerfile default; rocm720 = ROCm 7.2.0)'
|
||||||
|
required: false
|
||||||
|
type: choice
|
||||||
|
default: ''
|
||||||
|
options:
|
||||||
|
- ''
|
||||||
|
- rocm720
|
||||||
|
aiter_ref:
|
||||||
|
description: 'Override AITER commit (optional, leave empty to use Dockerfile default)'
|
||||||
|
required: false
|
||||||
|
type: string
|
||||||
|
default: ''
|
||||||
|
continue_on_error:
|
||||||
|
description: 'Continue on error (do not fail the workflow on test failures)'
|
||||||
|
required: false
|
||||||
|
type: boolean
|
||||||
|
default: false
|
||||||
|
workflow_call:
|
||||||
|
inputs:
|
||||||
|
ref:
|
||||||
|
description: 'Git ref (branch, tag, or SHA) to test. If not provided, uses the default branch.'
|
||||||
|
required: false
|
||||||
|
type: string
|
||||||
|
default: ''
|
||||||
|
rocm_version:
|
||||||
|
description: 'ROCm container variant (empty = Dockerfile default; rocm720 = ROCm 7.2.0)'
|
||||||
|
required: false
|
||||||
|
type: string
|
||||||
|
default: ''
|
||||||
|
aiter_ref:
|
||||||
|
description: 'Override AITER commit (optional, leave empty to use Dockerfile default)'
|
||||||
|
required: false
|
||||||
|
type: string
|
||||||
|
default: ''
|
||||||
|
continue_on_error:
|
||||||
|
description: 'Continue on error (do not fail the workflow on test failures)'
|
||||||
|
required: false
|
||||||
|
type: boolean
|
||||||
|
default: false
|
||||||
|
|
||||||
|
env:
|
||||||
|
AITER_COMMIT_OVERRIDE: ${{ inputs.aiter_ref }}
|
||||||
|
DOCKERHUB_AMD_USERNAME: ${{ secrets.DOCKERHUB_AMD_USERNAME }}
|
||||||
|
DOCKERHUB_AMD_TOKEN: ${{ secrets.DOCKERHUB_AMD_TOKEN }}
|
||||||
|
|
||||||
|
concurrency:
|
||||||
|
group: pr-test-amd-extra-${{ github.event_name }}-${{ github.head_ref || github.ref_name || 'default' }}-${{ inputs.ref || 'all' }}
|
||||||
|
cancel-in-progress: ${{ github.event_name != 'workflow_call' }}
|
||||||
|
|
||||||
|
permissions:
|
||||||
|
actions: write
|
||||||
|
contents: read
|
||||||
|
issues: read
|
||||||
|
pull-requests: read
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
# =============================================== PR Gate ====================================================
|
||||||
|
# Runtime live-fetch label gate (mirrors pr-test-extra.yml's call-gate):
|
||||||
|
# requires both `run-ci` and `run-ci-extra`. A failure here cascades to
|
||||||
|
# every test job via `needs`, so a PR without the labels ends in one red
|
||||||
|
# ~30s gate job plus a row of skipped jobs instead of consuming AMD runners.
|
||||||
|
#
|
||||||
|
# The job-level `if` only filters the `labeled` event type so that adding an
|
||||||
|
# unrelated label doesn't dispatch a full run; the actual label-presence
|
||||||
|
# gate is enforced at runtime inside pr-gate.yml.
|
||||||
|
call-gate:
|
||||||
|
if: |
|
||||||
|
github.event_name != 'pull_request' ||
|
||||||
|
github.event.action != 'labeled' ||
|
||||||
|
github.event.label.name == 'run-ci' ||
|
||||||
|
github.event.label.name == 'run-ci-extra'
|
||||||
|
uses: ./.github/workflows/pr-gate.yml
|
||||||
|
with:
|
||||||
|
require-run-ci: true
|
||||||
|
require-run-ci-extra: true
|
||||||
|
secrets: inherit
|
||||||
|
|
||||||
|
# =============================================== extra-a (1-gpu-small) ===============================================
|
||||||
|
# Single unpartitioned job: the 21 onboarded unit tests total ~233s, so the
|
||||||
|
# expensive per-job setup (container bring-up + sgl-kernel ROCm build + dep
|
||||||
|
# install, several minutes) dominates. Partitioning would multiply that
|
||||||
|
# setup across scarce AMD GPUs to shave only a couple minutes of test time,
|
||||||
|
# so one GPU running the whole suite sequentially is the better trade.
|
||||||
|
extra-a-test-1-gpu-small-amd:
|
||||||
|
name: ${{ format('extra-a-test-1-gpu-small-amd{0} (linux-{1}-1gpu-sglang)', inputs.rocm_version && format('-{0}', inputs.rocm_version) || '', inputs.runner_arch || 'mi325') }}
|
||||||
|
needs: [call-gate]
|
||||||
|
if: ${{ !cancelled() && needs.call-gate.result == 'success' }}
|
||||||
|
runs-on: ${{ format('linux-{0}-1gpu-sglang', inputs.runner_arch || 'mi325') }}
|
||||||
|
steps:
|
||||||
|
- name: Checkout code
|
||||||
|
uses: actions/checkout@v4
|
||||||
|
with:
|
||||||
|
ref: ${{ inputs.ref || github.sha }}
|
||||||
|
|
||||||
|
- name: Ensure VRAM is clear
|
||||||
|
run: bash scripts/ci/amd/ensure_vram_clear.sh rocm
|
||||||
|
|
||||||
|
- name: Start CI container
|
||||||
|
# `rocm_version` (e.g. rocm720) selects an alternate ROCm container; empty uses the Dockerfile default.
|
||||||
|
run: bash scripts/ci/amd/amd_ci_start_container.sh ${{ inputs.rocm_version && format('--rocm-version {0}', inputs.rocm_version) || '' }}
|
||||||
|
env:
|
||||||
|
GITHUB_WORKSPACE: ${{ github.workspace }}
|
||||||
|
|
||||||
|
- name: Install dependencies
|
||||||
|
run: bash scripts/ci/amd/amd_ci_install_dependency.sh
|
||||||
|
|
||||||
|
- name: Run test
|
||||||
|
timeout-minutes: 45
|
||||||
|
run: |
|
||||||
|
bash scripts/ci/amd/amd_ci_exec.sh -w "/sglang-checkout/test" python3 run_suite.py --hw amd --suite extra-a-test-1-gpu-small-amd --timeout-per-file 1800 ${{ inputs.continue_on_error == true && '--continue-on-error' || '' }}
|
||||||
|
|
||||||
|
# =============================================== aggregator ====================================================
|
||||||
|
# Single fan-in job so branch protection / notifications depend on one job
|
||||||
|
# rather than every matrix leg. Fails if any dependent failed or was
|
||||||
|
# cancelled; `skipped` (e.g. PR without the opt-in labels) is not a failure.
|
||||||
|
pr-test-amd-extra-finish:
|
||||||
|
needs:
|
||||||
|
[
|
||||||
|
call-gate,
|
||||||
|
extra-a-test-1-gpu-small-amd,
|
||||||
|
]
|
||||||
|
if: always()
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
steps:
|
||||||
|
- name: Check all dependent job statuses
|
||||||
|
run: |
|
||||||
|
json_needs='${{ toJson(needs) }}'
|
||||||
|
job_names=$(echo "$json_needs" | jq -r 'keys_unsorted[]')
|
||||||
|
for job in $job_names; do
|
||||||
|
result=$(echo "$json_needs" | jq -r --arg j "$job" '.[$j].result')
|
||||||
|
echo "$job: $result"
|
||||||
|
if [[ "$result" == "failure" || "$result" == "cancelled" ]]; then
|
||||||
|
echo "The above jobs failed."
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
done
|
||||||
|
echo "All jobs completed successfully"
|
||||||
|
exit 0
|
||||||
@@ -203,6 +203,26 @@ jobs:
|
|||||||
- "python/pyproject_rocm.toml"
|
- "python/pyproject_rocm.toml"
|
||||||
- "python/pyproject_other.toml"
|
- "python/pyproject_other.toml"
|
||||||
|
|
||||||
|
# =============================================== extra (scheduled) ====================================================
|
||||||
|
# ROCm 7.2 mirror of pr-test-amd.yml's `call-pr-test-amd-extra`: chain the
|
||||||
|
# label-gated AMD extra tier into this workflow's daily schedule, but in a
|
||||||
|
# ROCm 7.2 container (`rocm_version: rocm720`). On `schedule` (and
|
||||||
|
# run_all_tests dispatch) the extra suite runs on `main` without the
|
||||||
|
# `run-ci-extra` label (pr-gate.yml only enforces labels on pull_request
|
||||||
|
# events). Targeted dispatches (target_stage set) are excluded. Not wired
|
||||||
|
# into any finish aggregator so the base rocm720 run never depends on it.
|
||||||
|
call-pr-test-amd-extra-rocm720:
|
||||||
|
if: |
|
||||||
|
(github.event_name == 'schedule' || inputs.run_all_tests == true) &&
|
||||||
|
!(inputs.target_stage || inputs.target_stage_select)
|
||||||
|
uses: ./.github/workflows/pr-test-amd-extra.yml
|
||||||
|
with:
|
||||||
|
ref: ${{ inputs.pr_head_sha || inputs.ref || '' }}
|
||||||
|
rocm_version: rocm720
|
||||||
|
aiter_ref: ${{ inputs.aiter_ref }}
|
||||||
|
continue_on_error: true
|
||||||
|
secrets: inherit
|
||||||
|
|
||||||
# =============================================== sgl-kernel ====================================================
|
# =============================================== sgl-kernel ====================================================
|
||||||
sgl-kernel-unit-test-amd-rocm720:
|
sgl-kernel-unit-test-amd-rocm720:
|
||||||
needs: [check-changes]
|
needs: [check-changes]
|
||||||
|
|||||||
@@ -187,6 +187,25 @@ jobs:
|
|||||||
- "python/pyproject_rocm.toml"
|
- "python/pyproject_rocm.toml"
|
||||||
- "python/pyproject_other.toml"
|
- "python/pyproject_other.toml"
|
||||||
|
|
||||||
|
# =============================================== extra (scheduled) ====================================================
|
||||||
|
# Chain the label-gated AMD extra tier into the scheduled run, mirroring
|
||||||
|
# pr-test.yml's `call-pr-test-extra`. On `schedule` (and run_all_tests
|
||||||
|
# dispatch) the extra suite runs on `main` without needing the
|
||||||
|
# `run-ci-extra` label (pr-gate.yml only enforces labels on pull_request
|
||||||
|
# events). Targeted /rerun-stage dispatches (target_stage set) are excluded.
|
||||||
|
# Not added to `pr-test-amd-finish` so the base AMD gate never depends on
|
||||||
|
# the opt-in extra suite.
|
||||||
|
call-pr-test-amd-extra:
|
||||||
|
if: |
|
||||||
|
(github.event_name == 'schedule' || inputs.run_all_tests == true) &&
|
||||||
|
!(inputs.target_stage || inputs.target_stage_select)
|
||||||
|
uses: ./.github/workflows/pr-test-amd-extra.yml
|
||||||
|
with:
|
||||||
|
ref: ${{ inputs.pr_head_sha || inputs.ref || '' }}
|
||||||
|
aiter_ref: ${{ inputs.aiter_ref }}
|
||||||
|
continue_on_error: true
|
||||||
|
secrets: inherit
|
||||||
|
|
||||||
# =============================================== sgl-kernel ====================================================
|
# =============================================== sgl-kernel ====================================================
|
||||||
sgl-kernel-unit-test-amd:
|
sgl-kernel-unit-test-amd:
|
||||||
name: ${{ format('sgl-kernel-unit-test-amd (linux-{0}-1gpu-sglang)', inputs.runner_arch || 'mi325') }}
|
name: ${{ format('sgl-kernel-unit-test-amd (linux-{0}-1gpu-sglang)', inputs.runner_arch || 'mi325') }}
|
||||||
|
|||||||
@@ -12,10 +12,11 @@ from sglang.srt.kv_canary.pool_patcher.buffer_alloc import (
|
|||||||
make_row_source,
|
make_row_source,
|
||||||
resolve_real_kv_read_bytes,
|
resolve_real_kv_read_bytes,
|
||||||
)
|
)
|
||||||
from sglang.test.ci.ci_register import register_cuda_ci
|
from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci
|
||||||
from sglang.test.test_utils import CustomTestCase
|
from sglang.test.test_utils import CustomTestCase
|
||||||
|
|
||||||
register_cuda_ci(est_time=10, stage="extra-a", runner_config="1-gpu-small")
|
register_cuda_ci(est_time=10, stage="extra-a", runner_config="1-gpu-small")
|
||||||
|
register_amd_ci(est_time=10, suite="extra-a-test-1-gpu-small-amd")
|
||||||
|
|
||||||
|
|
||||||
def _config(mode: RealKvHashMode) -> CanaryConfig:
|
def _config(mode: RealKvHashMode) -> CanaryConfig:
|
||||||
|
|||||||
@@ -21,11 +21,12 @@ from sglang.srt.kv_canary.expected_inputs import ExpectedInputs
|
|||||||
from sglang.srt.kv_canary.state import (
|
from sglang.srt.kv_canary.state import (
|
||||||
ViolationLog,
|
ViolationLog,
|
||||||
)
|
)
|
||||||
from sglang.test.ci.ci_register import register_cuda_ci
|
from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci
|
||||||
from sglang.test.kv_canary.fixtures import DEFAULT_DEVICE
|
from sglang.test.kv_canary.fixtures import DEFAULT_DEVICE
|
||||||
from sglang.test.test_utils import CustomTestCase
|
from sglang.test.test_utils import CustomTestCase
|
||||||
|
|
||||||
register_cuda_ci(est_time=20, stage="extra-a", runner_config="1-gpu-small")
|
register_cuda_ci(est_time=20, stage="extra-a", runner_config="1-gpu-small")
|
||||||
|
register_amd_ci(est_time=20, suite="extra-a-test-1-gpu-small-amd")
|
||||||
|
|
||||||
|
|
||||||
def _make_endpoint(*, device, kernel_kind=CanaryLaunchTag.HEAD_K_FULL, swa_lut=None):
|
def _make_endpoint(*, device, kernel_kind=CanaryLaunchTag.HEAD_K_FULL, swa_lut=None):
|
||||||
|
|||||||
@@ -6,10 +6,11 @@ from typing import cast
|
|||||||
import torch
|
import torch
|
||||||
|
|
||||||
from sglang.srt.kv_canary.runner.future_tensor import FutureTensors
|
from sglang.srt.kv_canary.runner.future_tensor import FutureTensors
|
||||||
from sglang.test.ci.ci_register import register_cuda_ci
|
from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci
|
||||||
from sglang.test.test_utils import CustomTestCase
|
from sglang.test.test_utils import CustomTestCase
|
||||||
|
|
||||||
register_cuda_ci(est_time=20, stage="extra-a", runner_config="1-gpu-small")
|
register_cuda_ci(est_time=20, stage="extra-a", runner_config="1-gpu-small")
|
||||||
|
register_amd_ci(est_time=20, suite="extra-a-test-1-gpu-small-amd")
|
||||||
|
|
||||||
|
|
||||||
class _FakeEvent:
|
class _FakeEvent:
|
||||||
|
|||||||
@@ -27,7 +27,7 @@ from sglang.srt.kv_canary.perturb.utils import (
|
|||||||
flip_first_byte_in_source,
|
flip_first_byte_in_source,
|
||||||
pick_target_group,
|
pick_target_group,
|
||||||
)
|
)
|
||||||
from sglang.test.ci.ci_register import register_cuda_ci
|
from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci
|
||||||
from sglang.test.kv_canary.fixtures import (
|
from sglang.test.kv_canary.fixtures import (
|
||||||
DEFAULT_DEVICE,
|
DEFAULT_DEVICE,
|
||||||
make_buffer_group,
|
make_buffer_group,
|
||||||
@@ -41,6 +41,7 @@ if TYPE_CHECKING:
|
|||||||
from sglang.srt.mem_cache.base_prefix_cache import BasePrefixCache
|
from sglang.srt.mem_cache.base_prefix_cache import BasePrefixCache
|
||||||
|
|
||||||
register_cuda_ci(est_time=10, stage="extra-a", runner_config="1-gpu-small")
|
register_cuda_ci(est_time=10, stage="extra-a", runner_config="1-gpu-small")
|
||||||
|
register_amd_ci(est_time=10, suite="extra-a-test-1-gpu-small-amd")
|
||||||
|
|
||||||
|
|
||||||
class TestParseTargetGroupKind(CustomTestCase):
|
class TestParseTargetGroupKind(CustomTestCase):
|
||||||
|
|||||||
@@ -6,7 +6,7 @@ from types import SimpleNamespace
|
|||||||
import torch
|
import torch
|
||||||
|
|
||||||
from sglang.srt.kv_canary.plan_input import PlanInput
|
from sglang.srt.kv_canary.plan_input import PlanInput
|
||||||
from sglang.test.ci.ci_register import register_cuda_ci
|
from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci
|
||||||
from sglang.test.kv_canary.fixtures import (
|
from sglang.test.kv_canary.fixtures import (
|
||||||
DEFAULT_DEVICE,
|
DEFAULT_DEVICE,
|
||||||
make_forward_batch,
|
make_forward_batch,
|
||||||
@@ -14,6 +14,7 @@ from sglang.test.kv_canary.fixtures import (
|
|||||||
from sglang.test.test_utils import CustomTestCase
|
from sglang.test.test_utils import CustomTestCase
|
||||||
|
|
||||||
register_cuda_ci(est_time=30, stage="extra-a", runner_config="1-gpu-small")
|
register_cuda_ci(est_time=30, stage="extra-a", runner_config="1-gpu-small")
|
||||||
|
register_amd_ci(est_time=30, suite="extra-a-test-1-gpu-small-amd")
|
||||||
|
|
||||||
|
|
||||||
def _make_static_plan_input(*, bs_capacity: int, device) -> PlanInput:
|
def _make_static_plan_input(*, bs_capacity: int, device) -> PlanInput:
|
||||||
|
|||||||
@@ -15,7 +15,7 @@ from sglang.jit_kernel.kv_canary.verify import (
|
|||||||
)
|
)
|
||||||
from sglang.srt.kv_canary.buffer_group import PoolKind
|
from sglang.srt.kv_canary.buffer_group import PoolKind
|
||||||
from sglang.srt.kv_canary.pool_patcher.api import attach_canary_buffers
|
from sglang.srt.kv_canary.pool_patcher.api import attach_canary_buffers
|
||||||
from sglang.test.ci.ci_register import register_cuda_ci
|
from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci
|
||||||
from sglang.test.kv_canary.fixtures import (
|
from sglang.test.kv_canary.fixtures import (
|
||||||
DEFAULT_DEVICE,
|
DEFAULT_DEVICE,
|
||||||
make_base_config,
|
make_base_config,
|
||||||
@@ -25,6 +25,7 @@ from sglang.test.kv_canary.fixtures import (
|
|||||||
from sglang.test.test_utils import CustomTestCase
|
from sglang.test.test_utils import CustomTestCase
|
||||||
|
|
||||||
register_cuda_ci(est_time=45, stage="extra-a", runner_config="1-gpu-small")
|
register_cuda_ci(est_time=45, stage="extra-a", runner_config="1-gpu-small")
|
||||||
|
register_amd_ci(est_time=45, suite="extra-a-test-1-gpu-small-amd")
|
||||||
|
|
||||||
|
|
||||||
class PoolPatcherHelper:
|
class PoolPatcherHelper:
|
||||||
|
|||||||
@@ -3,10 +3,11 @@ from __future__ import annotations
|
|||||||
import unittest
|
import unittest
|
||||||
|
|
||||||
from sglang.srt.kv_canary.pool_patcher.utils import wrap_method
|
from sglang.srt.kv_canary.pool_patcher.utils import wrap_method
|
||||||
from sglang.test.ci.ci_register import register_cuda_ci
|
from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci
|
||||||
from sglang.test.test_utils import CustomTestCase
|
from sglang.test.test_utils import CustomTestCase
|
||||||
|
|
||||||
register_cuda_ci(est_time=10, stage="extra-a", runner_config="1-gpu-small")
|
register_cuda_ci(est_time=10, stage="extra-a", runner_config="1-gpu-small")
|
||||||
|
register_amd_ci(est_time=10, suite="extra-a-test-1-gpu-small-amd")
|
||||||
|
|
||||||
|
|
||||||
class _FakeObj:
|
class _FakeObj:
|
||||||
|
|||||||
@@ -6,11 +6,12 @@ import torch
|
|||||||
|
|
||||||
from sglang.srt.kv_canary.radix_cache_walker import walk_radix_cache_for_canary
|
from sglang.srt.kv_canary.radix_cache_walker import walk_radix_cache_for_canary
|
||||||
from sglang.srt.mem_cache.swa_radix_cache import SWARadixCache, TreeNode
|
from sglang.srt.mem_cache.swa_radix_cache import SWARadixCache, TreeNode
|
||||||
from sglang.test.ci.ci_register import register_cuda_ci
|
from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci
|
||||||
from sglang.test.kv_canary.fixtures import DEFAULT_DEVICE, make_radix_cache
|
from sglang.test.kv_canary.fixtures import DEFAULT_DEVICE, make_radix_cache
|
||||||
from sglang.test.test_utils import CustomTestCase
|
from sglang.test.test_utils import CustomTestCase
|
||||||
|
|
||||||
register_cuda_ci(est_time=30, stage="extra-a", runner_config="1-gpu-small")
|
register_cuda_ci(est_time=30, stage="extra-a", runner_config="1-gpu-small")
|
||||||
|
register_amd_ci(est_time=30, suite="extra-a-test-1-gpu-small-amd")
|
||||||
|
|
||||||
|
|
||||||
class TestSelfUnitRadixWalker(CustomTestCase):
|
class TestSelfUnitRadixWalker(CustomTestCase):
|
||||||
|
|||||||
@@ -10,11 +10,12 @@ from sglang.srt.kv_canary.req_to_expected_token_ids_manager import (
|
|||||||
compute_req_all_ids_info,
|
compute_req_all_ids_info,
|
||||||
populate_req_to_expected_token_ids,
|
populate_req_to_expected_token_ids,
|
||||||
)
|
)
|
||||||
from sglang.test.ci.ci_register import register_cuda_ci
|
from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci
|
||||||
from sglang.test.kv_canary.fixtures import DEFAULT_DEVICE, make_forward_batch
|
from sglang.test.kv_canary.fixtures import DEFAULT_DEVICE, make_forward_batch
|
||||||
from sglang.test.test_utils import CustomTestCase
|
from sglang.test.test_utils import CustomTestCase
|
||||||
|
|
||||||
register_cuda_ci(est_time=15, stage="extra-a", runner_config="1-gpu-small")
|
register_cuda_ci(est_time=15, stage="extra-a", runner_config="1-gpu-small")
|
||||||
|
register_amd_ci(est_time=15, suite="extra-a-test-1-gpu-small-amd")
|
||||||
|
|
||||||
|
|
||||||
def _make_req(*, origin: list[int], output: list[int]) -> SimpleNamespace:
|
def _make_req(*, origin: list[int], output: list[int]) -> SimpleNamespace:
|
||||||
|
|||||||
@@ -11,7 +11,7 @@ from sglang.srt.kv_canary.config import CanaryConfig
|
|||||||
from sglang.srt.kv_canary.runner import stats_logger as stats_logger_module
|
from sglang.srt.kv_canary.runner import stats_logger as stats_logger_module
|
||||||
from sglang.srt.kv_canary.runner.health_checker import KernelRunCounterHealthChecker
|
from sglang.srt.kv_canary.runner.health_checker import KernelRunCounterHealthChecker
|
||||||
from sglang.srt.kv_canary.state import CanaryDeviceState
|
from sglang.srt.kv_canary.state import CanaryDeviceState
|
||||||
from sglang.test.ci.ci_register import register_cuda_ci
|
from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci
|
||||||
from sglang.test.kv_canary.runner_test_base import (
|
from sglang.test.kv_canary.runner_test_base import (
|
||||||
CanaryManagerTestCase,
|
CanaryManagerTestCase,
|
||||||
make_config,
|
make_config,
|
||||||
@@ -20,6 +20,7 @@ from sglang.test.kv_canary.runner_test_base import (
|
|||||||
from sglang.test.test_utils import CustomTestCase
|
from sglang.test.test_utils import CustomTestCase
|
||||||
|
|
||||||
register_cuda_ci(est_time=45, stage="extra-a", runner_config="1-gpu-small")
|
register_cuda_ci(est_time=45, stage="extra-a", runner_config="1-gpu-small")
|
||||||
|
register_amd_ci(est_time=45, suite="extra-a-test-1-gpu-small-amd")
|
||||||
|
|
||||||
|
|
||||||
class TestSelfUnitManagerHealth(CanaryManagerTestCase):
|
class TestSelfUnitManagerHealth(CanaryManagerTestCase):
|
||||||
|
|||||||
@@ -12,7 +12,7 @@ from sglang.srt.kv_canary import endpoint as endpoint_module
|
|||||||
from sglang.srt.kv_canary.expected_inputs import ExpectedInputs
|
from sglang.srt.kv_canary.expected_inputs import ExpectedInputs
|
||||||
from sglang.srt.kv_canary.runner import kernel_launcher as kernel_launcher_module
|
from sglang.srt.kv_canary.runner import kernel_launcher as kernel_launcher_module
|
||||||
from sglang.srt.kv_canary.state import ViolationLog
|
from sglang.srt.kv_canary.state import ViolationLog
|
||||||
from sglang.test.ci.ci_register import register_cuda_ci
|
from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci
|
||||||
from sglang.test.kv_canary.fixtures import make_buffer_group, make_forward_batch
|
from sglang.test.kv_canary.fixtures import make_buffer_group, make_forward_batch
|
||||||
from sglang.test.kv_canary.runner_test_base import (
|
from sglang.test.kv_canary.runner_test_base import (
|
||||||
CanaryManagerTestCase,
|
CanaryManagerTestCase,
|
||||||
@@ -21,6 +21,7 @@ from sglang.test.kv_canary.runner_test_base import (
|
|||||||
)
|
)
|
||||||
|
|
||||||
register_cuda_ci(est_time=45, stage="extra-a", runner_config="1-gpu-small")
|
register_cuda_ci(est_time=45, stage="extra-a", runner_config="1-gpu-small")
|
||||||
|
register_amd_ci(est_time=45, suite="extra-a-test-1-gpu-small-amd")
|
||||||
|
|
||||||
|
|
||||||
class TestManagerPerForward(CanaryManagerTestCase):
|
class TestManagerPerForward(CanaryManagerTestCase):
|
||||||
|
|||||||
@@ -15,12 +15,13 @@ from sglang.srt.kv_canary.runner.swa_divergence import (
|
|||||||
SwaDivergenceReporter,
|
SwaDivergenceReporter,
|
||||||
compute_swa_full_idx_divergence,
|
compute_swa_full_idx_divergence,
|
||||||
)
|
)
|
||||||
from sglang.test.ci.ci_register import register_cuda_ci
|
from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci
|
||||||
from sglang.test.kv_canary.fixtures import make_buffer_group
|
from sglang.test.kv_canary.fixtures import make_buffer_group
|
||||||
from sglang.test.kv_canary.runner_test_base import CanaryManagerTestCase, make_manager
|
from sglang.test.kv_canary.runner_test_base import CanaryManagerTestCase, make_manager
|
||||||
from sglang.test.test_utils import CustomTestCase
|
from sglang.test.test_utils import CustomTestCase
|
||||||
|
|
||||||
register_cuda_ci(est_time=45, stage="extra-a", runner_config="1-gpu-small")
|
register_cuda_ci(est_time=45, stage="extra-a", runner_config="1-gpu-small")
|
||||||
|
register_amd_ci(est_time=45, suite="extra-a-test-1-gpu-small-amd")
|
||||||
|
|
||||||
_DEVICE = torch.device("cuda")
|
_DEVICE = torch.device("cuda")
|
||||||
|
|
||||||
|
|||||||
@@ -4,7 +4,7 @@ import unittest
|
|||||||
from unittest.mock import patch
|
from unittest.mock import patch
|
||||||
|
|
||||||
from sglang.srt.kv_canary import endpoint as endpoint_module
|
from sglang.srt.kv_canary import endpoint as endpoint_module
|
||||||
from sglang.test.ci.ci_register import register_cuda_ci
|
from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci
|
||||||
from sglang.test.kv_canary.fixtures import (
|
from sglang.test.kv_canary.fixtures import (
|
||||||
make_forward_batch,
|
make_forward_batch,
|
||||||
make_radix_cache,
|
make_radix_cache,
|
||||||
@@ -17,6 +17,7 @@ from sglang.test.kv_canary.runner_test_base import (
|
|||||||
)
|
)
|
||||||
|
|
||||||
register_cuda_ci(est_time=45, stage="extra-a", runner_config="1-gpu-small")
|
register_cuda_ci(est_time=45, stage="extra-a", runner_config="1-gpu-small")
|
||||||
|
register_amd_ci(est_time=45, suite="extra-a-test-1-gpu-small-amd")
|
||||||
|
|
||||||
|
|
||||||
def _run_one_cycle(manager, forward_batch) -> None:
|
def _run_one_cycle(manager, forward_batch) -> None:
|
||||||
|
|||||||
@@ -5,7 +5,7 @@ import unittest
|
|||||||
import torch
|
import torch
|
||||||
|
|
||||||
from sglang.srt.kv_canary.sweep_plan_builder import build_verify_plan_radix_sweep
|
from sglang.srt.kv_canary.sweep_plan_builder import build_verify_plan_radix_sweep
|
||||||
from sglang.test.ci.ci_register import register_cuda_ci
|
from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci
|
||||||
from sglang.test.kv_canary.fixtures import (
|
from sglang.test.kv_canary.fixtures import (
|
||||||
DEFAULT_DEVICE,
|
DEFAULT_DEVICE,
|
||||||
make_radix_cache,
|
make_radix_cache,
|
||||||
@@ -14,6 +14,7 @@ from sglang.test.kv_canary.fixtures import (
|
|||||||
from sglang.test.test_utils import CustomTestCase
|
from sglang.test.test_utils import CustomTestCase
|
||||||
|
|
||||||
register_cuda_ci(est_time=30, stage="extra-a", runner_config="1-gpu-small")
|
register_cuda_ci(est_time=30, stage="extra-a", runner_config="1-gpu-small")
|
||||||
|
register_amd_ci(est_time=30, suite="extra-a-test-1-gpu-small-amd")
|
||||||
|
|
||||||
|
|
||||||
class TestSelfUnitSweepPlanBuilder(CustomTestCase):
|
class TestSelfUnitSweepPlanBuilder(CustomTestCase):
|
||||||
|
|||||||
@@ -9,11 +9,12 @@ from sglang.srt.kv_canary.expected_inputs import ExpectedInputs
|
|||||||
from sglang.srt.kv_canary.token_oracle.oracle import HashOracle
|
from sglang.srt.kv_canary.token_oracle.oracle import HashOracle
|
||||||
from sglang.srt.kv_canary.token_oracle.oracle_manager import TokenOracleManager
|
from sglang.srt.kv_canary.token_oracle.oracle_manager import TokenOracleManager
|
||||||
from sglang.srt.model_executor.forward_batch_info import ForwardMode
|
from sglang.srt.model_executor.forward_batch_info import ForwardMode
|
||||||
from sglang.test.ci.ci_register import register_cuda_ci
|
from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci
|
||||||
from sglang.test.kv_canary.fixtures import DEFAULT_DEVICE
|
from sglang.test.kv_canary.fixtures import DEFAULT_DEVICE
|
||||||
from sglang.test.test_utils import CustomTestCase
|
from sglang.test.test_utils import CustomTestCase
|
||||||
|
|
||||||
register_cuda_ci(est_time=1, stage="extra-a", runner_config="1-gpu-small")
|
register_cuda_ci(est_time=1, stage="extra-a", runner_config="1-gpu-small")
|
||||||
|
register_amd_ci(est_time=1, suite="extra-a-test-1-gpu-small-amd")
|
||||||
|
|
||||||
|
|
||||||
class TestTokenOracleManager(CustomTestCase):
|
class TestTokenOracleManager(CustomTestCase):
|
||||||
|
|||||||
@@ -15,10 +15,11 @@ from sglang.srt.kv_canary.runner.violation_reporter import (
|
|||||||
ViolationReporter,
|
ViolationReporter,
|
||||||
_format_violation,
|
_format_violation,
|
||||||
)
|
)
|
||||||
from sglang.test.ci.ci_register import register_cuda_ci
|
from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci
|
||||||
from sglang.test.test_utils import CustomTestCase
|
from sglang.test.test_utils import CustomTestCase
|
||||||
|
|
||||||
register_cuda_ci(est_time=5, stage="extra-a", runner_config="1-gpu-small")
|
register_cuda_ci(est_time=5, stage="extra-a", runner_config="1-gpu-small")
|
||||||
|
register_amd_ci(est_time=5, suite="extra-a-test-1-gpu-small-amd")
|
||||||
|
|
||||||
|
|
||||||
def _make_row(
|
def _make_row(
|
||||||
|
|||||||
@@ -12,11 +12,12 @@ from sglang.srt.model_executor.forward_batch_info import (
|
|||||||
ForwardMode,
|
ForwardMode,
|
||||||
_stable_hash_str_to_i64,
|
_stable_hash_str_to_i64,
|
||||||
)
|
)
|
||||||
from sglang.test.ci.ci_register import register_cuda_ci
|
from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci
|
||||||
from sglang.test.mock_model.utils import mock_model_server_args, mock_model_server_env
|
from sglang.test.mock_model.utils import mock_model_server_args, mock_model_server_env
|
||||||
from sglang.test.test_utils import CustomTestCase
|
from sglang.test.test_utils import CustomTestCase
|
||||||
|
|
||||||
register_cuda_ci(est_time=60, stage="extra-a", runner_config="1-gpu-small")
|
register_cuda_ci(est_time=60, stage="extra-a", runner_config="1-gpu-small")
|
||||||
|
register_amd_ci(est_time=60, suite="extra-a-test-1-gpu-small-amd")
|
||||||
|
|
||||||
|
|
||||||
@dataclasses.dataclass
|
@dataclasses.dataclass
|
||||||
|
|||||||
@@ -9,10 +9,11 @@ os.environ["SGLANG_KV_CANARY_ENABLE_TOKEN_ORACLE"] = "1"
|
|||||||
from sglang.srt.kv_canary.token_oracle.install import install_token_oracle_from_env
|
from sglang.srt.kv_canary.token_oracle.install import install_token_oracle_from_env
|
||||||
from sglang.srt.kv_canary.token_oracle.oracle import HashOracle
|
from sglang.srt.kv_canary.token_oracle.oracle import HashOracle
|
||||||
from sglang.srt.layers.sampler import _CUSTOM_SAMPLER_FACTORIES
|
from sglang.srt.layers.sampler import _CUSTOM_SAMPLER_FACTORIES
|
||||||
from sglang.test.ci.ci_register import register_cuda_ci
|
from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci
|
||||||
from sglang.test.test_utils import CustomTestCase
|
from sglang.test.test_utils import CustomTestCase
|
||||||
|
|
||||||
register_cuda_ci(est_time=60, stage="extra-a", runner_config="1-gpu-small")
|
register_cuda_ci(est_time=60, stage="extra-a", runner_config="1-gpu-small")
|
||||||
|
register_amd_ci(est_time=60, suite="extra-a-test-1-gpu-small-amd")
|
||||||
|
|
||||||
|
|
||||||
def _make_server_args(*, sampling_backend: str) -> SimpleNamespace:
|
def _make_server_args(*, sampling_backend: str) -> SimpleNamespace:
|
||||||
|
|||||||
@@ -10,10 +10,11 @@ from sglang.srt.kv_canary.token_oracle.oracle import (
|
|||||||
HashOracle,
|
HashOracle,
|
||||||
_splitmix64_tensor,
|
_splitmix64_tensor,
|
||||||
)
|
)
|
||||||
from sglang.test.ci.ci_register import register_cuda_ci
|
from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci
|
||||||
from sglang.test.test_utils import CustomTestCase
|
from sglang.test.test_utils import CustomTestCase
|
||||||
|
|
||||||
register_cuda_ci(est_time=60, stage="extra-a", runner_config="1-gpu-small")
|
register_cuda_ci(est_time=60, stage="extra-a", runner_config="1-gpu-small")
|
||||||
|
register_amd_ci(est_time=60, suite="extra-a-test-1-gpu-small-amd")
|
||||||
|
|
||||||
|
|
||||||
_U64_MASK: int = (1 << 64) - 1
|
_U64_MASK: int = (1 << 64) - 1
|
||||||
|
|||||||
@@ -7,10 +7,11 @@ import torch
|
|||||||
|
|
||||||
from sglang.jit_kernel.kv_canary.verify_ref import splitmix64
|
from sglang.jit_kernel.kv_canary.verify_ref import splitmix64
|
||||||
from sglang.srt.kv_canary.token_oracle.oracle import HashOracle
|
from sglang.srt.kv_canary.token_oracle.oracle import HashOracle
|
||||||
from sglang.test.ci.ci_register import register_cuda_ci
|
from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci
|
||||||
from sglang.test.test_utils import CustomTestCase
|
from sglang.test.test_utils import CustomTestCase
|
||||||
|
|
||||||
register_cuda_ci(est_time=30, stage="extra-a", runner_config="1-gpu-small")
|
register_cuda_ci(est_time=30, stage="extra-a", runner_config="1-gpu-small")
|
||||||
|
register_amd_ci(est_time=30, suite="extra-a-test-1-gpu-small-amd")
|
||||||
|
|
||||||
|
|
||||||
class TestHashOracleTorchVsRef(CustomTestCase):
|
class TestHashOracleTorchVsRef(CustomTestCase):
|
||||||
|
|||||||
@@ -18,10 +18,11 @@ from sglang.srt.kv_canary.token_oracle.oracle import HashOracle
|
|||||||
from sglang.srt.kv_canary.token_oracle.sampler import install_oracle_sampler
|
from sglang.srt.kv_canary.token_oracle.sampler import install_oracle_sampler
|
||||||
from sglang.srt.layers.sampler import _CUSTOM_SAMPLER_FACTORIES
|
from sglang.srt.layers.sampler import _CUSTOM_SAMPLER_FACTORIES
|
||||||
from sglang.srt.server_args import SAMPLING_BACKEND_CHOICES
|
from sglang.srt.server_args import SAMPLING_BACKEND_CHOICES
|
||||||
from sglang.test.ci.ci_register import register_cuda_ci
|
from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci
|
||||||
from sglang.test.test_utils import CustomTestCase
|
from sglang.test.test_utils import CustomTestCase
|
||||||
|
|
||||||
register_cuda_ci(est_time=60, stage="extra-a", runner_config="1-gpu-small")
|
register_cuda_ci(est_time=60, stage="extra-a", runner_config="1-gpu-small")
|
||||||
|
register_amd_ci(est_time=60, suite="extra-a-test-1-gpu-small-amd")
|
||||||
|
|
||||||
|
|
||||||
class TestInstallOracleSampler(CustomTestCase):
|
class TestInstallOracleSampler(CustomTestCase):
|
||||||
|
|||||||
@@ -43,6 +43,14 @@ PER_COMMIT_SUITES = {
|
|||||||
"stage-c-test-4-gpu-amd",
|
"stage-c-test-4-gpu-amd",
|
||||||
"stage-c-test-large-8-gpu-amd",
|
"stage-c-test-large-8-gpu-amd",
|
||||||
"stage-c-test-large-8-gpu-amd-mi35x",
|
"stage-c-test-large-8-gpu-amd-mi35x",
|
||||||
|
# extra-a: label-gated PR opt-in suite in pr-test-amd-extra.yml
|
||||||
|
# (mirror of CUDA extra-a; tests stay tagged per-commit but only
|
||||||
|
# dispatch when the PR carries the `run-ci-extra` label). Only the
|
||||||
|
# 1-gpu-small mock-model / kv_canary *unit* tests are onboarded so
|
||||||
|
# far; the canary *e2e* tests (1-/2-gpu-large) need the canary JIT
|
||||||
|
# kernel ported to ROCm first, so those suites are intentionally
|
||||||
|
# not yet registered for AMD.
|
||||||
|
"extra-a-test-1-gpu-small-amd",
|
||||||
],
|
],
|
||||||
HWBackend.MUSA: [],
|
HWBackend.MUSA: [],
|
||||||
HWBackend.CUDA: [
|
HWBackend.CUDA: [
|
||||||
|
|||||||
Reference in New Issue
Block a user