[AMD] Add 4-GPU test suite for MI325 runners (#20294)

This commit is contained in:
Michael
2026-03-24 14:04:49 -07:00
committed by GitHub
parent c4db64c16b
commit 6cb1c2d53d
12 changed files with 252 additions and 13 deletions
@@ -32,6 +32,7 @@ on:
- nightly-accuracy-2-gpu-vlm-rocm720
- nightly-perf-2-gpu-text-rocm720
- nightly-perf-2-gpu-vlm-rocm720
- nightly-4-gpu-rocm720
- nightly-accuracy-8-gpu-rocm720
- nightly-8-gpu-grok1-int4-rocm720
- nightly-8-gpu-grok2-rocm720
@@ -247,6 +248,37 @@ jobs:
echo "$(<github_summary.md )" >> $GITHUB_STEP_SUMMARY || true
exit ${TEST_EXIT_CODE:-0}
# ============================================== MI30x ROCm 7.2 4-GPU Tests ==============================================
# 4-GPU Nightly Tests - Dumper/Comparator E2E, VLM Encoder DP (ROCm 7.2)
nightly-4-gpu-rocm720:
if: (github.repository == 'sgl-project/sglang' || github.event_name == 'pull_request') && (!(inputs.job_filter || inputs.job_select) || (inputs.job_filter || inputs.job_select) == 'all' || contains(format(',{0},', inputs.job_filter || inputs.job_select), ',nightly-4-gpu-rocm720,'))
runs-on: linux-mi325-4gpu-sglang
steps:
- name: Checkout code
uses: actions/checkout@v4
with:
ref: ${{ inputs.ref || github.ref }}
- name: Setup docker (ROCm 7.2)
run: |
touch github_summary.md
bash scripts/ci/amd/amd_ci_start_container.sh --rocm-version rocm720
env:
GITHUB_WORKSPACE: ${{ github.workspace }}
- name: Install dependencies
run: bash scripts/ci/amd/amd_ci_install_dependency.sh
- name: Nightly Test ROCm 7.2 (4-GPU)
timeout-minutes: 120
run: |
> github_summary.md
bash scripts/ci/amd/amd_ci_exec.sh -w /sglang-checkout/test \
-e GITHUB_STEP_SUMMARY="/sglang-checkout/github_summary.md" \
python3 run_suite.py --hw amd --suite nightly-amd-4-gpu --nightly --continue-on-error --timeout-per-file 3600 || TEST_EXIT_CODE=$?
echo "$(<github_summary.md )" >> $GITHUB_STEP_SUMMARY || true
exit ${TEST_EXIT_CODE:-0}
# 8-GPU Accuracy Tests - GPT-OSS, Grok1-FP8 (ROCm 7.2)
nightly-accuracy-8-gpu-rocm720:
if: (github.repository == 'sgl-project/sglang' || github.event_name == 'pull_request') && (!(inputs.job_filter || inputs.job_select) || (inputs.job_filter || inputs.job_select) == 'all' || contains(format(',{0},', inputs.job_filter || inputs.job_select), ',nightly-accuracy-8-gpu-rocm720,'))
@@ -1317,6 +1349,8 @@ jobs:
# MI30x ROCm 7.2 Performance Tests
- nightly-perf-2-gpu-text-rocm720
- nightly-perf-2-gpu-vlm-rocm720
# MI30x ROCm 7.2 4-GPU Tests
- nightly-4-gpu-rocm720
- nightly-accuracy-8-gpu-rocm720
# MI30x ROCm 7.2 Combined Accuracy + Performance Tests
- nightly-8-gpu-grok1-int4-rocm720
+34
View File
@@ -32,6 +32,7 @@ on:
- nightly-accuracy-2-gpu-vlm
- nightly-perf-2-gpu-text
- nightly-perf-2-gpu-vlm
- nightly-4-gpu
- nightly-accuracy-8-gpu
- nightly-8-gpu-grok1-int4
- nightly-8-gpu-grok2
@@ -252,6 +253,37 @@ jobs:
echo "$(<github_summary.md )" >> $GITHUB_STEP_SUMMARY || true
exit ${TEST_EXIT_CODE:-0}
# ============================================== MI30x 4-GPU Tests ==============================================
# 4-GPU Nightly Tests - Dumper/Comparator E2E, VLM Encoder DP
nightly-4-gpu:
if: (github.repository == 'sgl-project/sglang' || github.event_name == 'pull_request') && (!(inputs.job_filter || inputs.job_select) || (inputs.job_filter || inputs.job_select) == 'all' || contains(format(',{0},', inputs.job_filter || inputs.job_select), ',nightly-4-gpu,'))
runs-on: linux-mi325-4gpu-sglang
steps:
- name: Checkout code
uses: actions/checkout@v4
with:
ref: ${{ inputs.ref || github.ref }}
- name: Setup docker
run: |
touch github_summary.md
bash scripts/ci/amd/amd_ci_start_container.sh
env:
GITHUB_WORKSPACE: ${{ github.workspace }}
- name: Install dependencies
run: bash scripts/ci/amd/amd_ci_install_dependency.sh
- name: Nightly Test (4-GPU)
timeout-minutes: 120
run: |
> github_summary.md
bash scripts/ci/amd/amd_ci_exec.sh -w /sglang-checkout/test \
-e GITHUB_STEP_SUMMARY="/sglang-checkout/github_summary.md" \
python3 run_suite.py --hw amd --suite nightly-amd-4-gpu --nightly --continue-on-error --timeout-per-file 3600 || TEST_EXIT_CODE=$?
echo "$(<github_summary.md )" >> $GITHUB_STEP_SUMMARY || true
exit ${TEST_EXIT_CODE:-0}
# 8-GPU Accuracy Tests - GPT-OSS, Grok1-FP8 (accuracy only)
nightly-accuracy-8-gpu:
if: (github.repository == 'sgl-project/sglang' || github.event_name == 'pull_request') && (!(inputs.job_filter || inputs.job_select) || (inputs.job_filter || inputs.job_select) == 'all' || contains(format(',{0},', inputs.job_filter || inputs.job_select), ',nightly-accuracy-8-gpu,'))
@@ -1320,6 +1352,8 @@ jobs:
# MI30x Accuracy Tests
- nightly-accuracy-2-gpu
- nightly-accuracy-2-gpu-vlm
# MI30x 4-GPU Tests
- nightly-4-gpu
- nightly-accuracy-8-gpu
# MI30x Performance Tests - excluded from check (perf failures don't block CI)
# - nightly-perf-2-gpu-text
+58
View File
@@ -47,6 +47,7 @@ on:
- stage-c-test-large-8-gpu-amd
- stage-c-test-large-8-gpu-amd-mi35x
- stage-b-test-large-8-gpu-disaggregation-amd
- stage-c-test-4-gpu-amd
target_stage:
description: "Or type comma-separated stage names (overrides dropdown if non-empty)"
required: false
@@ -775,6 +776,62 @@ jobs:
free -h
stage-c-test-4-gpu-amd:
needs: [check-changes, stage-b-test-small-1-gpu-amd, stage-b-test-large-2-gpu-amd]
if: |
always() &&
(
(contains(format(',{0},', inputs.target_stage || inputs.target_stage_select), ',stage-c-test-4-gpu-amd,')) ||
(
!(inputs.target_stage || inputs.target_stage_select) &&
(!failure() && !cancelled()) &&
((needs.check-changes.outputs.main_package == 'true') || (needs.check-changes.outputs.sgl_kernel == 'true'))
)
)
strategy:
fail-fast: false
matrix:
runner: [linux-mi325-4gpu-sglang]
part: [0, 1]
runs-on: ${{matrix.runner}}
steps:
- name: Checkout code
uses: actions/checkout@v4
with:
ref: ${{ inputs.pr_head_sha || inputs.ref || github.sha }}
- name: Ensure VRAM is clear
run: bash scripts/ensure_vram_clear.sh rocm
- name: Start CI container
run: bash scripts/ci/amd/amd_ci_start_container.sh --rocm-version rocm720
env:
GITHUB_WORKSPACE: ${{ github.workspace }}
- name: Install dependencies
run: bash scripts/ci/amd/amd_ci_install_dependency.sh
- name: Run test
timeout-minutes: 60
run: |
bash scripts/ci/amd/amd_ci_exec.sh \
-e NCCL_CUMEM_ENABLE=0 \
-e NCCL_NVLS_ENABLE=0 \
-e RCCL_MSCCL_ENABLE=0 \
-e SGLANG_USE_ROCM700A=1 \
-w "/sglang-checkout/test" \
python3 run_suite.py \
--hw amd \
--suite stage-c-test-4-gpu-amd \
--auto-partition-id ${{ matrix.part }} \
--auto-partition-size 2 \
--timeout-per-file 1800 \
--enable-retry \
--max-attempts 2 \
--retry-wait-seconds 120 \
--retry-timeout-increase 0 \
${{ inputs.continue_on_error && '--continue-on-error' || '' }}
stage-c-test-large-8-gpu-amd:
needs: [check-changes]
if: |
@@ -992,6 +1049,7 @@ jobs:
stage-b-test-1-gpu-large-amd,
stage-b-test-2-gpu-large-amd,
stage-b-test-large-8-gpu-35x-disaggregation-amd,
stage-c-test-4-gpu-amd,
stage-c-test-large-8-gpu-amd,
stage-c-test-large-8-gpu-amd-mi35x,
]
+58
View File
@@ -42,6 +42,7 @@ on:
- stage-b-test-2-gpu-large-amd
- multimodal-gen-test-1-gpu-amd
- multimodal-gen-test-2-gpu-amd
- stage-c-test-4-gpu-amd
- stage-c-test-large-8-gpu-amd
- stage-c-test-large-8-gpu-amd-mi35x
- stage-b-test-large-8-gpu-disaggregation-amd
@@ -778,6 +779,62 @@ jobs:
free -h
stage-c-test-4-gpu-amd:
needs: [check-changes, call-gate, stage-b-test-small-1-gpu-amd, stage-b-test-large-2-gpu-amd]
if: |
always() &&
(
(contains(format(',{0},', inputs.target_stage || inputs.target_stage_select), ',stage-c-test-4-gpu-amd,')) ||
(
!(inputs.target_stage || inputs.target_stage_select) &&
(!failure() && !cancelled()) &&
((needs.check-changes.outputs.main_package == 'true') || (needs.check-changes.outputs.sgl_kernel == 'true'))
)
)
strategy:
fail-fast: false
matrix:
runner: [linux-mi325-4gpu-sglang]
part: [0, 1]
runs-on: ${{matrix.runner}}
steps:
- name: Checkout code
uses: actions/checkout@v4
with:
ref: ${{ inputs.pr_head_sha || inputs.ref || github.sha }}
- name: Ensure VRAM is clear
run: bash scripts/ensure_vram_clear.sh rocm
- name: Start CI container
run: bash scripts/ci/amd/amd_ci_start_container.sh
env:
GITHUB_WORKSPACE: ${{ github.workspace }}
- name: Install dependencies
run: bash scripts/ci/amd/amd_ci_install_dependency.sh
- name: Run test
timeout-minutes: 60
run: |
bash scripts/ci/amd/amd_ci_exec.sh \
-e NCCL_CUMEM_ENABLE=0 \
-e NCCL_NVLS_ENABLE=0 \
-e RCCL_MSCCL_ENABLE=0 \
-e SGLANG_USE_ROCM700A=1 \
-w "/sglang-checkout/test" \
python3 run_suite.py \
--hw amd \
--suite stage-c-test-4-gpu-amd \
--auto-partition-id ${{ matrix.part }} \
--auto-partition-size 2 \
--timeout-per-file 1800 \
--enable-retry \
--max-attempts 2 \
--retry-wait-seconds 120 \
--retry-timeout-increase 0 \
${{ inputs.continue_on_error && '--continue-on-error' || '' }}
stage-c-test-large-8-gpu-amd:
needs: [check-changes, call-gate, stage-b-test-1-gpu-small-amd, stage-b-test-2-gpu-large-amd]
if: |
@@ -997,6 +1054,7 @@ jobs:
stage-b-test-1-gpu-large-amd,
stage-b-test-2-gpu-large-amd,
stage-b-test-large-8-gpu-35x-disaggregation-amd,
stage-c-test-4-gpu-amd,
stage-c-test-large-8-gpu-amd,
stage-c-test-large-8-gpu-amd-mi35x,
]
@@ -28,7 +28,7 @@ pytestmark = pytest.mark.filterwarnings(
)
from sglang.srt.utils import kill_process_tree
from sglang.test.ci.ci_register import register_cuda_ci
from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci
from sglang.test.test_utils import (
DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH,
DEFAULT_URL_FOR_TEST,
@@ -36,6 +36,12 @@ from sglang.test.test_utils import (
)
register_cuda_ci(est_time=300, suite="nightly-4-gpu", nightly=True)
register_amd_ci(
est_time=300,
suite="nightly-amd-4-gpu",
nightly=True,
disabled="TP=2 vs TP=4 numerical mismatch on AMD (comparator fails tolerance check)",
)
MODEL = "Qwen/Qwen3-30B-A3B"
BASELINE_TP = 2
@@ -5,7 +5,7 @@ import requests
from sglang.lang.chat_template import get_chat_template_by_model_path
from sglang.srt.utils import kill_process_tree
from sglang.test.ci.ci_register import register_cuda_ci
from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci
from sglang.test.few_shot_gsm8k import run_eval as run_eval_few_shot_gsm8k
from sglang.test.kits.ebnf_constrained_kit import EBNFConstrainedMixin
from sglang.test.kits.json_constrained_kit import JSONConstrainedMixin
@@ -24,8 +24,13 @@ from sglang.test.test_utils import (
)
register_cuda_ci(est_time=350, suite="stage-c-test-4-gpu-h100")
register_amd_ci(est_time=350, suite="stage-c-test-4-gpu-amd")
@unittest.skipIf(
is_in_amd_ci(),
"DeepSeek MLA forward_mla NameError on AMD (batched_gemm not defined)",
)
class TestDPAttentionDP2TP4(
CustomTestCase,
JSONConstrainedMixin,
@@ -66,6 +71,10 @@ class TestDPAttentionDP2TP4(
self.assertGreater(metrics["score"], 0.8)
@unittest.skipIf(
is_in_amd_ci(),
"DeepSeek MTP forward_mla NameError on AMD + needs 8 GPUs",
)
class TestDPAttentionDP2TP2DeepseekV3MTP(
CustomTestCase,
JSONConstrainedMixin,
@@ -131,6 +140,10 @@ class TestDPAttentionDP2TP2DeepseekV3MTP(
self.assertGreater(avg_spec_accept_length, 2.5)
@unittest.skipIf(
is_in_amd_ci(),
"Qwen3-VL-30B-A3B-Instruct OOMs at TP=4 DP=2 on MI325 4-GPU runners",
)
class TestDPAttentionDP2TP4VLM(CustomTestCase):
@classmethod
def setUpClass(cls):
@@ -15,7 +15,7 @@ import requests
from sglang.bench_one_batch_server import BenchArgs as OneBatchBenchArgs
from sglang.srt.server_args import ServerArgs
from sglang.srt.utils import kill_process_tree
from sglang.test.ci.ci_register import register_cuda_ci
from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci
from sglang.test.few_shot_gsm8k import run_eval as run_eval_few_shot_gsm8k
from sglang.test.run_eval import run_eval
from sglang.test.test_utils import (
@@ -26,12 +26,14 @@ from sglang.test.test_utils import (
DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH,
DEFAULT_URL_FOR_TEST,
CustomTestCase,
is_in_amd_ci,
is_in_ci,
popen_launch_server,
run_bench_one_batch_server,
)
register_cuda_ci(est_time=650, suite="stage-c-test-4-gpu-h100")
register_amd_ci(est_time=650, suite="stage-c-test-4-gpu-amd")
class TestPPAccuracy(unittest.TestCase):
@@ -69,7 +71,11 @@ class TestPPAccuracy(unittest.TestCase):
metrics = run_eval_few_shot_gsm8k(args)
print(f"{metrics=}")
self.assertGreater(metrics["accuracy"], 0.74)
if is_in_amd_ci():
# AMD triton backend produces slightly lower accuracy than FA3 on NVIDIA
self.assertGreater(metrics["accuracy"], 0.70)
else:
self.assertGreater(metrics["accuracy"], 0.74)
# Wait a little bit so that the memory check happens.
time.sleep(4)
@@ -97,6 +103,7 @@ class TestPPAccuracy(unittest.TestCase):
assert len(output_top_logprobs) == 16
@unittest.skipIf(is_in_amd_ci(), "MLA model with DP attention not yet supported on AMD")
class TestDPAttentionDP2PP2(CustomTestCase):
@classmethod
def setUpClass(cls):
@@ -136,6 +143,10 @@ class TestDPAttentionDP2PP2(CustomTestCase):
self.assertGreater(metrics["score"], 0.8)
@unittest.skipIf(
is_in_amd_ci(),
"VLM PP accuracy too low on AMD (0.48-0.50 with both aiter and triton)",
)
class TestQwenVLPPAccuracy(unittest.TestCase):
@classmethod
def setUpClass(cls):
@@ -244,6 +255,7 @@ class TestQwenPPAccuracy(unittest.TestCase):
)
@unittest.skipIf(is_in_amd_ci(), "PP consistency too flaky on AMD 4-GPU runners")
class TestQwenPPTieWeightsAccuracy(unittest.TestCase):
@classmethod
def setUpClass(cls):
@@ -350,6 +362,7 @@ class TestQwenMoePPAccuracy(unittest.TestCase):
)
@unittest.skipIf(is_in_amd_ci(), "PP consistency too flaky on AMD 4-GPU runners")
class TestQwen35PPAccuracy(unittest.TestCase):
@classmethod
def setUpClass(cls):
@@ -12,7 +12,7 @@ from torch.distributed.device_mesh import init_device_mesh
from transformers import AutoModelForCausalLM
from sglang.srt.entrypoints.engine import Engine as SglangEngine
from sglang.test.ci.ci_register import register_cuda_ci
from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci
from sglang.test.test_utils import (
DEFAULT_SMALL_MODEL_NAME_FOR_TEST,
DEFAULT_SMALL_MODEL_NAME_FOR_TEST_BASE,
@@ -21,6 +21,11 @@ from sglang.test.test_utils import (
)
register_cuda_ci(est_time=64, suite="stage-c-test-4-gpu-h100")
register_amd_ci(
est_time=64,
suite="stage-c-test-4-gpu-amd",
disabled="torch_memory_saver incompatible with ROCm (libcuda.so.1 not found)",
)
TEST_SUITE = dict(
model_path=DEFAULT_SMALL_MODEL_NAME_FOR_TEST,
@@ -12,7 +12,7 @@ from sglang.srt.layers.moe.routed_experts_capturer import (
extract_routed_experts_from_meta_info,
)
from sglang.srt.utils import kill_process_tree
from sglang.test.ci.ci_register import register_cuda_ci
from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci
from sglang.test.test_utils import (
DEFAULT_ENABLE_ROUTED_EXPERTS_MODEL_NAME_FOR_TEST,
DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH,
@@ -22,6 +22,11 @@ from sglang.test.test_utils import (
)
register_cuda_ci(est_time=360, suite="stage-c-test-4-gpu-h100")
register_amd_ci(
est_time=360,
suite="stage-c-test-4-gpu-amd",
disabled="TP=4 DP=4 routed expert mismatch >15% on AMD; needs TP/DP tuning + concurrency reduction",
)
SHAREGPT_URL = (
"https://huggingface.co/datasets/anon8231489123/"
@@ -4,7 +4,7 @@ from types import SimpleNamespace
import requests
from sglang.srt.environ import envs
from sglang.test.ci.ci_register import register_cuda_ci
from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci
from sglang.test.few_shot_gsm8k import run_eval as run_eval_few_shot_gsm8k
from sglang.test.send_one import BenchArgs, send_one_prompt
from sglang.test.test_utils import (
@@ -22,6 +22,7 @@ from sglang.test.test_utils import (
# EAGLE3 with DP attention (tp=2, dp=2, requires 4 GPUs)
register_cuda_ci(est_time=200, suite="stage-c-test-4-gpu-h100")
register_amd_ci(est_time=200, suite="stage-c-test-4-gpu-amd")
class TestEAGLE3EngineDPAttention(CustomTestCase):
@@ -50,7 +51,7 @@ class TestEAGLE3EngineDPAttention(CustomTestCase):
"--moe-dense-tp-size",
"1",
"--attention-backend",
"fa3",
"triton" if is_in_amd_ci() else "fa3",
"--mem-fraction-static",
"0.75",
"--cuda-graph-max-bs",
@@ -106,9 +107,17 @@ class TestEAGLE3EngineDPAttention(CustomTestCase):
f'{metrics["accuracy"]=:.3f}\n'
f"{avg_spec_accept_length=:.2f}\n"
)
self.assertGreater(metrics["accuracy"], 0.91)
if is_in_amd_ci():
# AMD triton backend produces slightly lower accuracy than FA3 on NVIDIA
self.assertGreater(metrics["accuracy"], 0.88)
else:
self.assertGreater(metrics["accuracy"], 0.91)
if avg_spec_accept_length is not None:
self.assertGreater(avg_spec_accept_length, 2.5)
if is_in_amd_ci():
# AMD triton backend produces slightly lower accept length than FA3 on NVIDIA
self.assertGreater(avg_spec_accept_length, 2.0)
else:
self.assertGreater(avg_spec_accept_length, 2.5)
def test_bs_1_speed(self):
"""Test batch size 1 speed with EAGLE3 DP Attention"""
+5 -3
View File
@@ -8,17 +8,19 @@ import unittest
from types import SimpleNamespace
from sglang.srt.utils import kill_process_tree
from sglang.test.ci.ci_register import register_cuda_ci
from sglang.test.ci.ci_register import register_amd_ci, register_cuda_ci
from sglang.test.kits.mmmu_vlm_kit import _run_lmms_eval_with_retry
from sglang.test.test_utils import (
DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH,
DEFAULT_URL_FOR_TEST,
CustomTestCase,
is_in_amd_ci,
is_in_ci,
popen_launch_server,
)
register_cuda_ci(est_time=500, suite="nightly-4-gpu", nightly=True)
register_amd_ci(est_time=500, suite="nightly-amd-4-gpu", nightly=True)
MODELS = [
SimpleNamespace(model="Qwen/Qwen2.5-VL-72B-Instruct", mmmu_accuracy=0.55),
@@ -127,8 +129,8 @@ class TestVLMEncoderDP(CustomTestCase):
process_env = os.environ.copy()
if custom_env:
process_env.update(custom_env)
# if test vlm with cuda_ipc feature, open this env_var
process_env["SGLANG_USE_CUDA_IPC_TRANSPORT"] = "1"
if not is_in_amd_ci():
process_env["SGLANG_USE_CUDA_IPC_TRANSPORT"] = "1"
# Prepare stdout/stderr redirection if needed
stdout_file = None
+2
View File
@@ -32,6 +32,7 @@ PER_COMMIT_SUITES = {
"stage-b-test-large-8-gpu-35x-disaggregation-amd",
"stage-b-test-1-gpu-large-amd",
"stage-b-test-2-gpu-large-amd",
"stage-c-test-4-gpu-amd",
"stage-c-test-large-8-gpu-amd",
"stage-c-test-large-8-gpu-amd-mi35x",
],
@@ -87,6 +88,7 @@ NIGHTLY_SUITES = {
"nightly-amd-1-gpu",
"nightly-amd-1-gpu-mi35x",
"nightly-amd-1-gpu-zimage-turbo",
"nightly-amd-4-gpu",
"nightly-amd-8-gpu",
"nightly-amd-vlm",
# MI35x 8-GPU suite (different model configs)