ci: add 4-GPU mi35x runner and rebalance off the saturated 8-GPU pool (#28745)

Co-authored-by: michaelzhang-ai <michaelzhang-ai@users.noreply.github.com>
This commit is contained in:
Michael
2026-06-19 15:41:19 -07:00
committed by GitHub
co-authored by michaelzhang-ai
parent 3a574846ff
commit 13aab2fc06
6 changed files with 109 additions and 14 deletions
@@ -16,7 +16,7 @@ from sglang.test.test_utils import (
write_github_step_summary,
)
register_amd_ci(est_time=3600, suite="stage-c-test-large-8-gpu-amd-mi35x")
register_amd_ci(est_time=3600, suite="stage-c-test-4-gpu-amd-mi35x")
DEEPSEEK_R1_MODEL_PATH = "amd/DeepSeek-R1-MXFP4-Preview"
SERVER_LAUNCH_TIMEOUT = 1800
@@ -30,12 +30,12 @@ class TestDeepseekR1MXFP4(CustomTestCase):
# Workaround: AITER custom all-gather corrupts CUDA-graph IPC buffer
# registration and triggers a decode-time "Memory access fault" on
# MI35x TP=8. Disable until the AITER-side fix lands (see PR body).
# MI35x. Disable until the AITER-side fix lands (see PR body).
envs.SGLANG_USE_AITER_AG.set(False)
other_args = [
"--tp",
"8",
"4",
"--chunked-prefill-size",
"131072",
"--model-loader-extra-config",
@@ -90,7 +90,8 @@ class TestDeepseekR1MXFP4(CustomTestCase):
write_github_step_summary(
f"### test_bs_1_speed (deepseek-r1-mxfp4)\n" f"{speed=:.2f} token/s\n"
)
self.assertGreater(speed, 75)
# Report-only: the previous >75 tok/s gate was calibrated for TP=8.
# Decode throughput at TP=4 differs; re-calibrate before re-enabling.
class TestDeepseekR1MXFP4MTP(CustomTestCase):
@@ -105,7 +106,7 @@ class TestDeepseekR1MXFP4MTP(CustomTestCase):
other_args = [
"--tp",
"8",
"4",
"--chunked-prefill-size",
"131072",
"--speculative-algorithm",
@@ -175,7 +176,8 @@ class TestDeepseekR1MXFP4MTP(CustomTestCase):
f"{speed=:.2f} token/s\n"
)
self.assertGreater(acc_length, 2.04)
self.assertGreater(speed, 150)
# Report-only: the previous >150 tok/s gate was calibrated for TP=8.
# Decode throughput at TP=4 differs; re-calibrate before re-enabling.
if __name__ == "__main__":
+1
View File
@@ -41,6 +41,7 @@ PER_COMMIT_SUITES = {
"jit-kernel-unit-test-amd",
"sgl-kernel-unit-test-2-gpu-amd",
"stage-c-test-4-gpu-amd",
"stage-c-test-4-gpu-amd-mi35x",
"stage-c-test-large-8-gpu-amd",
"stage-c-test-large-8-gpu-amd-mi35x",
# extra-a: label-gated PR opt-in suites in pr-test-amd-extra.yml