fix: adding matrix partitioning for h200 and b200 nightly tests (#17091)
This commit is contained in:
@@ -98,6 +98,10 @@ jobs:
|
|||||||
nightly-test-general-8-gpu-h200:
|
nightly-test-general-8-gpu-h200:
|
||||||
if: github.repository == 'sgl-project/sglang' && (inputs.job_filter == '' || inputs.job_filter == 'all' || inputs.job_filter == 'nightly-test-general-8-gpu-h200')
|
if: github.repository == 'sgl-project/sglang' && (inputs.job_filter == '' || inputs.job_filter == 'all' || inputs.job_filter == 'nightly-test-general-8-gpu-h200')
|
||||||
runs-on: 8-gpu-h200
|
runs-on: 8-gpu-h200
|
||||||
|
strategy:
|
||||||
|
fail-fast: false
|
||||||
|
matrix:
|
||||||
|
partition: [0, 1, 2]
|
||||||
env:
|
env:
|
||||||
RUNNER_LABELS: 8-gpu-h200
|
RUNNER_LABELS: 8-gpu-h200
|
||||||
steps:
|
steps:
|
||||||
@@ -120,7 +124,7 @@ jobs:
|
|||||||
IS_H200: "1"
|
IS_H200: "1"
|
||||||
run: |
|
run: |
|
||||||
cd test
|
cd test
|
||||||
python3 run_suite.py --hw cuda --suite nightly-8-gpu-common --nightly --timeout-per-file=18000 --continue-on-error
|
python3 run_suite.py --hw cuda --suite nightly-8-gpu-common --nightly --timeout-per-file=18000 --continue-on-error --auto-partition-id=${{ matrix.partition }} --auto-partition-size=3
|
||||||
|
|
||||||
- name: Run test
|
- name: Run test
|
||||||
timeout-minutes: 30
|
timeout-minutes: 30
|
||||||
@@ -156,8 +160,12 @@ jobs:
|
|||||||
|
|
||||||
# General tests - 8 GPU B200
|
# General tests - 8 GPU B200
|
||||||
nightly-test-general-8-gpu-b200:
|
nightly-test-general-8-gpu-b200:
|
||||||
if: github.repository == 'sgl-project/sglang' && (inputs.job_filter == '' || inputs.job_filter == 'all' || inputs.job_filter == 'nightly-test-general-8-gpu-h20')
|
if: github.repository == 'sgl-project/sglang' && (inputs.job_filter == '' || inputs.job_filter == 'all' || inputs.job_filter == 'nightly-test-general-8-gpu-b200')
|
||||||
runs-on: 8-gpu-b200
|
runs-on: 8-gpu-b200
|
||||||
|
strategy:
|
||||||
|
fail-fast: false
|
||||||
|
matrix:
|
||||||
|
partition: [0, 1, 2]
|
||||||
steps:
|
steps:
|
||||||
- name: Checkout code
|
- name: Checkout code
|
||||||
uses: actions/checkout@v4
|
uses: actions/checkout@v4
|
||||||
@@ -177,7 +185,7 @@ jobs:
|
|||||||
GPU_CONFIG: "8-gpu-b200"
|
GPU_CONFIG: "8-gpu-b200"
|
||||||
run: |
|
run: |
|
||||||
cd test
|
cd test
|
||||||
IS_BLACKWELL=1 python3 run_suite.py --hw cuda --suite nightly-8-gpu-common --nightly --timeout-per-file=12000 --continue-on-error
|
IS_BLACKWELL=1 python3 run_suite.py --hw cuda --suite nightly-8-gpu-common --nightly --timeout-per-file=12000 --continue-on-error --auto-partition-id=${{ matrix.partition }} --auto-partition-size=3
|
||||||
|
|
||||||
# Text model accuracy tests
|
# Text model accuracy tests
|
||||||
nightly-test-text-accuracy-2-gpu-runner:
|
nightly-test-text-accuracy-2-gpu-runner:
|
||||||
|
|||||||
@@ -7,7 +7,7 @@ from sglang.test.run_combined_tests import run_combined_tests
|
|||||||
from sglang.test.test_utils import ModelLaunchSettings
|
from sglang.test.test_utils import ModelLaunchSettings
|
||||||
|
|
||||||
# Runs on both H200 and B200 via nightly-8-gpu-common suite
|
# Runs on both H200 and B200 via nightly-8-gpu-common suite
|
||||||
register_cuda_ci(est_time=12000, suite="nightly-8-gpu-common", nightly=True)
|
register_cuda_ci(est_time=5400, suite="nightly-8-gpu-common", nightly=True)
|
||||||
|
|
||||||
DEEPSEEK_V31_MODEL_PATH = "deepseek-ai/DeepSeek-V3.1"
|
DEEPSEEK_V31_MODEL_PATH = "deepseek-ai/DeepSeek-V3.1"
|
||||||
|
|
||||||
|
|||||||
@@ -6,7 +6,7 @@ from sglang.test.performance_test_runner import PerformanceTestParams
|
|||||||
from sglang.test.run_combined_tests import run_combined_tests
|
from sglang.test.run_combined_tests import run_combined_tests
|
||||||
from sglang.test.test_utils import ModelLaunchSettings, is_blackwell_system
|
from sglang.test.test_utils import ModelLaunchSettings, is_blackwell_system
|
||||||
|
|
||||||
register_cuda_ci(est_time=18000, suite="nightly-8-gpu-common", nightly=True)
|
register_cuda_ci(est_time=5400, suite="nightly-8-gpu-common", nightly=True)
|
||||||
|
|
||||||
DEEPSEEK_V32_MODEL_PATH = "deepseek-ai/DeepSeek-V3.2"
|
DEEPSEEK_V32_MODEL_PATH = "deepseek-ai/DeepSeek-V3.2"
|
||||||
|
|
||||||
|
|||||||
@@ -5,7 +5,7 @@ from sglang.test.ci.ci_register import register_cuda_ci
|
|||||||
from sglang.test.run_combined_tests import run_combined_tests
|
from sglang.test.run_combined_tests import run_combined_tests
|
||||||
from sglang.test.test_utils import ModelLaunchSettings, is_blackwell_system
|
from sglang.test.test_utils import ModelLaunchSettings, is_blackwell_system
|
||||||
|
|
||||||
register_cuda_ci(est_time=18000, suite="nightly-8-gpu-common", nightly=True)
|
register_cuda_ci(est_time=5400, suite="nightly-8-gpu-common", nightly=True)
|
||||||
|
|
||||||
DEEPSEEK_V32_EXP_MODEL_PATH = "deepseek-ai/DeepSeek-V3.2-Exp"
|
DEEPSEEK_V32_EXP_MODEL_PATH = "deepseek-ai/DeepSeek-V3.2-Exp"
|
||||||
|
|
||||||
|
|||||||
@@ -7,7 +7,7 @@ from sglang.test.run_combined_tests import run_combined_tests
|
|||||||
from sglang.test.test_utils import ModelLaunchSettings
|
from sglang.test.test_utils import ModelLaunchSettings
|
||||||
|
|
||||||
# Runs on both H200 and B200 via nightly-8-gpu-common suite
|
# Runs on both H200 and B200 via nightly-8-gpu-common suite
|
||||||
register_cuda_ci(est_time=12000, suite="nightly-8-gpu-common", nightly=True)
|
register_cuda_ci(est_time=1800, suite="nightly-8-gpu-common", nightly=True)
|
||||||
|
|
||||||
GLM_4_6_MODEL_PATH = "zai-org/GLM-4.6"
|
GLM_4_6_MODEL_PATH = "zai-org/GLM-4.6"
|
||||||
|
|
||||||
|
|||||||
@@ -7,7 +7,7 @@ from sglang.test.run_combined_tests import run_combined_tests
|
|||||||
from sglang.test.test_utils import ModelLaunchSettings
|
from sglang.test.test_utils import ModelLaunchSettings
|
||||||
|
|
||||||
# Runs on both H200 and B200 via nightly-8-gpu-common suite
|
# Runs on both H200 and B200 via nightly-8-gpu-common suite
|
||||||
register_cuda_ci(est_time=12000, suite="nightly-8-gpu-common", nightly=True)
|
register_cuda_ci(est_time=1800, suite="nightly-8-gpu-common", nightly=True)
|
||||||
|
|
||||||
GLM_4_6_FP8_MODEL_PATH = "zai-org/GLM-4.6-FP8"
|
GLM_4_6_FP8_MODEL_PATH = "zai-org/GLM-4.6-FP8"
|
||||||
|
|
||||||
|
|||||||
@@ -7,7 +7,7 @@ from sglang.test.run_combined_tests import run_combined_tests
|
|||||||
from sglang.test.test_utils import ModelLaunchSettings
|
from sglang.test.test_utils import ModelLaunchSettings
|
||||||
|
|
||||||
# Runs on both H200 and B200 via nightly-8-gpu-common suite
|
# Runs on both H200 and B200 via nightly-8-gpu-common suite
|
||||||
register_cuda_ci(est_time=12000, suite="nightly-8-gpu-common", nightly=True)
|
register_cuda_ci(est_time=1800, suite="nightly-8-gpu-common", nightly=True)
|
||||||
|
|
||||||
KIMI_K2_THINKING_MODEL_PATH = "moonshotai/Kimi-K2-Thinking"
|
KIMI_K2_THINKING_MODEL_PATH = "moonshotai/Kimi-K2-Thinking"
|
||||||
|
|
||||||
|
|||||||
@@ -7,7 +7,7 @@ from sglang.test.run_combined_tests import run_combined_tests
|
|||||||
from sglang.test.test_utils import ModelLaunchSettings
|
from sglang.test.test_utils import ModelLaunchSettings
|
||||||
|
|
||||||
# Runs on both H200 and B200 via nightly-8-gpu-common suite
|
# Runs on both H200 and B200 via nightly-8-gpu-common suite
|
||||||
register_cuda_ci(est_time=12000, suite="nightly-8-gpu-common", nightly=True)
|
register_cuda_ci(est_time=1800, suite="nightly-8-gpu-common", nightly=True)
|
||||||
|
|
||||||
LLAMA4_MODEL_PATH = "meta-llama/Llama-4-Scout-17B-16E-Instruct"
|
LLAMA4_MODEL_PATH = "meta-llama/Llama-4-Scout-17B-16E-Instruct"
|
||||||
|
|
||||||
|
|||||||
@@ -7,7 +7,7 @@ from sglang.test.run_combined_tests import run_combined_tests
|
|||||||
from sglang.test.test_utils import ModelLaunchSettings
|
from sglang.test.test_utils import ModelLaunchSettings
|
||||||
|
|
||||||
# Runs on both H200 and B200 via nightly-8-gpu-common suite
|
# Runs on both H200 and B200 via nightly-8-gpu-common suite
|
||||||
register_cuda_ci(est_time=12000, suite="nightly-8-gpu-common", nightly=True)
|
register_cuda_ci(est_time=1800, suite="nightly-8-gpu-common", nightly=True)
|
||||||
|
|
||||||
MINIMAX_M2_MODEL_PATH = "MiniMaxAI/MiniMax-M2"
|
MINIMAX_M2_MODEL_PATH = "MiniMaxAI/MiniMax-M2"
|
||||||
|
|
||||||
|
|||||||
@@ -9,7 +9,7 @@ from sglang.test.test_utils import ModelLaunchSettings, is_blackwell_system
|
|||||||
|
|
||||||
# Runs on both H200 and B200 via nightly-8-gpu-common suite
|
# Runs on both H200 and B200 via nightly-8-gpu-common suite
|
||||||
# Note: trtllm_mla backend may have hardware-specific behavior
|
# Note: trtllm_mla backend may have hardware-specific behavior
|
||||||
register_cuda_ci(est_time=12000, suite="nightly-8-gpu-common", nightly=True)
|
register_cuda_ci(est_time=1800, suite="nightly-8-gpu-common", nightly=True)
|
||||||
|
|
||||||
MISTRAL_LARGE3_MODEL_PATH = "mistralai/Mistral-Large-3-675B-Instruct-2512"
|
MISTRAL_LARGE3_MODEL_PATH = "mistralai/Mistral-Large-3-675B-Instruct-2512"
|
||||||
MISTRAL_LARGE3_EAGLE_MODEL_PATH = "mistralai/Mistral-Large-3-675B-Instruct-2512-Eagle"
|
MISTRAL_LARGE3_EAGLE_MODEL_PATH = "mistralai/Mistral-Large-3-675B-Instruct-2512-Eagle"
|
||||||
|
|||||||
@@ -7,7 +7,7 @@ from sglang.test.run_combined_tests import run_combined_tests
|
|||||||
from sglang.test.test_utils import ModelLaunchSettings, is_blackwell_system
|
from sglang.test.test_utils import ModelLaunchSettings, is_blackwell_system
|
||||||
|
|
||||||
# Runs on both H200 and B200 via nightly-8-gpu-common suite
|
# Runs on both H200 and B200 via nightly-8-gpu-common suite
|
||||||
register_cuda_ci(est_time=12000, suite="nightly-8-gpu-common", nightly=True)
|
register_cuda_ci(est_time=1800, suite="nightly-8-gpu-common", nightly=True)
|
||||||
|
|
||||||
QWEN3_235B_MODEL_PATH = "Qwen/Qwen3-235B-A22B-Instruct-2507"
|
QWEN3_235B_MODEL_PATH = "Qwen/Qwen3-235B-A22B-Instruct-2507"
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user