[CI] Update B200 est_times to prevent timeouts on slower machine (#22609)
Co-authored-by: Alison Shao <alison.shao@MacBook-Pro-D2W773R9CD.local>
This commit is contained in:
co-authored by
Alison Shao
parent
f21d23a211
commit
d6c9d9116b
@@ -4,7 +4,7 @@ from sglang.test.ci.ci_register import register_cuda_ci
|
|||||||
from sglang.test.gpt_oss_common import BaseTestGptOss
|
from sglang.test.gpt_oss_common import BaseTestGptOss
|
||||||
|
|
||||||
register_cuda_ci(est_time=328, suite="stage-c-test-4-gpu-h100")
|
register_cuda_ci(est_time=328, suite="stage-c-test-4-gpu-h100")
|
||||||
register_cuda_ci(est_time=312, suite="stage-c-test-4-gpu-b200")
|
register_cuda_ci(est_time=740, suite="stage-c-test-4-gpu-b200")
|
||||||
|
|
||||||
|
|
||||||
class TestGptOss4Gpu(BaseTestGptOss):
|
class TestGptOss4Gpu(BaseTestGptOss):
|
||||||
|
|||||||
@@ -11,7 +11,7 @@ from sglang.test.test_utils import (
|
|||||||
popen_launch_server,
|
popen_launch_server,
|
||||||
)
|
)
|
||||||
|
|
||||||
register_cuda_ci(est_time=294, suite="stage-c-test-4-gpu-b200")
|
register_cuda_ci(est_time=710, suite="stage-c-test-4-gpu-b200")
|
||||||
|
|
||||||
NEMOTRON_3_SUPER_NVFP4_MODEL = "nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-NVFP4"
|
NEMOTRON_3_SUPER_NVFP4_MODEL = "nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-NVFP4"
|
||||||
|
|
||||||
|
|||||||
@@ -13,7 +13,7 @@ from sglang.test.test_utils import (
|
|||||||
)
|
)
|
||||||
|
|
||||||
# FlashAttention4 integration test (requires SM 100+ / Blackwell B200)
|
# FlashAttention4 integration test (requires SM 100+ / Blackwell B200)
|
||||||
register_cuda_ci(est_time=259, suite="stage-b-test-4-gpu-b200")
|
register_cuda_ci(est_time=332, suite="stage-b-test-4-gpu-b200")
|
||||||
|
|
||||||
|
|
||||||
@unittest.skipIf(get_device_sm() < 100, "Test requires CUDA SM 100 or higher")
|
@unittest.skipIf(get_device_sm() < 100, "Test requires CUDA SM 100 or higher")
|
||||||
|
|||||||
@@ -16,7 +16,7 @@ except ImportError:
|
|||||||
CuteDslMoEWrapper = None
|
CuteDslMoEWrapper = None
|
||||||
convert_sf_to_mma_layout = None
|
convert_sf_to_mma_layout = None
|
||||||
|
|
||||||
register_cuda_ci(est_time=13, suite="stage-c-test-4-gpu-b200")
|
register_cuda_ci(est_time=590, suite="stage-c-test-4-gpu-b200")
|
||||||
|
|
||||||
SKIP_TEST = torch.cuda.get_device_capability() < (10, 0)
|
SKIP_TEST = torch.cuda.get_device_capability() < (10, 0)
|
||||||
SKIP_REASON = "Nvfp4 Requires compute capability of 10 or above."
|
SKIP_REASON = "Nvfp4 Requires compute capability of 10 or above."
|
||||||
|
|||||||
@@ -14,7 +14,7 @@ from sglang.test.test_utils import (
|
|||||||
write_github_step_summary,
|
write_github_step_summary,
|
||||||
)
|
)
|
||||||
|
|
||||||
register_cuda_ci(est_time=1146, suite="stage-c-test-4-gpu-b200")
|
register_cuda_ci(est_time=1380, suite="stage-c-test-4-gpu-b200")
|
||||||
|
|
||||||
FULL_DEEPSEEK_V3_FP4_MODEL_PATH = "nvidia/DeepSeek-V3-0324-FP4"
|
FULL_DEEPSEEK_V3_FP4_MODEL_PATH = "nvidia/DeepSeek-V3-0324-FP4"
|
||||||
SERVER_LAUNCH_TIMEOUT = 1200
|
SERVER_LAUNCH_TIMEOUT = 1200
|
||||||
|
|||||||
@@ -12,7 +12,7 @@ from sglang.test.test_utils import (
|
|||||||
try_cached_model,
|
try_cached_model,
|
||||||
)
|
)
|
||||||
|
|
||||||
register_cuda_ci(est_time=302, suite="stage-c-test-4-gpu-b200")
|
register_cuda_ci(est_time=630, suite="stage-c-test-4-gpu-b200")
|
||||||
|
|
||||||
MODEL_PATH = "Qwen/Qwen3-4B-Instruct-2507-FP8"
|
MODEL_PATH = "Qwen/Qwen3-4B-Instruct-2507-FP8"
|
||||||
MXFP8_MODEL_PATH = "zianglih/Qwen3-4B-Instruct-2507-MXFP8"
|
MXFP8_MODEL_PATH = "zianglih/Qwen3-4B-Instruct-2507-MXFP8"
|
||||||
|
|||||||
@@ -12,7 +12,7 @@ from sglang.test.test_utils import (
|
|||||||
try_cached_model,
|
try_cached_model,
|
||||||
)
|
)
|
||||||
|
|
||||||
register_cuda_ci(est_time=322, suite="stage-c-test-4-gpu-b200")
|
register_cuda_ci(est_time=550, suite="stage-c-test-4-gpu-b200")
|
||||||
|
|
||||||
MODEL_PATH = "nvidia/Llama-3.1-8B-Instruct-NVFP4"
|
MODEL_PATH = "nvidia/Llama-3.1-8B-Instruct-NVFP4"
|
||||||
|
|
||||||
|
|||||||
@@ -16,7 +16,7 @@ from sglang.test.test_utils import (
|
|||||||
write_github_step_summary,
|
write_github_step_summary,
|
||||||
)
|
)
|
||||||
|
|
||||||
register_cuda_ci(est_time=416, suite="stage-b-test-4-gpu-b200")
|
register_cuda_ci(est_time=510, suite="stage-b-test-4-gpu-b200")
|
||||||
|
|
||||||
FULL_DEEPSEEK_V3_FP4_MODEL_PATH = "nvidia/DeepSeek-V3-0324-FP4"
|
FULL_DEEPSEEK_V3_FP4_MODEL_PATH = "nvidia/DeepSeek-V3-0324-FP4"
|
||||||
SERVER_LAUNCH_TIMEOUT = 1200
|
SERVER_LAUNCH_TIMEOUT = 1200
|
||||||
|
|||||||
@@ -17,7 +17,7 @@ from sglang.test.test_utils import (
|
|||||||
)
|
)
|
||||||
|
|
||||||
# EAGLE with DP attention on B200 (tp=2, dp=2, requires 4 B200 GPUs)
|
# EAGLE with DP attention on B200 (tp=2, dp=2, requires 4 B200 GPUs)
|
||||||
register_cuda_ci(est_time=68, suite="stage-c-test-4-gpu-b200")
|
register_cuda_ci(est_time=136, suite="stage-c-test-4-gpu-b200")
|
||||||
|
|
||||||
|
|
||||||
def test_gsm8k(base_url: str, model: str):
|
def test_gsm8k(base_url: str, model: str):
|
||||||
|
|||||||
Reference in New Issue
Block a user