diff --git a/python/sglang/test/test_utils.py b/python/sglang/test/test_utils.py index 5ebbca864..1c65776bc 100644 --- a/python/sglang/test/test_utils.py +++ b/python/sglang/test/test_utils.py @@ -2223,12 +2223,6 @@ class ModelLaunchSettings: self.extra_args.append(fixed_arg) -class ModelEvalMetrics: - def __init__(self, accuracy: float, eval_time: float): - self.accuracy = accuracy - self.eval_time = eval_time - - def extract_trace_link_from_bench_one_batch_server_output(output: str) -> str: match = re.search(r"\[Profile\]\((.*?)\)", output) if match: diff --git a/test/registered/8-gpu-models/test_llama4.py b/test/manual/8-gpu-models/test_llama4.py similarity index 80% rename from test/registered/8-gpu-models/test_llama4.py rename to test/manual/8-gpu-models/test_llama4.py index b76ca9604..beda94dc8 100644 --- a/test/registered/8-gpu-models/test_llama4.py +++ b/test/manual/8-gpu-models/test_llama4.py @@ -1,19 +1,22 @@ +"""Moved out of test/registered/8-gpu-models/. + +Originally registered with `register_cuda_ci(...)` on the nightly 8-gpu-h200 and +8-gpu-b200 suites. Moved here because nobody serves Llama 4 any more, and the CI +HF account has no access to meta-llama/Llama-4-Scout-17B-16E-Instruct either, so +it had been skipping for a while. Run with +`python3 test/manual/8-gpu-models/test_llama4.py`. +""" + import unittest from sglang.test.accuracy_test_runner import AccuracyTestParams -from sglang.test.ci.ci_register import register_cuda_ci from sglang.test.performance_test_runner import PerformanceTestParams from sglang.test.run_combined_tests import run_combined_tests from sglang.test.test_utils import ModelLaunchSettings -# Runs on both H200 and B200: registered once per runner_config below -register_cuda_ci(est_time=1800, stage="nightly", runner_config="8-gpu-h200") -register_cuda_ci(est_time=1800, stage="nightly", runner_config="8-gpu-b200") - LLAMA4_MODEL_PATH = "meta-llama/Llama-4-Scout-17B-16E-Instruct" -@unittest.skip("Blocked: Missing HF token permission for Llama 4 model") class TestLlama4(unittest.TestCase): """Unified test class for Llama-4-Scout performance and accuracy. diff --git a/test/manual/nightly/test_text_models_gsm8k_eval.py b/test/manual/nightly/test_text_models_gsm8k_eval.py deleted file mode 100644 index 8cd62e604..000000000 --- a/test/manual/nightly/test_text_models_gsm8k_eval.py +++ /dev/null @@ -1,124 +0,0 @@ -import json -import unittest -import warnings -from types import SimpleNamespace - -from sglang.srt.utils import kill_process_tree -from sglang.test.run_eval import run_eval -from sglang.test.test_utils import ( - DEFAULT_MODEL_NAME_FOR_NIGHTLY_EVAL_FP8_TP1, - DEFAULT_MODEL_NAME_FOR_NIGHTLY_EVAL_FP8_TP2, - DEFAULT_MODEL_NAME_FOR_NIGHTLY_EVAL_TP1, - DEFAULT_MODEL_NAME_FOR_NIGHTLY_EVAL_TP2, - DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH, - DEFAULT_URL_FOR_TEST, - ModelLaunchSettings, - check_evaluation_test_results, - parse_models, - popen_launch_server, - write_results_to_json, -) - -MODEL_SCORE_THRESHOLDS = { - "meta-llama/Llama-3.1-8B-Instruct": 0.82, - "mistralai/Mistral-7B-Instruct-v0.3": 0.58, - "deepseek-ai/DeepSeek-Coder-V2-Lite-Instruct": 0.85, - "google/gemma-2-27b-it": 0.91, - "meta-llama/Llama-3.1-70B-Instruct": 0.95, - "mistralai/Mixtral-8x7B-Instruct-v0.1": 0.616, - "Qwen/Qwen2-57B-A14B-Instruct": 0.86, - "neuralmagic/Meta-Llama-3.1-8B-Instruct-FP8": 0.83, - "neuralmagic/Mistral-7B-Instruct-v0.3-FP8": 0.54, - "neuralmagic/DeepSeek-Coder-V2-Lite-Instruct-FP8": 0.835, - "zai-org/GLM-4.5-Air-FP8": 0.75, - # The threshold of neuralmagic/gemma-2-2b-it-FP8 should be 0.6, but this model has some accuracy regression. - # The fix is tracked at https://github.com/sgl-project/sglang/issues/4324, we set it to 0.50, for now, to make CI green. - "neuralmagic/gemma-2-2b-it-FP8": 0.50, - "neuralmagic/Meta-Llama-3.1-70B-Instruct-FP8": 0.94, - "neuralmagic/Mixtral-8x7B-Instruct-v0.1-FP8": 0.65, - "neuralmagic/Qwen2-72B-Instruct-FP8": 0.94, - "neuralmagic/Qwen2-57B-A14B-Instruct-FP8": 0.82, -} - - -# Do not use `CustomTestCase` since `test_mgsm_en_all_models` does not want retry -class TestNightlyGsm8KEval(unittest.TestCase): - @classmethod - def setUpClass(cls): - cls.models = [] - models_tp1 = parse_models( - DEFAULT_MODEL_NAME_FOR_NIGHTLY_EVAL_TP1 - ) + parse_models(DEFAULT_MODEL_NAME_FOR_NIGHTLY_EVAL_FP8_TP1) - for model_path in models_tp1: - cls.models.append(ModelLaunchSettings(model_path, tp_size=1)) - - models_tp2 = parse_models( - DEFAULT_MODEL_NAME_FOR_NIGHTLY_EVAL_TP2 - ) + parse_models(DEFAULT_MODEL_NAME_FOR_NIGHTLY_EVAL_FP8_TP2) - for model_path in models_tp2: - cls.models.append(ModelLaunchSettings(model_path, tp_size=2)) - - cls.base_url = DEFAULT_URL_FOR_TEST - - def test_mgsm_en_all_models(self): - warnings.filterwarnings( - "ignore", category=ResourceWarning, message="unclosed.*socket" - ) - is_first = True - all_results = [] - for model_setup in self.models: - with self.subTest(model=model_setup.model_path): - other_args = list(model_setup.extra_args) - - if model_setup.model_path == "meta-llama/Llama-3.1-70B-Instruct": - other_args.extend(["--mem-fraction-static", "0.9"]) - - process = popen_launch_server( - model=model_setup.model_path, - other_args=other_args, - base_url=self.base_url, - timeout=DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH, - ) - - try: - args = SimpleNamespace( - base_url=self.base_url, - model=model_setup.model_path, - eval_name="mgsm_en", - num_examples=None, - num_threads=1024, - ) - - metrics = run_eval(args) - print( - f"{'=' * 42}\n{model_setup.model_path} - metrics={metrics} score={metrics['score']}\n{'=' * 42}\n" - ) - - write_results_to_json( - model_setup.model_path, metrics, "w" if is_first else "a" - ) - is_first = False - - # 0.0 for empty latency - all_results.append((model_setup.model_path, metrics["score"], 0.0)) - finally: - kill_process_tree(process.pid) - - try: - with open("results.json", "r") as f: - print("\nFinal Results from results.json:") - print(json.dumps(json.load(f), indent=2)) - except Exception as e: - print(f"Error reading results.json: {e}") - - # Check all scores after collecting all results - check_evaluation_test_results( - all_results, - self.__class__.__name__, - model_accuracy_thresholds=MODEL_SCORE_THRESHOLDS, - model_count=len(self.models), - ) - - -if __name__ == "__main__": - unittest.main() diff --git a/test/manual/nightly/test_text_models_perf.py b/test/manual/nightly/test_text_models_perf.py deleted file mode 100644 index f4161b838..000000000 --- a/test/manual/nightly/test_text_models_perf.py +++ /dev/null @@ -1,59 +0,0 @@ -import unittest - -from sglang.test.nightly_utils import NightlyBenchmarkRunner -from sglang.test.test_utils import ( - DEFAULT_URL_FOR_TEST, - ModelLaunchSettings, - _parse_int_list_env, - parse_models, -) - -RESULT_DIR = "performance_results_text_models" - - -class TestNightlyTextModelsPerformance(unittest.TestCase): - @classmethod - def setUpClass(cls): - cls.models = [] - # TODO: replace with DEFAULT_MODEL_NAME_FOR_NIGHTLY_EVAL_TP1 or other model lists - for model_path in parse_models("meta-llama/Llama-3.1-8B-Instruct"): - cls.models.append(ModelLaunchSettings(model_path, tp_size=1)) - for model_path in parse_models("Qwen/Qwen2-57B-A14B-Instruct"): - cls.models.append(ModelLaunchSettings(model_path, tp_size=2)) - # (parse_models(DEFAULT_MODEL_NAME_FOR_NIGHTLY_EVAL_TP1), False, False), - # (parse_models(DEFAULT_MODEL_NAME_FOR_NIGHTLY_EVAL_TP2), False, True), - # (parse_models(DEFAULT_MODEL_NAME_FOR_NIGHTLY_EVAL_FP8_TP1), True, False), - # (parse_models(DEFAULT_MODEL_NAME_FOR_NIGHTLY_EVAL_FP8_TP2), True, True), - cls.base_url = DEFAULT_URL_FOR_TEST - cls.batch_sizes = [1, 1, 8, 16, 64] - cls.input_lens = tuple(_parse_int_list_env("NIGHTLY_INPUT_LENS", "4096")) - cls.output_lens = tuple(_parse_int_list_env("NIGHTLY_OUTPUT_LENS", "512")) - cls.runner = NightlyBenchmarkRunner(RESULT_DIR, cls.__name__, cls.base_url) - cls.runner.setup_result_directory() - - def test_bench_one_batch(self): - all_model_succeed = True - - for model_setup in self.models: - with self.subTest(model=model_setup.model_path): - results, success = self.runner.run_benchmark_for_model( - model_path=model_setup.model_path, - batch_sizes=self.batch_sizes, - input_lens=self.input_lens, - output_lens=self.output_lens, - other_args=model_setup.extra_args, - ) - - if not success: - all_model_succeed = False - - self.runner.add_report(results) - - self.runner.write_final_report() - - if not all_model_succeed: - raise AssertionError("Some models failed the perf tests.") - - -if __name__ == "__main__": - unittest.main() diff --git a/test/manual/nightly/test_vlms_mmmu_eval.py b/test/manual/nightly/test_vlms_mmmu_eval.py deleted file mode 100644 index aa2b43bd1..000000000 --- a/test/manual/nightly/test_vlms_mmmu_eval.py +++ /dev/null @@ -1,127 +0,0 @@ -import json -import unittest -import warnings -from types import SimpleNamespace - -from sglang.srt.utils import kill_process_tree -from sglang.test.run_eval import run_eval -from sglang.test.test_utils import ( - DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH, - DEFAULT_URL_FOR_TEST, - ModelEvalMetrics, - ModelLaunchSettings, - check_evaluation_test_results, - popen_launch_server, - write_results_to_json, -) - -MODEL_THRESHOLDS = { - # Conservative thresholds on 100 MMMU samples, especially for latency thresholds - ModelLaunchSettings("deepseek-ai/deepseek-vl2-small"): ModelEvalMetrics( - 0.330, 56.1 - ), - ModelLaunchSettings("deepseek-ai/Janus-Pro-7B"): ModelEvalMetrics(0.285, 40.3), - ModelLaunchSettings("Efficient-Large-Model/NVILA-8B-hf"): ModelEvalMetrics( - 0.270, 56.7 - ), - ModelLaunchSettings("Efficient-Large-Model/NVILA-Lite-2B-hf"): ModelEvalMetrics( - 0.270, 23.8 - ), - ModelLaunchSettings("google/gemma-3-4b-it"): ModelEvalMetrics(0.360, 10.9), - ModelLaunchSettings("google/gemma-3n-E4B-it"): ModelEvalMetrics(0.360, 17.7), - ModelLaunchSettings("mistral-community/pixtral-12b"): ModelEvalMetrics(0.360, 16.6), - ModelLaunchSettings("moonshotai/Kimi-VL-A3B-Instruct"): ModelEvalMetrics( - 0.330, 22.3 - ), - ModelLaunchSettings("openbmb/MiniCPM-o-2_6"): ModelEvalMetrics(0.330, 29.3), - ModelLaunchSettings("openbmb/MiniCPM-v-2_6"): ModelEvalMetrics(0.259, 36.3), - ModelLaunchSettings("OpenGVLab/InternVL2_5-2B"): ModelEvalMetrics(0.300, 17.0), - ModelLaunchSettings("Qwen/Qwen2-VL-7B-Instruct"): ModelEvalMetrics(0.310, 83.3), - ModelLaunchSettings("Qwen/Qwen2.5-VL-7B-Instruct"): ModelEvalMetrics(0.340, 31.9), - ModelLaunchSettings( - "Qwen/Qwen3-VL-30B-A3B-Instruct", extra_args=["--tp=2"] - ): ModelEvalMetrics(0.29, 37.0), - ModelLaunchSettings( - "unsloth/Mistral-Small-3.1-24B-Instruct-2503" - ): ModelEvalMetrics(0.310, 16.7), - ModelLaunchSettings("XiaomiMiMo/MiMo-VL-7B-RL"): ModelEvalMetrics(0.28, 32.0), - ModelLaunchSettings("zai-org/GLM-4.1V-9B-Thinking"): ModelEvalMetrics(0.280, 30.4), -} - - -class TestNightlyVLMMmmuEval(unittest.TestCase): - @classmethod - def setUpClass(cls): - cls.models = list(MODEL_THRESHOLDS.keys()) - cls.base_url = DEFAULT_URL_FOR_TEST - - def test_mmmu_vlm_models(self): - warnings.filterwarnings( - "ignore", category=ResourceWarning, message="unclosed.*socket" - ) - is_first = True - all_results = [] - - for model in self.models: - model_path = model.model_path - with self.subTest(model=model_path): - process = popen_launch_server( - model=model_path, - base_url=self.base_url, - other_args=model.extra_args, - timeout=DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH, - ) - try: - args = SimpleNamespace( - base_url=self.base_url, - model=model_path, - eval_name="mmmu", - num_examples=100, - num_threads=64, - max_tokens=30, - ) - - args.return_latency = True - - metrics, latency = run_eval(args) - - metrics["score"] = round(metrics["score"], 4) - metrics["latency"] = round(latency, 4) - print( - f"{'=' * 42}\n{model_path} - metrics={metrics} score={metrics['score']}\n{'=' * 42}\n" - ) - - write_results_to_json(model_path, metrics, "w" if is_first else "a") - is_first = False - - all_results.append( - (model_path, metrics["score"], metrics["latency"]) - ) - finally: - kill_process_tree(process.pid) - - try: - with open("results.json", "r") as f: - print("\nFinal Results from results.json:") - print(json.dumps(json.load(f), indent=2)) - except Exception as e: - print(f"Error reading results: {e}") - - model_accuracy_thresholds = { - model.model_path: threshold.accuracy - for model, threshold in MODEL_THRESHOLDS.items() - } - model_latency_thresholds = { - model.model_path: threshold.eval_time - for model, threshold in MODEL_THRESHOLDS.items() - } - check_evaluation_test_results( - all_results, - self.__class__.__name__, - model_accuracy_thresholds=model_accuracy_thresholds, - model_latency_thresholds=model_latency_thresholds, - ) - - -if __name__ == "__main__": - unittest.main() diff --git a/test/manual/nightly/test_vlms_perf.py b/test/manual/nightly/test_vlms_perf.py deleted file mode 100644 index 872cd09bb..000000000 --- a/test/manual/nightly/test_vlms_perf.py +++ /dev/null @@ -1,87 +0,0 @@ -import os -import unittest -import warnings - -from sglang.test.nightly_utils import NightlyBenchmarkRunner -from sglang.test.test_utils import ( - DEFAULT_URL_FOR_TEST, - ModelLaunchSettings, - _parse_int_list_env, - parse_models, -) - -RESULT_DIR = "performance_results_vlms" - -MODEL_DEFAULTS = [ - # Keep conservative defaults. Can be overridden by env NIGHTLY_VLM_MODELS - ModelLaunchSettings( - "Qwen/Qwen2.5-VL-7B-Instruct", - extra_args=["--mem-fraction-static=0.7"], - ), - ModelLaunchSettings( - "google/gemma-3-27b-it", - ), - ModelLaunchSettings("Qwen/Qwen3-VL-30B-A3B-Instruct", extra_args=["--tp=2"]), - # "OpenGVLab/InternVL2_5-2B", - # buggy in official transformers impl - # "openbmb/MiniCPM-V-2_6", -] - - -class TestNightlyVLMModelsPerformance(unittest.TestCase): - @classmethod - def setUpClass(cls): - warnings.filterwarnings( - "ignore", category=ResourceWarning, message="unclosed.*socket" - ) - - nightly_vlm_models_str = os.environ.get("NIGHTLY_VLM_MODELS") - if nightly_vlm_models_str: - cls.models = [] - model_paths = parse_models(nightly_vlm_models_str) - for model_path in model_paths: - cls.models.append(ModelLaunchSettings(model_path)) - else: - cls.models = MODEL_DEFAULTS - - cls.base_url = DEFAULT_URL_FOR_TEST - - cls.batch_sizes = _parse_int_list_env("NIGHTLY_VLM_BATCH_SIZES", "1,1,2,8,16") - cls.input_lens = tuple(_parse_int_list_env("NIGHTLY_VLM_INPUT_LENS", "4096")) - cls.output_lens = tuple(_parse_int_list_env("NIGHTLY_VLM_OUTPUT_LENS", "512")) - cls.runner = NightlyBenchmarkRunner(RESULT_DIR, cls.__name__, cls.base_url) - cls.runner.setup_result_directory() - - def test_bench_one_batch(self): - all_model_succeed = True - - for model_setup in self.models: - with self.subTest(model=model_setup.model_path): - # VLMs need additional benchmark args for dataset and trust-remote-code - extra_bench_args = [ - "--trust-remote-code", - "--dataset-name=mmmu", - ] - - results, success = self.runner.run_benchmark_for_model( - model_path=model_setup.model_path, - batch_sizes=self.batch_sizes, - input_lens=self.input_lens, - output_lens=self.output_lens, - other_args=model_setup.extra_args, - extra_bench_args=extra_bench_args, - ) - - if not success: - all_model_succeed = False - - self.runner.add_report(results) - - self.runner.write_final_report() - - if not all_model_succeed: - raise AssertionError("Some models failed the perf tests.") - - -if __name__ == "__main__": - unittest.main()