diff --git a/.github/workflows/nightly-test-nvidia.yml b/.github/workflows/nightly-test-nvidia.yml index 06952e0a1..82f9f44dc 100644 --- a/.github/workflows/nightly-test-nvidia.yml +++ b/.github/workflows/nightly-test-nvidia.yml @@ -204,33 +204,12 @@ jobs: if: always() timeout-minutes: 300 env: - TRACE_BASE_URL: https://raw.githubusercontent.com/sglang-bot/sglang-ci-data/main/traces/${{ github.run_id }} - PERFETTO_RELAY_URL: ${{ vars.PERFETTO_RELAY_URL }} GPU_CONFIG: "8-gpu-h200" IS_H200: "1" run: | cd test python3 run_suite.py --hw cuda --suite nightly-8-gpu-common --nightly --timeout-per-file=18000 --continue-on-error --auto-partition-id=${{ matrix.partition }} --auto-partition-size=4 - - name: Publish traces to storage repo - if: always() - continue-on-error: true - env: - GITHUB_TOKEN: ${{ secrets.GH_PAT_FOR_NIGHTLY_CI_DATA }} - GITHUB_RUN_ID: ${{ github.run_id }} - GITHUB_RUN_NUMBER: ${{ github.run_number }} - run: | - TRACE_ARGS="" - for dir in test/performance_profiles_*/; do - [ -d "$dir" ] && TRACE_ARGS="$TRACE_ARGS --traces-dir $dir" - done - if [ -n "$TRACE_ARGS" ]; then - python3 scripts/ci/utils/publish_traces.py $TRACE_ARGS - find test/performance_profiles_*/ -name '*.json.gz' -delete - else - echo "No trace directories found, skipping publish" - fi - - name: Run test timeout-minutes: 30 env: @@ -247,7 +226,7 @@ jobs: --partition ${{ matrix.partition }} \ --run-id ${{ github.run_id }} \ --output test/metrics-8gpu-h200-partition-${{ matrix.partition }}.json \ - --search-dir test/performance_profiles_8_gpu \ + --search-dir test/performance_results_8_gpu \ --search-dir test - name: Upload partition metrics @@ -318,32 +297,11 @@ jobs: if: always() timeout-minutes: 200 env: - TRACE_BASE_URL: https://raw.githubusercontent.com/sglang-bot/sglang-ci-data/main/traces/${{ github.run_id }} - PERFETTO_RELAY_URL: ${{ vars.PERFETTO_RELAY_URL }} GPU_CONFIG: "8-gpu-b200" run: | cd test python3 run_suite.py --hw cuda --suite nightly-8-gpu-common --nightly --timeout-per-file=12000 --continue-on-error --auto-partition-id=${{ matrix.partition }} --auto-partition-size=4 - - name: Publish traces to storage repo - if: always() - continue-on-error: true - env: - GITHUB_TOKEN: ${{ secrets.GH_PAT_FOR_NIGHTLY_CI_DATA }} - GITHUB_RUN_ID: ${{ github.run_id }} - GITHUB_RUN_NUMBER: ${{ github.run_number }} - run: | - TRACE_ARGS="" - for dir in test/performance_profiles_*/; do - [ -d "$dir" ] && TRACE_ARGS="$TRACE_ARGS --traces-dir $dir" - done - if [ -n "$TRACE_ARGS" ]; then - python3 scripts/ci/utils/publish_traces.py $TRACE_ARGS - find test/performance_profiles_*/ -name '*.json.gz' -delete - else - echo "No trace directories found, skipping publish" - fi - - name: Collect performance metrics if: always() run: | @@ -352,7 +310,7 @@ jobs: --partition ${{ matrix.partition }} \ --run-id ${{ github.run_id }} \ --output test/metrics-8gpu-b200-partition-${{ matrix.partition }}.json \ - --search-dir test/performance_profiles_8_gpu \ + --search-dir test/performance_results_8_gpu \ --search-dir test - name: Upload partition metrics @@ -413,22 +371,12 @@ jobs: - name: Run performance test for text models timeout-minutes: 30 env: - TRACE_BASE_URL: https://raw.githubusercontent.com/sglang-bot/sglang-ci-data/main/traces/${{ github.run_id }} - PERFETTO_RELAY_URL: ${{ vars.PERFETTO_RELAY_URL }} GPU_CONFIG: "2-gpu-h100" run: | cd test - rm -rf performance_profiles_text_models/ + rm -rf performance_results_text_models/ python3 run_suite.py --hw cuda --suite nightly-perf-text-2-gpu --nightly --continue-on-error --timeout-per-file 3600 - - name: Publish traces to storage repo - env: - GITHUB_TOKEN: ${{ secrets.GH_PAT_FOR_NIGHTLY_CI_DATA }} - GITHUB_RUN_ID: ${{ github.run_id }} - GITHUB_RUN_NUMBER: ${{ github.run_number }} - run: | - python3 scripts/ci/utils/publish_traces.py --traces-dir test/performance_profiles_text_models - - uses: ./.github/actions/upload-cuda-coredumps if: failure() @@ -476,22 +424,12 @@ jobs: - name: Run perf test for VLM models (MMMU) timeout-minutes: 30 env: - TRACE_BASE_URL: https://raw.githubusercontent.com/sglang-bot/sglang-ci-data/main/traces/${{ github.run_id }} - PERFETTO_RELAY_URL: ${{ vars.PERFETTO_RELAY_URL }} GPU_CONFIG: "2-gpu-h100" run: | cd test - rm -rf performance_profiles_vlms/ + rm -rf performance_results_vlms/ python3 run_suite.py --hw cuda --suite nightly-perf-vlm-2-gpu --nightly --continue-on-error --timeout-per-file 3600 - - name: Publish traces to storage repo - env: - GITHUB_TOKEN: ${{ secrets.GH_PAT_FOR_NIGHTLY_CI_DATA }} - GITHUB_RUN_ID: ${{ github.run_id }} - GITHUB_RUN_NUMBER: ${{ github.run_number }} - run: | - python3 scripts/ci/utils/publish_traces.py --traces-dir test/performance_profiles_vlms - - uses: ./.github/actions/upload-cuda-coredumps if: failure() @@ -514,32 +452,11 @@ jobs: - name: Run test timeout-minutes: 200 env: - TRACE_BASE_URL: https://raw.githubusercontent.com/sglang-bot/sglang-ci-data/main/traces/${{ github.run_id }} - PERFETTO_RELAY_URL: ${{ vars.PERFETTO_RELAY_URL }} GPU_CONFIG: "4-gpu-b200" run: | cd test python3 run_suite.py --hw cuda --suite nightly-4-gpu-b200 --nightly --continue-on-error --timeout-per-file 12000 - - name: Publish traces to storage repo - if: always() - continue-on-error: true - env: - GITHUB_TOKEN: ${{ secrets.GH_PAT_FOR_NIGHTLY_CI_DATA }} - GITHUB_RUN_ID: ${{ github.run_id }} - GITHUB_RUN_NUMBER: ${{ github.run_number }} - run: | - TRACE_ARGS="" - for dir in test/performance_profiles_*/; do - [ -d "$dir" ] && TRACE_ARGS="$TRACE_ARGS --traces-dir $dir" - done - if [ -n "$TRACE_ARGS" ]; then - python3 scripts/ci/utils/publish_traces.py $TRACE_ARGS - find test/performance_profiles_*/ -name '*.json.gz' -delete - else - echo "No trace directories found, skipping publish" - fi - - uses: ./.github/actions/upload-cuda-coredumps if: failure() @@ -577,32 +494,11 @@ jobs: - name: Run test timeout-minutes: 600 env: - TRACE_BASE_URL: https://raw.githubusercontent.com/sglang-bot/sglang-ci-data/main/traces/${{ github.run_id }} - PERFETTO_RELAY_URL: ${{ vars.PERFETTO_RELAY_URL }} GPU_CONFIG: "4-gpu-gb300" run: | cd test python3 run_suite.py --hw cuda --suite ${{ matrix.suite }} --nightly --continue-on-error --timeout-per-file 7200 - - name: Publish traces to storage repo - if: always() - continue-on-error: true - env: - GITHUB_TOKEN: ${{ secrets.GH_PAT_FOR_NIGHTLY_CI_DATA }} - GITHUB_RUN_ID: ${{ github.run_id }} - GITHUB_RUN_NUMBER: ${{ github.run_number }} - run: | - TRACE_ARGS="" - for dir in test/performance_profiles_*/; do - [ -d "$dir" ] && TRACE_ARGS="$TRACE_ARGS --traces-dir $dir" - done - if [ -n "$TRACE_ARGS" ]; then - python3 scripts/ci/utils/publish_traces.py $TRACE_ARGS - find test/performance_profiles_*/ -name '*.json.gz' -delete - else - echo "No trace directories found, skipping publish" - fi - - uses: ./.github/actions/upload-cuda-coredumps if: failure() diff --git a/python/sglang/test/nightly_bench_utils.py b/python/sglang/test/nightly_bench_utils.py index 88648aa44..b6cfcd79c 100644 --- a/python/sglang/test/nightly_bench_utils.py +++ b/python/sglang/test/nightly_bench_utils.py @@ -1,12 +1,9 @@ import json -import logging import os from typing import List, Optional from pydantic import BaseModel -logger = logging.getLogger(__name__) - # TODO: # There is huge redundancy between BenchmarkResult and BenchOneCaseResult, and redundancy between to_markdown_row, generate_markdown_report, get_report_summary. # We should refactor them to reduce the code duplication. @@ -33,17 +30,7 @@ class BenchmarkResult(BaseModel): profile_link_decode: Optional[str] = None server_args: Optional[List[str]] = None - @staticmethod - def help_str() -> str: - return f""" -Note: To view the traces through perfetto-ui, please: - 1. open with Google Chrome - 2. allow popup -""" - - def to_markdown_row( - self, trace_dir, base_url: str = "", relay_base: str = "" - ) -> str: + def to_markdown_row(self) -> str: """Convert this benchmark result to a markdown table row.""" hourly_cost_per_gpu = 2 # $2/hour for one H100 @@ -54,42 +41,11 @@ Note: To view the traces through perfetto-ui, please: input_cost = 1e6 / (self.input_throughput * input_util) / 3600 * hourly_cost output_cost = 1e6 / self.output_throughput / 3600 * hourly_cost - def get_perfetto_relay_link_from_trace_file(trace_file: str): - from urllib.parse import quote - - rel_path = os.path.relpath(trace_file, trace_dir) - raw_file_link = f"{base_url}/{rel_path}" - relay_link = ( - f"{relay_base}?src={quote(raw_file_link, safe='')}" - if relay_base - else raw_file_link - ) - return relay_link - - # Handle profile links - profile_link = "NA | NA" - if self.profile_link_extend or self.profile_link_decode: - # Create a combined link or use the first available one - trace_files = [self.profile_link_extend, self.profile_link_decode] - if any(trace_file is None for trace_file in trace_files): - logger.error("Some trace files are None", f"{trace_files=}") - trace_files_relay_links = [ - ( - f"[trace]({get_perfetto_relay_link_from_trace_file(trace_file)})" - if trace_file - else "N/A" - ) - for trace_file in trace_files - ] - - profile_link = " | ".join(trace_files_relay_links) - - # Build the row - return f"| {self.batch_size} | {self.input_len} | {self.latency:.2f} | {self.input_throughput:.2f} | {self.output_throughput:.2f} | {accept_length} | {itl:.2f} | {input_cost:.2f} | {output_cost:.2f} | {profile_link} |\n" + return f"| {self.batch_size} | {self.input_len} | {self.latency:.2f} | {self.input_throughput:.2f} | {self.output_throughput:.2f} | {accept_length} | {itl:.2f} | {input_cost:.2f} | {output_cost:.2f} |\n" def generate_markdown_report( - trace_dir, results: List[BenchmarkResult], variant: Optional[str] = None + results: List[BenchmarkResult], variant: Optional[str] = None ) -> str: """Generate a markdown report from a list of BenchmarkResult object from a single run.""" # Build model header with run_name if it's not "default" @@ -107,17 +63,12 @@ def generate_markdown_report( summary = f"### {model_header}\n" - summary += "| batch size | input len | latency (s) | input throughput (tok/s) | output throughput (tok/s) | acc length | ITL (ms) | input cost ($/1M) | output cost ($/1M) | profile (extend) | profile (decode)|\n" - summary += "| ---------- | --------- | ----------- | ------------------------- | ------------------------- | ---------- | -------- | ----------------- | ------------------ | ---------------- | --------------- |\n" + summary += "| batch size | input len | latency (s) | input throughput (tok/s) | output throughput (tok/s) | acc length | ITL (ms) | input cost ($/1M) | output cost ($/1M) |\n" + summary += "| ---------- | --------- | ----------- | ------------------------- | ------------------------- | ---------- | -------- | ----------------- | ------------------ |\n" # all results should share the same isl & osl for result in results: - base_url = os.getenv("TRACE_BASE_URL", "").rstrip("/") - relay_base = os.getenv( - "PERFETTO_RELAY_URL", - "", - ).rstrip("/") - summary += result.to_markdown_row(trace_dir, base_url, relay_base) + summary += result.to_markdown_row() return summary diff --git a/python/sglang/test/nightly_utils.py b/python/sglang/test/nightly_utils.py index a26faea2f..a1eb5f0d6 100644 --- a/python/sglang/test/nightly_utils.py +++ b/python/sglang/test/nightly_utils.py @@ -1,4 +1,4 @@ -"""Utilities for running nightly performance benchmarks with profiling.""" +"""Utilities for running nightly performance benchmarks.""" import json import os @@ -19,16 +19,16 @@ from sglang.test.test_utils import ( class NightlyBenchmarkRunner: - """Helper class for running nightly performance benchmarks with profiling. + """Helper class for running nightly performance benchmarks. This class encapsulates common patterns used across nightly performance tests, - including profile directory management, benchmark command construction, + including result directory management, benchmark command construction, result parsing, and report generation. """ def __init__( self, - profile_dir: str, + result_dir: str, test_name: str, base_url: str, gpu_config: str = None, @@ -36,12 +36,12 @@ class NightlyBenchmarkRunner: """Initialize the benchmark runner. Args: - profile_dir: Directory to store performance profiles + result_dir: Directory to store benchmark results test_name: Name of the test (used for reporting) base_url: Base URL for the server gpu_config: Optional GPU configuration string (e.g., "2-gpu-h100", "8-gpu-b200") """ - self.profile_dir = profile_dir + self.result_dir = result_dir self.test_name = test_name self.base_url = base_url self.gpu_config = gpu_config or os.environ.get("GPU_CONFIG", "") @@ -51,38 +51,32 @@ class NightlyBenchmarkRunner: if self.gpu_config: header += f" ({self.gpu_config})" header += "\n" - self.full_report = header + BenchmarkResult.help_str() + self.full_report = header - def setup_profile_directory(self) -> None: - """Create the profile directory if it doesn't exist.""" - os.makedirs(self.profile_dir, exist_ok=True) + def setup_result_directory(self) -> None: + """Create the result directory if it doesn't exist.""" + os.makedirs(self.result_dir, exist_ok=True) - def generate_profile_filename( - self, model_path: str, variant: str = "" - ) -> Tuple[str, str]: - """Generate unique profile filename and path for the model. + def generate_result_filename(self, model_path: str, variant: str = "") -> str: + """Generate a unique result filename for the model. Args: model_path: Path to the model (e.g., "deepseek-ai/DeepSeek-V3.1") variant: Optional variant suffix (e.g., "basic", "mtp", "dsa") Returns: - Tuple of (profile_path_prefix, json_output_file) + Path to the JSON result file """ timestamp = int(time.time()) model_safe_name = model_path.replace("/", "_") # Build filename with optional variant if variant: - profile_filename = f"{model_safe_name}_{variant}_{timestamp}" json_filename = f"results_{model_safe_name}_{variant}_{timestamp}.json" else: - profile_filename = f"{model_safe_name}_{timestamp}" json_filename = f"results_{model_safe_name}_{timestamp}.json" - profile_path_prefix = os.path.join(self.profile_dir, profile_filename) - - return profile_path_prefix, json_filename + return os.path.join(self.result_dir, json_filename) def build_benchmark_command( self, @@ -90,11 +84,9 @@ class NightlyBenchmarkRunner: batch_sizes: List[int], input_lens: Tuple[int, ...], output_lens: Tuple[int, ...], - profile_path_prefix: str, json_output_file: str, extra_args: Optional[List[str]] = None, server_args: Optional[List[str]] = None, - enable_profile: bool = True, ) -> List[str]: """Build the benchmark command with all required arguments. @@ -103,11 +95,9 @@ class NightlyBenchmarkRunner: batch_sizes: List of batch sizes to test input_lens: Tuple of input lengths to test output_lens: Tuple of output lengths to test - profile_path_prefix: Prefix for profile output files json_output_file: Path to JSON output file extra_args: Optional extra arguments to append to command server_args: Optional server launch arguments to record in metrics - enable_profile: Whether to enable profiling (default True for NVIDIA) Returns: List of command arguments ready for subprocess.run() @@ -132,17 +122,6 @@ class NightlyBenchmarkRunner: "--trust-remote-code", ] - # Add profiling flags only if enabled (disabled for AMD tests) - if enable_profile and profile_path_prefix: - command.extend( - [ - "--profile", - "--profile-by-stage", - "--profile-output-dir", - profile_path_prefix, - ] - ) - if extra_args: command.extend(extra_args) @@ -227,7 +206,6 @@ class NightlyBenchmarkRunner: other_args: Optional[List[str]] = None, variant: str = "", extra_bench_args: Optional[List[str]] = None, - enable_profile: bool = True, timeout: Optional[int] = None, env: Optional[dict] = None, ) -> Tuple[List[BenchmarkResult], bool, Optional[float]]: @@ -235,7 +213,7 @@ class NightlyBenchmarkRunner: This method handles: - Server launch and cleanup - - Profile filename generation + - Result filename generation - Benchmark command construction and execution - Result loading and parsing - Fetching speculative decoding accept length (for MTP/EAGLE) @@ -248,7 +226,6 @@ class NightlyBenchmarkRunner: other_args: Arguments to pass to server launch variant: Optional variant suffix (e.g., "basic", "mtp") extra_bench_args: Extra arguments for the benchmark command - enable_profile: Whether to enable profiling (default True for NVIDIA) timeout: Optional timeout for server launch (defaults to DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH) env: Environment dict for subprocess @@ -275,9 +252,7 @@ class NightlyBenchmarkRunner: ) # Generate filenames - profile_path_prefix, json_output_file = self.generate_profile_filename( - model_path, variant - ) + json_output_file = self.generate_result_filename(model_path, variant) # Build and run benchmark command # Prepare extra args with run_name if variant is specified @@ -290,11 +265,9 @@ class NightlyBenchmarkRunner: batch_sizes, input_lens, output_lens, - profile_path_prefix, json_output_file, extra_args=bench_args, server_args=other_args, - enable_profile=enable_profile, ) result, cmd_success = self.run_benchmark_command(command, model_description) @@ -346,7 +319,7 @@ class NightlyBenchmarkRunner: results: List of BenchmarkResult objects to add to report """ if results: - report_part = generate_markdown_report(self.profile_dir, results, variant) + report_part = generate_markdown_report(results, variant) self.full_report += report_part + "\n" def write_final_report(self) -> None: diff --git a/python/sglang/test/performance_test_runner.py b/python/sglang/test/performance_test_runner.py index ad4447ddf..5f1f2b055 100644 --- a/python/sglang/test/performance_test_runner.py +++ b/python/sglang/test/performance_test_runner.py @@ -13,7 +13,7 @@ class PerformanceTestParams: batch_sizes: List[int] = field(default_factory=lambda: [1, 8, 16]) input_lens: Tuple[int, ...] = (8192,) output_lens: Tuple[int, ...] = (512,) - profile_dir: Optional[str] = None # None = auto-generate based on is_vlm + result_dir: Optional[str] = None # None = auto-generate based on is_vlm dataset_name: str = "mmmu" # For VLM perf test # MTP/EAGLE speculative decoding: minimum accept length threshold (None = no validation) spec_accept_length_threshold: Optional[float] = None @@ -86,9 +86,8 @@ def run_performance_test( perf_runner.add_report(results, variant=model.variant) print(f"✓ Performance test succeeded for {model.model_path}") - # The cumulative /server_info accept length is reset by the cache - # flush before the profiling phase, so it can be missing here. Fall - # back to the per-run accept lengths captured during benchmarking. + # Fall back to the per-run accept lengths captured during benchmarking + # when the cumulative /server_info metric is unavailable. if avg_spec_accept_length is None: run_accept_lengths = [ r.acc_length @@ -155,7 +154,7 @@ def run_performance_test( def run_performance_for_models( models: List[ModelLaunchSettings], - profile_dir: str, + result_dir: str, test_name: str, base_url: Optional[str] = None, batch_sizes: List[int] = None, @@ -168,7 +167,7 @@ def run_performance_for_models( Args: models: List of ModelLaunchSettings to test - profile_dir: Directory for performance profiles + result_dir: Directory for performance results test_name: Name for the test (used in reports) base_url: Server base URL (default: DEFAULT_URL_FOR_TEST) batch_sizes: Batch sizes for perf test @@ -188,11 +187,11 @@ def run_performance_for_models( # Setup performance runner perf_runner = NightlyBenchmarkRunner( - profile_dir=profile_dir, + result_dir=result_dir, test_name=test_name, base_url=base_url, ) - perf_runner.setup_profile_directory() + perf_runner.setup_result_directory() all_results = [] all_passed = True diff --git a/python/sglang/test/run_combined_tests.py b/python/sglang/test/run_combined_tests.py index fa419b3ff..0a02e3165 100644 --- a/python/sglang/test/run_combined_tests.py +++ b/python/sglang/test/run_combined_tests.py @@ -76,18 +76,16 @@ def run_combined_tests( # Set up performance parameters if run_perf: perf = performance_params - profile_dir = perf.profile_dir or ( - "performance_profiles_vlms" - if is_vlm - else "performance_profiles_text_models" + result_dir = perf.result_dir or ( + "performance_results_vlms" if is_vlm else "performance_results_text_models" ) perf_runner = NightlyBenchmarkRunner( - profile_dir=profile_dir, + result_dir=result_dir, test_name=test_name, base_url=base_url, ) - perf_runner.setup_profile_directory() + perf_runner.setup_result_directory() else: perf_runner = None diff --git a/scripts/ci/utils/save_metrics.py b/scripts/ci/utils/save_metrics.py index 90d481ede..46aaaf4da 100755 --- a/scripts/ci/utils/save_metrics.py +++ b/scripts/ci/utils/save_metrics.py @@ -1,7 +1,7 @@ #!/usr/bin/env python3 """Collect and save performance metrics from nightly benchmark results. -This script reads benchmark result JSON files from performance profile directories +This script reads benchmark result JSON files from performance result directories and saves them with metadata for artifact collection in CI. Usage: @@ -223,9 +223,9 @@ def main(): # Default search directories if none specified search_dirs = args.search_dirs or [ - "test/performance_profiles_8_gpu", - "test/performance_profiles_text_models", - "test/performance_profiles_vlms", + "test/performance_results_8_gpu", + "test/performance_results_text_models", + "test/performance_results_vlms", "test", ".", ] diff --git a/test/manual/nightly/test_deepseek_v31_perf.py b/test/manual/nightly/test_deepseek_v31_perf.py index b04cfb6d2..34a8cde8b 100644 --- a/test/manual/nightly/test_deepseek_v31_perf.py +++ b/test/manual/nightly/test_deepseek_v31_perf.py @@ -4,7 +4,7 @@ from sglang.test.nightly_utils import NightlyBenchmarkRunner from sglang.test.test_utils import DEFAULT_URL_FOR_TEST, _parse_int_list_env DEEPSEEK_V31_MODEL_PATH = "deepseek-ai/DeepSeek-V3.1" -PROFILE_DIR = "performance_profiles_deepseek_v31" +RESULT_DIR = "performance_results_deepseek_v31" class TestNightlyDeepseekV31Performance(unittest.TestCase): @@ -50,8 +50,8 @@ class TestNightlyDeepseekV31Performance(unittest.TestCase): }, ] - cls.runner = NightlyBenchmarkRunner(PROFILE_DIR, cls.__name__, cls.base_url) - cls.runner.setup_profile_directory() + cls.runner = NightlyBenchmarkRunner(RESULT_DIR, cls.__name__, cls.base_url) + cls.runner.setup_result_directory() def test_bench_one_batch(self): failed_variants = [] diff --git a/test/manual/nightly/test_deepseek_v32_perf.py b/test/manual/nightly/test_deepseek_v32_perf.py index b5d6811d2..bb64468c9 100644 --- a/test/manual/nightly/test_deepseek_v32_perf.py +++ b/test/manual/nightly/test_deepseek_v32_perf.py @@ -4,7 +4,7 @@ from sglang.test.nightly_utils import NightlyBenchmarkRunner from sglang.test.test_utils import DEFAULT_URL_FOR_TEST, _parse_int_list_env DEEPSEEK_V32_MODEL_PATH = "deepseek-ai/DeepSeek-V3.2" -PROFILE_DIR = "performance_profiles_deepseek_v32" +RESULT_DIR = "performance_results_deepseek_v32" class TestNightlyDeepseekV32Performance(unittest.TestCase): @@ -91,8 +91,8 @@ class TestNightlyDeepseekV32Performance(unittest.TestCase): }, ] - cls.runner = NightlyBenchmarkRunner(PROFILE_DIR, cls.__name__, cls.base_url) - cls.runner.setup_profile_directory() + cls.runner = NightlyBenchmarkRunner(RESULT_DIR, cls.__name__, cls.base_url) + cls.runner.setup_result_directory() def test_bench_one_batch(self): failed_variants = [] diff --git a/test/manual/nightly/test_text_models_perf.py b/test/manual/nightly/test_text_models_perf.py index f83483ac2..f4161b838 100644 --- a/test/manual/nightly/test_text_models_perf.py +++ b/test/manual/nightly/test_text_models_perf.py @@ -8,7 +8,7 @@ from sglang.test.test_utils import ( parse_models, ) -PROFILE_DIR = "performance_profiles_text_models" +RESULT_DIR = "performance_results_text_models" class TestNightlyTextModelsPerformance(unittest.TestCase): @@ -28,8 +28,8 @@ class TestNightlyTextModelsPerformance(unittest.TestCase): cls.batch_sizes = [1, 1, 8, 16, 64] cls.input_lens = tuple(_parse_int_list_env("NIGHTLY_INPUT_LENS", "4096")) cls.output_lens = tuple(_parse_int_list_env("NIGHTLY_OUTPUT_LENS", "512")) - cls.runner = NightlyBenchmarkRunner(PROFILE_DIR, cls.__name__, cls.base_url) - cls.runner.setup_profile_directory() + cls.runner = NightlyBenchmarkRunner(RESULT_DIR, cls.__name__, cls.base_url) + cls.runner.setup_result_directory() def test_bench_one_batch(self): all_model_succeed = True diff --git a/test/manual/nightly/test_vlms_perf.py b/test/manual/nightly/test_vlms_perf.py index 0d4c2801b..872cd09bb 100644 --- a/test/manual/nightly/test_vlms_perf.py +++ b/test/manual/nightly/test_vlms_perf.py @@ -10,7 +10,7 @@ from sglang.test.test_utils import ( parse_models, ) -PROFILE_DIR = "performance_profiles_vlms" +RESULT_DIR = "performance_results_vlms" MODEL_DEFAULTS = [ # Keep conservative defaults. Can be overridden by env NIGHTLY_VLM_MODELS @@ -49,8 +49,8 @@ class TestNightlyVLMModelsPerformance(unittest.TestCase): cls.batch_sizes = _parse_int_list_env("NIGHTLY_VLM_BATCH_SIZES", "1,1,2,8,16") cls.input_lens = tuple(_parse_int_list_env("NIGHTLY_VLM_INPUT_LENS", "4096")) cls.output_lens = tuple(_parse_int_list_env("NIGHTLY_VLM_OUTPUT_LENS", "512")) - cls.runner = NightlyBenchmarkRunner(PROFILE_DIR, cls.__name__, cls.base_url) - cls.runner.setup_profile_directory() + cls.runner = NightlyBenchmarkRunner(RESULT_DIR, cls.__name__, cls.base_url) + cls.runner.setup_result_directory() def test_bench_one_batch(self): all_model_succeed = True diff --git a/test/manual/test_deepseek_v31.py b/test/manual/test_deepseek_v31.py index 543879b17..3a81d6809 100644 --- a/test/manual/test_deepseek_v31.py +++ b/test/manual/test_deepseek_v31.py @@ -61,7 +61,7 @@ class TestDeepseekV31(unittest.TestCase): dataset="gsm8k", baseline_accuracy=0.935 ), performance_params=PerformanceTestParams( - profile_dir="performance_profiles_deepseek_v31", + result_dir="performance_results_deepseek_v31", ), ) diff --git a/test/manual/test_glm_46_fp8.py b/test/manual/test_glm_46_fp8.py index 815ad33f4..28bea3eb7 100644 --- a/test/manual/test_glm_46_fp8.py +++ b/test/manual/test_glm_46_fp8.py @@ -50,7 +50,7 @@ class TestGLM46FP8(unittest.TestCase): test_name="GLM-4.6-FP8", accuracy_params=AccuracyTestParams(dataset="gsm8k", baseline_accuracy=0.80), performance_params=PerformanceTestParams( - profile_dir="performance_profiles_glm_4_6_fp8", + result_dir="performance_results_glm_4_6_fp8", ), ) diff --git a/test/manual/test_qwen3_235b.py b/test/manual/test_qwen3_235b.py index 41cf71a47..19606a819 100644 --- a/test/manual/test_qwen3_235b.py +++ b/test/manual/test_qwen3_235b.py @@ -61,7 +61,7 @@ class TestQwen3235BFP8(unittest.TestCase): test_name="Qwen3-235B-FP8", accuracy_params=AccuracyTestParams(dataset="gsm8k", baseline_accuracy=0.88), performance_params=PerformanceTestParams( - profile_dir="performance_profiles_qwen3_235b_fp8", + result_dir="performance_results_qwen3_235b_fp8", ), ) diff --git a/test/registered/8-gpu-models/test_glm52_fp8.py b/test/registered/8-gpu-models/test_glm52_fp8.py index dca35ed6d..47cb78ed5 100644 --- a/test/registered/8-gpu-models/test_glm52_fp8.py +++ b/test/registered/8-gpu-models/test_glm52_fp8.py @@ -59,7 +59,7 @@ class TestGlm52Fp8(unittest.TestCase): test_name="GLM-5.2-FP8", accuracy_params=AccuracyTestParams(dataset="gsm8k", baseline_accuracy=0.92), performance_params=PerformanceTestParams( - profile_dir="performance_profiles_glm_52_fp8", + result_dir="performance_results_glm_52_fp8", ), ) diff --git a/test/registered/8-gpu-models/test_glm_46.py b/test/registered/8-gpu-models/test_glm_46.py index dc22744b4..71786fcfc 100644 --- a/test/registered/8-gpu-models/test_glm_46.py +++ b/test/registered/8-gpu-models/test_glm_46.py @@ -43,7 +43,7 @@ class TestGLM46(unittest.TestCase): test_name="GLM-4.6", accuracy_params=AccuracyTestParams(dataset="gsm8k", baseline_accuracy=0.80), performance_params=PerformanceTestParams( - profile_dir="performance_profiles_glm_4_6", + result_dir="performance_results_glm_4_6", ), ) diff --git a/test/registered/8-gpu-models/test_gpt_oss_120b.py b/test/registered/8-gpu-models/test_gpt_oss_120b.py index ca003cfa1..1b6791dff 100644 --- a/test/registered/8-gpu-models/test_gpt_oss_120b.py +++ b/test/registered/8-gpu-models/test_gpt_oss_120b.py @@ -74,7 +74,7 @@ class TestGptOss120B(unittest.TestCase): test_name="GPT-OSS-120B", accuracy_params=None, performance_params=PerformanceTestParams( - profile_dir="performance_profiles_gpt_oss_120b", + result_dir="performance_results_gpt_oss_120b", ), ) diff --git a/test/registered/8-gpu-models/test_inkling_nvfp4_nightly.py b/test/registered/8-gpu-models/test_inkling_nvfp4_nightly.py index bda8229d1..e86ba5a88 100644 --- a/test/registered/8-gpu-models/test_inkling_nvfp4_nightly.py +++ b/test/registered/8-gpu-models/test_inkling_nvfp4_nightly.py @@ -70,7 +70,7 @@ class TestInklingNVFP4Nightly(unittest.TestCase): repeat=1, ), performance_params=PerformanceTestParams( - profile_dir="performance_profiles_inkling_nvfp4", + result_dir="performance_results_inkling_nvfp4", ), ) diff --git a/test/registered/8-gpu-models/test_kimi_k25.py b/test/registered/8-gpu-models/test_kimi_k25.py index a160e8211..fc6b75c90 100644 --- a/test/registered/8-gpu-models/test_kimi_k25.py +++ b/test/registered/8-gpu-models/test_kimi_k25.py @@ -54,7 +54,7 @@ class TestKimiK25(unittest.TestCase): test_name="Kimi-K2.5", accuracy_params=AccuracyTestParams(dataset="gsm8k", baseline_accuracy=0.92), performance_params=PerformanceTestParams( - profile_dir="performance_profiles_kimi_k25", + result_dir="performance_results_kimi_k25", ), ) diff --git a/test/registered/8-gpu-models/test_llama4.py b/test/registered/8-gpu-models/test_llama4.py index 5f8b7fab1..d5e1e8eab 100644 --- a/test/registered/8-gpu-models/test_llama4.py +++ b/test/registered/8-gpu-models/test_llama4.py @@ -47,7 +47,7 @@ class TestLlama4(unittest.TestCase): test_name="Llama-4-Scout", accuracy_params=AccuracyTestParams(dataset="gsm8k", baseline_accuracy=0.9), performance_params=PerformanceTestParams( - profile_dir="performance_profiles_llama4", + result_dir="performance_results_llama4", ), ) diff --git a/test/registered/8-gpu-models/test_longcat_flash_lite_fp8.py b/test/registered/8-gpu-models/test_longcat_flash_lite_fp8.py index feec8d0e7..be3b8e000 100644 --- a/test/registered/8-gpu-models/test_longcat_flash_lite_fp8.py +++ b/test/registered/8-gpu-models/test_longcat_flash_lite_fp8.py @@ -58,7 +58,7 @@ class TestLongCatFlashLiteFp8(unittest.TestCase): num_examples=200, ), performance_params=PerformanceTestParams( - profile_dir="performance_profiles_longcat_flash_lite_fp8", + result_dir="performance_results_longcat_flash_lite_fp8", ), ) diff --git a/test/registered/8-gpu-models/test_minimax_m25.py b/test/registered/8-gpu-models/test_minimax_m25.py index 59091a5a6..b9e6d0adf 100644 --- a/test/registered/8-gpu-models/test_minimax_m25.py +++ b/test/registered/8-gpu-models/test_minimax_m25.py @@ -54,7 +54,7 @@ class TestMiniMaxM25(unittest.TestCase): test_name="MiniMax-M2.5", accuracy_params=AccuracyTestParams(dataset="gsm8k", baseline_accuracy=0.80), performance_params=PerformanceTestParams( - profile_dir="performance_profiles_minimax_m25", + result_dir="performance_results_minimax_m25", ), ) diff --git a/test/registered/8-gpu-models/test_mistral_large3.py b/test/registered/8-gpu-models/test_mistral_large3.py index 58587d45e..9e20e1cd9 100644 --- a/test/registered/8-gpu-models/test_mistral_large3.py +++ b/test/registered/8-gpu-models/test_mistral_large3.py @@ -89,7 +89,7 @@ class TestMistralLarge3(unittest.TestCase): test_name="Mistral-Large-3", accuracy_params=AccuracyTestParams(dataset="gsm8k", baseline_accuracy=0.85), performance_params=PerformanceTestParams( - profile_dir="performance_profiles_mistral_large3", + result_dir="performance_results_mistral_large3", ), ) diff --git a/test/registered/8-gpu-models/test_nvidia_nemotron_3_super_nightly.py b/test/registered/8-gpu-models/test_nvidia_nemotron_3_super_nightly.py index cd299032d..e8536cdcf 100644 --- a/test/registered/8-gpu-models/test_nvidia_nemotron_3_super_nightly.py +++ b/test/registered/8-gpu-models/test_nvidia_nemotron_3_super_nightly.py @@ -92,7 +92,7 @@ class TestNvidiaNemotron3SuperNightly(unittest.TestCase): repeat=1, ), performance_params=PerformanceTestParams( - profile_dir="performance_profiles_nemotron_3_super_bf16", + result_dir="performance_results_nemotron_3_super_bf16", ), ) @@ -129,7 +129,7 @@ class TestNvidiaNemotron3SuperNightly(unittest.TestCase): repeat=1, ), performance_params=PerformanceTestParams( - profile_dir="performance_profiles_nemotron_3_super_nvfp4", + result_dir="performance_results_nemotron_3_super_nvfp4", ), ) diff --git a/test/registered/8-gpu-models/test_qwen35.py b/test/registered/8-gpu-models/test_qwen35.py index f8e48e757..32ca62bd0 100644 --- a/test/registered/8-gpu-models/test_qwen35.py +++ b/test/registered/8-gpu-models/test_qwen35.py @@ -71,7 +71,7 @@ class TestQwen35(unittest.TestCase): num_examples=200, ), performance_params=PerformanceTestParams( - profile_dir="performance_profiles_qwen35", + result_dir="performance_results_qwen35", ), ) diff --git a/test/registered/amd/perf/mi30x/test_deepseek_v31_perf.py b/test/registered/amd/perf/mi30x/test_deepseek_v31_perf.py index eca184072..62a90e2fd 100644 --- a/test/registered/amd/perf/mi30x/test_deepseek_v31_perf.py +++ b/test/registered/amd/perf/mi30x/test_deepseek_v31_perf.py @@ -22,7 +22,7 @@ register_amd_ci(est_time=18000, suite="nightly-perf-8-gpu-deepseek-v31", nightly def generate_simple_markdown_report(results: List[BenchmarkResult]) -> str: - """Generate a simplified markdown report without traces and cost columns. + """Generate a simplified markdown report without cost columns. Skips the first result if it's a warmup run (duplicate batch_size). """ @@ -56,7 +56,7 @@ def generate_simple_markdown_report(results: List[BenchmarkResult]) -> str: DEEPSEEK_V31_MODEL_PATH = os.environ.get( "DEEPSEEK_V31_MODEL_PATH", "deepseek-ai/DeepSeek-V3.1" ) -PROFILE_DIR = "performance_profiles_deepseek_v31" +RESULT_DIR = "performance_results_deepseek_v31" class TestNightlyDeepseekV31Performance(unittest.TestCase): @@ -109,9 +109,9 @@ class TestNightlyDeepseekV31Performance(unittest.TestCase): }, ] - cls.runner = NightlyBenchmarkRunner(PROFILE_DIR, cls.__name__, cls.base_url) - cls.runner.setup_profile_directory() - # Override full_report to remove traces help text + cls.runner = NightlyBenchmarkRunner(RESULT_DIR, cls.__name__, cls.base_url) + cls.runner.setup_result_directory() + # Set the report header for this test cls.runner.full_report = f"## {cls.__name__}\n" def test_bench_one_batch(self): @@ -129,7 +129,6 @@ class TestNightlyDeepseekV31Performance(unittest.TestCase): other_args=variant_config["other_args"], variant=variant_config["name"], extra_bench_args=["--trust-remote-code"], - enable_profile=False, # Disable profiling for AMD tests ) results = result_tuple[0] success = result_tuple[1] @@ -137,7 +136,7 @@ class TestNightlyDeepseekV31Performance(unittest.TestCase): if not success: failed_variants.append(variant_config["name"]) - # Use simplified report format without traces + # Use the simplified report format if results: self.runner.full_report += ( generate_simple_markdown_report(results) + "\n" diff --git a/test/registered/amd/perf/mi30x/test_deepseek_v32_basic_perf_amd.py b/test/registered/amd/perf/mi30x/test_deepseek_v32_basic_perf_amd.py index 9b78008f1..9b82276cf 100644 --- a/test/registered/amd/perf/mi30x/test_deepseek_v32_basic_perf_amd.py +++ b/test/registered/amd/perf/mi30x/test_deepseek_v32_basic_perf_amd.py @@ -26,7 +26,7 @@ register_amd_ci( def generate_simple_markdown_report(results: List[BenchmarkResult]) -> str: - """Generate a simplified markdown report without traces and cost columns. + """Generate a simplified markdown report without cost columns. Skips the first result if it's a warmup run (duplicate batch_size). """ @@ -60,7 +60,7 @@ def generate_simple_markdown_report(results: List[BenchmarkResult]) -> str: DEEPSEEK_V32_MODEL_PATH = os.environ.get( "DEEPSEEK_V32_MODEL_PATH", "deepseek-ai/DeepSeek-V3.2" ) -PROFILE_DIR = "performance_profiles_deepseek_v32_basic_mi325" +RESULT_DIR = "performance_results_deepseek_v32_basic_mi325" class TestNightlyDeepseekV32BasicPerformance(unittest.TestCase): @@ -99,9 +99,9 @@ class TestNightlyDeepseekV32BasicPerformance(unittest.TestCase): "env_vars": {"SGLANG_USE_AITER": "1"}, } - cls.runner = NightlyBenchmarkRunner(PROFILE_DIR, cls.__name__, cls.base_url) - cls.runner.setup_profile_directory() - # Override full_report to remove traces help text + cls.runner = NightlyBenchmarkRunner(RESULT_DIR, cls.__name__, cls.base_url) + cls.runner.setup_result_directory() + # Set the report header for this test cls.runner.full_report = f"## {cls.__name__}\n" def test_bench_one_batch(self): @@ -115,7 +115,6 @@ class TestNightlyDeepseekV32BasicPerformance(unittest.TestCase): other_args=self.variant_config["other_args"], variant=self.variant_config["name"], extra_bench_args=["--trust-remote-code"], - enable_profile=False, # Disable profiling for AMD tests ) results = result_tuple[0] success = result_tuple[1] @@ -125,7 +124,7 @@ class TestNightlyDeepseekV32BasicPerformance(unittest.TestCase): if avg_spec_accept_length is not None: print(f" avg_spec_accept_length={avg_spec_accept_length:.2f}") - # Use simplified report format without traces + # Use the simplified report format if results: self.runner.full_report += ( generate_simple_markdown_report(results) + "\n" diff --git a/test/registered/amd/perf/mi30x/test_deepseek_v32_mtp_perf_amd.py b/test/registered/amd/perf/mi30x/test_deepseek_v32_mtp_perf_amd.py index 0dc0c7b52..7342ab3a8 100644 --- a/test/registered/amd/perf/mi30x/test_deepseek_v32_mtp_perf_amd.py +++ b/test/registered/amd/perf/mi30x/test_deepseek_v32_mtp_perf_amd.py @@ -27,7 +27,7 @@ register_amd_ci( def generate_simple_markdown_report(results: List[BenchmarkResult]) -> str: - """Generate a simplified markdown report without traces and cost columns. + """Generate a simplified markdown report without cost columns. Skips the first result if it's a warmup run (duplicate batch_size). """ @@ -61,7 +61,7 @@ def generate_simple_markdown_report(results: List[BenchmarkResult]) -> str: DEEPSEEK_V32_MODEL_PATH = os.environ.get( "DEEPSEEK_V32_MODEL_PATH", "deepseek-ai/DeepSeek-V3.2" ) -PROFILE_DIR = "performance_profiles_deepseek_v32_mtp_mi325" +RESULT_DIR = "performance_results_deepseek_v32_mtp_mi325" class TestNightlyDeepseekV32MTPPerformance(unittest.TestCase): @@ -108,9 +108,9 @@ class TestNightlyDeepseekV32MTPPerformance(unittest.TestCase): "env_vars": {"SGLANG_USE_AITER": "1"}, } - cls.runner = NightlyBenchmarkRunner(PROFILE_DIR, cls.__name__, cls.base_url) - cls.runner.setup_profile_directory() - # Override full_report to remove traces help text + cls.runner = NightlyBenchmarkRunner(RESULT_DIR, cls.__name__, cls.base_url) + cls.runner.setup_result_directory() + # Set the report header for this test cls.runner.full_report = f"## {cls.__name__}\n" def test_bench_one_batch(self): @@ -124,7 +124,6 @@ class TestNightlyDeepseekV32MTPPerformance(unittest.TestCase): other_args=self.variant_config["other_args"], variant=self.variant_config["name"], extra_bench_args=["--trust-remote-code"], - enable_profile=False, # Disable profiling for AMD tests ) results = result_tuple[0] success = result_tuple[1] @@ -134,7 +133,7 @@ class TestNightlyDeepseekV32MTPPerformance(unittest.TestCase): if avg_spec_accept_length is not None: print(f" avg_spec_accept_length={avg_spec_accept_length:.2f}") - # Use simplified report format without traces + # Use the simplified report format if results: self.runner.full_report += ( generate_simple_markdown_report(results) + "\n" diff --git a/test/registered/amd/perf/mi30x/test_deepseek_v3_perf.py b/test/registered/amd/perf/mi30x/test_deepseek_v3_perf.py index 6f0cd52a1..95133c3a6 100644 --- a/test/registered/amd/perf/mi30x/test_deepseek_v3_perf.py +++ b/test/registered/amd/perf/mi30x/test_deepseek_v3_perf.py @@ -22,7 +22,7 @@ register_amd_ci(est_time=18000, suite="nightly-perf-8-gpu-deepseek-v3", nightly= def generate_simple_markdown_report(results: List[BenchmarkResult]) -> str: - """Generate a simplified markdown report without traces and cost columns.""" + """Generate a simplified markdown report without cost columns.""" model_header = results[0].model_path if results[0].run_name and results[0].run_name != "default": model_header += f" ({results[0].run_name})" @@ -46,7 +46,7 @@ def generate_simple_markdown_report(results: List[BenchmarkResult]) -> str: DEEPSEEK_V3_MODEL_PATH = os.environ.get( "DEEPSEEK_V3_MODEL_PATH", "deepseek-ai/DeepSeek-V3-0324" ) -PROFILE_DIR = "performance_profiles_deepseek_v3" +RESULT_DIR = "performance_results_deepseek_v3" class TestNightlyDeepseekV3Performance(unittest.TestCase): @@ -99,9 +99,9 @@ class TestNightlyDeepseekV3Performance(unittest.TestCase): }, ] - cls.runner = NightlyBenchmarkRunner(PROFILE_DIR, cls.__name__, cls.base_url) - cls.runner.setup_profile_directory() - # Override full_report to remove traces help text + cls.runner = NightlyBenchmarkRunner(RESULT_DIR, cls.__name__, cls.base_url) + cls.runner.setup_result_directory() + # Set the report header for this test cls.runner.full_report = f"## {cls.__name__}\n" def test_bench_one_batch(self): @@ -119,7 +119,6 @@ class TestNightlyDeepseekV3Performance(unittest.TestCase): other_args=variant_config["other_args"], variant=variant_config["name"], extra_bench_args=["--trust-remote-code"], - enable_profile=False, # Disable profiling for AMD tests ) results = result_tuple[0] success = result_tuple[1] @@ -127,7 +126,7 @@ class TestNightlyDeepseekV3Performance(unittest.TestCase): if not success: failed_variants.append(variant_config["name"]) - # Use simplified report format without traces + # Use the simplified report format if results: self.runner.full_report += ( generate_simple_markdown_report(results) + "\n" diff --git a/test/registered/amd/perf/mi30x/test_glm51_perf_amd.py b/test/registered/amd/perf/mi30x/test_glm51_perf_amd.py index 4a2d43004..439958641 100644 --- a/test/registered/amd/perf/mi30x/test_glm51_perf_amd.py +++ b/test/registered/amd/perf/mi30x/test_glm51_perf_amd.py @@ -47,7 +47,7 @@ def generate_simple_markdown_report(results: List[BenchmarkResult]) -> str: GLM51_MODEL_PATH = os.environ.get("GLM51_MODEL_PATH", "zai-org/GLM-5.1-FP8") -PROFILE_DIR = "performance_profiles_glm51" +RESULT_DIR = "performance_results_glm51" class TestNightlyGLM51Performance(unittest.TestCase): @@ -94,8 +94,8 @@ class TestNightlyGLM51Performance(unittest.TestCase): }, } - cls.runner = NightlyBenchmarkRunner(PROFILE_DIR, cls.__name__, cls.base_url) - cls.runner.setup_profile_directory() + cls.runner = NightlyBenchmarkRunner(RESULT_DIR, cls.__name__, cls.base_url) + cls.runner.setup_result_directory() cls.runner.full_report = f"## {cls.__name__}\n" def test_bench_glm51(self): @@ -113,7 +113,6 @@ class TestNightlyGLM51Performance(unittest.TestCase): other_args=self.model_config["other_args"], variant=self.model_config["name"], extra_bench_args=["--trust-remote-code"], - enable_profile=False, timeout=5400, ) results = result_tuple[0] diff --git a/test/registered/amd/perf/mi30x/test_glm5_perf_amd.py b/test/registered/amd/perf/mi30x/test_glm5_perf_amd.py index 216d8ac21..4707d5d3e 100644 --- a/test/registered/amd/perf/mi30x/test_glm5_perf_amd.py +++ b/test/registered/amd/perf/mi30x/test_glm5_perf_amd.py @@ -48,7 +48,7 @@ def generate_simple_markdown_report(results: List[BenchmarkResult]) -> str: GLM5_MODEL_PATH = os.environ.get("GLM5_MODEL_PATH", "zai-org/GLM-5-FP8") -PROFILE_DIR = "performance_profiles_glm5" +RESULT_DIR = "performance_results_glm5" class TestNightlyGLM5Performance(unittest.TestCase): @@ -95,8 +95,8 @@ class TestNightlyGLM5Performance(unittest.TestCase): }, } - cls.runner = NightlyBenchmarkRunner(PROFILE_DIR, cls.__name__, cls.base_url) - cls.runner.setup_profile_directory() + cls.runner = NightlyBenchmarkRunner(RESULT_DIR, cls.__name__, cls.base_url) + cls.runner.setup_result_directory() cls.runner.full_report = f"## {cls.__name__}\n" def test_bench_glm5(self): @@ -115,7 +115,6 @@ class TestNightlyGLM5Performance(unittest.TestCase): other_args=self.model_config["other_args"], variant=self.model_config["name"], extra_bench_args=["--trust-remote-code"], - enable_profile=False, timeout=5400, ) results = result_tuple[0] diff --git a/test/registered/amd/perf/mi30x/test_grok1_fp8_perf.py b/test/registered/amd/perf/mi30x/test_grok1_fp8_perf.py index 7e04096eb..e80b5a0ce 100644 --- a/test/registered/amd/perf/mi30x/test_grok1_fp8_perf.py +++ b/test/registered/amd/perf/mi30x/test_grok1_fp8_perf.py @@ -24,7 +24,7 @@ register_amd_ci(est_time=1500, suite="nightly-perf-8-gpu-grok1-fp8", nightly=Tru def generate_simple_markdown_report(results: List[BenchmarkResult]) -> str: - """Generate a simplified markdown report without traces and cost columns.""" + """Generate a simplified markdown report without cost columns.""" model_header = results[0].model_path if results[0].run_name and results[0].run_name != "default": model_header += f" ({results[0].run_name})" @@ -47,7 +47,7 @@ def generate_simple_markdown_report(results: List[BenchmarkResult]) -> str: # Model and tokenizer paths can be overridden via environment variables GROK1_MODEL_PATH = os.environ.get("GROK1_MODEL_PATH", "lmzheng/grok-1") GROK1_TOKENIZER_PATH = os.environ.get("GROK1_TOKENIZER_PATH", "Xenova/grok-1-tokenizer") -PROFILE_DIR = "performance_profiles_grok1_fp8" +RESULT_DIR = "performance_results_grok1_fp8" class TestNightlyGrok1FP8Performance(unittest.TestCase): @@ -87,8 +87,8 @@ class TestNightlyGrok1FP8Performance(unittest.TestCase): }, } - cls.runner = NightlyBenchmarkRunner(PROFILE_DIR, cls.__name__, cls.base_url) - cls.runner.setup_profile_directory() + cls.runner = NightlyBenchmarkRunner(RESULT_DIR, cls.__name__, cls.base_url) + cls.runner.setup_result_directory() cls.runner.full_report = f"## {cls.__name__}\n" def test_bench_grok1_fp8(self): @@ -109,7 +109,6 @@ class TestNightlyGrok1FP8Performance(unittest.TestCase): other_args=self.model_config["other_args"], variant=self.model_config["name"], extra_bench_args=["--trust-remote-code"], - enable_profile=False, # Disable profiling for AMD tests ) results = result_tuple[0] success = result_tuple[1] diff --git a/test/registered/amd/perf/mi30x/test_grok1_int4_perf.py b/test/registered/amd/perf/mi30x/test_grok1_int4_perf.py index 07c67c0da..c4b892fca 100644 --- a/test/registered/amd/perf/mi30x/test_grok1_int4_perf.py +++ b/test/registered/amd/perf/mi30x/test_grok1_int4_perf.py @@ -24,7 +24,7 @@ register_amd_ci(est_time=1500, suite="nightly-perf-8-gpu-grok1-int4", nightly=Tr def generate_simple_markdown_report(results: List[BenchmarkResult]) -> str: - """Generate a simplified markdown report without traces and cost columns. + """Generate a simplified markdown report without cost columns. Skips the first result if it's a warmup run (duplicate batch_size). """ @@ -57,7 +57,7 @@ def generate_simple_markdown_report(results: List[BenchmarkResult]) -> str: # Model and tokenizer paths can be overridden via environment variables GROK1_MODEL_PATH = os.environ.get("GROK1_MODEL_PATH", "amd/grok-1-W4A8KV8") GROK1_TOKENIZER_PATH = os.environ.get("GROK1_TOKENIZER_PATH", "Xenova/grok-1-tokenizer") -PROFILE_DIR = "performance_profiles_grok1_int4" +RESULT_DIR = "performance_results_grok1_int4" class TestNightlyGrok1INT4Performance(unittest.TestCase): @@ -97,8 +97,8 @@ class TestNightlyGrok1INT4Performance(unittest.TestCase): }, } - cls.runner = NightlyBenchmarkRunner(PROFILE_DIR, cls.__name__, cls.base_url) - cls.runner.setup_profile_directory() + cls.runner = NightlyBenchmarkRunner(RESULT_DIR, cls.__name__, cls.base_url) + cls.runner.setup_result_directory() cls.runner.full_report = f"## {cls.__name__}\n" def test_bench_grok1_int4(self): @@ -119,7 +119,6 @@ class TestNightlyGrok1INT4Performance(unittest.TestCase): other_args=self.model_config["other_args"], variant=self.model_config["name"], extra_bench_args=["--trust-remote-code"], - enable_profile=False, # Disable profiling for AMD tests ) results = result_tuple[0] success = result_tuple[1] diff --git a/test/registered/amd/perf/mi30x/test_grok2_perf.py b/test/registered/amd/perf/mi30x/test_grok2_perf.py index af089ff50..e2bd82c8e 100644 --- a/test/registered/amd/perf/mi30x/test_grok2_perf.py +++ b/test/registered/amd/perf/mi30x/test_grok2_perf.py @@ -24,7 +24,7 @@ register_amd_ci(est_time=1500, suite="nightly-perf-8-gpu-grok2", nightly=True) def generate_simple_markdown_report(results: List[BenchmarkResult]) -> str: - """Generate a simplified markdown report without traces and cost columns. + """Generate a simplified markdown report without cost columns. Skips the first result if it's a warmup run (duplicate batch_size). """ @@ -59,7 +59,7 @@ GROK2_MODEL_PATH = os.environ.get("GROK2_MODEL_PATH", "xai-org/grok-2") GROK2_TOKENIZER_PATH = os.environ.get( "GROK2_TOKENIZER_PATH", "alvarobartt/grok-2-tokenizer" ) -PROFILE_DIR = "performance_profiles_grok2" +RESULT_DIR = "performance_results_grok2" class TestNightlyGrok2Performance(unittest.TestCase): @@ -99,8 +99,8 @@ class TestNightlyGrok2Performance(unittest.TestCase): }, } - cls.runner = NightlyBenchmarkRunner(PROFILE_DIR, cls.__name__, cls.base_url) - cls.runner.setup_profile_directory() + cls.runner = NightlyBenchmarkRunner(RESULT_DIR, cls.__name__, cls.base_url) + cls.runner.setup_result_directory() cls.runner.full_report = f"## {cls.__name__}\n" def test_bench_grok2(self): @@ -121,7 +121,6 @@ class TestNightlyGrok2Performance(unittest.TestCase): other_args=self.model_config["other_args"], variant=self.model_config["name"], extra_bench_args=["--trust-remote-code"], - enable_profile=False, # Disable profiling for AMD tests ) results = result_tuple[0] success = result_tuple[1] diff --git a/test/registered/amd/perf/mi30x/test_kimi_k26_perf_amd.py b/test/registered/amd/perf/mi30x/test_kimi_k26_perf_amd.py index f14d85e04..4b68b31d1 100644 --- a/test/registered/amd/perf/mi30x/test_kimi_k26_perf_amd.py +++ b/test/registered/amd/perf/mi30x/test_kimi_k26_perf_amd.py @@ -28,7 +28,7 @@ register_amd_ci(est_time=5400, suite="nightly-perf-8-gpu-kimi-k26", nightly=True def generate_simple_markdown_report(results: List[BenchmarkResult]) -> str: - """Generate a simplified markdown report without traces and cost columns. + """Generate a simplified markdown report without cost columns. Skips the first result if it's a warmup run (duplicate batch_size). """ @@ -58,7 +58,7 @@ def generate_simple_markdown_report(results: List[BenchmarkResult]) -> str: KIMI_K26_MODEL_PATH = os.environ.get("KIMI_K26_MODEL_PATH", "moonshotai/Kimi-K2.6") -PROFILE_DIR = "performance_profiles_kimi_k26" +RESULT_DIR = "performance_results_kimi_k26" class TestNightlyKimiK26Performance(unittest.TestCase): @@ -101,8 +101,8 @@ class TestNightlyKimiK26Performance(unittest.TestCase): }, } - cls.runner = NightlyBenchmarkRunner(PROFILE_DIR, cls.__name__, cls.base_url) - cls.runner.setup_profile_directory() + cls.runner = NightlyBenchmarkRunner(RESULT_DIR, cls.__name__, cls.base_url) + cls.runner.setup_result_directory() cls.runner.full_report = f"## {cls.__name__}\n" def test_bench_kimi_k26(self): @@ -121,7 +121,6 @@ class TestNightlyKimiK26Performance(unittest.TestCase): other_args=self.model_config["other_args"], variant=self.model_config["name"], extra_bench_args=["--trust-remote-code"], - enable_profile=False, timeout=5400, ) results = result_tuple[0] diff --git a/test/registered/amd/perf/mi30x/test_minimax_m25_perf_amd.py b/test/registered/amd/perf/mi30x/test_minimax_m25_perf_amd.py index ace3c8cef..ccd1c3063 100644 --- a/test/registered/amd/perf/mi30x/test_minimax_m25_perf_amd.py +++ b/test/registered/amd/perf/mi30x/test_minimax_m25_perf_amd.py @@ -23,7 +23,7 @@ register_amd_ci(est_time=5400, suite="nightly-perf-8-gpu-minimax-m25", nightly=T def generate_simple_markdown_report(results: List[BenchmarkResult]) -> str: - """Generate a simplified markdown report without traces and cost columns. + """Generate a simplified markdown report without cost columns. Skips the first result if it's a warmup run (duplicate batch_size). """ @@ -55,7 +55,7 @@ def generate_simple_markdown_report(results: List[BenchmarkResult]) -> str: MINIMAX_M25_MODEL_PATH = os.environ.get( "MINIMAX_M25_MODEL_PATH", "MiniMaxAI/MiniMax-M2.5" ) -PROFILE_DIR = "performance_profiles_minimax_m25" +RESULT_DIR = "performance_results_minimax_m25" class TestNightlyMiniMaxM25Performance(unittest.TestCase): @@ -94,8 +94,8 @@ class TestNightlyMiniMaxM25Performance(unittest.TestCase): }, } - cls.runner = NightlyBenchmarkRunner(PROFILE_DIR, cls.__name__, cls.base_url) - cls.runner.setup_profile_directory() + cls.runner = NightlyBenchmarkRunner(RESULT_DIR, cls.__name__, cls.base_url) + cls.runner.setup_result_directory() cls.runner.full_report = f"## {cls.__name__}\n" def test_bench_minimax_m25(self): @@ -115,7 +115,6 @@ class TestNightlyMiniMaxM25Performance(unittest.TestCase): other_args=self.model_config["other_args"], variant=self.model_config["name"], extra_bench_args=["--trust-remote-code"], - enable_profile=False, timeout=5400, ) results = result_tuple[0] diff --git a/test/registered/amd/perf/mi30x/test_minimax_m27_perf_amd.py b/test/registered/amd/perf/mi30x/test_minimax_m27_perf_amd.py index 6981432ea..7755518e7 100644 --- a/test/registered/amd/perf/mi30x/test_minimax_m27_perf_amd.py +++ b/test/registered/amd/perf/mi30x/test_minimax_m27_perf_amd.py @@ -23,7 +23,7 @@ register_amd_ci(est_time=5400, suite="nightly-perf-8-gpu-minimax-m27", nightly=T def generate_simple_markdown_report(results: List[BenchmarkResult]) -> str: - """Generate a simplified markdown report without traces and cost columns. + """Generate a simplified markdown report without cost columns. Skips the first result if it's a warmup run (duplicate batch_size). """ @@ -55,7 +55,7 @@ def generate_simple_markdown_report(results: List[BenchmarkResult]) -> str: MINIMAX_M27_MODEL_PATH = os.environ.get( "MINIMAX_M27_MODEL_PATH", "MiniMaxAI/MiniMax-M2.7" ) -PROFILE_DIR = "performance_profiles_minimax_m27" +RESULT_DIR = "performance_results_minimax_m27" class TestNightlyMiniMaxM27Performance(unittest.TestCase): @@ -94,8 +94,8 @@ class TestNightlyMiniMaxM27Performance(unittest.TestCase): }, } - cls.runner = NightlyBenchmarkRunner(PROFILE_DIR, cls.__name__, cls.base_url) - cls.runner.setup_profile_directory() + cls.runner = NightlyBenchmarkRunner(RESULT_DIR, cls.__name__, cls.base_url) + cls.runner.setup_result_directory() cls.runner.full_report = f"## {cls.__name__}\n" def test_bench_minimax_m27(self): @@ -115,7 +115,6 @@ class TestNightlyMiniMaxM27Performance(unittest.TestCase): other_args=self.model_config["other_args"], variant=self.model_config["name"], extra_bench_args=["--trust-remote-code"], - enable_profile=False, timeout=5400, ) results = result_tuple[0] diff --git a/test/registered/amd/perf/mi30x/test_qwen35_fp8_perf_amd.py b/test/registered/amd/perf/mi30x/test_qwen35_fp8_perf_amd.py index be5314a64..66d93159a 100644 --- a/test/registered/amd/perf/mi30x/test_qwen35_fp8_perf_amd.py +++ b/test/registered/amd/perf/mi30x/test_qwen35_fp8_perf_amd.py @@ -24,7 +24,7 @@ register_amd_ci(est_time=5400, suite="nightly-perf-8-gpu-qwen35-fp8", nightly=Tr def generate_simple_markdown_report(results: List[BenchmarkResult]) -> str: - """Generate a simplified markdown report without traces and cost columns. + """Generate a simplified markdown report without cost columns. Skips the first result if it's a warmup run (duplicate batch_size). """ @@ -56,7 +56,7 @@ def generate_simple_markdown_report(results: List[BenchmarkResult]) -> str: QWEN35_FP8_MODEL_PATH = os.environ.get( "QWEN35_FP8_MODEL_PATH", "Qwen/Qwen3.5-397B-A17B-FP8" ) -PROFILE_DIR = "performance_profiles_qwen35_fp8" +RESULT_DIR = "performance_results_qwen35_fp8" class TestNightlyQwen35Fp8Performance(unittest.TestCase): @@ -94,8 +94,8 @@ class TestNightlyQwen35Fp8Performance(unittest.TestCase): }, } - cls.runner = NightlyBenchmarkRunner(PROFILE_DIR, cls.__name__, cls.base_url) - cls.runner.setup_profile_directory() + cls.runner = NightlyBenchmarkRunner(RESULT_DIR, cls.__name__, cls.base_url) + cls.runner.setup_result_directory() cls.runner.full_report = f"## {cls.__name__}\n" def test_bench_qwen35_fp8(self): @@ -114,7 +114,6 @@ class TestNightlyQwen35Fp8Performance(unittest.TestCase): other_args=self.model_config["other_args"], variant=self.model_config["name"], extra_bench_args=["--trust-remote-code"], - enable_profile=False, timeout=5400, ) results = result_tuple[0] diff --git a/test/registered/amd/perf/mi30x/test_text_models_perf_amd.py b/test/registered/amd/perf/mi30x/test_text_models_perf_amd.py index 66b90a52f..913dfb6ce 100644 --- a/test/registered/amd/perf/mi30x/test_text_models_perf_amd.py +++ b/test/registered/amd/perf/mi30x/test_text_models_perf_amd.py @@ -25,11 +25,11 @@ from sglang.test.test_utils import ( # Register for AMD CI - Text models benchmark (~60 min) register_amd_ci(est_time=3600, suite="nightly-amd-perf-text-2-gpu", nightly=True) -PROFILE_DIR = "performance_profiles_text_models_amd" +RESULT_DIR = "performance_results_text_models_amd" def generate_simple_markdown_report(results: List[BenchmarkResult]) -> str: - """Generate a simplified markdown report without traces and cost columns. + """Generate a simplified markdown report without cost columns. Skips the first result if it's a warmup run (duplicate batch_size). """ @@ -89,8 +89,8 @@ class TestNightlyTextModelsPerfAMD(unittest.TestCase): cls.batch_sizes = [1, 1, 8, 16, 64] cls.input_lens = tuple(_parse_int_list_env("NIGHTLY_INPUT_LENS", "4096")) cls.output_lens = tuple(_parse_int_list_env("NIGHTLY_OUTPUT_LENS", "512")) - cls.runner = NightlyBenchmarkRunner(PROFILE_DIR, cls.__name__, cls.base_url) - cls.runner.setup_profile_directory() + cls.runner = NightlyBenchmarkRunner(RESULT_DIR, cls.__name__, cls.base_url) + cls.runner.setup_result_directory() cls.runner.full_report = f"## {cls.__name__}\n" def test_bench_one_batch(self): @@ -110,7 +110,6 @@ class TestNightlyTextModelsPerfAMD(unittest.TestCase): input_lens=self.input_lens, output_lens=self.output_lens, other_args=other_args, - enable_profile=False, # Disable profiling for AMD tests ) results = result_tuple[0] success = result_tuple[1] diff --git a/test/registered/amd/perf/mi30x/test_vlms_perf_amd.py b/test/registered/amd/perf/mi30x/test_vlms_perf_amd.py index fe638ae97..9088fe39c 100644 --- a/test/registered/amd/perf/mi30x/test_vlms_perf_amd.py +++ b/test/registered/amd/perf/mi30x/test_vlms_perf_amd.py @@ -26,7 +26,7 @@ from sglang.test.test_utils import ( # Register for AMD CI - VLM models benchmark (~120 min) register_amd_ci(est_time=7200, suite="nightly-amd-perf-vlm-2-gpu", nightly=True) -PROFILE_DIR = "performance_profiles_vlms_amd" +RESULT_DIR = "performance_results_vlms_amd" # VLM models suitable for AMD MODEL_DEFAULTS = [ @@ -42,7 +42,7 @@ MODEL_DEFAULTS = [ def generate_simple_markdown_report(results: List[BenchmarkResult]) -> str: - """Generate a simplified markdown report without traces and cost columns. + """Generate a simplified markdown report without cost columns. Skips the first result if it's a warmup run (duplicate batch_size). """ @@ -95,8 +95,8 @@ class TestNightlyVLMsPerfAMD(unittest.TestCase): cls.batch_sizes = _parse_int_list_env("NIGHTLY_VLM_BATCH_SIZES", "1,1,2,8,16") cls.input_lens = tuple(_parse_int_list_env("NIGHTLY_VLM_INPUT_LENS", "4096")) cls.output_lens = tuple(_parse_int_list_env("NIGHTLY_VLM_OUTPUT_LENS", "512")) - cls.runner = NightlyBenchmarkRunner(PROFILE_DIR, cls.__name__, cls.base_url) - cls.runner.setup_profile_directory() + cls.runner = NightlyBenchmarkRunner(RESULT_DIR, cls.__name__, cls.base_url) + cls.runner.setup_result_directory() cls.runner.full_report = f"## {cls.__name__}\n" def test_bench_one_batch(self): @@ -123,7 +123,6 @@ class TestNightlyVLMsPerfAMD(unittest.TestCase): output_lens=self.output_lens, other_args=other_args, extra_bench_args=extra_bench_args, - enable_profile=False, # Disable profiling for AMD tests ) results = result_tuple[0] success = result_tuple[1] diff --git a/test/registered/amd/perf/mi35x/test_deepseek_r1_mxfp4_ar_fusion_perf_mi35x.py b/test/registered/amd/perf/mi35x/test_deepseek_r1_mxfp4_ar_fusion_perf_mi35x.py index 0091d961d..5a51ecd06 100644 --- a/test/registered/amd/perf/mi35x/test_deepseek_r1_mxfp4_ar_fusion_perf_mi35x.py +++ b/test/registered/amd/perf/mi35x/test_deepseek_r1_mxfp4_ar_fusion_perf_mi35x.py @@ -24,7 +24,7 @@ register_amd_ci( def generate_simple_markdown_report(results: List[BenchmarkResult]) -> str: - """Generate a simplified markdown report without traces and cost columns. + """Generate a simplified markdown report without cost columns. Skips the first result if it's a warmup run (duplicate batch_size). """ @@ -54,7 +54,7 @@ def generate_simple_markdown_report(results: List[BenchmarkResult]) -> str: return summary -PROFILE_DIR = "performance_profiles_deepseek_r1_mxfp4_ar_fusion_mi35x" +RESULT_DIR = "performance_results_deepseek_r1_mxfp4_ar_fusion_mi35x" class TestDeepseekR1MXFP4ArFusionPerfMI35x(unittest.TestCase): @@ -90,8 +90,8 @@ class TestDeepseekR1MXFP4ArFusionPerfMI35x(unittest.TestCase): }, ] - cls.runner = NightlyBenchmarkRunner(PROFILE_DIR, cls.__name__, cls.base_url) - cls.runner.setup_profile_directory() + cls.runner = NightlyBenchmarkRunner(RESULT_DIR, cls.__name__, cls.base_url) + cls.runner.setup_result_directory() cls.runner.full_report = f"## {cls.__name__}\n" def test_bench_one_batch(self): @@ -109,7 +109,6 @@ class TestDeepseekR1MXFP4ArFusionPerfMI35x(unittest.TestCase): other_args=variant_config["other_args"], variant=variant_config["name"], extra_bench_args=["--trust-remote-code"], - enable_profile=False, ) results = result_tuple[0] success = result_tuple[1] diff --git a/test/registered/amd/perf/mi35x/test_deepseek_r1_mxfp4_kv_fp8_perf_mi35x.py b/test/registered/amd/perf/mi35x/test_deepseek_r1_mxfp4_kv_fp8_perf_mi35x.py index 31dd968c6..71b1e0fb2 100644 --- a/test/registered/amd/perf/mi35x/test_deepseek_r1_mxfp4_kv_fp8_perf_mi35x.py +++ b/test/registered/amd/perf/mi35x/test_deepseek_r1_mxfp4_kv_fp8_perf_mi35x.py @@ -24,7 +24,7 @@ register_amd_ci( def generate_simple_markdown_report(results: List[BenchmarkResult]) -> str: - """Generate a simplified markdown report without traces and cost columns. + """Generate a simplified markdown report without cost columns. Skips the first result if it's a warmup run (duplicate batch_size). """ @@ -54,7 +54,7 @@ def generate_simple_markdown_report(results: List[BenchmarkResult]) -> str: return summary -PROFILE_DIR = "performance_profiles_deepseek_r1_mxfp4_kv_fp8_mi35x" +RESULT_DIR = "performance_results_deepseek_r1_mxfp4_kv_fp8_mi35x" class TestDeepseekR1MXFP4KvFp8PerfMI35x(unittest.TestCase): @@ -91,8 +91,8 @@ class TestDeepseekR1MXFP4KvFp8PerfMI35x(unittest.TestCase): }, ] - cls.runner = NightlyBenchmarkRunner(PROFILE_DIR, cls.__name__, cls.base_url) - cls.runner.setup_profile_directory() + cls.runner = NightlyBenchmarkRunner(RESULT_DIR, cls.__name__, cls.base_url) + cls.runner.setup_result_directory() cls.runner.full_report = f"## {cls.__name__}\n" def test_bench_one_batch(self): @@ -110,7 +110,6 @@ class TestDeepseekR1MXFP4KvFp8PerfMI35x(unittest.TestCase): other_args=variant_config["other_args"], variant=variant_config["name"], extra_bench_args=["--trust-remote-code"], - enable_profile=False, ) results = result_tuple[0] success = result_tuple[1] diff --git a/test/registered/amd/perf/mi35x/test_deepseek_r1_mxfp4_perf_mi35x.py b/test/registered/amd/perf/mi35x/test_deepseek_r1_mxfp4_perf_mi35x.py index c1a4864f4..c7cca2a2b 100644 --- a/test/registered/amd/perf/mi35x/test_deepseek_r1_mxfp4_perf_mi35x.py +++ b/test/registered/amd/perf/mi35x/test_deepseek_r1_mxfp4_perf_mi35x.py @@ -21,7 +21,7 @@ register_amd_ci( def generate_simple_markdown_report(results: List[BenchmarkResult]) -> str: - """Generate a simplified markdown report without traces and cost columns. + """Generate a simplified markdown report without cost columns. Skips the first result if it's a warmup run (duplicate batch_size). """ @@ -51,7 +51,7 @@ def generate_simple_markdown_report(results: List[BenchmarkResult]) -> str: return summary -PROFILE_DIR = "performance_profiles_deepseek_r1_mxfp4_mi35x" +RESULT_DIR = "performance_results_deepseek_r1_mxfp4_mi35x" class TestDeepseekR1MXFP4PerfMI35x(unittest.TestCase): @@ -88,9 +88,9 @@ class TestDeepseekR1MXFP4PerfMI35x(unittest.TestCase): }, ] - cls.runner = NightlyBenchmarkRunner(PROFILE_DIR, cls.__name__, cls.base_url) - cls.runner.setup_profile_directory() - # Override full_report to remove traces help text + cls.runner = NightlyBenchmarkRunner(RESULT_DIR, cls.__name__, cls.base_url) + cls.runner.setup_result_directory() + # Set the report header for this test cls.runner.full_report = f"## {cls.__name__}\n" def test_bench_one_batch(self): @@ -108,7 +108,6 @@ class TestDeepseekR1MXFP4PerfMI35x(unittest.TestCase): other_args=variant_config["other_args"], variant=variant_config["name"], extra_bench_args=["--trust-remote-code"], - enable_profile=False, # Disable profiling for AMD tests ) results = result_tuple[0] success = result_tuple[1] @@ -116,7 +115,7 @@ class TestDeepseekR1MXFP4PerfMI35x(unittest.TestCase): if not success: failed_variants.append(variant_config["name"]) - # Use simplified report format without traces + # Use the simplified report format if results: self.runner.full_report += ( generate_simple_markdown_report(results) + "\n" diff --git a/test/registered/amd/perf/mi35x/test_deepseek_v32_basic_perf_mi35x.py b/test/registered/amd/perf/mi35x/test_deepseek_v32_basic_perf_mi35x.py index d537e7928..712e7abf5 100644 --- a/test/registered/amd/perf/mi35x/test_deepseek_v32_basic_perf_mi35x.py +++ b/test/registered/amd/perf/mi35x/test_deepseek_v32_basic_perf_mi35x.py @@ -26,7 +26,7 @@ register_amd_ci( def generate_simple_markdown_report(results: List[BenchmarkResult]) -> str: - """Generate a simplified markdown report without traces and cost columns. + """Generate a simplified markdown report without cost columns. Skips the first result if it's a warmup run (duplicate batch_size). """ @@ -60,7 +60,7 @@ def generate_simple_markdown_report(results: List[BenchmarkResult]) -> str: DEEPSEEK_V32_MODEL_PATH = os.environ.get( "DEEPSEEK_V32_MODEL_PATH", "deepseek-ai/DeepSeek-V3.2" ) -PROFILE_DIR = "performance_profiles_deepseek_v32_basic" +RESULT_DIR = "performance_results_deepseek_v32_basic" class TestNightlyDeepseekV32BasicPerformance(unittest.TestCase): @@ -98,9 +98,9 @@ class TestNightlyDeepseekV32BasicPerformance(unittest.TestCase): ], } - cls.runner = NightlyBenchmarkRunner(PROFILE_DIR, cls.__name__, cls.base_url) - cls.runner.setup_profile_directory() - # Override full_report to remove traces help text + cls.runner = NightlyBenchmarkRunner(RESULT_DIR, cls.__name__, cls.base_url) + cls.runner.setup_result_directory() + # Set the report header for this test cls.runner.full_report = f"## {cls.__name__}\n" def test_bench_one_batch(self): @@ -114,13 +114,12 @@ class TestNightlyDeepseekV32BasicPerformance(unittest.TestCase): other_args=self.variant_config["other_args"], variant=self.variant_config["name"], extra_bench_args=["--trust-remote-code"], - enable_profile=False, # Disable profiling for AMD tests timeout=5400, # Extended timeout for large model loading ) results = result_tuple[0] success = result_tuple[1] - # Use simplified report format without traces + # Use the simplified report format if results: self.runner.full_report += ( generate_simple_markdown_report(results) + "\n" diff --git a/test/registered/amd/perf/mi35x/test_deepseek_v32_mtp_perf_mi35x.py b/test/registered/amd/perf/mi35x/test_deepseek_v32_mtp_perf_mi35x.py index ddfdea753..7c5f9c852 100644 --- a/test/registered/amd/perf/mi35x/test_deepseek_v32_mtp_perf_mi35x.py +++ b/test/registered/amd/perf/mi35x/test_deepseek_v32_mtp_perf_mi35x.py @@ -32,7 +32,7 @@ register_amd_ci( def generate_simple_markdown_report(results: List[BenchmarkResult]) -> str: - """Generate a simplified markdown report without traces and cost columns. + """Generate a simplified markdown report without cost columns. Skips the first result if it's a warmup run (duplicate batch_size). """ @@ -82,9 +82,7 @@ def _run_benchmark_with_timeout( timeout=timeout, ) try: - profile_path_prefix, json_output_file = runner.generate_profile_filename( - model_path, variant - ) + json_output_file = runner.generate_result_filename(model_path, variant) bench_args = list(extra_bench_args) if extra_bench_args else [] if variant: bench_args.extend(["--run-name", variant]) @@ -93,10 +91,8 @@ def _run_benchmark_with_timeout( batch_sizes, input_lens, output_lens, - profile_path_prefix, json_output_file, extra_args=bench_args, - enable_profile=False, # Disable profiling for AMD tests ) _, cmd_success = runner.run_benchmark_command(command, model_description) if not cmd_success: @@ -113,7 +109,7 @@ def _run_benchmark_with_timeout( DEEPSEEK_V32_MODEL_PATH = os.environ.get( "DEEPSEEK_V32_MODEL_PATH", "deepseek-ai/DeepSeek-V3.2" ) -PROFILE_DIR = "performance_profiles_deepseek_v32_mtp" +RESULT_DIR = "performance_results_deepseek_v32_mtp" SERVER_LAUNCH_TIMEOUT = 5400 @@ -160,9 +156,9 @@ class TestNightlyDeepseekV32MTPPerformance(unittest.TestCase): ], } - cls.runner = NightlyBenchmarkRunner(PROFILE_DIR, cls.__name__, cls.base_url) - cls.runner.setup_profile_directory() - # Override full_report to remove traces help text + cls.runner = NightlyBenchmarkRunner(RESULT_DIR, cls.__name__, cls.base_url) + cls.runner.setup_result_directory() + # Set the report header for this test cls.runner.full_report = f"## {cls.__name__}\n" def test_bench_one_batch(self): @@ -187,7 +183,7 @@ class TestNightlyDeepseekV32MTPPerformance(unittest.TestCase): if avg_spec_accept_length is not None: print(f" avg_spec_accept_length={avg_spec_accept_length:.2f}") - # Use simplified report format without traces + # Use the simplified report format if results: self.runner.full_report += ( generate_simple_markdown_report(results) + "\n" diff --git a/test/registered/amd/perf/mi35x/test_glm51_perf_mi35x.py b/test/registered/amd/perf/mi35x/test_glm51_perf_mi35x.py index 333a3d76c..e6ac67c05 100644 --- a/test/registered/amd/perf/mi35x/test_glm51_perf_mi35x.py +++ b/test/registered/amd/perf/mi35x/test_glm51_perf_mi35x.py @@ -45,7 +45,7 @@ def generate_simple_markdown_report(results: List[BenchmarkResult]) -> str: GLM51_MODEL_PATH = os.environ.get("GLM51_MODEL_PATH", "zai-org/GLM-5.1-FP8") -PROFILE_DIR = "performance_profiles_glm51_mi35x" +RESULT_DIR = "performance_results_glm51_mi35x" class TestGLM51PerfMI35x(unittest.TestCase): @@ -96,8 +96,8 @@ class TestGLM51PerfMI35x(unittest.TestCase): } os.environ.setdefault("SGLANG_BENCH_TIMEOUT", "3600") - cls.runner = NightlyBenchmarkRunner(PROFILE_DIR, cls.__name__, cls.base_url) - cls.runner.setup_profile_directory() + cls.runner = NightlyBenchmarkRunner(RESULT_DIR, cls.__name__, cls.base_url) + cls.runner.setup_result_directory() cls.runner.full_report = f"## {cls.__name__}\n" def test_glm51_perf(self): @@ -115,7 +115,6 @@ class TestGLM51PerfMI35x(unittest.TestCase): other_args=self.model_config["other_args"], variant=self.model_config["name"], extra_bench_args=["--trust-remote-code"], - enable_profile=False, timeout=5400, ) results = result_tuple[0] diff --git a/test/registered/amd/perf/mi35x/test_glm5_mxfp4_perf_mi35x.py b/test/registered/amd/perf/mi35x/test_glm5_mxfp4_perf_mi35x.py index f576224c8..e2f7b413f 100644 --- a/test/registered/amd/perf/mi35x/test_glm5_mxfp4_perf_mi35x.py +++ b/test/registered/amd/perf/mi35x/test_glm5_mxfp4_perf_mi35x.py @@ -25,7 +25,7 @@ register_amd_ci( def generate_simple_markdown_report(results: List[BenchmarkResult]) -> str: - """Generate a simplified markdown report without traces and cost columns. + """Generate a simplified markdown report without cost columns. Skips the first result if it's a warmup run (duplicate batch_size). """ @@ -58,7 +58,7 @@ def generate_simple_markdown_report(results: List[BenchmarkResult]) -> str: return summary -PROFILE_DIR = "performance_profiles_glm5_mxfp4_mi35x" +RESULT_DIR = "performance_results_glm5_mxfp4_mi35x" class TestGLM5MXFP4PerfMI35x(unittest.TestCase): @@ -99,8 +99,8 @@ class TestGLM5MXFP4PerfMI35x(unittest.TestCase): }, ] - cls.runner = NightlyBenchmarkRunner(PROFILE_DIR, cls.__name__, cls.base_url) - cls.runner.setup_profile_directory() + cls.runner = NightlyBenchmarkRunner(RESULT_DIR, cls.__name__, cls.base_url) + cls.runner.setup_result_directory() cls.runner.full_report = f"## {cls.__name__}\n" def test_bench_one_batch(self): @@ -124,7 +124,6 @@ class TestGLM5MXFP4PerfMI35x(unittest.TestCase): other_args=variant_config["other_args"], variant=variant_config["name"], extra_bench_args=["--trust-remote-code"], - enable_profile=False, ) results = result_tuple[0] success = result_tuple[1] diff --git a/test/registered/amd/perf/mi35x/test_glm5_perf_mi35x.py b/test/registered/amd/perf/mi35x/test_glm5_perf_mi35x.py index 4f1ee92ab..cb2af1eb0 100644 --- a/test/registered/amd/perf/mi35x/test_glm5_perf_mi35x.py +++ b/test/registered/amd/perf/mi35x/test_glm5_perf_mi35x.py @@ -44,7 +44,7 @@ def generate_simple_markdown_report(results: List[BenchmarkResult]) -> str: GLM5_MODEL_PATH = os.environ.get("GLM5_MODEL_PATH", "zai-org/GLM-5-FP8") -PROFILE_DIR = "performance_profiles_glm5_mi35x" +RESULT_DIR = "performance_results_glm5_mi35x" class TestGLM5PerfMI35x(unittest.TestCase): @@ -94,8 +94,8 @@ class TestGLM5PerfMI35x(unittest.TestCase): } os.environ.setdefault("SGLANG_BENCH_TIMEOUT", "3600") - cls.runner = NightlyBenchmarkRunner(PROFILE_DIR, cls.__name__, cls.base_url) - cls.runner.setup_profile_directory() + cls.runner = NightlyBenchmarkRunner(RESULT_DIR, cls.__name__, cls.base_url) + cls.runner.setup_result_directory() cls.runner.full_report = f"## {cls.__name__}\n" def test_glm5_perf(self): @@ -114,7 +114,6 @@ class TestGLM5PerfMI35x(unittest.TestCase): other_args=self.model_config["other_args"], variant=self.model_config["name"], extra_bench_args=["--trust-remote-code"], - enable_profile=False, timeout=5400, ) results = result_tuple[0] diff --git a/test/registered/amd/perf/mi35x/test_grok1_int4_perf_mi35x.py b/test/registered/amd/perf/mi35x/test_grok1_int4_perf_mi35x.py index 0e23f7b73..027a60600 100644 --- a/test/registered/amd/perf/mi35x/test_grok1_int4_perf_mi35x.py +++ b/test/registered/amd/perf/mi35x/test_grok1_int4_perf_mi35x.py @@ -21,7 +21,7 @@ register_amd_ci( def generate_simple_markdown_report(results: List[BenchmarkResult]) -> str: - """Generate a simplified markdown report without traces and cost columns. + """Generate a simplified markdown report without cost columns. Skips the first result if it's a warmup run (duplicate batch_size). """ @@ -54,7 +54,7 @@ def generate_simple_markdown_report(results: List[BenchmarkResult]) -> str: # Model and tokenizer paths can be overridden via environment variables GROK1_MODEL_PATH = os.environ.get("GROK1_MODEL_PATH", "amd/grok-1-W4A8KV8") GROK1_TOKENIZER_PATH = os.environ.get("GROK1_TOKENIZER_PATH", "Xenova/grok-1-tokenizer") -PROFILE_DIR = "performance_profiles_grok1_int4_mi35x" +RESULT_DIR = "performance_results_grok1_int4_mi35x" class TestGrok1INT4PerfMI35x(unittest.TestCase): @@ -90,8 +90,8 @@ class TestGrok1INT4PerfMI35x(unittest.TestCase): }, } - cls.runner = NightlyBenchmarkRunner(PROFILE_DIR, cls.__name__, cls.base_url) - cls.runner.setup_profile_directory() + cls.runner = NightlyBenchmarkRunner(RESULT_DIR, cls.__name__, cls.base_url) + cls.runner.setup_result_directory() cls.runner.full_report = f"## {cls.__name__}\n" def test_grok1_int4_perf(self): @@ -112,7 +112,6 @@ class TestGrok1INT4PerfMI35x(unittest.TestCase): other_args=self.model_config["other_args"], variant=self.model_config["name"], extra_bench_args=["--trust-remote-code"], - enable_profile=False, # Disable profiling for AMD tests ) results = result_tuple[0] success = result_tuple[1] diff --git a/test/registered/amd/perf/mi35x/test_grok2_perf_mi35x.py b/test/registered/amd/perf/mi35x/test_grok2_perf_mi35x.py index 62dc28a00..b199a010f 100644 --- a/test/registered/amd/perf/mi35x/test_grok2_perf_mi35x.py +++ b/test/registered/amd/perf/mi35x/test_grok2_perf_mi35x.py @@ -19,7 +19,7 @@ register_amd_ci(est_time=1500, suite="nightly-perf-8-gpu-mi35x-grok2", nightly=T def generate_simple_markdown_report(results: List[BenchmarkResult]) -> str: - """Generate a simplified markdown report without traces and cost columns. + """Generate a simplified markdown report without cost columns. Skips the first result if it's a warmup run (duplicate batch_size). """ @@ -54,7 +54,7 @@ GROK2_MODEL_PATH = os.environ.get("GROK2_MODEL_PATH", "xai-org/grok-2") GROK2_TOKENIZER_PATH = os.environ.get( "GROK2_TOKENIZER_PATH", "alvarobartt/grok-2-tokenizer" ) -PROFILE_DIR = "performance_profiles_grok2_mi35x" +RESULT_DIR = "performance_results_grok2_mi35x" class TestGrok2PerfMI35x(unittest.TestCase): @@ -90,8 +90,8 @@ class TestGrok2PerfMI35x(unittest.TestCase): }, } - cls.runner = NightlyBenchmarkRunner(PROFILE_DIR, cls.__name__, cls.base_url) - cls.runner.setup_profile_directory() + cls.runner = NightlyBenchmarkRunner(RESULT_DIR, cls.__name__, cls.base_url) + cls.runner.setup_result_directory() cls.runner.full_report = f"## {cls.__name__}\n" def test_grok2_perf(self): @@ -112,7 +112,6 @@ class TestGrok2PerfMI35x(unittest.TestCase): other_args=self.model_config["other_args"], variant=self.model_config["name"], extra_bench_args=["--trust-remote-code"], - enable_profile=False, # Disable profiling for AMD tests ) results = result_tuple[0] success = result_tuple[1] diff --git a/test/registered/amd/perf/mi35x/test_kimi_k26_perf_mi35x.py b/test/registered/amd/perf/mi35x/test_kimi_k26_perf_mi35x.py index 69feda327..543a004bd 100644 --- a/test/registered/amd/perf/mi35x/test_kimi_k26_perf_mi35x.py +++ b/test/registered/amd/perf/mi35x/test_kimi_k26_perf_mi35x.py @@ -28,7 +28,7 @@ register_amd_ci(est_time=5400, suite="nightly-perf-8-gpu-mi35x-kimi-k26", nightl def generate_simple_markdown_report(results: List[BenchmarkResult]) -> str: - """Generate a simplified markdown report without traces and cost columns. + """Generate a simplified markdown report without cost columns. Skips the first result if it's a warmup run (duplicate batch_size). """ @@ -58,7 +58,7 @@ def generate_simple_markdown_report(results: List[BenchmarkResult]) -> str: KIMI_K26_MODEL_PATH = os.environ.get("KIMI_K26_MODEL_PATH", "moonshotai/Kimi-K2.6") -PROFILE_DIR = "performance_profiles_kimi_k26_mi35x" +RESULT_DIR = "performance_results_kimi_k26_mi35x" class TestNightlyKimiK26PerformanceMI35x(unittest.TestCase): @@ -101,8 +101,8 @@ class TestNightlyKimiK26PerformanceMI35x(unittest.TestCase): }, } - cls.runner = NightlyBenchmarkRunner(PROFILE_DIR, cls.__name__, cls.base_url) - cls.runner.setup_profile_directory() + cls.runner = NightlyBenchmarkRunner(RESULT_DIR, cls.__name__, cls.base_url) + cls.runner.setup_result_directory() cls.runner.full_report = f"## {cls.__name__}\n" def test_bench_kimi_k26(self): @@ -121,7 +121,6 @@ class TestNightlyKimiK26PerformanceMI35x(unittest.TestCase): other_args=self.model_config["other_args"], variant=self.model_config["name"], extra_bench_args=["--trust-remote-code"], - enable_profile=False, timeout=5400, ) results = result_tuple[0] diff --git a/test/registered/amd/perf/mi35x/test_minimax_m25_perf_mi35x.py b/test/registered/amd/perf/mi35x/test_minimax_m25_perf_mi35x.py index 776fbde1c..a001f80de 100644 --- a/test/registered/amd/perf/mi35x/test_minimax_m25_perf_mi35x.py +++ b/test/registered/amd/perf/mi35x/test_minimax_m25_perf_mi35x.py @@ -25,7 +25,7 @@ register_amd_ci( def generate_simple_markdown_report(results: List[BenchmarkResult]) -> str: - """Generate a simplified markdown report without traces and cost columns. + """Generate a simplified markdown report without cost columns. Skips the first result if it's a warmup run (duplicate batch_size). """ @@ -57,7 +57,7 @@ def generate_simple_markdown_report(results: List[BenchmarkResult]) -> str: MINIMAX_M25_MODEL_PATH = os.environ.get( "MINIMAX_M25_MODEL_PATH", "MiniMaxAI/MiniMax-M2.5" ) -PROFILE_DIR = "performance_profiles_minimax_m25_mi35x" +RESULT_DIR = "performance_results_minimax_m25_mi35x" class TestNightlyMiniMaxM25PerformanceMI35x(unittest.TestCase): @@ -96,8 +96,8 @@ class TestNightlyMiniMaxM25PerformanceMI35x(unittest.TestCase): }, } - cls.runner = NightlyBenchmarkRunner(PROFILE_DIR, cls.__name__, cls.base_url) - cls.runner.setup_profile_directory() + cls.runner = NightlyBenchmarkRunner(RESULT_DIR, cls.__name__, cls.base_url) + cls.runner.setup_result_directory() cls.runner.full_report = f"## {cls.__name__}\n" def test_bench_minimax_m25(self): @@ -117,7 +117,6 @@ class TestNightlyMiniMaxM25PerformanceMI35x(unittest.TestCase): other_args=self.model_config["other_args"], variant=self.model_config["name"], extra_bench_args=["--trust-remote-code"], - enable_profile=False, timeout=5400, ) results = result_tuple[0] diff --git a/test/registered/amd/perf/mi35x/test_minimax_m27_perf_mi35x.py b/test/registered/amd/perf/mi35x/test_minimax_m27_perf_mi35x.py index 0cf9f60cc..f82892023 100644 --- a/test/registered/amd/perf/mi35x/test_minimax_m27_perf_mi35x.py +++ b/test/registered/amd/perf/mi35x/test_minimax_m27_perf_mi35x.py @@ -25,7 +25,7 @@ register_amd_ci( def generate_simple_markdown_report(results: List[BenchmarkResult]) -> str: - """Generate a simplified markdown report without traces and cost columns. + """Generate a simplified markdown report without cost columns. Skips the first result if it's a warmup run (duplicate batch_size). """ @@ -57,7 +57,7 @@ def generate_simple_markdown_report(results: List[BenchmarkResult]) -> str: MINIMAX_M27_MODEL_PATH = os.environ.get( "MINIMAX_M27_MODEL_PATH", "MiniMaxAI/MiniMax-M2.7" ) -PROFILE_DIR = "performance_profiles_minimax_m27_mi35x" +RESULT_DIR = "performance_results_minimax_m27_mi35x" class TestNightlyMiniMaxM27PerformanceMI35x(unittest.TestCase): @@ -96,8 +96,8 @@ class TestNightlyMiniMaxM27PerformanceMI35x(unittest.TestCase): }, } - cls.runner = NightlyBenchmarkRunner(PROFILE_DIR, cls.__name__, cls.base_url) - cls.runner.setup_profile_directory() + cls.runner = NightlyBenchmarkRunner(RESULT_DIR, cls.__name__, cls.base_url) + cls.runner.setup_result_directory() cls.runner.full_report = f"## {cls.__name__}\n" def test_bench_minimax_m27(self): @@ -117,7 +117,6 @@ class TestNightlyMiniMaxM27PerformanceMI35x(unittest.TestCase): other_args=self.model_config["other_args"], variant=self.model_config["name"], extra_bench_args=["--trust-remote-code"], - enable_profile=False, timeout=5400, ) results = result_tuple[0] diff --git a/test/registered/amd/perf/mi35x/test_qwen35_fp8_perf_mi35x.py b/test/registered/amd/perf/mi35x/test_qwen35_fp8_perf_mi35x.py index 1a71acd3d..1e30d8890 100644 --- a/test/registered/amd/perf/mi35x/test_qwen35_fp8_perf_mi35x.py +++ b/test/registered/amd/perf/mi35x/test_qwen35_fp8_perf_mi35x.py @@ -21,7 +21,7 @@ register_amd_ci( def generate_simple_markdown_report(results: List[BenchmarkResult]) -> str: - """Generate a simplified markdown report without traces and cost columns. + """Generate a simplified markdown report without cost columns. Skips the first result if it's a warmup run (duplicate batch_size). """ @@ -53,7 +53,7 @@ def generate_simple_markdown_report(results: List[BenchmarkResult]) -> str: QWEN35_FP8_MODEL_PATH = os.environ.get( "QWEN35_FP8_MODEL_PATH", "Qwen/Qwen3.5-397B-A17B-FP8" ) -PROFILE_DIR = "performance_profiles_qwen35_fp8_mi35x" +RESULT_DIR = "performance_results_qwen35_fp8_mi35x" class TestQwen35Fp8PerfMI35x(unittest.TestCase): @@ -87,8 +87,8 @@ class TestQwen35Fp8PerfMI35x(unittest.TestCase): }, } - cls.runner = NightlyBenchmarkRunner(PROFILE_DIR, cls.__name__, cls.base_url) - cls.runner.setup_profile_directory() + cls.runner = NightlyBenchmarkRunner(RESULT_DIR, cls.__name__, cls.base_url) + cls.runner.setup_result_directory() cls.runner.full_report = f"## {cls.__name__}\n" def test_qwen35_fp8_perf(self): @@ -107,7 +107,6 @@ class TestQwen35Fp8PerfMI35x(unittest.TestCase): other_args=self.model_config["other_args"], variant=self.model_config["name"], extra_bench_args=["--trust-remote-code"], - enable_profile=False, timeout=5400, ) results = result_tuple[0] diff --git a/test/registered/gb300/test_deepseek_v4_pro_fp4.py b/test/registered/gb300/test_deepseek_v4_pro_fp4.py index 3509a9d58..615e6bc83 100644 --- a/test/registered/gb300/test_deepseek_v4_pro_fp4.py +++ b/test/registered/gb300/test_deepseek_v4_pro_fp4.py @@ -136,7 +136,7 @@ class TestDeepSeekV4ProFp4(unittest.TestCase): accuracy_params=accuracy_params, performance_params=PerformanceTestParams( batch_sizes=PERFORMANCE_BATCH_SIZES[variant.variant], - profile_dir="performance_profiles_gb300", + result_dir="performance_results_gb300", ), ) except AssertionError as e: diff --git a/test/registered/gb300/test_glm52_nvfp4.py b/test/registered/gb300/test_glm52_nvfp4.py index 11c143c22..1795fb0a4 100644 --- a/test/registered/gb300/test_glm52_nvfp4.py +++ b/test/registered/gb300/test_glm52_nvfp4.py @@ -61,7 +61,7 @@ class TestGlm52Nvfp4(unittest.TestCase): test_name="GLM-5.2-NVFP4", accuracy_params=AccuracyTestParams(dataset="gsm8k", baseline_accuracy=0.92), performance_params=PerformanceTestParams( - profile_dir="performance_profiles_gb300", + result_dir="performance_results_gb300", ), ) diff --git a/test/registered/gb300/test_kimi_k25.py b/test/registered/gb300/test_kimi_k25.py index b2a0ffe0f..8937263d4 100644 --- a/test/registered/gb300/test_kimi_k25.py +++ b/test/registered/gb300/test_kimi_k25.py @@ -54,7 +54,7 @@ class TestKimiK25(unittest.TestCase): dataset="mmmu-pro", baseline_accuracy=0.69, repeat=1, max_tokens=32768 ), performance_params=PerformanceTestParams( - profile_dir="performance_profiles_gb300", + result_dir="performance_results_gb300", ), ) diff --git a/test/registered/gb300/test_kimi_k25_nvfp4.py b/test/registered/gb300/test_kimi_k25_nvfp4.py index d6b073b0f..094654e7e 100644 --- a/test/registered/gb300/test_kimi_k25_nvfp4.py +++ b/test/registered/gb300/test_kimi_k25_nvfp4.py @@ -69,7 +69,7 @@ class TestKimiK25Nvfp4(unittest.TestCase): dataset="mmmu-pro", baseline_accuracy=0.69, repeat=1, max_tokens=32768 ), performance_params=PerformanceTestParams( - profile_dir="performance_profiles_gb300", + result_dir="performance_results_gb300", ), ) diff --git a/test/registered/gb300/test_qwen35_fp8.py b/test/registered/gb300/test_qwen35_fp8.py index 9c8a21bae..adc372586 100644 --- a/test/registered/gb300/test_qwen35_fp8.py +++ b/test/registered/gb300/test_qwen35_fp8.py @@ -65,7 +65,7 @@ class TestQwen35Fp8(unittest.TestCase): dataset="mmmu-pro", baseline_accuracy=0.76, repeat=1, max_tokens=32768 ), performance_params=PerformanceTestParams( - profile_dir="performance_profiles_gb300", + result_dir="performance_results_gb300", ), ) diff --git a/test/registered/gb300/test_qwen35_nvfp4.py b/test/registered/gb300/test_qwen35_nvfp4.py index abaf640f4..e8d47f42a 100644 --- a/test/registered/gb300/test_qwen35_nvfp4.py +++ b/test/registered/gb300/test_qwen35_nvfp4.py @@ -74,7 +74,7 @@ class TestQwen35Nvfp4(unittest.TestCase): dataset="mmmu-pro", baseline_accuracy=0.76, repeat=1, max_tokens=32768 ), performance_params=PerformanceTestParams( - profile_dir="performance_profiles_gb300", + result_dir="performance_results_gb300", ), ) diff --git a/test/registered/perf/test_dpsk_v3_fp4_4gpu_perf.py b/test/registered/perf/test_dpsk_v3_fp4_4gpu_perf.py index 387be756b..f9aec8612 100644 --- a/test/registered/perf/test_dpsk_v3_fp4_4gpu_perf.py +++ b/test/registered/perf/test_dpsk_v3_fp4_4gpu_perf.py @@ -68,7 +68,7 @@ class TestDeepseekR1FP4Unified(unittest.TestCase): api="completion", ), performance_params=PerformanceTestParams( - profile_dir="performance_profiles_deepseek_v3_fp4", + result_dir="performance_results_deepseek_v3_fp4", ), ) diff --git a/test/registered/perf/test_gpt_oss_4gpu_perf.py b/test/registered/perf/test_gpt_oss_4gpu_perf.py index 652cd5ccf..d4a048fd6 100644 --- a/test/registered/perf/test_gpt_oss_4gpu_perf.py +++ b/test/registered/perf/test_gpt_oss_4gpu_perf.py @@ -6,7 +6,7 @@ from sglang.test.test_utils import DEFAULT_URL_FOR_TEST register_cuda_ci(est_time=600, suite="nightly-4-gpu-b200", nightly=True) -PROFILE_DIR = "performance_profiles_gpt_oss_4gpu" +RESULT_DIR = "performance_results_gpt_oss_4gpu" class TestNightlyGptOss4GpuPerformance(unittest.TestCase): @@ -29,8 +29,8 @@ class TestNightlyGptOss4GpuPerformance(unittest.TestCase): cls.batch_sizes = [1, 1, 8, 16, 64] cls.input_lens = (4096,) cls.output_lens = (512,) - cls.runner = NightlyBenchmarkRunner(PROFILE_DIR, cls.__name__, cls.base_url) - cls.runner.setup_profile_directory() + cls.runner = NightlyBenchmarkRunner(RESULT_DIR, cls.__name__, cls.base_url) + cls.runner.setup_result_directory() def test_bench_one_batch(self): all_model_succeed = True diff --git a/test/registered/perf/test_text_models_perf.py b/test/registered/perf/test_text_models_perf.py index 83689eef9..d4f538950 100644 --- a/test/registered/perf/test_text_models_perf.py +++ b/test/registered/perf/test_text_models_perf.py @@ -11,7 +11,7 @@ from sglang.test.test_utils import ( register_cuda_ci(est_time=3600, suite="nightly-perf-text-2-gpu", nightly=True) -PROFILE_DIR = "performance_profiles_text_models" +RESULT_DIR = "performance_results_text_models" class TestNightlyTextModelsPerformance(unittest.TestCase): @@ -31,8 +31,8 @@ class TestNightlyTextModelsPerformance(unittest.TestCase): cls.batch_sizes = [1, 1, 8, 16, 64] cls.input_lens = tuple(_parse_int_list_env("NIGHTLY_INPUT_LENS", "4096")) cls.output_lens = tuple(_parse_int_list_env("NIGHTLY_OUTPUT_LENS", "512")) - cls.runner = NightlyBenchmarkRunner(PROFILE_DIR, cls.__name__, cls.base_url) - cls.runner.setup_profile_directory() + cls.runner = NightlyBenchmarkRunner(RESULT_DIR, cls.__name__, cls.base_url) + cls.runner.setup_result_directory() def test_bench_one_batch(self): all_model_succeed = True diff --git a/test/registered/perf/test_vlms_perf.py b/test/registered/perf/test_vlms_perf.py index d9c7db1ee..433805a77 100644 --- a/test/registered/perf/test_vlms_perf.py +++ b/test/registered/perf/test_vlms_perf.py @@ -13,7 +13,7 @@ from sglang.test.test_utils import ( register_cuda_ci(est_time=7200, suite="nightly-perf-vlm-2-gpu", nightly=True) -PROFILE_DIR = "performance_profiles_vlms" +RESULT_DIR = "performance_results_vlms" MODEL_DEFAULTS = [ # Keep conservative defaults. Can be overridden by env NIGHTLY_VLM_MODELS @@ -52,8 +52,8 @@ class TestNightlyVLMModelsPerformance(unittest.TestCase): cls.batch_sizes = _parse_int_list_env("NIGHTLY_VLM_BATCH_SIZES", "1,1,2,8,16") cls.input_lens = tuple(_parse_int_list_env("NIGHTLY_VLM_INPUT_LENS", "4096")) cls.output_lens = tuple(_parse_int_list_env("NIGHTLY_VLM_OUTPUT_LENS", "512")) - cls.runner = NightlyBenchmarkRunner(PROFILE_DIR, cls.__name__, cls.base_url) - cls.runner.setup_profile_directory() + cls.runner = NightlyBenchmarkRunner(RESULT_DIR, cls.__name__, cls.base_url) + cls.runner.setup_result_directory() def test_bench_one_batch(self): all_model_succeed = True diff --git a/test/registered/quant/test_kimi_k25_nvfp4_eagle.py b/test/registered/quant/test_kimi_k25_nvfp4_eagle.py index 8d7756aad..c5073abb8 100644 --- a/test/registered/quant/test_kimi_k25_nvfp4_eagle.py +++ b/test/registered/quant/test_kimi_k25_nvfp4_eagle.py @@ -59,7 +59,7 @@ class TestKimiK25Nvfp4Eagle(unittest.TestCase): performance_params=PerformanceTestParams( batch_sizes=[1, 8, 16], spec_accept_length_threshold=2.8, - profile_dir="performance_profiles_kimi_k25_nvfp4_eagle", + result_dir="performance_results_kimi_k25_nvfp4_eagle", ), ) diff --git a/test/registered/quant/test_kimi_k26_nvfp4_dflash.py b/test/registered/quant/test_kimi_k26_nvfp4_dflash.py index f8fd0aaa8..47722386f 100644 --- a/test/registered/quant/test_kimi_k26_nvfp4_dflash.py +++ b/test/registered/quant/test_kimi_k26_nvfp4_dflash.py @@ -61,7 +61,7 @@ class TestKimiK26Nvfp4Dflash(unittest.TestCase): performance_params=PerformanceTestParams( batch_sizes=[1, 8, 16], spec_accept_length_threshold=2.0, - profile_dir="performance_profiles_kimi_k26_nvfp4_dflash", + result_dir="performance_results_kimi_k26_nvfp4_dflash", ), )