From c670609ac5e4b73ede1643b1956762153f2afb38 Mon Sep 17 00:00:00 2001 From: Elizaveta Martirosian Date: Thu, 4 Jun 2026 00:14:37 +0300 Subject: [PATCH] [NPU] Diffusion CI Ground Truth Generation (NPU) (#24630) Co-authored-by: github-actions[bot] Co-authored-by: Elizaveta Martirosian Co-authored-by: Elizaveta Martirosian Co-authored-by: ronnie_zheng --- .github/CODEOWNERS | 1 + .github/workflows/diffusion-ci-gt-gen-npu.yml | 214 +++++++++++++ .github/workflows/pr-test-npu.yml | 62 +--- .../sglang/multimodal_gen/test/run_suite.py | 131 ++------ .../multimodal_gen/test/run_suite_npu.py | 299 ------------------ .../server/ascend/perf_baselines_npu.json | 220 ++++++------- .../test/server/ascend/test_server_8_npu.py | 31 -- .../server/ascend/testcase_configs_npu.py | 73 ++++- .../multimodal_gen/test/server/gpu_cases.py | 96 ++++++ .../sglang/multimodal_gen/test/test_utils.py | 8 + .../diffusion/compute_diffusion_partitions.py | 14 +- .../utils/diffusion/diffusion_case_parser.py | 12 + 12 files changed, 551 insertions(+), 610 deletions(-) create mode 100644 .github/workflows/diffusion-ci-gt-gen-npu.yml delete mode 100644 python/sglang/multimodal_gen/test/run_suite_npu.py delete mode 100644 python/sglang/multimodal_gen/test/server/ascend/test_server_8_npu.py diff --git a/.github/CODEOWNERS b/.github/CODEOWNERS index 511e94097..041aa99a6 100644 --- a/.github/CODEOWNERS +++ b/.github/CODEOWNERS @@ -10,6 +10,7 @@ /python/sglang/multimodal_gen/runtime/cache @DefTruth /python/sglang/multimodal_gen/runtime/layers @mickqian @yhyang201 @BBuf @yingluosanqian @ping1jing2 /python/sglang/multimodal_gen/runtime/models/dits @mickqian @yhyang201 @BBuf @yingluosanqian @ping1jing2 +/python/sglang/multimodal_gen/test/server/ascend @ping1jing2 @ssshinigami @Makcum888e @e-martirosian /python/sglang/srt/batch_invariant_ops @Fridge003 @hebiao064 /python/sglang/srt/compilation @hebiao064 @Oasis-Git /python/sglang/srt/constrained @hnyls2002 @DarkSharpness diff --git a/.github/workflows/diffusion-ci-gt-gen-npu.yml b/.github/workflows/diffusion-ci-gt-gen-npu.yml new file mode 100644 index 000000000..b4561b9ba --- /dev/null +++ b/.github/workflows/diffusion-ci-gt-gen-npu.yml @@ -0,0 +1,214 @@ +name: Diffusion CI Ground Truth Generation (NPU) + +on: + workflow_dispatch: + inputs: + ref: + description: 'Git ref to checkout' + required: false + default: '' + type: string + case_ids: + description: 'Specific case IDs to run (space-separated, optional)' + required: false + default: '' + type: string + output_name: + description: 'Custom local output/artifact folder name. Leave empty to use defaults.' + required: false + default: '' + type: string + publish_target_dir: + description: 'Remote target directory in sgl-project/ci-data. Leave empty to use diffusion-ci/consistency_gt/sglang_generated/ascend.' + required: false + default: '' + type: string + + +concurrency: + group: diffusion-ci-gt-gen-${{ github.ref }}-${{ inputs.output_name || inputs.case_ids || 'default' }} + cancel-in-progress: true + +permissions: + contents: write + actions: read + +env: + SGLANG_IS_IN_CI: true + OUTPUT_NAME: ${{ inputs.output_name || 'diffusion-ci-outputs' }} + PUBLISH_TARGET_DIR: ${{ inputs.publish_target_dir || 'diffusion-ci/consistency_gt/sglang_generated/ascend' }} + +jobs: + + compute-diffusion-partitions: + if: github.repository == 'sgl-project/sglang' + runs-on: ubuntu-latest + outputs: + matrix-1npu: ${{ steps.compute.outputs.matrix-1npu }} + matrix-2npu: ${{ steps.compute.outputs.matrix-2npu }} + partition-count-1npu: ${{ steps.compute.outputs['partition-count-1npu'] }} + partition-count-2npu: ${{ steps.compute.outputs['partition-count-2npu'] }} + plan-1npu: ${{ steps.compute.outputs.plan-1npu }} + plan-2npu: ${{ steps.compute.outputs.plan-2npu }} + steps: + - name: Checkout code + uses: actions/checkout@v4 + with: + ref: ${{ inputs.ref || github.ref }} + + - name: Set up Python + uses: actions/setup-python@v5 + with: + python-version: '3.10' + + - name: Compute partitions + env: + USE_NPU_CONFIGS: true + id: compute + run: | + python scripts/ci/utils/diffusion/compute_diffusion_partitions.py \ + --min-time 1200 \ + --target-time 1800 \ + --max-time 2400 \ + --max-partitions 10 \ + --parametrized-only + + multimodal-diffusion-gen-1npu: + needs: compute-diffusion-partitions + if: | + needs.compute-diffusion-partitions.result == 'success' && + needs.compute-diffusion-partitions.outputs.matrix-1npu != '{"include":[]}' + runs-on: linux-aarch64-a3-2 + strategy: + fail-fast: false + matrix: ${{ fromJson(needs.compute-diffusion-partitions.outputs.matrix-1npu) }} + timeout-minutes: 150 + container: + image: swr.cn-southwest-2.myhuaweicloud.com/base_image/ascend-ci/cann:9.0.0-a3-ubuntu22.04-py3.11 + steps: + - name: Checkout code + uses: actions/checkout@v4 + + - name: Mark repository safe + run: | + git config --system --add safe.directory ${GITHUB_WORKSPACE} + + - name: Install dependencies + env: + TORCH_CACHE_URL: "http://cache-service.nginx-pypi-cache.svc.cluster.local/whl/cpu" + PYPI_CACHE_URL: "http://cache-service.nginx-pypi-cache.svc.cluster.local/pypi/simple" + UV_INDEX_URL: "http://cache-service.nginx-pypi-cache.svc.cluster.local/pypi/simple" + GITHUB_PROXY_URL: "https://gh-proxy.test.osinfra.cn/" + RUSTUP_CACHE_URL: "http://cache-service.nginx-pypi-cache.svc.cluster.local:8082" + run: | + # speed up by using infra cache services + CACHING_URL="cache-service.nginx-pypi-cache.svc.cluster.local" + sed -Ei "s@(ports|archive).ubuntu.com@${CACHING_URL}:8081@g" /etc/apt/sources.list + pip config set global.index-url http://${CACHING_URL}/pypi/simple + pip config set global.trusted-host "${CACHING_URL}" + + bash scripts/ci/npu/npu_ci_install_dependency.sh a3 diffusion + + - name: Generate outputs + env: + PARTITION_PLAN_JSON: ${{ needs.compute-diffusion-partitions.outputs.plan-1npu }} + SGLANG_USE_MODELSCOPE: true + SGLANG_IS_IN_CI: true + HF_ENDPOINT: https://hf-mirror.com + TORCH_EXTENSIONS_DIR: /tmp/torch_extensions + PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True" + STREAMS_PER_DEVICE: 32 + run: | + cd python + python -m sglang.multimodal_gen.test.scripts.gen_diffusion_ci_outputs \ + --suite 1-npu \ + --partition-id ${{ matrix.part }} \ + --total-partitions ${{ needs.compute-diffusion-partitions.outputs['partition-count-1npu'] }} \ + --partition-plan-json "$PARTITION_PLAN_JSON" \ + --out-dir ./${{ env.OUTPUT_NAME }} \ + ${{ inputs.case_ids != '' && format('--case-ids {0}', inputs.case_ids) || '' }} + + - name: Upload artifact + uses: actions/upload-artifact@v4 + with: + name: ${{ env.OUTPUT_NAME }}-1npu-part${{ matrix.part }} + path: python/${{ env.OUTPUT_NAME }} + retention-days: 7 + + - name: Publish GT images to sgl-project/ci-data + env: + GITHUB_TOKEN: ${{ secrets.GH_PAT_FOR_NIGHTLY_CI_DATA }} + run: | + python scripts/ci/utils/diffusion/publish_diffusion_gt.py \ + --source-dir python/${{ env.OUTPUT_NAME }} \ + --target-dir "${{ env.PUBLISH_TARGET_DIR }}" + + multimodal-diffusion-gen-2npu: + needs: compute-diffusion-partitions + if: | + needs.compute-diffusion-partitions.result == 'success' && + needs.compute-diffusion-partitions.outputs.matrix-2npu != '{"include":[]}' + runs-on: linux-aarch64-a3-16 + strategy: + fail-fast: false + matrix: ${{ fromJson(needs.compute-diffusion-partitions.outputs.matrix-2npu) }} + timeout-minutes: 150 + container: + image: swr.cn-southwest-2.myhuaweicloud.com/base_image/ascend-ci/cann:9.0.0-a3-ubuntu22.04-py3.11 + steps: + - name: Checkout code + uses: actions/checkout@v4 + + - name: Mark repository safe + run: | + git config --system --add safe.directory ${GITHUB_WORKSPACE} + + - name: Install dependencies + env: + TORCH_CACHE_URL: "http://cache-service.nginx-pypi-cache.svc.cluster.local/whl/cpu" + PYPI_CACHE_URL: "http://cache-service.nginx-pypi-cache.svc.cluster.local/pypi/simple" + UV_INDEX_URL: "http://cache-service.nginx-pypi-cache.svc.cluster.local/pypi/simple" + GITHUB_PROXY_URL: "https://gh-proxy.test.osinfra.cn/" + RUSTUP_CACHE_URL: "http://cache-service.nginx-pypi-cache.svc.cluster.local:8082" + run: | + # speed up by using infra cache services + CACHING_URL="cache-service.nginx-pypi-cache.svc.cluster.local" + sed -Ei "s@(ports|archive).ubuntu.com@${CACHING_URL}:8081@g" /etc/apt/sources.list + pip config set global.index-url http://${CACHING_URL}/pypi/simple + pip config set global.trusted-host "${CACHING_URL}" + + bash scripts/ci/npu/npu_ci_install_dependency.sh a3 diffusion + + - name: Generate outputs + env: + PARTITION_PLAN_JSON: ${{ needs.compute-diffusion-partitions.outputs.plan-2npu }} + SGLANG_USE_MODELSCOPE: true + SGLANG_IS_IN_CI: true + HF_ENDPOINT: https://hf-mirror.com + TORCH_EXTENSIONS_DIR: /tmp/torch_extensions + PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True" + STREAMS_PER_DEVICE: 32 + run: | + cd python + python -m sglang.multimodal_gen.test.scripts.gen_diffusion_ci_outputs \ + --suite 2-npu \ + --partition-id ${{ matrix.part }} \ + --total-partitions ${{ needs.compute-diffusion-partitions.outputs['partition-count-2npu'] }} \ + --partition-plan-json "$PARTITION_PLAN_JSON" \ + --out-dir ./${{ env.OUTPUT_NAME }} \ + ${{ inputs.case_ids != '' && format('--case-ids {0}', inputs.case_ids) || '' }} + + - name: Upload artifact + uses: actions/upload-artifact@v4 + with: + name: ${{ env.OUTPUT_NAME }}-2npu-part${{ matrix.part }} + path: python/${{ env.OUTPUT_NAME }} + retention-days: 7 + + - name: Publish GT images to sgl-project/ci-data + env: + GITHUB_TOKEN: ${{ secrets.GH_PAT_FOR_NIGHTLY_CI_DATA }} + run: | + python scripts/ci/utils/diffusion/publish_diffusion_gt.py \ + --source-dir python/${{ env.OUTPUT_NAME }} \ + --target-dir "${{ env.PUBLISH_TARGET_DIR }}" diff --git a/.github/workflows/pr-test-npu.yml b/.github/workflows/pr-test-npu.yml index 445fe3fd9..962adcf74 100644 --- a/.github/workflows/pr-test-npu.yml +++ b/.github/workflows/pr-test-npu.yml @@ -341,7 +341,7 @@ jobs: SGLANG_DIFFUSION_ARTIFACT_DIR: ${{ github.workspace }}/diffusion-failures run: | cd python - python3 sglang/multimodal_gen/test/run_suite_npu.py --suite 1-npu + python3 sglang/multimodal_gen/test/run_suite.py --suite 1-npu - name: Upload diffusion failure artifacts if: always() @@ -398,7 +398,7 @@ jobs: SGLANG_DIFFUSION_ARTIFACT_DIR: ${{ github.workspace }}/diffusion-failures run: | cd python - python3 sglang/multimodal_gen/test/run_suite_npu.py --suite 2-npu + python3 sglang/multimodal_gen/test/run_suite.py --suite 2-npu - name: Upload diffusion failure artifacts if: always() @@ -409,63 +409,6 @@ jobs: if-no-files-found: ignore retention-days: 7 - multimodal-gen-test-8-npu-a3: - needs: [check-changes, pr-gate, set-image-config] - if: needs.check-changes.outputs.multimodal_gen == 'true' - runs-on: linux-aarch64-a3-8 - container: - image: ${{ needs.set-image-config.outputs.CANN_image_a3 }} - steps: - - name: Checkout code - uses: actions/checkout@v4 - - - name: Mark repository safe - run: | - git config --system --add safe.directory ${GITHUB_WORKSPACE} - - - name: Install dependencies - env: - TORCH_CACHE_URL: "http://cache-service.nginx-pypi-cache.svc.cluster.local/whl/cpu" - PYPI_CACHE_URL: "http://cache-service.nginx-pypi-cache.svc.cluster.local/pypi/simple" - UV_INDEX_URL: "http://cache-service.nginx-pypi-cache.svc.cluster.local/pypi/simple" - GITHUB_PROXY_URL: "https://gh-proxy.test.osinfra.cn/" - RUSTUP_CACHE_URL: "http://cache-service.nginx-pypi-cache.svc.cluster.local:8082" - run: | - # speed up by using infra cache services - CACHING_URL="cache-service.nginx-pypi-cache.svc.cluster.local" - sed -Ei "s@(ports|archive).ubuntu.com@${CACHING_URL}:8081@g" /etc/apt/sources.list - pip config set global.index-url http://${CACHING_URL}/pypi/simple - pip config set global.trusted-host "${CACHING_URL}" - - bash scripts/ci/npu/npu_ci_install_dependency.sh a3 - # copy required file from our daily cache - cp ~/.cache/modelscope/hub/datasets/otavia/ShareGPT_Vicuna_unfiltered/ShareGPT_V3_unfiltered_cleaned_split.json /tmp - # copy gsm8k dataset - cp ~/.cache/modelscope/hub/datasets/tmp/test.jsonl /tmp - - - name: Run test - timeout-minutes: 60 - env: - SGLANG_USE_MODELSCOPE: true - SGLANG_IS_IN_CI: true - HF_ENDPOINT: https://hf-mirror.com - TORCH_EXTENSIONS_DIR: /tmp/torch_extensions - PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True" - STREAMS_PER_DEVICE: 32 - SGLANG_DIFFUSION_ARTIFACT_DIR: ${{ github.workspace }}/diffusion-failures - run: | - cd python - python3 sglang/multimodal_gen/test/run_suite_npu.py --suite 8-npu - - - name: Upload diffusion failure artifacts - if: always() - uses: actions/upload-artifact@v4 - with: - name: diffusion-failures-npu-8-${{ github.run_attempt }} - path: diffusion-failures/ - if-no-files-found: ignore - retention-days: 7 - pr-test-npu-finish: needs: [ @@ -478,7 +421,6 @@ jobs: multimodal-gen-test-1-npu-a3, multimodal-gen-test-2-npu-a3, - multimodal-gen-test-8-npu-a3, ] if: always() runs-on: ubuntu-latest diff --git a/python/sglang/multimodal_gen/test/run_suite.py b/python/sglang/multimodal_gen/test/run_suite.py index 50ff4527f..29f8ba4ca 100644 --- a/python/sglang/multimodal_gen/test/run_suite.py +++ b/python/sglang/multimodal_gen/test/run_suite.py @@ -19,117 +19,52 @@ from pathlib import Path import tabulate +from sglang.multimodal_gen.runtime.platforms import current_platform from sglang.multimodal_gen.runtime.utils.logging_utils import init_logger from sglang.multimodal_gen.test.partitioning import ( PartitionItem, partition_items_by_lpt, ) -from sglang.multimodal_gen.test.server.gpu_cases import ( - ONE_GPU_CASES, - TWO_GPU_CASES, -) from sglang.multimodal_gen.test.server.testcase_configs import ( BASELINE_CONFIG, DiffusionTestCase, ) -logger = init_logger(__name__) - -DEFAULT_EST_TIME_SECONDS = 300.0 -STARTUP_OVERHEAD_SECONDS = 120.0 -DEFAULT_STANDALONE_EST_TIME_SECONDS = 300.0 - -_UPDATE_WEIGHTS_FROM_DISK_TEST_FILE = "test_update_weights_from_disk.py" -_UPDATE_WEIGHTS_MODEL_PAIR_ENV = "SGLANG_MMGEN_UPDATE_WEIGHTS_PAIR" -_UPDATE_WEIGHTS_MODEL_PAIR_IDS = ( - "FLUX.2-klein-base-4B", - "Qwen-Image", -) - - -def _discover_unit_tests() -> list[str]: - unit_dir = Path(__file__).resolve().parent / "unit" - if not unit_dir.is_dir(): - return [] - return sorted( - f"../unit/{f.name}" for f in unit_dir.glob("test_*.py") if f.is_file() +# TODO: remove duplicated code +if current_platform.is_npu(): + from sglang.multimodal_gen.test.server.ascend.testcase_configs_npu import ( + _UPDATE_WEIGHTS_FROM_DISK_TEST_FILE, + COMPONENT_ACCURACY_SUITES, + DEFAULT_EST_TIME_SECONDS, + DEFAULT_STANDALONE_EST_TIME_SECONDS, + FILE_SUITES, + PARAMETRIZED_CASE_GROUPS, + STANDALONE_FILES, + STARTUP_OVERHEAD_SECONDS, + SUITES, + ) +else: + from sglang.multimodal_gen.test.server.gpu_cases import ( # noqa: F401 It is used by ci scripts + _UPDATE_WEIGHTS_FROM_DISK_TEST_FILE, + _UPDATE_WEIGHTS_MODEL_PAIR_ENV, + _UPDATE_WEIGHTS_MODEL_PAIR_IDS, + COMPONENT_ACCURACY_FILE_NUM_GPUS, + COMPONENT_ACCURACY_SUITES, + DEFAULT_EST_TIME_SECONDS, + DEFAULT_STANDALONE_EST_TIME_SECONDS, + FILE_SUITES, + ONE_GPU_CASES, + PARAMETRIZED_CASE_GROUPS, + STANDALONE_FILE_EST_TIMES, + STANDALONE_FILES, + STARTUP_OVERHEAD_SECONDS, + STRICT_SUITES, + SUITES, + TWO_GPU_CASES, ) -FILE_SUITES = { - "unit": _discover_unit_tests(), - "component-accuracy": [ - "test_component_accuracy_1_gpu.py", - "test_component_accuracy_2_gpu.py", - ], - "component-accuracy-1-gpu": [ - "test_component_accuracy_1_gpu.py", - ], - "component-accuracy-2-gpu": [ - "test_component_accuracy_2_gpu.py", - ], - "1-gpu-b200": [ - "test_server_b200.py", - ], -} - -PARAMETRIZED_CASE_GROUPS = { - "1-gpu": [ - ("test_server_1_gpu.py", ONE_GPU_CASES), - ], - "2-gpu": [ - ("test_server_2_gpu.py", TWO_GPU_CASES), - ], -} - -STANDALONE_FILES = { - "1-gpu": [ - "../cli/test_generate_t2i_perf.py", - # Temporarily disabled: 24 timeout failures since 2026-04-09 across - # multimodal-gen-test-1-gpu. Re-enable after the flakiness is fixed. - # "test_update_weights_from_disk.py", - ], - "2-gpu": [ - "test_disagg_server.py", - ], -} - -# New standalone files may omit an estimate once to learn the real CI runtime. -# CI will use a fallback estimate for sharding, run the test, then print a -# measured value that must be copied into STANDALONE_FILE_EST_TIMES. -STANDALONE_FILE_EST_TIMES = { - "1-gpu": { - "../cli/test_generate_t2i_perf.py": 240.0, - # See STANDALONE_FILES note above — temporarily disabled. - # "test_update_weights_from_disk.py": 480.0, - }, - "2-gpu": { - # Two disagg clusters × (~3 min startup + ~1 min generate) ≈ 8 min. - # Raise if CI reports a higher measured time. - "test_disagg_server.py": 600.0, - }, -} - -# Backward-compatible suite view for scripts that still operate on file lists. -SUITES = { - **FILE_SUITES, - **{ - suite: [filename for filename, _ in case_groups] - + STANDALONE_FILES.get(suite, []) - for suite, case_groups in PARAMETRIZED_CASE_GROUPS.items() - }, -} - -STRICT_SUITES = {"unit"} -COMPONENT_ACCURACY_SUITES = { - "component-accuracy", - "component-accuracy-1-gpu", - "component-accuracy-2-gpu", -} -COMPONENT_ACCURACY_FILE_NUM_GPUS = { - "test_component_accuracy_1_gpu.py": 1, - "test_component_accuracy_2_gpu.py": 2, -} +logger = init_logger(__name__) @dataclass(frozen=True) diff --git a/python/sglang/multimodal_gen/test/run_suite_npu.py b/python/sglang/multimodal_gen/test/run_suite_npu.py deleted file mode 100644 index ae31de20f..000000000 --- a/python/sglang/multimodal_gen/test/run_suite_npu.py +++ /dev/null @@ -1,299 +0,0 @@ -""" -Test runner for multimodal_gen that manages test suites and parallel execution. - -Usage: - python3 run_suite_npu.py --suite --partition-id --total-partitions - -Example: - python3 run_suite_npu.py --suite 1-npu --partition-id 0 --total-partitions 4 -""" - -import argparse -import os -import subprocess -import sys -from pathlib import Path - -import tabulate - -from sglang.multimodal_gen.runtime.utils.logging_utils import init_logger - -logger = init_logger(__name__) - -SUITES = { - "1-npu": [ - "ascend/test_server_1_npu.py", - # add new 1-npu test files here - ], - "2-npu": [ - "ascend/test_server_2_npu.py", - # add new 2-npu test files here - ], - "8-npu": [ - "ascend/test_server_8_npu.py", - # add new 8-npu test files here - ], -} - - -def parse_args(): - parser = argparse.ArgumentParser(description="Run multimodal_gen test suite") - parser.add_argument( - "--suite", - type=str, - required=True, - choices=list(SUITES.keys()), - help="The test suite to run (valid names are defined in SUITES)", - ) - parser.add_argument( - "--partition-id", - type=int, - default=0, - help="Index of the current partition (for parallel execution)", - ) - parser.add_argument( - "--total-partitions", - type=int, - default=1, - help="Total number of partitions", - ) - parser.add_argument( - "--base-dir", - type=str, - default="server", - help="Base directory for tests relative to this script's parent", - ) - parser.add_argument( - "-k", - "--filter", - type=str, - default=None, - help="Pytest filter expression (passed to pytest -k)", - ) - parser.add_argument( - "--continue-on-error", - action="store_true", - default=False, - help="Continue running remaining tests even if one fails (for CI consistency; pytest already continues by default)", - ) - return parser.parse_args() - - -def collect_test_items(files, filter_expr=None): - """Collect test item node IDs from the given files using pytest --collect-only.""" - cmd = [sys.executable, "-m", "pytest", "--collect-only", "-q"] - if filter_expr: - cmd.extend(["-k", filter_expr]) - cmd.extend(files) - - print(f"Collecting tests with command: {' '.join(cmd)}") - result = subprocess.run(cmd, capture_output=True, text=True) - - # Check for collection errors - # pytest exit codes: - # 0: success - # 1: tests collected but some had errors during collection - # 2: test execution interrupted - # 3: internal error - # 4: command line usage error - # 5: no tests collected (may be expected with filters) - if result.returncode not in (0, 5): - error_msg = ( - f"pytest --collect-only failed with exit code {result.returncode}\n" - f"Command: {' '.join(cmd)}\n" - ) - if result.stderr: - error_msg += f"stderr:\n{result.stderr}\n" - if result.stdout: - error_msg += f"stdout:\n{result.stdout}\n" - logger.error(error_msg) - raise RuntimeError(error_msg) - - if result.returncode == 5: - print( - "No tests were collected (exit code 5). This may be expected with filters." - ) - - # Parse the output to extract test node IDs - # pytest -q outputs lines like: test_file.py::TestClass::test_method[param] - test_items = [] - for line in result.stdout.strip().split("\n"): - line = line.strip() - # Skip empty lines and summary lines - if line and "::" in line and not line.startswith(("=", "-", " ")): - # Handle lines that might have extra info after the test ID - test_id = line.split()[0] if " " in line else line - if "::" in test_id: - test_items.append(test_id) - - print(f"Collected {len(test_items)} test items") - return test_items - - -def run_pytest(files, filter_expr=None, exitfirst=False): - if not files: - print("No files to run.") - return 0 - - base_cmd = [sys.executable, "-m", "pytest", "-s", "-v"] - if exitfirst: - base_cmd.append("-x") - - # Add pytest -k filter if provided - if filter_expr: - base_cmd.extend(["-k", filter_expr]) - - max_retries = 6 - # retry if the perf assertion failed, for {max_retries} times - for i in range(max_retries + 1): - cmd = list(base_cmd) - if i > 0: - cmd.append("--last-failed") - # Always include files to constrain test discovery scope - # This prevents pytest from scanning the entire rootdir and - # discovering unrelated tests that may have missing dependencies - cmd.extend(files) - - if i > 0: - print( - f"Performance assertion failed. Retrying ({i}/{max_retries}) with --last-failed..." - ) - - print(f"Running command: {' '.join(cmd)}") - - process = subprocess.Popen( - cmd, - stdout=subprocess.PIPE, - stderr=subprocess.STDOUT, - bufsize=0, - ) - - output_bytes = bytearray() - while True: - chunk = process.stdout.read(4096) - if not chunk: - break - sys.stdout.buffer.write(chunk) - sys.stdout.buffer.flush() - output_bytes.extend(chunk) - - process.wait() - returncode = process.returncode - - if returncode == 0: - return 0 - - # Exit code 5 means no tests were collected/selected - treat as success - # when using filters, since some partitions may have all tests filtered out - if returncode == 5: - print( - "No tests collected (exit code 5). This is expected when filters " - "deselect all tests in a partition. Treating as success." - ) - return 0 - - # check if the failure is due to an assertion in test_server_utils.py - full_output = output_bytes.decode("utf-8", errors="replace") - is_perf_assertion = ( - "multimodal_gen/test/server/test_server_utils.py" in full_output - and "AssertionError" in full_output - ) - - is_flaky_ci_assertion = ( - "SafetensorError" in full_output - or "FileNotFoundError" in full_output - or "TimeoutError" in full_output - ) - - is_oom_error = ( - "out of memory" in full_output.lower() - or "oom killer" in full_output.lower() - ) - - if not (is_perf_assertion or is_flaky_ci_assertion or is_oom_error): - return returncode - - print(f"Max retry exceeded") - return returncode - - -def main(): - args = parse_args() - - # 1. resolve base path - current_file_path = Path(__file__).resolve() - test_root_dir = current_file_path.parent - target_dir = test_root_dir / args.base_dir - - if not target_dir.exists(): - print(f"Error: Target directory {target_dir} does not exist.") - sys.exit(1) - - # 2. get files from suite - suite_files_rel = SUITES[args.suite] - - suite_files_abs = [] - for f_rel in suite_files_rel: - f_abs = target_dir / f_rel - if not f_abs.exists(): - msg = f"Test file {f_rel} not found in {target_dir}." - print(f"Warning: {msg} Skipping.") - continue - suite_files_abs.append(str(f_abs)) - - if not suite_files_abs: - print(f"No valid test files found for suite '{args.suite}'.") - sys.exit(0) - - # 3. collect all test items and partition by items (not files) - all_test_items = collect_test_items(suite_files_abs, filter_expr=args.filter) - - if not all_test_items: - print(f"No test items found for suite '{args.suite}'.") - sys.exit(0) - - # Partition by test items - my_items = [ - item - for i, item in enumerate(all_test_items) - if i % args.total_partitions == args.partition_id - ] - - # Print test info at beginning (similar to test/run_suite.py pretty_print_tests) - partition_info = f"{args.partition_id + 1}/{args.total_partitions} (0-based id={args.partition_id})" - headers = ["Suite", "Partition"] - rows = [[args.suite, partition_info]] - msg = tabulate.tabulate(rows, headers=headers, tablefmt="psql") + "\n" - msg += f"✅ Enabled {len(my_items)} test(s):\n" - for item in my_items: - msg += f" - {item}\n" - print(msg, flush=True) - print( - f"Suite: {args.suite} | Partition: {args.partition_id}/{args.total_partitions}" - ) - print(f"Selected {len(suite_files_abs)} files:") - for f in suite_files_abs: - print(f" - {os.path.basename(f)}") - - if not my_items: - print("No items assigned to this partition. Exiting success.") - sys.exit(0) - - print(f"Running {len(my_items)} items in this shard: {', '.join(my_items)}") - - # 4. execute with the specific test items - # Fast-fail: stop on first failure unless --continue-on-error is set - exit_code = run_pytest(my_items, exitfirst=not args.continue_on_error) - - # Print tests again at the end for visibility - msg = "\n" + tabulate.tabulate(rows, headers=headers, tablefmt="psql") + "\n" - msg += f"✅ Executed {len(my_items)} test(s):\n" - for item in my_items: - msg += f" - {item}\n" - print(msg, flush=True) - - sys.exit(exit_code) - - -if __name__ == "__main__": - main() diff --git a/python/sglang/multimodal_gen/test/server/ascend/perf_baselines_npu.json b/python/sglang/multimodal_gen/test/server/ascend/perf_baselines_npu.json index 798d4e957..0706a3f5e 100644 --- a/python/sglang/multimodal_gen/test/server/ascend/perf_baselines_npu.json +++ b/python/sglang/multimodal_gen/test/server/ascend/perf_baselines_npu.json @@ -1,7 +1,7 @@ { "metadata": { "model": "Diffusion Server", - "hardware": "CI A2 64GB pool", + "hardware": "Ascend A3", "description": "Reference numbers captured from the CI diffusion server baseline run" }, "scenarios": { @@ -72,69 +72,69 @@ }, "flux_2_image_t2i_2npu": { "stages_ms": { - "InputValidationStage": 0.06, - "TextEncodingStage": 5628.31, + "InputValidationStage": 0.08, + "TextEncodingStage": 192.4, "ImageVAEEncodingStage": 0.01, - "LatentPreparationStage": 0.75, - "TimestepPreparationStage": 30.68, - "DenoisingStage": 55002.26, - "DecodingStage": 43.73 + "LatentPreparationStage": 0.97, + "TimestepPreparationStage": 34.65, + "DenoisingStage": 45390.41, + "DecodingStage": 5.32 }, "denoise_step_ms": { - "0": 110.35, - "1": 301.82, - "2": 1139.81, - "3": 1114.17, - "4": 1099.34, - "5": 1099.12, - "6": 1100.16, - "7": 1099.67, - "8": 1099.09, - "9": 1089.81, - "10": 1109.73, - "11": 1099.97, - "12": 1100.26, - "13": 1099.67, - "14": 1099.79, - "15": 1099.6, - "16": 1100.16, - "17": 1099.87, - "18": 1100.02, - "19": 1099.34, - "20": 1099.6, - "21": 1099.45, - "22": 1100.2, - "23": 1099.29, - "24": 1098.86, - "25": 1090.38, - "26": 1109.19, - "27": 1099.67, - "28": 1100.06, - "29": 1099.22, - "30": 1100.08, - "31": 1098.86, - "32": 1099.73, - "33": 1099.11, - "34": 1100.13, - "35": 1103.97, - "36": 1095.26, - "37": 1099.38, - "38": 1099.34, - "39": 1099.17, - "40": 1100.08, - "41": 1089.89, - "42": 1106.69, - "43": 1102.57, - "44": 1100.17, - "45": 1099.21, - "46": 1100.42, - "47": 1099.38, - "48": 1099.59, - "49": 1099.47 + "0": 84.23, + "1": 80.0, + "2": 874.6, + "3": 918.81, + "4": 900.4, + "5": 910.67, + "6": 903.36, + "7": 904.97, + "8": 906.84, + "9": 906.39, + "10": 904.99, + "11": 909.96, + "12": 901.67, + "13": 908.8, + "14": 902.93, + "15": 906.17, + "16": 906.67, + "17": 905.89, + "18": 906.9, + "19": 907.1, + "20": 905.31, + "21": 907.93, + "22": 903.68, + "23": 904.49, + "24": 905.73, + "25": 907.66, + "26": 906.71, + "27": 912.69, + "28": 901.4, + "29": 909.9, + "30": 901.72, + "31": 904.35, + "32": 905.61, + "33": 905.97, + "34": 906.5, + "35": 921.45, + "36": 892.58, + "37": 909.03, + "38": 903.82, + "39": 906.25, + "40": 905.18, + "41": 905.06, + "42": 906.21, + "43": 911.62, + "44": 901.04, + "45": 909.09, + "46": 904.29, + "47": 905.16, + "48": 907.35, + "49": 905.65 }, - "expected_e2e_ms": 64195.08, - "expected_avg_denoise_ms": 1065.0, - "expected_median_denoise_ms": 1099.63 + "expected_e2e_ms": 46557.7, + "expected_avg_denoise_ms": 872.7, + "expected_median_denoise_ms": 905.81 }, "wan2_1_t2v_1.3b_1_npu": { "stages_ms": { @@ -203,61 +203,61 @@ "expected_median_denoise_ms": 537.54, "estimated_full_test_time_s": 157.8 }, - "wan2_2_t2v_14b_w8a8_8npu": { + "wan2_2_t2v_14b_w8a8_2npu": { "stages_ms": { - "InputValidationStage": 0.14, - "TextEncodingStage": 3020.73, - "LatentPreparationStage": 0.19, - "TimestepPreparationStage": 5.01, - "DenoisingStage": 82744.33, - "DecodingStage": 932.41, + "InputValidationStage": 0.09, + "TextEncodingStage": 2789.3, + "LatentPreparationStage": 0.28, + "TimestepPreparationStage": 3.19, + "DenoisingStage": 187650.19, + "DecodingStage": 3491.88, "per_frame_generation": null }, "denoise_step_ms": { - "0": 1232.32, - "1": 2091.77, - "2": 2097.62, - "3": 2087.53, - "4": 2088.54, - "5": 2087.96, - "6": 2088.28, - "7": 2089.77, - "8": 2101.9, - "9": 2088.73, - "10": 2088.04, - "11": 2087.53, - "12": 2088.89, - "13": 2087.09, - "14": 2088.25, - "15": 2087.96, - "16": 2088.24, - "17": 2088.45, - "18": 2104.7, - "19": 2088.44, - "20": 2087.19, - "21": 2088.19, - "22": 2088.37, - "23": 2087.6, - "24": 2088.13, - "25": 2088.06, - "26": 2126.23, - "27": 2089.92, - "28": 2087.37, - "29": 2089.21, - "30": 2088.29, - "31": 2087.89, - "32": 2073.1, - "33": 2086.71, - "34": 2087.88, - "35": 2088.64, - "36": 2088.1, - "37": 2089.14, - "38": 2087.5, - "39": 2087.86 + "0": 1415.84, + "1": 4801.54, + "2": 4781.57, + "3": 4770.19, + "4": 4808.86, + "5": 4737.55, + "6": 4774.02, + "7": 4774.85, + "8": 4773.25, + "9": 4775.33, + "10": 4771.31, + "11": 4773.07, + "12": 4773.18, + "13": 4772.77, + "14": 4773.68, + "15": 4771.91, + "16": 4776.16, + "17": 4773.88, + "18": 4769.87, + "19": 4772.1, + "20": 4774.55, + "21": 4772.54, + "22": 4772.39, + "23": 4776.14, + "24": 4772.84, + "25": 4772.82, + "26": 4789.34, + "27": 4792.58, + "28": 4807.52, + "29": 4740.66, + "30": 4773.46, + "31": 4774.19, + "32": 4772.8, + "33": 4774.84, + "34": 4772.52, + "35": 4773.77, + "36": 4775.43, + "37": 4771.58, + "38": 4772.68, + "39": 4769.1 }, - "expected_e2e_ms": 86719.57, - "expected_avg_denoise_ms": 2068.43, - "expected_median_denoise_ms": 2088.21 + "expected_e2e_ms": 193947.19, + "expected_avg_denoise_ms": 4691.07, + "expected_median_denoise_ms": 4773.22 }, "qwen_image_t2i_2npu": { "stages_ms": { diff --git a/python/sglang/multimodal_gen/test/server/ascend/test_server_8_npu.py b/python/sglang/multimodal_gen/test/server/ascend/test_server_8_npu.py deleted file mode 100644 index 30ae51f37..000000000 --- a/python/sglang/multimodal_gen/test/server/ascend/test_server_8_npu.py +++ /dev/null @@ -1,31 +0,0 @@ -""" -Config-driven diffusion performance test with pytest parametrization. - - -If the actual run is significantly better than the baseline, the improved cases with their updated baseline will be printed -""" - -from __future__ import annotations - -import pytest - -from sglang.multimodal_gen.runtime.utils.logging_utils import init_logger -from sglang.multimodal_gen.test.server.ascend.testcase_configs_npu import ( - EIGHT_NPU_CASES, -) -from sglang.multimodal_gen.test.server.test_server_common import ( # noqa: F401 - DiffusionServerBase, - diffusion_server, -) -from sglang.multimodal_gen.test.server.testcase_configs import DiffusionTestCase - -logger = init_logger(__name__) - - -class TestDiffusionServerEightNpu(DiffusionServerBase): - """Performance tests for 8-NPU diffusion cases.""" - - @pytest.fixture(params=EIGHT_NPU_CASES, ids=lambda c: c.id) - def case(self, request) -> DiffusionTestCase: - """Provide a DiffusionTestCase for each 8-NPU test.""" - return request.param diff --git a/python/sglang/multimodal_gen/test/server/ascend/testcase_configs_npu.py b/python/sglang/multimodal_gen/test/server/ascend/testcase_configs_npu.py index 0acadac21..6b32e8bc6 100644 --- a/python/sglang/multimodal_gen/test/server/ascend/testcase_configs_npu.py +++ b/python/sglang/multimodal_gen/test/server/ascend/testcase_configs_npu.py @@ -1,3 +1,5 @@ +import os + from sglang.multimodal_gen.test.server.testcase_configs import ( T2V_PROMPT, DiffusionSamplingParams, @@ -6,12 +8,31 @@ from sglang.multimodal_gen.test.server.testcase_configs import ( T2I_sampling_params, ) +MODEL_WEIGHTS_DIR = "/root/.cache/modelscope/hub/models/" + +FLUX_1_DEV_WEIGHTS_PATH = os.path.join( + MODEL_WEIGHTS_DIR, "black-forest-labs/FLUX.1-dev" +) +FLUX_2_DEV_WEIGHTS_PATH = os.path.join( + MODEL_WEIGHTS_DIR, "black-forest-labs/FLUX.2-dev" +) +QWEN_IMAGE_WEIGHTS_PATH = os.path.join(MODEL_WEIGHTS_DIR, "Qwen/Qwen-Image") +WAN2_1_T2V_1_3B_DIFFUSERS_WEIGHTS_PATH = os.path.join( + MODEL_WEIGHTS_DIR, "Wan-AI/Wan2.1-T2V-1.3B-Diffusers" +) +WAN2_2_T2V_A14B_DIFFUSERS_W8A8_WEIGHTS_PATH = os.path.join( + MODEL_WEIGHTS_DIR, "Eco-Tech/Wan2.2-T2V-A14B-Diffusers-w8a8" +) + +EXTRAS_DISABLE_WARMUP = ["--server-warmup", "false"] + ONE_NPU_CASES: list[DiffusionTestCase] = [ # === Text to Image (T2I) === DiffusionTestCase( "flux_image_t2i_npu", DiffusionServerArgs( - model_path="/root/.cache/modelscope/hub/models/black-forest-labs/FLUX.1-dev", + model_path=FLUX_1_DEV_WEIGHTS_PATH, + extras=EXTRAS_DISABLE_WARMUP, ), T2I_sampling_params, run_consistency_check=False, @@ -20,7 +41,8 @@ ONE_NPU_CASES: list[DiffusionTestCase] = [ DiffusionTestCase( "wan2_1_t2v_1.3b_1_npu", DiffusionServerArgs( - model_path="/root/.cache/modelscope/hub/models/Wan-AI/Wan2.1-T2V-1.3B-Diffusers", + model_path=WAN2_1_T2V_1_3B_DIFFUSERS_WEIGHTS_PATH, + extras=EXTRAS_DISABLE_WARMUP, ), DiffusionSamplingParams( prompt=T2V_PROMPT, @@ -34,9 +56,10 @@ TWO_NPU_CASES: list[DiffusionTestCase] = [ DiffusionTestCase( "flux_2_image_t2i_2npu", DiffusionServerArgs( - model_path="/root/.cache/modelscope/hub/models/black-forest-labs/FLUX.2-dev", + model_path=FLUX_2_DEV_WEIGHTS_PATH, num_gpus=2, tp_size=2, + extras=EXTRAS_DISABLE_WARMUP, ), T2I_sampling_params, run_consistency_check=False, @@ -44,26 +67,25 @@ TWO_NPU_CASES: list[DiffusionTestCase] = [ DiffusionTestCase( "qwen_image_t2i_2npu", DiffusionServerArgs( - model_path="/root/.cache/modelscope/hub/models/Qwen/Qwen-Image", + model_path=QWEN_IMAGE_WEIGHTS_PATH, num_gpus=2, # test ring attn ulysses_degree=1, ring_degree=2, + extras=EXTRAS_DISABLE_WARMUP, ), T2I_sampling_params, run_consistency_check=False, ), -] - -EIGHT_NPU_CASES: list[DiffusionTestCase] = [ # === Text to Video (T2V) === DiffusionTestCase( - "wan2_2_t2v_14b_w8a8_8npu", + "wan2_2_t2v_14b_w8a8_2npu", DiffusionServerArgs( - model_path="/root/.cache/modelscope/hub/models/Eco-Tech/Wan2.2-T2V-A14B-Diffusers-w8a8", - num_gpus=8, - tp_size=4, + model_path=WAN2_2_T2V_A14B_DIFFUSERS_W8A8_WEIGHTS_PATH, + num_gpus=2, + tp_size=1, ulysses_degree=2, + extras=EXTRAS_DISABLE_WARMUP, ), DiffusionSamplingParams( prompt=T2V_PROMPT, @@ -71,3 +93,32 @@ EIGHT_NPU_CASES: list[DiffusionTestCase] = [ run_consistency_check=False, ), ] + +DEFAULT_EST_TIME_SECONDS = 300.0 +STARTUP_OVERHEAD_SECONDS = 120.0 +DEFAULT_STANDALONE_EST_TIME_SECONDS = 300.0 + +SUITES = { + "1-npu": [ + "ascend/test_server_1_npu.py", + # add new 1-npu test files here + ], + "2-npu": [ + "ascend/test_server_2_npu.py", + # add new 2-npu test files here + ], +} + +PARAMETRIZED_CASE_GROUPS = { + "1-npu": [ + ("ascend/test_server_1_npu.py", ONE_NPU_CASES), + ], + "2-npu": [ + ("ascend/test_server_2_npu.py", TWO_NPU_CASES), + ], +} + +FILE_SUITES = {} +STANDALONE_FILES = {} +COMPONENT_ACCURACY_SUITES = {} +_UPDATE_WEIGHTS_FROM_DISK_TEST_FILE = None diff --git a/python/sglang/multimodal_gen/test/server/gpu_cases.py b/python/sglang/multimodal_gen/test/server/gpu_cases.py index 018bcead4..9f622374f 100644 --- a/python/sglang/multimodal_gen/test/server/gpu_cases.py +++ b/python/sglang/multimodal_gen/test/server/gpu_cases.py @@ -787,3 +787,99 @@ if not current_platform.is_hip(): ONE_GPU_CASES += ONE_GPU_MODELOPT_FP8_CASES TWO_GPU_CASES = _with_default_num_gpus(TWO_GPU_CASES, 2) + + +def _discover_unit_tests() -> list[str]: + unit_dir = Path(__file__).resolve().parent.parent / "unit" + if not unit_dir.is_dir(): + return [] + return sorted( + f"../unit/{f.name}" for f in unit_dir.glob("test_*.py") if f.is_file() + ) + + +FILE_SUITES = { + "unit": _discover_unit_tests(), + "component-accuracy": [ + "test_component_accuracy_1_gpu.py", + "test_component_accuracy_2_gpu.py", + ], + "component-accuracy-1-gpu": [ + "test_component_accuracy_1_gpu.py", + ], + "component-accuracy-2-gpu": [ + "test_component_accuracy_2_gpu.py", + ], + "1-gpu-b200": [ + "test_server_b200.py", + ], +} + +PARAMETRIZED_CASE_GROUPS = { + "1-gpu": [ + ("test_server_1_gpu.py", ONE_GPU_CASES), + ], + "2-gpu": [ + ("test_server_2_gpu.py", TWO_GPU_CASES), + ], +} + +STANDALONE_FILES = { + "1-gpu": [ + "../cli/test_generate_t2i_perf.py", + # Temporarily disabled: 24 timeout failures since 2026-04-09 across + # multimodal-gen-test-1-gpu. Re-enable after the flakiness is fixed. + # "test_update_weights_from_disk.py", + ], + "2-gpu": [ + "test_disagg_server.py", + ], +} + +# New standalone files may omit an estimate once to learn the real CI runtime. +# CI will use a fallback estimate for sharding, run the test, then print a +# measured value that must be copied into STANDALONE_FILE_EST_TIMES. +STANDALONE_FILE_EST_TIMES = { + "1-gpu": { + "../cli/test_generate_t2i_perf.py": 240.0, + # See STANDALONE_FILES note above — temporarily disabled. + # "test_update_weights_from_disk.py": 480.0, + }, + "2-gpu": { + # Two disagg clusters × (~3 min startup + ~1 min generate) ≈ 8 min. + # Raise if CI reports a higher measured time. + "test_disagg_server.py": 600.0, + }, +} + +# Backward-compatible suite view for scripts that still operate on file lists. +SUITES = { + **FILE_SUITES, + **{ + suite: [filename for filename, _ in case_groups] + + STANDALONE_FILES.get(suite, []) + for suite, case_groups in PARAMETRIZED_CASE_GROUPS.items() + }, +} + +STRICT_SUITES = {"unit"} +COMPONENT_ACCURACY_SUITES = { + "component-accuracy", + "component-accuracy-1-gpu", + "component-accuracy-2-gpu", +} +COMPONENT_ACCURACY_FILE_NUM_GPUS = { + "test_component_accuracy_1_gpu.py": 1, + "test_component_accuracy_2_gpu.py": 2, +} + +DEFAULT_EST_TIME_SECONDS = 300.0 +STARTUP_OVERHEAD_SECONDS = 120.0 +DEFAULT_STANDALONE_EST_TIME_SECONDS = 300.0 + +_UPDATE_WEIGHTS_FROM_DISK_TEST_FILE = "test_update_weights_from_disk.py" +_UPDATE_WEIGHTS_MODEL_PAIR_ENV = "SGLANG_MMGEN_UPDATE_WEIGHTS_PAIR" +_UPDATE_WEIGHTS_MODEL_PAIR_IDS = ( + "FLUX.2-klein-base-4B", + "Qwen-Image", +) diff --git a/python/sglang/multimodal_gen/test/test_utils.py b/python/sglang/multimodal_gen/test/test_utils.py index 6451ae425..c1b4a2ca3 100644 --- a/python/sglang/multimodal_gen/test/test_utils.py +++ b/python/sglang/multimodal_gen/test/test_utils.py @@ -45,10 +45,18 @@ SGL_TEST_FILES_OFFICIAL_CONSISTENCY_GT_BASE = ( SGL_TEST_FILES_SGLANG_CONSISTENCY_GT_BASE = ( f"{SGL_TEST_FILES_CONSISTENCY_GT_ROOT}/sglang_generated" ) +SGL_TEST_FILES_OFFICIAL_CONSISTENCY_GT_BASE_ASCEND = ( + f"{SGL_TEST_FILES_CONSISTENCY_GT_ROOT}/official_generated/ascend" +) +SGL_TEST_FILES_SGLANG_CONSISTENCY_GT_BASE_ASCEND = ( + f"{SGL_TEST_FILES_CONSISTENCY_GT_ROOT}/sglang_generated/ascend" +) SGL_TEST_FILES_CONSISTENCY_GT_BASE = SGL_TEST_FILES_SGLANG_CONSISTENCY_GT_BASE SGL_TEST_FILES_CONSISTENCY_GT_BASES = ( SGL_TEST_FILES_OFFICIAL_CONSISTENCY_GT_BASE, SGL_TEST_FILES_SGLANG_CONSISTENCY_GT_BASE, + SGL_TEST_FILES_OFFICIAL_CONSISTENCY_GT_BASE_ASCEND, + SGL_TEST_FILES_SGLANG_CONSISTENCY_GT_BASE_ASCEND, ) # LTX cases listed here compare against official-generated GT. SGL_TEST_FILES_OFFICIAL_CONSISTENCY_GT_CASES = frozenset( diff --git a/scripts/ci/utils/diffusion/compute_diffusion_partitions.py b/scripts/ci/utils/diffusion/compute_diffusion_partitions.py index c4d304ab5..013635a05 100755 --- a/scripts/ci/utils/diffusion/compute_diffusion_partitions.py +++ b/scripts/ci/utils/diffusion/compute_diffusion_partitions.py @@ -38,6 +38,12 @@ def _load_partitioning_helpers(): PartitionItem, partition_items_by_lpt = _load_partitioning_helpers() SUITE_OUTPUT_NAMES = {"1-gpu": "1gpu", "2-gpu": "2gpu", "1-gpu-b200": "b200"} + +USE_NPU_CONFIGS = os.getenv("USE_NPU_CONFIGS", "0").lower() in ("1", "true") + +if USE_NPU_CONFIGS: + SUITE_OUTPUT_NAMES = {"1-npu": "1npu", "2-npu": "2npu"} + DEFAULT_STANDALONE_EST_TIME_SECONDS = 300.0 @@ -261,7 +267,13 @@ def main(): print(f"Error: Run suite not found: {run_suite_path}") sys.exit(1) try: - case_config_path = resolve_case_config_path(repo_root, run_suite_path) + if USE_NPU_CONFIGS: + case_config_path = ( + repo_root + / "python/sglang/multimodal_gen/test/server/ascend/testcase_configs_npu.py" + ) + else: + case_config_path = resolve_case_config_path(repo_root, run_suite_path) except (RuntimeError, FileNotFoundError) as exc: print(f"Error: {exc}") sys.exit(1) diff --git a/scripts/ci/utils/diffusion/diffusion_case_parser.py b/scripts/ci/utils/diffusion/diffusion_case_parser.py index 74159e668..6383be85f 100755 --- a/scripts/ci/utils/diffusion/diffusion_case_parser.py +++ b/scripts/ci/utils/diffusion/diffusion_case_parser.py @@ -16,6 +16,7 @@ Usage: import ast import json +import os from dataclasses import dataclass from pathlib import Path from typing import Dict, List, Optional @@ -44,6 +45,17 @@ STARTUP_OVERHEAD_SECONDS = 120.0 BASELINE_REL_PATH = "python/sglang/multimodal_gen/test/server/perf_baselines.json" RUN_SUITE_REL_PATH = "python/sglang/multimodal_gen/test/run_suite.py" +USE_NPU_CONFIGS = os.getenv("USE_NPU_CONFIGS", "0").lower() in ("1", "true") + +if USE_NPU_CONFIGS: + BASELINE_REL_PATH = ( + "python/sglang/multimodal_gen/test/server/perf_baselines_npu.json" + ) + CASE_LIST_TO_SUITE = { + "ONE_NPU_CASES": "1-npu", + "TWO_NPU_CASES": "2-npu", + } + @dataclass class DiffusionCaseInfo: