[NPU] Diffusion CI Ground Truth Generation (NPU) (#24630)

Co-authored-by: github-actions[bot] <github-actions[bot]@users.noreply.github.com>
Co-authored-by: Elizaveta Martirosian <you@example.com>
Co-authored-by: Elizaveta Martirosian <elizaveta.martirosian@gmail.com>
Co-authored-by: ronnie_zheng <zl19940307@163.com>
This commit is contained in:
Elizaveta Martirosian
2026-06-04 00:14:37 +03:00
committed by GitHub
co-authored by github-actions[bot] Elizaveta Martirosian Elizaveta Martirosian ronnie_zheng
parent e485ad6ac1
commit c670609ac5
12 changed files with 551 additions and 610 deletions
+1
View File
@@ -10,6 +10,7 @@
/python/sglang/multimodal_gen/runtime/cache @DefTruth /python/sglang/multimodal_gen/runtime/cache @DefTruth
/python/sglang/multimodal_gen/runtime/layers @mickqian @yhyang201 @BBuf @yingluosanqian @ping1jing2 /python/sglang/multimodal_gen/runtime/layers @mickqian @yhyang201 @BBuf @yingluosanqian @ping1jing2
/python/sglang/multimodal_gen/runtime/models/dits @mickqian @yhyang201 @BBuf @yingluosanqian @ping1jing2 /python/sglang/multimodal_gen/runtime/models/dits @mickqian @yhyang201 @BBuf @yingluosanqian @ping1jing2
/python/sglang/multimodal_gen/test/server/ascend @ping1jing2 @ssshinigami @Makcum888e @e-martirosian
/python/sglang/srt/batch_invariant_ops @Fridge003 @hebiao064 /python/sglang/srt/batch_invariant_ops @Fridge003 @hebiao064
/python/sglang/srt/compilation @hebiao064 @Oasis-Git /python/sglang/srt/compilation @hebiao064 @Oasis-Git
/python/sglang/srt/constrained @hnyls2002 @DarkSharpness /python/sglang/srt/constrained @hnyls2002 @DarkSharpness
@@ -0,0 +1,214 @@
name: Diffusion CI Ground Truth Generation (NPU)
on:
workflow_dispatch:
inputs:
ref:
description: 'Git ref to checkout'
required: false
default: ''
type: string
case_ids:
description: 'Specific case IDs to run (space-separated, optional)'
required: false
default: ''
type: string
output_name:
description: 'Custom local output/artifact folder name. Leave empty to use defaults.'
required: false
default: ''
type: string
publish_target_dir:
description: 'Remote target directory in sgl-project/ci-data. Leave empty to use diffusion-ci/consistency_gt/sglang_generated/ascend.'
required: false
default: ''
type: string
concurrency:
group: diffusion-ci-gt-gen-${{ github.ref }}-${{ inputs.output_name || inputs.case_ids || 'default' }}
cancel-in-progress: true
permissions:
contents: write
actions: read
env:
SGLANG_IS_IN_CI: true
OUTPUT_NAME: ${{ inputs.output_name || 'diffusion-ci-outputs' }}
PUBLISH_TARGET_DIR: ${{ inputs.publish_target_dir || 'diffusion-ci/consistency_gt/sglang_generated/ascend' }}
jobs:
compute-diffusion-partitions:
if: github.repository == 'sgl-project/sglang'
runs-on: ubuntu-latest
outputs:
matrix-1npu: ${{ steps.compute.outputs.matrix-1npu }}
matrix-2npu: ${{ steps.compute.outputs.matrix-2npu }}
partition-count-1npu: ${{ steps.compute.outputs['partition-count-1npu'] }}
partition-count-2npu: ${{ steps.compute.outputs['partition-count-2npu'] }}
plan-1npu: ${{ steps.compute.outputs.plan-1npu }}
plan-2npu: ${{ steps.compute.outputs.plan-2npu }}
steps:
- name: Checkout code
uses: actions/checkout@v4
with:
ref: ${{ inputs.ref || github.ref }}
- name: Set up Python
uses: actions/setup-python@v5
with:
python-version: '3.10'
- name: Compute partitions
env:
USE_NPU_CONFIGS: true
id: compute
run: |
python scripts/ci/utils/diffusion/compute_diffusion_partitions.py \
--min-time 1200 \
--target-time 1800 \
--max-time 2400 \
--max-partitions 10 \
--parametrized-only
multimodal-diffusion-gen-1npu:
needs: compute-diffusion-partitions
if: |
needs.compute-diffusion-partitions.result == 'success' &&
needs.compute-diffusion-partitions.outputs.matrix-1npu != '{"include":[]}'
runs-on: linux-aarch64-a3-2
strategy:
fail-fast: false
matrix: ${{ fromJson(needs.compute-diffusion-partitions.outputs.matrix-1npu) }}
timeout-minutes: 150
container:
image: swr.cn-southwest-2.myhuaweicloud.com/base_image/ascend-ci/cann:9.0.0-a3-ubuntu22.04-py3.11
steps:
- name: Checkout code
uses: actions/checkout@v4
- name: Mark repository safe
run: |
git config --system --add safe.directory ${GITHUB_WORKSPACE}
- name: Install dependencies
env:
TORCH_CACHE_URL: "http://cache-service.nginx-pypi-cache.svc.cluster.local/whl/cpu"
PYPI_CACHE_URL: "http://cache-service.nginx-pypi-cache.svc.cluster.local/pypi/simple"
UV_INDEX_URL: "http://cache-service.nginx-pypi-cache.svc.cluster.local/pypi/simple"
GITHUB_PROXY_URL: "https://gh-proxy.test.osinfra.cn/"
RUSTUP_CACHE_URL: "http://cache-service.nginx-pypi-cache.svc.cluster.local:8082"
run: |
# speed up by using infra cache services
CACHING_URL="cache-service.nginx-pypi-cache.svc.cluster.local"
sed -Ei "s@(ports|archive).ubuntu.com@${CACHING_URL}:8081@g" /etc/apt/sources.list
pip config set global.index-url http://${CACHING_URL}/pypi/simple
pip config set global.trusted-host "${CACHING_URL}"
bash scripts/ci/npu/npu_ci_install_dependency.sh a3 diffusion
- name: Generate outputs
env:
PARTITION_PLAN_JSON: ${{ needs.compute-diffusion-partitions.outputs.plan-1npu }}
SGLANG_USE_MODELSCOPE: true
SGLANG_IS_IN_CI: true
HF_ENDPOINT: https://hf-mirror.com
TORCH_EXTENSIONS_DIR: /tmp/torch_extensions
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
STREAMS_PER_DEVICE: 32
run: |
cd python
python -m sglang.multimodal_gen.test.scripts.gen_diffusion_ci_outputs \
--suite 1-npu \
--partition-id ${{ matrix.part }} \
--total-partitions ${{ needs.compute-diffusion-partitions.outputs['partition-count-1npu'] }} \
--partition-plan-json "$PARTITION_PLAN_JSON" \
--out-dir ./${{ env.OUTPUT_NAME }} \
${{ inputs.case_ids != '' && format('--case-ids {0}', inputs.case_ids) || '' }}
- name: Upload artifact
uses: actions/upload-artifact@v4
with:
name: ${{ env.OUTPUT_NAME }}-1npu-part${{ matrix.part }}
path: python/${{ env.OUTPUT_NAME }}
retention-days: 7
- name: Publish GT images to sgl-project/ci-data
env:
GITHUB_TOKEN: ${{ secrets.GH_PAT_FOR_NIGHTLY_CI_DATA }}
run: |
python scripts/ci/utils/diffusion/publish_diffusion_gt.py \
--source-dir python/${{ env.OUTPUT_NAME }} \
--target-dir "${{ env.PUBLISH_TARGET_DIR }}"
multimodal-diffusion-gen-2npu:
needs: compute-diffusion-partitions
if: |
needs.compute-diffusion-partitions.result == 'success' &&
needs.compute-diffusion-partitions.outputs.matrix-2npu != '{"include":[]}'
runs-on: linux-aarch64-a3-16
strategy:
fail-fast: false
matrix: ${{ fromJson(needs.compute-diffusion-partitions.outputs.matrix-2npu) }}
timeout-minutes: 150
container:
image: swr.cn-southwest-2.myhuaweicloud.com/base_image/ascend-ci/cann:9.0.0-a3-ubuntu22.04-py3.11
steps:
- name: Checkout code
uses: actions/checkout@v4
- name: Mark repository safe
run: |
git config --system --add safe.directory ${GITHUB_WORKSPACE}
- name: Install dependencies
env:
TORCH_CACHE_URL: "http://cache-service.nginx-pypi-cache.svc.cluster.local/whl/cpu"
PYPI_CACHE_URL: "http://cache-service.nginx-pypi-cache.svc.cluster.local/pypi/simple"
UV_INDEX_URL: "http://cache-service.nginx-pypi-cache.svc.cluster.local/pypi/simple"
GITHUB_PROXY_URL: "https://gh-proxy.test.osinfra.cn/"
RUSTUP_CACHE_URL: "http://cache-service.nginx-pypi-cache.svc.cluster.local:8082"
run: |
# speed up by using infra cache services
CACHING_URL="cache-service.nginx-pypi-cache.svc.cluster.local"
sed -Ei "s@(ports|archive).ubuntu.com@${CACHING_URL}:8081@g" /etc/apt/sources.list
pip config set global.index-url http://${CACHING_URL}/pypi/simple
pip config set global.trusted-host "${CACHING_URL}"
bash scripts/ci/npu/npu_ci_install_dependency.sh a3 diffusion
- name: Generate outputs
env:
PARTITION_PLAN_JSON: ${{ needs.compute-diffusion-partitions.outputs.plan-2npu }}
SGLANG_USE_MODELSCOPE: true
SGLANG_IS_IN_CI: true
HF_ENDPOINT: https://hf-mirror.com
TORCH_EXTENSIONS_DIR: /tmp/torch_extensions
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
STREAMS_PER_DEVICE: 32
run: |
cd python
python -m sglang.multimodal_gen.test.scripts.gen_diffusion_ci_outputs \
--suite 2-npu \
--partition-id ${{ matrix.part }} \
--total-partitions ${{ needs.compute-diffusion-partitions.outputs['partition-count-2npu'] }} \
--partition-plan-json "$PARTITION_PLAN_JSON" \
--out-dir ./${{ env.OUTPUT_NAME }} \
${{ inputs.case_ids != '' && format('--case-ids {0}', inputs.case_ids) || '' }}
- name: Upload artifact
uses: actions/upload-artifact@v4
with:
name: ${{ env.OUTPUT_NAME }}-2npu-part${{ matrix.part }}
path: python/${{ env.OUTPUT_NAME }}
retention-days: 7
- name: Publish GT images to sgl-project/ci-data
env:
GITHUB_TOKEN: ${{ secrets.GH_PAT_FOR_NIGHTLY_CI_DATA }}
run: |
python scripts/ci/utils/diffusion/publish_diffusion_gt.py \
--source-dir python/${{ env.OUTPUT_NAME }} \
--target-dir "${{ env.PUBLISH_TARGET_DIR }}"
+2 -60
View File
@@ -341,7 +341,7 @@ jobs:
SGLANG_DIFFUSION_ARTIFACT_DIR: ${{ github.workspace }}/diffusion-failures SGLANG_DIFFUSION_ARTIFACT_DIR: ${{ github.workspace }}/diffusion-failures
run: | run: |
cd python cd python
python3 sglang/multimodal_gen/test/run_suite_npu.py --suite 1-npu python3 sglang/multimodal_gen/test/run_suite.py --suite 1-npu
- name: Upload diffusion failure artifacts - name: Upload diffusion failure artifacts
if: always() if: always()
@@ -398,7 +398,7 @@ jobs:
SGLANG_DIFFUSION_ARTIFACT_DIR: ${{ github.workspace }}/diffusion-failures SGLANG_DIFFUSION_ARTIFACT_DIR: ${{ github.workspace }}/diffusion-failures
run: | run: |
cd python cd python
python3 sglang/multimodal_gen/test/run_suite_npu.py --suite 2-npu python3 sglang/multimodal_gen/test/run_suite.py --suite 2-npu
- name: Upload diffusion failure artifacts - name: Upload diffusion failure artifacts
if: always() if: always()
@@ -409,63 +409,6 @@ jobs:
if-no-files-found: ignore if-no-files-found: ignore
retention-days: 7 retention-days: 7
multimodal-gen-test-8-npu-a3:
needs: [check-changes, pr-gate, set-image-config]
if: needs.check-changes.outputs.multimodal_gen == 'true'
runs-on: linux-aarch64-a3-8
container:
image: ${{ needs.set-image-config.outputs.CANN_image_a3 }}
steps:
- name: Checkout code
uses: actions/checkout@v4
- name: Mark repository safe
run: |
git config --system --add safe.directory ${GITHUB_WORKSPACE}
- name: Install dependencies
env:
TORCH_CACHE_URL: "http://cache-service.nginx-pypi-cache.svc.cluster.local/whl/cpu"
PYPI_CACHE_URL: "http://cache-service.nginx-pypi-cache.svc.cluster.local/pypi/simple"
UV_INDEX_URL: "http://cache-service.nginx-pypi-cache.svc.cluster.local/pypi/simple"
GITHUB_PROXY_URL: "https://gh-proxy.test.osinfra.cn/"
RUSTUP_CACHE_URL: "http://cache-service.nginx-pypi-cache.svc.cluster.local:8082"
run: |
# speed up by using infra cache services
CACHING_URL="cache-service.nginx-pypi-cache.svc.cluster.local"
sed -Ei "s@(ports|archive).ubuntu.com@${CACHING_URL}:8081@g" /etc/apt/sources.list
pip config set global.index-url http://${CACHING_URL}/pypi/simple
pip config set global.trusted-host "${CACHING_URL}"
bash scripts/ci/npu/npu_ci_install_dependency.sh a3
# copy required file from our daily cache
cp ~/.cache/modelscope/hub/datasets/otavia/ShareGPT_Vicuna_unfiltered/ShareGPT_V3_unfiltered_cleaned_split.json /tmp
# copy gsm8k dataset
cp ~/.cache/modelscope/hub/datasets/tmp/test.jsonl /tmp
- name: Run test
timeout-minutes: 60
env:
SGLANG_USE_MODELSCOPE: true
SGLANG_IS_IN_CI: true
HF_ENDPOINT: https://hf-mirror.com
TORCH_EXTENSIONS_DIR: /tmp/torch_extensions
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
STREAMS_PER_DEVICE: 32
SGLANG_DIFFUSION_ARTIFACT_DIR: ${{ github.workspace }}/diffusion-failures
run: |
cd python
python3 sglang/multimodal_gen/test/run_suite_npu.py --suite 8-npu
- name: Upload diffusion failure artifacts
if: always()
uses: actions/upload-artifact@v4
with:
name: diffusion-failures-npu-8-${{ github.run_attempt }}
path: diffusion-failures/
if-no-files-found: ignore
retention-days: 7
pr-test-npu-finish: pr-test-npu-finish:
needs: needs:
[ [
@@ -478,7 +421,6 @@ jobs:
multimodal-gen-test-1-npu-a3, multimodal-gen-test-1-npu-a3,
multimodal-gen-test-2-npu-a3, multimodal-gen-test-2-npu-a3,
multimodal-gen-test-8-npu-a3,
] ]
if: always() if: always()
runs-on: ubuntu-latest runs-on: ubuntu-latest
+33 -98
View File
@@ -19,117 +19,52 @@ from pathlib import Path
import tabulate import tabulate
from sglang.multimodal_gen.runtime.platforms import current_platform
from sglang.multimodal_gen.runtime.utils.logging_utils import init_logger from sglang.multimodal_gen.runtime.utils.logging_utils import init_logger
from sglang.multimodal_gen.test.partitioning import ( from sglang.multimodal_gen.test.partitioning import (
PartitionItem, PartitionItem,
partition_items_by_lpt, partition_items_by_lpt,
) )
from sglang.multimodal_gen.test.server.gpu_cases import (
ONE_GPU_CASES,
TWO_GPU_CASES,
)
from sglang.multimodal_gen.test.server.testcase_configs import ( from sglang.multimodal_gen.test.server.testcase_configs import (
BASELINE_CONFIG, BASELINE_CONFIG,
DiffusionTestCase, DiffusionTestCase,
) )
logger = init_logger(__name__) # TODO: remove duplicated code
if current_platform.is_npu():
DEFAULT_EST_TIME_SECONDS = 300.0 from sglang.multimodal_gen.test.server.ascend.testcase_configs_npu import (
STARTUP_OVERHEAD_SECONDS = 120.0 _UPDATE_WEIGHTS_FROM_DISK_TEST_FILE,
DEFAULT_STANDALONE_EST_TIME_SECONDS = 300.0 COMPONENT_ACCURACY_SUITES,
DEFAULT_EST_TIME_SECONDS,
_UPDATE_WEIGHTS_FROM_DISK_TEST_FILE = "test_update_weights_from_disk.py" DEFAULT_STANDALONE_EST_TIME_SECONDS,
_UPDATE_WEIGHTS_MODEL_PAIR_ENV = "SGLANG_MMGEN_UPDATE_WEIGHTS_PAIR" FILE_SUITES,
_UPDATE_WEIGHTS_MODEL_PAIR_IDS = ( PARAMETRIZED_CASE_GROUPS,
"FLUX.2-klein-base-4B", STANDALONE_FILES,
"Qwen-Image", STARTUP_OVERHEAD_SECONDS,
) SUITES,
)
else:
def _discover_unit_tests() -> list[str]: from sglang.multimodal_gen.test.server.gpu_cases import ( # noqa: F401 It is used by ci scripts
unit_dir = Path(__file__).resolve().parent / "unit" _UPDATE_WEIGHTS_FROM_DISK_TEST_FILE,
if not unit_dir.is_dir(): _UPDATE_WEIGHTS_MODEL_PAIR_ENV,
return [] _UPDATE_WEIGHTS_MODEL_PAIR_IDS,
return sorted( COMPONENT_ACCURACY_FILE_NUM_GPUS,
f"../unit/{f.name}" for f in unit_dir.glob("test_*.py") if f.is_file() COMPONENT_ACCURACY_SUITES,
DEFAULT_EST_TIME_SECONDS,
DEFAULT_STANDALONE_EST_TIME_SECONDS,
FILE_SUITES,
ONE_GPU_CASES,
PARAMETRIZED_CASE_GROUPS,
STANDALONE_FILE_EST_TIMES,
STANDALONE_FILES,
STARTUP_OVERHEAD_SECONDS,
STRICT_SUITES,
SUITES,
TWO_GPU_CASES,
) )
FILE_SUITES = { logger = init_logger(__name__)
"unit": _discover_unit_tests(),
"component-accuracy": [
"test_component_accuracy_1_gpu.py",
"test_component_accuracy_2_gpu.py",
],
"component-accuracy-1-gpu": [
"test_component_accuracy_1_gpu.py",
],
"component-accuracy-2-gpu": [
"test_component_accuracy_2_gpu.py",
],
"1-gpu-b200": [
"test_server_b200.py",
],
}
PARAMETRIZED_CASE_GROUPS = {
"1-gpu": [
("test_server_1_gpu.py", ONE_GPU_CASES),
],
"2-gpu": [
("test_server_2_gpu.py", TWO_GPU_CASES),
],
}
STANDALONE_FILES = {
"1-gpu": [
"../cli/test_generate_t2i_perf.py",
# Temporarily disabled: 24 timeout failures since 2026-04-09 across
# multimodal-gen-test-1-gpu. Re-enable after the flakiness is fixed.
# "test_update_weights_from_disk.py",
],
"2-gpu": [
"test_disagg_server.py",
],
}
# New standalone files may omit an estimate once to learn the real CI runtime.
# CI will use a fallback estimate for sharding, run the test, then print a
# measured value that must be copied into STANDALONE_FILE_EST_TIMES.
STANDALONE_FILE_EST_TIMES = {
"1-gpu": {
"../cli/test_generate_t2i_perf.py": 240.0,
# See STANDALONE_FILES note above — temporarily disabled.
# "test_update_weights_from_disk.py": 480.0,
},
"2-gpu": {
# Two disagg clusters × (~3 min startup + ~1 min generate) ≈ 8 min.
# Raise if CI reports a higher measured time.
"test_disagg_server.py": 600.0,
},
}
# Backward-compatible suite view for scripts that still operate on file lists.
SUITES = {
**FILE_SUITES,
**{
suite: [filename for filename, _ in case_groups]
+ STANDALONE_FILES.get(suite, [])
for suite, case_groups in PARAMETRIZED_CASE_GROUPS.items()
},
}
STRICT_SUITES = {"unit"}
COMPONENT_ACCURACY_SUITES = {
"component-accuracy",
"component-accuracy-1-gpu",
"component-accuracy-2-gpu",
}
COMPONENT_ACCURACY_FILE_NUM_GPUS = {
"test_component_accuracy_1_gpu.py": 1,
"test_component_accuracy_2_gpu.py": 2,
}
@dataclass(frozen=True) @dataclass(frozen=True)
@@ -1,299 +0,0 @@
"""
Test runner for multimodal_gen that manages test suites and parallel execution.
Usage:
python3 run_suite_npu.py --suite <suite_name> --partition-id <id> --total-partitions <num>
Example:
python3 run_suite_npu.py --suite 1-npu --partition-id 0 --total-partitions 4
"""
import argparse
import os
import subprocess
import sys
from pathlib import Path
import tabulate
from sglang.multimodal_gen.runtime.utils.logging_utils import init_logger
logger = init_logger(__name__)
SUITES = {
"1-npu": [
"ascend/test_server_1_npu.py",
# add new 1-npu test files here
],
"2-npu": [
"ascend/test_server_2_npu.py",
# add new 2-npu test files here
],
"8-npu": [
"ascend/test_server_8_npu.py",
# add new 8-npu test files here
],
}
def parse_args():
parser = argparse.ArgumentParser(description="Run multimodal_gen test suite")
parser.add_argument(
"--suite",
type=str,
required=True,
choices=list(SUITES.keys()),
help="The test suite to run (valid names are defined in SUITES)",
)
parser.add_argument(
"--partition-id",
type=int,
default=0,
help="Index of the current partition (for parallel execution)",
)
parser.add_argument(
"--total-partitions",
type=int,
default=1,
help="Total number of partitions",
)
parser.add_argument(
"--base-dir",
type=str,
default="server",
help="Base directory for tests relative to this script's parent",
)
parser.add_argument(
"-k",
"--filter",
type=str,
default=None,
help="Pytest filter expression (passed to pytest -k)",
)
parser.add_argument(
"--continue-on-error",
action="store_true",
default=False,
help="Continue running remaining tests even if one fails (for CI consistency; pytest already continues by default)",
)
return parser.parse_args()
def collect_test_items(files, filter_expr=None):
"""Collect test item node IDs from the given files using pytest --collect-only."""
cmd = [sys.executable, "-m", "pytest", "--collect-only", "-q"]
if filter_expr:
cmd.extend(["-k", filter_expr])
cmd.extend(files)
print(f"Collecting tests with command: {' '.join(cmd)}")
result = subprocess.run(cmd, capture_output=True, text=True)
# Check for collection errors
# pytest exit codes:
# 0: success
# 1: tests collected but some had errors during collection
# 2: test execution interrupted
# 3: internal error
# 4: command line usage error
# 5: no tests collected (may be expected with filters)
if result.returncode not in (0, 5):
error_msg = (
f"pytest --collect-only failed with exit code {result.returncode}\n"
f"Command: {' '.join(cmd)}\n"
)
if result.stderr:
error_msg += f"stderr:\n{result.stderr}\n"
if result.stdout:
error_msg += f"stdout:\n{result.stdout}\n"
logger.error(error_msg)
raise RuntimeError(error_msg)
if result.returncode == 5:
print(
"No tests were collected (exit code 5). This may be expected with filters."
)
# Parse the output to extract test node IDs
# pytest -q outputs lines like: test_file.py::TestClass::test_method[param]
test_items = []
for line in result.stdout.strip().split("\n"):
line = line.strip()
# Skip empty lines and summary lines
if line and "::" in line and not line.startswith(("=", "-", " ")):
# Handle lines that might have extra info after the test ID
test_id = line.split()[0] if " " in line else line
if "::" in test_id:
test_items.append(test_id)
print(f"Collected {len(test_items)} test items")
return test_items
def run_pytest(files, filter_expr=None, exitfirst=False):
if not files:
print("No files to run.")
return 0
base_cmd = [sys.executable, "-m", "pytest", "-s", "-v"]
if exitfirst:
base_cmd.append("-x")
# Add pytest -k filter if provided
if filter_expr:
base_cmd.extend(["-k", filter_expr])
max_retries = 6
# retry if the perf assertion failed, for {max_retries} times
for i in range(max_retries + 1):
cmd = list(base_cmd)
if i > 0:
cmd.append("--last-failed")
# Always include files to constrain test discovery scope
# This prevents pytest from scanning the entire rootdir and
# discovering unrelated tests that may have missing dependencies
cmd.extend(files)
if i > 0:
print(
f"Performance assertion failed. Retrying ({i}/{max_retries}) with --last-failed..."
)
print(f"Running command: {' '.join(cmd)}")
process = subprocess.Popen(
cmd,
stdout=subprocess.PIPE,
stderr=subprocess.STDOUT,
bufsize=0,
)
output_bytes = bytearray()
while True:
chunk = process.stdout.read(4096)
if not chunk:
break
sys.stdout.buffer.write(chunk)
sys.stdout.buffer.flush()
output_bytes.extend(chunk)
process.wait()
returncode = process.returncode
if returncode == 0:
return 0
# Exit code 5 means no tests were collected/selected - treat as success
# when using filters, since some partitions may have all tests filtered out
if returncode == 5:
print(
"No tests collected (exit code 5). This is expected when filters "
"deselect all tests in a partition. Treating as success."
)
return 0
# check if the failure is due to an assertion in test_server_utils.py
full_output = output_bytes.decode("utf-8", errors="replace")
is_perf_assertion = (
"multimodal_gen/test/server/test_server_utils.py" in full_output
and "AssertionError" in full_output
)
is_flaky_ci_assertion = (
"SafetensorError" in full_output
or "FileNotFoundError" in full_output
or "TimeoutError" in full_output
)
is_oom_error = (
"out of memory" in full_output.lower()
or "oom killer" in full_output.lower()
)
if not (is_perf_assertion or is_flaky_ci_assertion or is_oom_error):
return returncode
print(f"Max retry exceeded")
return returncode
def main():
args = parse_args()
# 1. resolve base path
current_file_path = Path(__file__).resolve()
test_root_dir = current_file_path.parent
target_dir = test_root_dir / args.base_dir
if not target_dir.exists():
print(f"Error: Target directory {target_dir} does not exist.")
sys.exit(1)
# 2. get files from suite
suite_files_rel = SUITES[args.suite]
suite_files_abs = []
for f_rel in suite_files_rel:
f_abs = target_dir / f_rel
if not f_abs.exists():
msg = f"Test file {f_rel} not found in {target_dir}."
print(f"Warning: {msg} Skipping.")
continue
suite_files_abs.append(str(f_abs))
if not suite_files_abs:
print(f"No valid test files found for suite '{args.suite}'.")
sys.exit(0)
# 3. collect all test items and partition by items (not files)
all_test_items = collect_test_items(suite_files_abs, filter_expr=args.filter)
if not all_test_items:
print(f"No test items found for suite '{args.suite}'.")
sys.exit(0)
# Partition by test items
my_items = [
item
for i, item in enumerate(all_test_items)
if i % args.total_partitions == args.partition_id
]
# Print test info at beginning (similar to test/run_suite.py pretty_print_tests)
partition_info = f"{args.partition_id + 1}/{args.total_partitions} (0-based id={args.partition_id})"
headers = ["Suite", "Partition"]
rows = [[args.suite, partition_info]]
msg = tabulate.tabulate(rows, headers=headers, tablefmt="psql") + "\n"
msg += f"✅ Enabled {len(my_items)} test(s):\n"
for item in my_items:
msg += f" - {item}\n"
print(msg, flush=True)
print(
f"Suite: {args.suite} | Partition: {args.partition_id}/{args.total_partitions}"
)
print(f"Selected {len(suite_files_abs)} files:")
for f in suite_files_abs:
print(f" - {os.path.basename(f)}")
if not my_items:
print("No items assigned to this partition. Exiting success.")
sys.exit(0)
print(f"Running {len(my_items)} items in this shard: {', '.join(my_items)}")
# 4. execute with the specific test items
# Fast-fail: stop on first failure unless --continue-on-error is set
exit_code = run_pytest(my_items, exitfirst=not args.continue_on_error)
# Print tests again at the end for visibility
msg = "\n" + tabulate.tabulate(rows, headers=headers, tablefmt="psql") + "\n"
msg += f"✅ Executed {len(my_items)} test(s):\n"
for item in my_items:
msg += f" - {item}\n"
print(msg, flush=True)
sys.exit(exit_code)
if __name__ == "__main__":
main()
@@ -1,7 +1,7 @@
{ {
"metadata": { "metadata": {
"model": "Diffusion Server", "model": "Diffusion Server",
"hardware": "CI A2 64GB pool", "hardware": "Ascend A3",
"description": "Reference numbers captured from the CI diffusion server baseline run" "description": "Reference numbers captured from the CI diffusion server baseline run"
}, },
"scenarios": { "scenarios": {
@@ -72,69 +72,69 @@
}, },
"flux_2_image_t2i_2npu": { "flux_2_image_t2i_2npu": {
"stages_ms": { "stages_ms": {
"InputValidationStage": 0.06, "InputValidationStage": 0.08,
"TextEncodingStage": 5628.31, "TextEncodingStage": 192.4,
"ImageVAEEncodingStage": 0.01, "ImageVAEEncodingStage": 0.01,
"LatentPreparationStage": 0.75, "LatentPreparationStage": 0.97,
"TimestepPreparationStage": 30.68, "TimestepPreparationStage": 34.65,
"DenoisingStage": 55002.26, "DenoisingStage": 45390.41,
"DecodingStage": 43.73 "DecodingStage": 5.32
}, },
"denoise_step_ms": { "denoise_step_ms": {
"0": 110.35, "0": 84.23,
"1": 301.82, "1": 80.0,
"2": 1139.81, "2": 874.6,
"3": 1114.17, "3": 918.81,
"4": 1099.34, "4": 900.4,
"5": 1099.12, "5": 910.67,
"6": 1100.16, "6": 903.36,
"7": 1099.67, "7": 904.97,
"8": 1099.09, "8": 906.84,
"9": 1089.81, "9": 906.39,
"10": 1109.73, "10": 904.99,
"11": 1099.97, "11": 909.96,
"12": 1100.26, "12": 901.67,
"13": 1099.67, "13": 908.8,
"14": 1099.79, "14": 902.93,
"15": 1099.6, "15": 906.17,
"16": 1100.16, "16": 906.67,
"17": 1099.87, "17": 905.89,
"18": 1100.02, "18": 906.9,
"19": 1099.34, "19": 907.1,
"20": 1099.6, "20": 905.31,
"21": 1099.45, "21": 907.93,
"22": 1100.2, "22": 903.68,
"23": 1099.29, "23": 904.49,
"24": 1098.86, "24": 905.73,
"25": 1090.38, "25": 907.66,
"26": 1109.19, "26": 906.71,
"27": 1099.67, "27": 912.69,
"28": 1100.06, "28": 901.4,
"29": 1099.22, "29": 909.9,
"30": 1100.08, "30": 901.72,
"31": 1098.86, "31": 904.35,
"32": 1099.73, "32": 905.61,
"33": 1099.11, "33": 905.97,
"34": 1100.13, "34": 906.5,
"35": 1103.97, "35": 921.45,
"36": 1095.26, "36": 892.58,
"37": 1099.38, "37": 909.03,
"38": 1099.34, "38": 903.82,
"39": 1099.17, "39": 906.25,
"40": 1100.08, "40": 905.18,
"41": 1089.89, "41": 905.06,
"42": 1106.69, "42": 906.21,
"43": 1102.57, "43": 911.62,
"44": 1100.17, "44": 901.04,
"45": 1099.21, "45": 909.09,
"46": 1100.42, "46": 904.29,
"47": 1099.38, "47": 905.16,
"48": 1099.59, "48": 907.35,
"49": 1099.47 "49": 905.65
}, },
"expected_e2e_ms": 64195.08, "expected_e2e_ms": 46557.7,
"expected_avg_denoise_ms": 1065.0, "expected_avg_denoise_ms": 872.7,
"expected_median_denoise_ms": 1099.63 "expected_median_denoise_ms": 905.81
}, },
"wan2_1_t2v_1.3b_1_npu": { "wan2_1_t2v_1.3b_1_npu": {
"stages_ms": { "stages_ms": {
@@ -203,61 +203,61 @@
"expected_median_denoise_ms": 537.54, "expected_median_denoise_ms": 537.54,
"estimated_full_test_time_s": 157.8 "estimated_full_test_time_s": 157.8
}, },
"wan2_2_t2v_14b_w8a8_8npu": { "wan2_2_t2v_14b_w8a8_2npu": {
"stages_ms": { "stages_ms": {
"InputValidationStage": 0.14, "InputValidationStage": 0.09,
"TextEncodingStage": 3020.73, "TextEncodingStage": 2789.3,
"LatentPreparationStage": 0.19, "LatentPreparationStage": 0.28,
"TimestepPreparationStage": 5.01, "TimestepPreparationStage": 3.19,
"DenoisingStage": 82744.33, "DenoisingStage": 187650.19,
"DecodingStage": 932.41, "DecodingStage": 3491.88,
"per_frame_generation": null "per_frame_generation": null
}, },
"denoise_step_ms": { "denoise_step_ms": {
"0": 1232.32, "0": 1415.84,
"1": 2091.77, "1": 4801.54,
"2": 2097.62, "2": 4781.57,
"3": 2087.53, "3": 4770.19,
"4": 2088.54, "4": 4808.86,
"5": 2087.96, "5": 4737.55,
"6": 2088.28, "6": 4774.02,
"7": 2089.77, "7": 4774.85,
"8": 2101.9, "8": 4773.25,
"9": 2088.73, "9": 4775.33,
"10": 2088.04, "10": 4771.31,
"11": 2087.53, "11": 4773.07,
"12": 2088.89, "12": 4773.18,
"13": 2087.09, "13": 4772.77,
"14": 2088.25, "14": 4773.68,
"15": 2087.96, "15": 4771.91,
"16": 2088.24, "16": 4776.16,
"17": 2088.45, "17": 4773.88,
"18": 2104.7, "18": 4769.87,
"19": 2088.44, "19": 4772.1,
"20": 2087.19, "20": 4774.55,
"21": 2088.19, "21": 4772.54,
"22": 2088.37, "22": 4772.39,
"23": 2087.6, "23": 4776.14,
"24": 2088.13, "24": 4772.84,
"25": 2088.06, "25": 4772.82,
"26": 2126.23, "26": 4789.34,
"27": 2089.92, "27": 4792.58,
"28": 2087.37, "28": 4807.52,
"29": 2089.21, "29": 4740.66,
"30": 2088.29, "30": 4773.46,
"31": 2087.89, "31": 4774.19,
"32": 2073.1, "32": 4772.8,
"33": 2086.71, "33": 4774.84,
"34": 2087.88, "34": 4772.52,
"35": 2088.64, "35": 4773.77,
"36": 2088.1, "36": 4775.43,
"37": 2089.14, "37": 4771.58,
"38": 2087.5, "38": 4772.68,
"39": 2087.86 "39": 4769.1
}, },
"expected_e2e_ms": 86719.57, "expected_e2e_ms": 193947.19,
"expected_avg_denoise_ms": 2068.43, "expected_avg_denoise_ms": 4691.07,
"expected_median_denoise_ms": 2088.21 "expected_median_denoise_ms": 4773.22
}, },
"qwen_image_t2i_2npu": { "qwen_image_t2i_2npu": {
"stages_ms": { "stages_ms": {
@@ -1,31 +0,0 @@
"""
Config-driven diffusion performance test with pytest parametrization.
If the actual run is significantly better than the baseline, the improved cases with their updated baseline will be printed
"""
from __future__ import annotations
import pytest
from sglang.multimodal_gen.runtime.utils.logging_utils import init_logger
from sglang.multimodal_gen.test.server.ascend.testcase_configs_npu import (
EIGHT_NPU_CASES,
)
from sglang.multimodal_gen.test.server.test_server_common import ( # noqa: F401
DiffusionServerBase,
diffusion_server,
)
from sglang.multimodal_gen.test.server.testcase_configs import DiffusionTestCase
logger = init_logger(__name__)
class TestDiffusionServerEightNpu(DiffusionServerBase):
"""Performance tests for 8-NPU diffusion cases."""
@pytest.fixture(params=EIGHT_NPU_CASES, ids=lambda c: c.id)
def case(self, request) -> DiffusionTestCase:
"""Provide a DiffusionTestCase for each 8-NPU test."""
return request.param
@@ -1,3 +1,5 @@
import os
from sglang.multimodal_gen.test.server.testcase_configs import ( from sglang.multimodal_gen.test.server.testcase_configs import (
T2V_PROMPT, T2V_PROMPT,
DiffusionSamplingParams, DiffusionSamplingParams,
@@ -6,12 +8,31 @@ from sglang.multimodal_gen.test.server.testcase_configs import (
T2I_sampling_params, T2I_sampling_params,
) )
MODEL_WEIGHTS_DIR = "/root/.cache/modelscope/hub/models/"
FLUX_1_DEV_WEIGHTS_PATH = os.path.join(
MODEL_WEIGHTS_DIR, "black-forest-labs/FLUX.1-dev"
)
FLUX_2_DEV_WEIGHTS_PATH = os.path.join(
MODEL_WEIGHTS_DIR, "black-forest-labs/FLUX.2-dev"
)
QWEN_IMAGE_WEIGHTS_PATH = os.path.join(MODEL_WEIGHTS_DIR, "Qwen/Qwen-Image")
WAN2_1_T2V_1_3B_DIFFUSERS_WEIGHTS_PATH = os.path.join(
MODEL_WEIGHTS_DIR, "Wan-AI/Wan2.1-T2V-1.3B-Diffusers"
)
WAN2_2_T2V_A14B_DIFFUSERS_W8A8_WEIGHTS_PATH = os.path.join(
MODEL_WEIGHTS_DIR, "Eco-Tech/Wan2.2-T2V-A14B-Diffusers-w8a8"
)
EXTRAS_DISABLE_WARMUP = ["--server-warmup", "false"]
ONE_NPU_CASES: list[DiffusionTestCase] = [ ONE_NPU_CASES: list[DiffusionTestCase] = [
# === Text to Image (T2I) === # === Text to Image (T2I) ===
DiffusionTestCase( DiffusionTestCase(
"flux_image_t2i_npu", "flux_image_t2i_npu",
DiffusionServerArgs( DiffusionServerArgs(
model_path="/root/.cache/modelscope/hub/models/black-forest-labs/FLUX.1-dev", model_path=FLUX_1_DEV_WEIGHTS_PATH,
extras=EXTRAS_DISABLE_WARMUP,
), ),
T2I_sampling_params, T2I_sampling_params,
run_consistency_check=False, run_consistency_check=False,
@@ -20,7 +41,8 @@ ONE_NPU_CASES: list[DiffusionTestCase] = [
DiffusionTestCase( DiffusionTestCase(
"wan2_1_t2v_1.3b_1_npu", "wan2_1_t2v_1.3b_1_npu",
DiffusionServerArgs( DiffusionServerArgs(
model_path="/root/.cache/modelscope/hub/models/Wan-AI/Wan2.1-T2V-1.3B-Diffusers", model_path=WAN2_1_T2V_1_3B_DIFFUSERS_WEIGHTS_PATH,
extras=EXTRAS_DISABLE_WARMUP,
), ),
DiffusionSamplingParams( DiffusionSamplingParams(
prompt=T2V_PROMPT, prompt=T2V_PROMPT,
@@ -34,9 +56,10 @@ TWO_NPU_CASES: list[DiffusionTestCase] = [
DiffusionTestCase( DiffusionTestCase(
"flux_2_image_t2i_2npu", "flux_2_image_t2i_2npu",
DiffusionServerArgs( DiffusionServerArgs(
model_path="/root/.cache/modelscope/hub/models/black-forest-labs/FLUX.2-dev", model_path=FLUX_2_DEV_WEIGHTS_PATH,
num_gpus=2, num_gpus=2,
tp_size=2, tp_size=2,
extras=EXTRAS_DISABLE_WARMUP,
), ),
T2I_sampling_params, T2I_sampling_params,
run_consistency_check=False, run_consistency_check=False,
@@ -44,26 +67,25 @@ TWO_NPU_CASES: list[DiffusionTestCase] = [
DiffusionTestCase( DiffusionTestCase(
"qwen_image_t2i_2npu", "qwen_image_t2i_2npu",
DiffusionServerArgs( DiffusionServerArgs(
model_path="/root/.cache/modelscope/hub/models/Qwen/Qwen-Image", model_path=QWEN_IMAGE_WEIGHTS_PATH,
num_gpus=2, num_gpus=2,
# test ring attn # test ring attn
ulysses_degree=1, ulysses_degree=1,
ring_degree=2, ring_degree=2,
extras=EXTRAS_DISABLE_WARMUP,
), ),
T2I_sampling_params, T2I_sampling_params,
run_consistency_check=False, run_consistency_check=False,
), ),
]
EIGHT_NPU_CASES: list[DiffusionTestCase] = [
# === Text to Video (T2V) === # === Text to Video (T2V) ===
DiffusionTestCase( DiffusionTestCase(
"wan2_2_t2v_14b_w8a8_8npu", "wan2_2_t2v_14b_w8a8_2npu",
DiffusionServerArgs( DiffusionServerArgs(
model_path="/root/.cache/modelscope/hub/models/Eco-Tech/Wan2.2-T2V-A14B-Diffusers-w8a8", model_path=WAN2_2_T2V_A14B_DIFFUSERS_W8A8_WEIGHTS_PATH,
num_gpus=8, num_gpus=2,
tp_size=4, tp_size=1,
ulysses_degree=2, ulysses_degree=2,
extras=EXTRAS_DISABLE_WARMUP,
), ),
DiffusionSamplingParams( DiffusionSamplingParams(
prompt=T2V_PROMPT, prompt=T2V_PROMPT,
@@ -71,3 +93,32 @@ EIGHT_NPU_CASES: list[DiffusionTestCase] = [
run_consistency_check=False, run_consistency_check=False,
), ),
] ]
DEFAULT_EST_TIME_SECONDS = 300.0
STARTUP_OVERHEAD_SECONDS = 120.0
DEFAULT_STANDALONE_EST_TIME_SECONDS = 300.0
SUITES = {
"1-npu": [
"ascend/test_server_1_npu.py",
# add new 1-npu test files here
],
"2-npu": [
"ascend/test_server_2_npu.py",
# add new 2-npu test files here
],
}
PARAMETRIZED_CASE_GROUPS = {
"1-npu": [
("ascend/test_server_1_npu.py", ONE_NPU_CASES),
],
"2-npu": [
("ascend/test_server_2_npu.py", TWO_NPU_CASES),
],
}
FILE_SUITES = {}
STANDALONE_FILES = {}
COMPONENT_ACCURACY_SUITES = {}
_UPDATE_WEIGHTS_FROM_DISK_TEST_FILE = None
@@ -787,3 +787,99 @@ if not current_platform.is_hip():
ONE_GPU_CASES += ONE_GPU_MODELOPT_FP8_CASES ONE_GPU_CASES += ONE_GPU_MODELOPT_FP8_CASES
TWO_GPU_CASES = _with_default_num_gpus(TWO_GPU_CASES, 2) TWO_GPU_CASES = _with_default_num_gpus(TWO_GPU_CASES, 2)
def _discover_unit_tests() -> list[str]:
unit_dir = Path(__file__).resolve().parent.parent / "unit"
if not unit_dir.is_dir():
return []
return sorted(
f"../unit/{f.name}" for f in unit_dir.glob("test_*.py") if f.is_file()
)
FILE_SUITES = {
"unit": _discover_unit_tests(),
"component-accuracy": [
"test_component_accuracy_1_gpu.py",
"test_component_accuracy_2_gpu.py",
],
"component-accuracy-1-gpu": [
"test_component_accuracy_1_gpu.py",
],
"component-accuracy-2-gpu": [
"test_component_accuracy_2_gpu.py",
],
"1-gpu-b200": [
"test_server_b200.py",
],
}
PARAMETRIZED_CASE_GROUPS = {
"1-gpu": [
("test_server_1_gpu.py", ONE_GPU_CASES),
],
"2-gpu": [
("test_server_2_gpu.py", TWO_GPU_CASES),
],
}
STANDALONE_FILES = {
"1-gpu": [
"../cli/test_generate_t2i_perf.py",
# Temporarily disabled: 24 timeout failures since 2026-04-09 across
# multimodal-gen-test-1-gpu. Re-enable after the flakiness is fixed.
# "test_update_weights_from_disk.py",
],
"2-gpu": [
"test_disagg_server.py",
],
}
# New standalone files may omit an estimate once to learn the real CI runtime.
# CI will use a fallback estimate for sharding, run the test, then print a
# measured value that must be copied into STANDALONE_FILE_EST_TIMES.
STANDALONE_FILE_EST_TIMES = {
"1-gpu": {
"../cli/test_generate_t2i_perf.py": 240.0,
# See STANDALONE_FILES note above — temporarily disabled.
# "test_update_weights_from_disk.py": 480.0,
},
"2-gpu": {
# Two disagg clusters × (~3 min startup + ~1 min generate) ≈ 8 min.
# Raise if CI reports a higher measured time.
"test_disagg_server.py": 600.0,
},
}
# Backward-compatible suite view for scripts that still operate on file lists.
SUITES = {
**FILE_SUITES,
**{
suite: [filename for filename, _ in case_groups]
+ STANDALONE_FILES.get(suite, [])
for suite, case_groups in PARAMETRIZED_CASE_GROUPS.items()
},
}
STRICT_SUITES = {"unit"}
COMPONENT_ACCURACY_SUITES = {
"component-accuracy",
"component-accuracy-1-gpu",
"component-accuracy-2-gpu",
}
COMPONENT_ACCURACY_FILE_NUM_GPUS = {
"test_component_accuracy_1_gpu.py": 1,
"test_component_accuracy_2_gpu.py": 2,
}
DEFAULT_EST_TIME_SECONDS = 300.0
STARTUP_OVERHEAD_SECONDS = 120.0
DEFAULT_STANDALONE_EST_TIME_SECONDS = 300.0
_UPDATE_WEIGHTS_FROM_DISK_TEST_FILE = "test_update_weights_from_disk.py"
_UPDATE_WEIGHTS_MODEL_PAIR_ENV = "SGLANG_MMGEN_UPDATE_WEIGHTS_PAIR"
_UPDATE_WEIGHTS_MODEL_PAIR_IDS = (
"FLUX.2-klein-base-4B",
"Qwen-Image",
)
@@ -45,10 +45,18 @@ SGL_TEST_FILES_OFFICIAL_CONSISTENCY_GT_BASE = (
SGL_TEST_FILES_SGLANG_CONSISTENCY_GT_BASE = ( SGL_TEST_FILES_SGLANG_CONSISTENCY_GT_BASE = (
f"{SGL_TEST_FILES_CONSISTENCY_GT_ROOT}/sglang_generated" f"{SGL_TEST_FILES_CONSISTENCY_GT_ROOT}/sglang_generated"
) )
SGL_TEST_FILES_OFFICIAL_CONSISTENCY_GT_BASE_ASCEND = (
f"{SGL_TEST_FILES_CONSISTENCY_GT_ROOT}/official_generated/ascend"
)
SGL_TEST_FILES_SGLANG_CONSISTENCY_GT_BASE_ASCEND = (
f"{SGL_TEST_FILES_CONSISTENCY_GT_ROOT}/sglang_generated/ascend"
)
SGL_TEST_FILES_CONSISTENCY_GT_BASE = SGL_TEST_FILES_SGLANG_CONSISTENCY_GT_BASE SGL_TEST_FILES_CONSISTENCY_GT_BASE = SGL_TEST_FILES_SGLANG_CONSISTENCY_GT_BASE
SGL_TEST_FILES_CONSISTENCY_GT_BASES = ( SGL_TEST_FILES_CONSISTENCY_GT_BASES = (
SGL_TEST_FILES_OFFICIAL_CONSISTENCY_GT_BASE, SGL_TEST_FILES_OFFICIAL_CONSISTENCY_GT_BASE,
SGL_TEST_FILES_SGLANG_CONSISTENCY_GT_BASE, SGL_TEST_FILES_SGLANG_CONSISTENCY_GT_BASE,
SGL_TEST_FILES_OFFICIAL_CONSISTENCY_GT_BASE_ASCEND,
SGL_TEST_FILES_SGLANG_CONSISTENCY_GT_BASE_ASCEND,
) )
# LTX cases listed here compare against official-generated GT. # LTX cases listed here compare against official-generated GT.
SGL_TEST_FILES_OFFICIAL_CONSISTENCY_GT_CASES = frozenset( SGL_TEST_FILES_OFFICIAL_CONSISTENCY_GT_CASES = frozenset(
@@ -38,6 +38,12 @@ def _load_partitioning_helpers():
PartitionItem, partition_items_by_lpt = _load_partitioning_helpers() PartitionItem, partition_items_by_lpt = _load_partitioning_helpers()
SUITE_OUTPUT_NAMES = {"1-gpu": "1gpu", "2-gpu": "2gpu", "1-gpu-b200": "b200"} SUITE_OUTPUT_NAMES = {"1-gpu": "1gpu", "2-gpu": "2gpu", "1-gpu-b200": "b200"}
USE_NPU_CONFIGS = os.getenv("USE_NPU_CONFIGS", "0").lower() in ("1", "true")
if USE_NPU_CONFIGS:
SUITE_OUTPUT_NAMES = {"1-npu": "1npu", "2-npu": "2npu"}
DEFAULT_STANDALONE_EST_TIME_SECONDS = 300.0 DEFAULT_STANDALONE_EST_TIME_SECONDS = 300.0
@@ -261,6 +267,12 @@ def main():
print(f"Error: Run suite not found: {run_suite_path}") print(f"Error: Run suite not found: {run_suite_path}")
sys.exit(1) sys.exit(1)
try: try:
if USE_NPU_CONFIGS:
case_config_path = (
repo_root
/ "python/sglang/multimodal_gen/test/server/ascend/testcase_configs_npu.py"
)
else:
case_config_path = resolve_case_config_path(repo_root, run_suite_path) case_config_path = resolve_case_config_path(repo_root, run_suite_path)
except (RuntimeError, FileNotFoundError) as exc: except (RuntimeError, FileNotFoundError) as exc:
print(f"Error: {exc}") print(f"Error: {exc}")
@@ -16,6 +16,7 @@ Usage:
import ast import ast
import json import json
import os
from dataclasses import dataclass from dataclasses import dataclass
from pathlib import Path from pathlib import Path
from typing import Dict, List, Optional from typing import Dict, List, Optional
@@ -44,6 +45,17 @@ STARTUP_OVERHEAD_SECONDS = 120.0
BASELINE_REL_PATH = "python/sglang/multimodal_gen/test/server/perf_baselines.json" BASELINE_REL_PATH = "python/sglang/multimodal_gen/test/server/perf_baselines.json"
RUN_SUITE_REL_PATH = "python/sglang/multimodal_gen/test/run_suite.py" RUN_SUITE_REL_PATH = "python/sglang/multimodal_gen/test/run_suite.py"
USE_NPU_CONFIGS = os.getenv("USE_NPU_CONFIGS", "0").lower() in ("1", "true")
if USE_NPU_CONFIGS:
BASELINE_REL_PATH = (
"python/sglang/multimodal_gen/test/server/perf_baselines_npu.json"
)
CASE_LIST_TO_SUITE = {
"ONE_NPU_CASES": "1-npu",
"TWO_NPU_CASES": "2-npu",
}
@dataclass @dataclass
class DiffusionCaseInfo: class DiffusionCaseInfo: