Fix CI by reverting incorrect metric check logic (#15004)

This commit is contained in:
Kangyan-Zhou
2025-12-12 10:07:38 -08:00
committed by GitHub
parent 526fd0082f
commit b243154614
4 changed files with 74 additions and 277 deletions
+22 -27
View File
@@ -27,7 +27,6 @@ concurrency:
env: env:
SGLANG_IS_IN_CI: true SGLANG_IS_IN_CI: true
SGLANG_CI_ENABLE_RETRY: true
jobs: jobs:
# =============================================== check changes ==================================================== # =============================================== check changes ====================================================
@@ -551,10 +550,10 @@ jobs:
CUSTOM_BUILD_SGL_KERNEL=${{needs.check-changes.outputs.sgl_kernel}} bash scripts/ci/ci_install_dependency.sh CUSTOM_BUILD_SGL_KERNEL=${{needs.check-changes.outputs.sgl_kernel}} bash scripts/ci/ci_install_dependency.sh
- name: Run test - name: Run test
timeout-minutes: 40 timeout-minutes: 30
run: | run: |
cd test/srt cd test/srt
python3 run_suite.py --suite quantization_test --enable-retry python3 run_suite.py --suite quantization_test
unit-test-backend-1-gpu: unit-test-backend-1-gpu:
needs: [check-changes, call-gate, stage-a-test-1] needs: [check-changes, call-gate, stage-a-test-1]
@@ -593,10 +592,10 @@ jobs:
CUSTOM_BUILD_SGL_KERNEL=${{needs.check-changes.outputs.sgl_kernel}} bash scripts/ci/ci_install_dependency.sh CUSTOM_BUILD_SGL_KERNEL=${{needs.check-changes.outputs.sgl_kernel}} bash scripts/ci/ci_install_dependency.sh
- name: Run test - name: Run test
timeout-minutes: 40 timeout-minutes: 30
run: | run: |
cd test/srt cd test/srt
python3 run_suite.py --suite per-commit-1-gpu --auto-partition-id ${{ matrix.part }} --auto-partition-size 15 --enable-retry python3 run_suite.py --suite per-commit-1-gpu --auto-partition-id ${{ matrix.part }} --auto-partition-size 15
unit-test-backend-2-gpu: unit-test-backend-2-gpu:
needs: [check-changes, call-gate, unit-test-backend-1-gpu] needs: [check-changes, call-gate, unit-test-backend-1-gpu]
@@ -634,10 +633,10 @@ jobs:
CUSTOM_BUILD_SGL_KERNEL=${{needs.check-changes.outputs.sgl_kernel}} bash scripts/ci/ci_install_dependency.sh CUSTOM_BUILD_SGL_KERNEL=${{needs.check-changes.outputs.sgl_kernel}} bash scripts/ci/ci_install_dependency.sh
- name: Run test - name: Run test
timeout-minutes: 40 timeout-minutes: 30
run: | run: |
cd test/srt cd test/srt
python3 run_suite.py --suite per-commit-2-gpu --auto-partition-id ${{ matrix.part }} --auto-partition-size 2 --enable-retry python3 run_suite.py --suite per-commit-2-gpu --auto-partition-id ${{ matrix.part }} --auto-partition-size 2
unit-test-backend-4-gpu: unit-test-backend-4-gpu:
needs: [check-changes, call-gate, unit-test-backend-2-gpu] needs: [check-changes, call-gate, unit-test-backend-2-gpu]
@@ -675,10 +674,10 @@ jobs:
CUSTOM_BUILD_SGL_KERNEL=${{needs.check-changes.outputs.sgl_kernel}} bash scripts/ci/ci_install_dependency.sh CUSTOM_BUILD_SGL_KERNEL=${{needs.check-changes.outputs.sgl_kernel}} bash scripts/ci/ci_install_dependency.sh
- name: Run test - name: Run test
timeout-minutes: 30 timeout-minutes: 20
run: | run: |
cd test/srt cd test/srt
python3 run_suite.py --suite per-commit-4-gpu --auto-partition-id ${{ matrix.part }} --auto-partition-size 3 --enable-retry python3 run_suite.py --suite per-commit-4-gpu --auto-partition-id ${{ matrix.part }} --auto-partition-size 3
unit-test-backend-8-gpu-h200: unit-test-backend-8-gpu-h200:
needs: [check-changes, call-gate, unit-test-backend-2-gpu] needs: [check-changes, call-gate, unit-test-backend-2-gpu]
@@ -722,10 +721,10 @@ jobs:
python3 -m sglang.compile_deep_gemm --model deepseek-ai/DeepSeek-V3-0324 --tp 8 --trust-remote-code python3 -m sglang.compile_deep_gemm --model deepseek-ai/DeepSeek-V3-0324 --tp 8 --trust-remote-code
- name: Run test - name: Run test
timeout-minutes: 30 timeout-minutes: 20
run: | run: |
cd test/srt cd test/srt
python3 run_suite.py --suite per-commit-8-gpu-h200 --auto-partition-id ${{ matrix.part }} --auto-partition-size 3 --enable-retry python3 run_suite.py --suite per-commit-8-gpu-h200 --auto-partition-id ${{ matrix.part }} --auto-partition-size 3
unit-test-backend-8-gpu-h20: unit-test-backend-8-gpu-h20:
needs: [check-changes, call-gate, unit-test-backend-2-gpu] needs: [check-changes, call-gate, unit-test-backend-2-gpu]
@@ -764,10 +763,10 @@ jobs:
CUSTOM_BUILD_SGL_KERNEL=${{needs.check-changes.outputs.sgl_kernel}} bash scripts/ci/ci_install_deepep.sh CUSTOM_BUILD_SGL_KERNEL=${{needs.check-changes.outputs.sgl_kernel}} bash scripts/ci/ci_install_deepep.sh
- name: Run test - name: Run test
timeout-minutes: 30 timeout-minutes: 20
run: | run: |
cd test/srt cd test/srt
python3 run_suite.py --suite per-commit-8-gpu-h20 --auto-partition-id ${{ matrix.part }} --auto-partition-size 2 --enable-retry python3 run_suite.py --suite per-commit-8-gpu-h20 --auto-partition-id ${{ matrix.part }} --auto-partition-size 2
performance-test-1-gpu-part-1: performance-test-1-gpu-part-1:
needs: [check-changes, call-gate, stage-a-test-1] needs: [check-changes, call-gate, stage-a-test-1]
@@ -1056,12 +1055,10 @@ jobs:
pip install -e . pip install -e .
- name: Evaluate accuracy - name: Evaluate accuracy
timeout-minutes: 25 timeout-minutes: 20
run: | run: |
cd test/srt cd test/srt
python3 test_eval_accuracy_large.py python3 test_eval_accuracy_large.py
env:
SGLANG_CI_ENABLE_RETRY: ${{ env.SGLANG_CI_ENABLE_RETRY }}
accuracy-test-2-gpu: accuracy-test-2-gpu:
needs: [check-changes, call-gate, accuracy-test-1-gpu] needs: [check-changes, call-gate, accuracy-test-1-gpu]
@@ -1098,12 +1095,10 @@ jobs:
pip install -e . pip install -e .
- name: Evaluate accuracy (TP=2) - name: Evaluate accuracy (TP=2)
timeout-minutes: 25 timeout-minutes: 20
run: | run: |
cd test/srt cd test/srt
python3 test_moe_eval_accuracy_large.py python3 test_moe_eval_accuracy_large.py
env:
SGLANG_CI_ENABLE_RETRY: ${{ env.SGLANG_CI_ENABLE_RETRY }}
unit-test-deepep-4-gpu: unit-test-deepep-4-gpu:
needs: [check-changes, call-gate, unit-test-backend-2-gpu] needs: [check-changes, call-gate, unit-test-backend-2-gpu]
@@ -1137,10 +1132,10 @@ jobs:
CUSTOM_BUILD_SGL_KERNEL=${{needs.check-changes.outputs.sgl_kernel}} bash scripts/ci/ci_install_deepep.sh CUSTOM_BUILD_SGL_KERNEL=${{needs.check-changes.outputs.sgl_kernel}} bash scripts/ci/ci_install_deepep.sh
- name: Run test - name: Run test
timeout-minutes: 30 timeout-minutes: 20
run: | run: |
cd test/srt cd test/srt
python3 run_suite.py --suite per-commit-4-gpu-deepep --enable-retry python3 run_suite.py --suite per-commit-4-gpu-deepep
unit-test-deepep-8-gpu: unit-test-deepep-8-gpu:
needs: [check-changes, call-gate, unit-test-backend-2-gpu] needs: [check-changes, call-gate, unit-test-backend-2-gpu]
@@ -1174,10 +1169,10 @@ jobs:
CUSTOM_BUILD_SGL_KERNEL=${{needs.check-changes.outputs.sgl_kernel}} bash scripts/ci/ci_install_deepep.sh CUSTOM_BUILD_SGL_KERNEL=${{needs.check-changes.outputs.sgl_kernel}} bash scripts/ci/ci_install_deepep.sh
- name: Run test - name: Run test
timeout-minutes: 30 timeout-minutes: 20
run: | run: |
cd test/srt cd test/srt
python3 run_suite.py --suite per-commit-8-gpu-h200-deepep --enable-retry python3 run_suite.py --suite per-commit-8-gpu-h200-deepep
unit-test-backend-4-gpu-b200: unit-test-backend-4-gpu-b200:
needs: [check-changes, call-gate, unit-test-backend-2-gpu] needs: [check-changes, call-gate, unit-test-backend-2-gpu]
@@ -1216,10 +1211,10 @@ jobs:
CUSTOM_BUILD_SGL_KERNEL=${{needs.check-changes.outputs.sgl_kernel}} IS_BLACKWELL=1 bash scripts/ci/ci_install_dependency.sh CUSTOM_BUILD_SGL_KERNEL=${{needs.check-changes.outputs.sgl_kernel}} IS_BLACKWELL=1 bash scripts/ci/ci_install_dependency.sh
- name: Run test - name: Run test
timeout-minutes: 40 timeout-minutes: 30
run: | run: |
cd test/srt cd test/srt
IS_BLACKWELL=1 python3 run_suite.py --suite per-commit-4-gpu-b200 --auto-partition-id ${{ matrix.part }} --auto-partition-size 3 --timeout-per-file 1800 --enable-retry IS_BLACKWELL=1 python3 run_suite.py --suite per-commit-4-gpu-b200 --auto-partition-id ${{ matrix.part }} --auto-partition-size 3 --timeout-per-file 1800
# TODO: Add gb200 tests back after the ci runner is fixed # TODO: Add gb200 tests back after the ci runner is fixed
# unit-test-backend-4-gpu-gb200: # unit-test-backend-4-gpu-gb200:
@@ -1256,10 +1251,10 @@ jobs:
# CUSTOM_BUILD_SGL_KERNEL=${{needs.check-changes.outputs.sgl_kernel}} IS_BLACKWELL=1 GRACE_BLACKWELL=1 bash scripts/ci/ci_install_deepep.sh # CUSTOM_BUILD_SGL_KERNEL=${{needs.check-changes.outputs.sgl_kernel}} IS_BLACKWELL=1 GRACE_BLACKWELL=1 bash scripts/ci/ci_install_deepep.sh
# - name: Run test # - name: Run test
# timeout-minutes: 55 # timeout-minutes: 45
# run: | # run: |
# cd test/srt # cd test/srt
# python3 run_suite.py --suite per-commit-4-gpu-gb200 --auto-partition-id 0 --auto-partition-size 1 --timeout-per-file 3600 --enable-retry # python3 run_suite.py --suite per-commit-4-gpu-gb200 --auto-partition-id 0 --auto-partition-size 1 --timeout-per-file 3600
pr-test-finish: pr-test-finish:
needs: needs:
+51 -203
View File
@@ -1,6 +1,4 @@
import logging
import os import os
import re
import subprocess import subprocess
import threading import threading
import time import time
@@ -9,8 +7,6 @@ from typing import Callable, List, Optional
from sglang.srt.utils.common import kill_process_tree from sglang.srt.utils.common import kill_process_tree
logger = logging.getLogger(__name__)
@dataclass @dataclass
class TestFile: class TestFile:
@@ -18,61 +14,6 @@ class TestFile:
estimated_time: float = 60 estimated_time: float = 60
# Patterns that indicate retriable accuracy/performance failures
RETRIABLE_PATTERNS = [
r"AssertionError:.*not greater than",
r"AssertionError:.*not less than",
r"AssertionError:.*not equal to",
r"AssertionError:.*!=.*expected",
r"accuracy",
r"score",
r"latency",
r"throughput",
]
# Patterns that indicate non-retriable failures (real code errors)
NON_RETRIABLE_PATTERNS = [
r"SyntaxError",
r"ImportError",
r"ModuleNotFoundError",
r"NameError",
r"TypeError",
r"AttributeError",
r"RuntimeError",
r"CUDA out of memory",
r"OOM",
r"Segmentation fault",
r"core dumped",
r"ConnectionRefusedError",
r"FileNotFoundError",
]
def is_retriable_failure(output: str) -> tuple[bool, str]:
"""
Determine if a test failure is retriable based on output patterns.
Returns:
tuple: (is_retriable, reason)
"""
# Check for non-retriable patterns first
for pattern in NON_RETRIABLE_PATTERNS:
if re.search(pattern, output, re.IGNORECASE):
return False, f"non-retriable error: {pattern}"
# Check for retriable patterns
for pattern in RETRIABLE_PATTERNS:
if re.search(pattern, output, re.IGNORECASE):
return True, f"retriable pattern: {pattern}"
# If we have an AssertionError but didn't match non-retriable, assume retriable
if re.search(r"AssertionError", output):
return True, "AssertionError (assuming retriable)"
# Default: not retriable
return False, "unknown failure type"
def run_with_timeout( def run_with_timeout(
func: Callable, func: Callable,
args: tuple = (), args: tuple = (),
@@ -97,21 +38,8 @@ def run_with_timeout(
return ret_value[0] return ret_value[0]
def write_github_step_summary(content: str):
"""Write content to GitHub Step Summary if available."""
summary_file = os.environ.get("GITHUB_STEP_SUMMARY")
if summary_file:
with open(summary_file, "a") as f:
f.write(content)
def run_unittest_files( def run_unittest_files(
files: List[TestFile], files: List[TestFile], timeout_per_file: float, continue_on_error: bool = False
timeout_per_file: float,
continue_on_error: bool = False,
enable_retry: bool = False,
max_attempts: int = 2,
retry_wait_seconds: int = 60,
): ):
""" """
Run a list of test files. Run a list of test files.
@@ -121,166 +49,86 @@ def run_unittest_files(
timeout_per_file: Timeout in seconds for each test file timeout_per_file: Timeout in seconds for each test file
continue_on_error: If True, continue running remaining tests even if one fails. continue_on_error: If True, continue running remaining tests even if one fails.
If False, stop at first failure (default behavior for PR tests). If False, stop at first failure (default behavior for PR tests).
enable_retry: If True, retry failed tests that appear to be accuracy/performance
assertion failures (not code errors).
max_attempts: Maximum number of attempts per file including initial run (default: 2).
retry_wait_seconds: Seconds to wait between retries (default: 60).
""" """
tic = time.perf_counter() tic = time.perf_counter()
success = True success = True
passed_tests = [] passed_tests = []
failed_tests = [] failed_tests = []
retried_tests = [] # Track which tests were retried
for i, file in enumerate(files): for i, file in enumerate(files):
filename, estimated_time = file.name, file.estimated_time filename, estimated_time = file.name, file.estimated_time
process = None process = None
output_lines = []
def run_one_file(filename, capture_output=False): def run_one_file(filename):
nonlocal process, output_lines nonlocal process
full_path = os.path.join(os.getcwd(), filename) filename = os.path.join(os.getcwd(), filename)
logger.info( print(
f".\n.\nBegin ({i}/{len(files) - 1}):\npython3 {full_path}\n.\n.\n" f".\n.\nBegin ({i}/{len(files) - 1}):\npython3 {filename}\n.\n.\n",
flush=True,
) )
file_tic = time.perf_counter() tic = time.perf_counter()
if capture_output: process = subprocess.Popen(
# Capture output for retry decision ["python3", filename], stdout=None, stderr=None, env=os.environ
process = subprocess.Popen( )
["python3", full_path], process.wait()
stdout=subprocess.PIPE, elapsed = time.perf_counter() - tic
stderr=subprocess.STDOUT,
env=os.environ,
text=True,
)
output_lines = []
for line in process.stdout:
logger.info(line.rstrip())
output_lines.append(line)
process.wait()
else:
process = subprocess.Popen(
["python3", full_path], stdout=None, stderr=None, env=os.environ
)
process.wait()
elapsed = time.perf_counter() - file_tic print(
f".\n.\nEnd ({i}/{len(files) - 1}):\n{filename=}, {elapsed=:.0f}, {estimated_time=}\n.\n.\n",
logger.info( flush=True,
f".\n.\nEnd ({i}/{len(files) - 1}):\n{filename=}, {elapsed=:.0f}, {estimated_time=}\n.\n.\n"
) )
return process.returncode return process.returncode
# Retry loop for each file try:
attempt = 1 ret_code = run_with_timeout(
file_passed = False run_one_file, args=(filename,), timeout=timeout_per_file
was_retried = False )
if ret_code != 0:
while attempt <= (max_attempts if enable_retry else 1): print(
if attempt > 1: f"\n✗ FAILED: {filename} returned exit code {ret_code}\n",
logger.info( flush=True,
f"\n[CI Retry] Attempt {attempt}/{max_attempts} for {filename}\n"
) )
was_retried = True success = False
failed_tests.append((filename, f"exit code {ret_code}"))
try: if not continue_on_error:
ret_code = run_with_timeout( # Stop at first failure for PR tests
run_one_file,
args=(filename,),
kwargs={"capture_output": enable_retry},
timeout=timeout_per_file,
)
if ret_code == 0:
file_passed = True
if was_retried:
logger.info(
f"\n✓ PASSED on retry (attempt {attempt}): {filename}\n"
)
retried_tests.append((filename, attempt, "passed"))
passed_tests.append(filename)
break break
else: # Otherwise continue to next test for nightly tests
# Check if we should retry else:
if enable_retry and attempt < max_attempts: passed_tests.append(filename)
output = "".join(output_lines) except TimeoutError:
is_retriable, reason = is_retriable_failure(output) kill_process_tree(process.pid)
time.sleep(5)
if is_retriable: print(
logger.info(f"\n[CI Retry] {filename} failed with {reason}") f"\n✗ TIMEOUT: {filename} after {timeout_per_file} seconds\n",
logger.info( flush=True,
f"[CI Retry] Waiting {retry_wait_seconds}s before retry...\n" )
)
time.sleep(retry_wait_seconds)
attempt += 1
continue
else:
logger.info(
f"\n[CI Retry] {filename} failed with {reason} - not retrying\n"
)
# No retry or not retriable
logger.info(
f"\n✗ FAILED: {filename} returned exit code {ret_code}\n"
)
if was_retried:
retried_tests.append((filename, attempt, "failed"))
failed_tests.append((filename, f"exit code {ret_code}"))
break
except TimeoutError:
kill_process_tree(process.pid)
time.sleep(5)
logger.info(
f"\n✗ TIMEOUT: {filename} after {timeout_per_file} seconds\n"
)
if was_retried:
retried_tests.append((filename, attempt, "timeout"))
failed_tests.append((filename, f"timeout after {timeout_per_file}s"))
break
if not file_passed:
success = False success = False
failed_tests.append((filename, f"timeout after {timeout_per_file}s"))
if not continue_on_error: if not continue_on_error:
# Stop at first timeout for PR tests
break break
# Otherwise continue to next test for nightly tests
elapsed_total = time.perf_counter() - tic
if success: if success:
logger.info(f"Success. Time elapsed: {elapsed_total:.2f}s") print(f"Success. Time elapsed: {time.perf_counter() - tic:.2f}s", flush=True)
else: else:
logger.info(f"Fail. Time elapsed: {elapsed_total:.2f}s") print(f"Fail. Time elapsed: {time.perf_counter() - tic:.2f}s", flush=True)
# Print summary # Print summary
logger.info(f"\n{'='*60}") print(f"\n{'='*60}", flush=True)
logger.info(f"Test Summary: {len(passed_tests)}/{len(files)} passed") print(f"Test Summary: {len(passed_tests)}/{len(files)} passed", flush=True)
if enable_retry and retried_tests: print(f"{'='*60}", flush=True)
logger.info(f"Retries: {len(retried_tests)} test(s) were retried")
logger.info(f"{'='*60}")
if passed_tests: if passed_tests:
logger.info("✓ PASSED:") print("✓ PASSED:", flush=True)
for test in passed_tests: for test in passed_tests:
logger.info(f" {test}") print(f" {test}", flush=True)
if failed_tests: if failed_tests:
logger.info("\n✗ FAILED:") print("\n✗ FAILED:", flush=True)
for test, reason in failed_tests: for test, reason in failed_tests:
logger.info(f" {test} ({reason})") print(f" {test} ({reason})", flush=True)
if retried_tests: print(f"{'='*60}\n", flush=True)
logger.info("\n↻ RETRIED:")
for test, attempts, result in retried_tests:
logger.info(f" {test} ({attempts} attempts, {result})")
logger.info(f"{'='*60}\n")
# Write GitHub Step Summary
if retried_tests:
summary = "\n### CI Retry Summary\n\n"
summary += "| Test File | Attempts | Result |\n"
summary += "|-----------|----------|--------|\n"
for test, attempts, result in retried_tests:
summary += f"| `{test}` | {attempts} | {result} |\n"
summary += "\n"
write_github_step_summary(summary)
return 0 if success else -1 return 0 if success else -1
-21
View File
@@ -136,9 +136,6 @@ def run_a_suite(args):
test_files, test_files,
timeout_per_file=args.timeout_per_file, timeout_per_file=args.timeout_per_file,
continue_on_error=args.continue_on_error, continue_on_error=args.continue_on_error,
enable_retry=args.enable_retry,
max_attempts=args.max_attempts,
retry_wait_seconds=args.retry_wait_seconds,
) )
@@ -181,24 +178,6 @@ def main():
type=int, type=int,
help="Use auto load balancing. The number of parts.", help="Use auto load balancing. The number of parts.",
) )
parser.add_argument(
"--enable-retry",
action="store_true",
default=False,
help="Enable smart retry for accuracy/performance assertion failures (not code errors)",
)
parser.add_argument(
"--max-attempts",
type=int,
default=2,
help="Maximum number of attempts per file including initial run (default: 2)",
)
parser.add_argument(
"--retry-wait-seconds",
type=int,
default=60,
help="Seconds to wait between retries (default: 60)",
)
args = parser.parse_args() args = parser.parse_args()
# Validate auto-partition arguments # Validate auto-partition arguments
+1 -26
View File
@@ -511,24 +511,6 @@ def main():
default=False, default=False,
help="Continue running remaining tests even if one fails (useful for nightly tests)", help="Continue running remaining tests even if one fails (useful for nightly tests)",
) )
arg_parser.add_argument(
"--enable-retry",
action="store_true",
default=False,
help="Enable smart retry for accuracy/performance assertion failures (not code errors)",
)
arg_parser.add_argument(
"--max-attempts",
type=int,
default=2,
help="Maximum number of attempts per file including initial run (default: 2)",
)
arg_parser.add_argument(
"--retry-wait-seconds",
type=int,
default=60,
help="Seconds to wait between retries (default: 60)",
)
args = arg_parser.parse_args() args = arg_parser.parse_args()
print(f"{args=}") print(f"{args=}")
@@ -544,14 +526,7 @@ def main():
print("The running tests are ", [f.name for f in files]) print("The running tests are ", [f.name for f in files])
exit_code = run_unittest_files( exit_code = run_unittest_files(files, args.timeout_per_file, args.continue_on_error)
files,
args.timeout_per_file,
args.continue_on_error,
enable_retry=args.enable_retry,
max_attempts=args.max_attempts,
retry_wait_seconds=args.retry_wait_seconds,
)
exit(exit_code) exit(exit_code)