Fix CI by reverting incorrect metric check logic (#15004)
This commit is contained in:
@@ -27,7 +27,6 @@ concurrency:
|
|||||||
|
|
||||||
env:
|
env:
|
||||||
SGLANG_IS_IN_CI: true
|
SGLANG_IS_IN_CI: true
|
||||||
SGLANG_CI_ENABLE_RETRY: true
|
|
||||||
|
|
||||||
jobs:
|
jobs:
|
||||||
# =============================================== check changes ====================================================
|
# =============================================== check changes ====================================================
|
||||||
@@ -551,10 +550,10 @@ jobs:
|
|||||||
CUSTOM_BUILD_SGL_KERNEL=${{needs.check-changes.outputs.sgl_kernel}} bash scripts/ci/ci_install_dependency.sh
|
CUSTOM_BUILD_SGL_KERNEL=${{needs.check-changes.outputs.sgl_kernel}} bash scripts/ci/ci_install_dependency.sh
|
||||||
|
|
||||||
- name: Run test
|
- name: Run test
|
||||||
timeout-minutes: 40
|
timeout-minutes: 30
|
||||||
run: |
|
run: |
|
||||||
cd test/srt
|
cd test/srt
|
||||||
python3 run_suite.py --suite quantization_test --enable-retry
|
python3 run_suite.py --suite quantization_test
|
||||||
|
|
||||||
unit-test-backend-1-gpu:
|
unit-test-backend-1-gpu:
|
||||||
needs: [check-changes, call-gate, stage-a-test-1]
|
needs: [check-changes, call-gate, stage-a-test-1]
|
||||||
@@ -593,10 +592,10 @@ jobs:
|
|||||||
CUSTOM_BUILD_SGL_KERNEL=${{needs.check-changes.outputs.sgl_kernel}} bash scripts/ci/ci_install_dependency.sh
|
CUSTOM_BUILD_SGL_KERNEL=${{needs.check-changes.outputs.sgl_kernel}} bash scripts/ci/ci_install_dependency.sh
|
||||||
|
|
||||||
- name: Run test
|
- name: Run test
|
||||||
timeout-minutes: 40
|
timeout-minutes: 30
|
||||||
run: |
|
run: |
|
||||||
cd test/srt
|
cd test/srt
|
||||||
python3 run_suite.py --suite per-commit-1-gpu --auto-partition-id ${{ matrix.part }} --auto-partition-size 15 --enable-retry
|
python3 run_suite.py --suite per-commit-1-gpu --auto-partition-id ${{ matrix.part }} --auto-partition-size 15
|
||||||
|
|
||||||
unit-test-backend-2-gpu:
|
unit-test-backend-2-gpu:
|
||||||
needs: [check-changes, call-gate, unit-test-backend-1-gpu]
|
needs: [check-changes, call-gate, unit-test-backend-1-gpu]
|
||||||
@@ -634,10 +633,10 @@ jobs:
|
|||||||
CUSTOM_BUILD_SGL_KERNEL=${{needs.check-changes.outputs.sgl_kernel}} bash scripts/ci/ci_install_dependency.sh
|
CUSTOM_BUILD_SGL_KERNEL=${{needs.check-changes.outputs.sgl_kernel}} bash scripts/ci/ci_install_dependency.sh
|
||||||
|
|
||||||
- name: Run test
|
- name: Run test
|
||||||
timeout-minutes: 40
|
timeout-minutes: 30
|
||||||
run: |
|
run: |
|
||||||
cd test/srt
|
cd test/srt
|
||||||
python3 run_suite.py --suite per-commit-2-gpu --auto-partition-id ${{ matrix.part }} --auto-partition-size 2 --enable-retry
|
python3 run_suite.py --suite per-commit-2-gpu --auto-partition-id ${{ matrix.part }} --auto-partition-size 2
|
||||||
|
|
||||||
unit-test-backend-4-gpu:
|
unit-test-backend-4-gpu:
|
||||||
needs: [check-changes, call-gate, unit-test-backend-2-gpu]
|
needs: [check-changes, call-gate, unit-test-backend-2-gpu]
|
||||||
@@ -675,10 +674,10 @@ jobs:
|
|||||||
CUSTOM_BUILD_SGL_KERNEL=${{needs.check-changes.outputs.sgl_kernel}} bash scripts/ci/ci_install_dependency.sh
|
CUSTOM_BUILD_SGL_KERNEL=${{needs.check-changes.outputs.sgl_kernel}} bash scripts/ci/ci_install_dependency.sh
|
||||||
|
|
||||||
- name: Run test
|
- name: Run test
|
||||||
timeout-minutes: 30
|
timeout-minutes: 20
|
||||||
run: |
|
run: |
|
||||||
cd test/srt
|
cd test/srt
|
||||||
python3 run_suite.py --suite per-commit-4-gpu --auto-partition-id ${{ matrix.part }} --auto-partition-size 3 --enable-retry
|
python3 run_suite.py --suite per-commit-4-gpu --auto-partition-id ${{ matrix.part }} --auto-partition-size 3
|
||||||
|
|
||||||
unit-test-backend-8-gpu-h200:
|
unit-test-backend-8-gpu-h200:
|
||||||
needs: [check-changes, call-gate, unit-test-backend-2-gpu]
|
needs: [check-changes, call-gate, unit-test-backend-2-gpu]
|
||||||
@@ -722,10 +721,10 @@ jobs:
|
|||||||
python3 -m sglang.compile_deep_gemm --model deepseek-ai/DeepSeek-V3-0324 --tp 8 --trust-remote-code
|
python3 -m sglang.compile_deep_gemm --model deepseek-ai/DeepSeek-V3-0324 --tp 8 --trust-remote-code
|
||||||
|
|
||||||
- name: Run test
|
- name: Run test
|
||||||
timeout-minutes: 30
|
timeout-minutes: 20
|
||||||
run: |
|
run: |
|
||||||
cd test/srt
|
cd test/srt
|
||||||
python3 run_suite.py --suite per-commit-8-gpu-h200 --auto-partition-id ${{ matrix.part }} --auto-partition-size 3 --enable-retry
|
python3 run_suite.py --suite per-commit-8-gpu-h200 --auto-partition-id ${{ matrix.part }} --auto-partition-size 3
|
||||||
|
|
||||||
unit-test-backend-8-gpu-h20:
|
unit-test-backend-8-gpu-h20:
|
||||||
needs: [check-changes, call-gate, unit-test-backend-2-gpu]
|
needs: [check-changes, call-gate, unit-test-backend-2-gpu]
|
||||||
@@ -764,10 +763,10 @@ jobs:
|
|||||||
CUSTOM_BUILD_SGL_KERNEL=${{needs.check-changes.outputs.sgl_kernel}} bash scripts/ci/ci_install_deepep.sh
|
CUSTOM_BUILD_SGL_KERNEL=${{needs.check-changes.outputs.sgl_kernel}} bash scripts/ci/ci_install_deepep.sh
|
||||||
|
|
||||||
- name: Run test
|
- name: Run test
|
||||||
timeout-minutes: 30
|
timeout-minutes: 20
|
||||||
run: |
|
run: |
|
||||||
cd test/srt
|
cd test/srt
|
||||||
python3 run_suite.py --suite per-commit-8-gpu-h20 --auto-partition-id ${{ matrix.part }} --auto-partition-size 2 --enable-retry
|
python3 run_suite.py --suite per-commit-8-gpu-h20 --auto-partition-id ${{ matrix.part }} --auto-partition-size 2
|
||||||
|
|
||||||
performance-test-1-gpu-part-1:
|
performance-test-1-gpu-part-1:
|
||||||
needs: [check-changes, call-gate, stage-a-test-1]
|
needs: [check-changes, call-gate, stage-a-test-1]
|
||||||
@@ -1056,12 +1055,10 @@ jobs:
|
|||||||
pip install -e .
|
pip install -e .
|
||||||
|
|
||||||
- name: Evaluate accuracy
|
- name: Evaluate accuracy
|
||||||
timeout-minutes: 25
|
timeout-minutes: 20
|
||||||
run: |
|
run: |
|
||||||
cd test/srt
|
cd test/srt
|
||||||
python3 test_eval_accuracy_large.py
|
python3 test_eval_accuracy_large.py
|
||||||
env:
|
|
||||||
SGLANG_CI_ENABLE_RETRY: ${{ env.SGLANG_CI_ENABLE_RETRY }}
|
|
||||||
|
|
||||||
accuracy-test-2-gpu:
|
accuracy-test-2-gpu:
|
||||||
needs: [check-changes, call-gate, accuracy-test-1-gpu]
|
needs: [check-changes, call-gate, accuracy-test-1-gpu]
|
||||||
@@ -1098,12 +1095,10 @@ jobs:
|
|||||||
pip install -e .
|
pip install -e .
|
||||||
|
|
||||||
- name: Evaluate accuracy (TP=2)
|
- name: Evaluate accuracy (TP=2)
|
||||||
timeout-minutes: 25
|
timeout-minutes: 20
|
||||||
run: |
|
run: |
|
||||||
cd test/srt
|
cd test/srt
|
||||||
python3 test_moe_eval_accuracy_large.py
|
python3 test_moe_eval_accuracy_large.py
|
||||||
env:
|
|
||||||
SGLANG_CI_ENABLE_RETRY: ${{ env.SGLANG_CI_ENABLE_RETRY }}
|
|
||||||
|
|
||||||
unit-test-deepep-4-gpu:
|
unit-test-deepep-4-gpu:
|
||||||
needs: [check-changes, call-gate, unit-test-backend-2-gpu]
|
needs: [check-changes, call-gate, unit-test-backend-2-gpu]
|
||||||
@@ -1137,10 +1132,10 @@ jobs:
|
|||||||
CUSTOM_BUILD_SGL_KERNEL=${{needs.check-changes.outputs.sgl_kernel}} bash scripts/ci/ci_install_deepep.sh
|
CUSTOM_BUILD_SGL_KERNEL=${{needs.check-changes.outputs.sgl_kernel}} bash scripts/ci/ci_install_deepep.sh
|
||||||
|
|
||||||
- name: Run test
|
- name: Run test
|
||||||
timeout-minutes: 30
|
timeout-minutes: 20
|
||||||
run: |
|
run: |
|
||||||
cd test/srt
|
cd test/srt
|
||||||
python3 run_suite.py --suite per-commit-4-gpu-deepep --enable-retry
|
python3 run_suite.py --suite per-commit-4-gpu-deepep
|
||||||
|
|
||||||
unit-test-deepep-8-gpu:
|
unit-test-deepep-8-gpu:
|
||||||
needs: [check-changes, call-gate, unit-test-backend-2-gpu]
|
needs: [check-changes, call-gate, unit-test-backend-2-gpu]
|
||||||
@@ -1174,10 +1169,10 @@ jobs:
|
|||||||
CUSTOM_BUILD_SGL_KERNEL=${{needs.check-changes.outputs.sgl_kernel}} bash scripts/ci/ci_install_deepep.sh
|
CUSTOM_BUILD_SGL_KERNEL=${{needs.check-changes.outputs.sgl_kernel}} bash scripts/ci/ci_install_deepep.sh
|
||||||
|
|
||||||
- name: Run test
|
- name: Run test
|
||||||
timeout-minutes: 30
|
timeout-minutes: 20
|
||||||
run: |
|
run: |
|
||||||
cd test/srt
|
cd test/srt
|
||||||
python3 run_suite.py --suite per-commit-8-gpu-h200-deepep --enable-retry
|
python3 run_suite.py --suite per-commit-8-gpu-h200-deepep
|
||||||
|
|
||||||
unit-test-backend-4-gpu-b200:
|
unit-test-backend-4-gpu-b200:
|
||||||
needs: [check-changes, call-gate, unit-test-backend-2-gpu]
|
needs: [check-changes, call-gate, unit-test-backend-2-gpu]
|
||||||
@@ -1216,10 +1211,10 @@ jobs:
|
|||||||
CUSTOM_BUILD_SGL_KERNEL=${{needs.check-changes.outputs.sgl_kernel}} IS_BLACKWELL=1 bash scripts/ci/ci_install_dependency.sh
|
CUSTOM_BUILD_SGL_KERNEL=${{needs.check-changes.outputs.sgl_kernel}} IS_BLACKWELL=1 bash scripts/ci/ci_install_dependency.sh
|
||||||
|
|
||||||
- name: Run test
|
- name: Run test
|
||||||
timeout-minutes: 40
|
timeout-minutes: 30
|
||||||
run: |
|
run: |
|
||||||
cd test/srt
|
cd test/srt
|
||||||
IS_BLACKWELL=1 python3 run_suite.py --suite per-commit-4-gpu-b200 --auto-partition-id ${{ matrix.part }} --auto-partition-size 3 --timeout-per-file 1800 --enable-retry
|
IS_BLACKWELL=1 python3 run_suite.py --suite per-commit-4-gpu-b200 --auto-partition-id ${{ matrix.part }} --auto-partition-size 3 --timeout-per-file 1800
|
||||||
|
|
||||||
# TODO: Add gb200 tests back after the ci runner is fixed
|
# TODO: Add gb200 tests back after the ci runner is fixed
|
||||||
# unit-test-backend-4-gpu-gb200:
|
# unit-test-backend-4-gpu-gb200:
|
||||||
@@ -1256,10 +1251,10 @@ jobs:
|
|||||||
# CUSTOM_BUILD_SGL_KERNEL=${{needs.check-changes.outputs.sgl_kernel}} IS_BLACKWELL=1 GRACE_BLACKWELL=1 bash scripts/ci/ci_install_deepep.sh
|
# CUSTOM_BUILD_SGL_KERNEL=${{needs.check-changes.outputs.sgl_kernel}} IS_BLACKWELL=1 GRACE_BLACKWELL=1 bash scripts/ci/ci_install_deepep.sh
|
||||||
|
|
||||||
# - name: Run test
|
# - name: Run test
|
||||||
# timeout-minutes: 55
|
# timeout-minutes: 45
|
||||||
# run: |
|
# run: |
|
||||||
# cd test/srt
|
# cd test/srt
|
||||||
# python3 run_suite.py --suite per-commit-4-gpu-gb200 --auto-partition-id 0 --auto-partition-size 1 --timeout-per-file 3600 --enable-retry
|
# python3 run_suite.py --suite per-commit-4-gpu-gb200 --auto-partition-id 0 --auto-partition-size 1 --timeout-per-file 3600
|
||||||
|
|
||||||
pr-test-finish:
|
pr-test-finish:
|
||||||
needs:
|
needs:
|
||||||
|
|||||||
@@ -1,6 +1,4 @@
|
|||||||
import logging
|
|
||||||
import os
|
import os
|
||||||
import re
|
|
||||||
import subprocess
|
import subprocess
|
||||||
import threading
|
import threading
|
||||||
import time
|
import time
|
||||||
@@ -9,8 +7,6 @@ from typing import Callable, List, Optional
|
|||||||
|
|
||||||
from sglang.srt.utils.common import kill_process_tree
|
from sglang.srt.utils.common import kill_process_tree
|
||||||
|
|
||||||
logger = logging.getLogger(__name__)
|
|
||||||
|
|
||||||
|
|
||||||
@dataclass
|
@dataclass
|
||||||
class TestFile:
|
class TestFile:
|
||||||
@@ -18,61 +14,6 @@ class TestFile:
|
|||||||
estimated_time: float = 60
|
estimated_time: float = 60
|
||||||
|
|
||||||
|
|
||||||
# Patterns that indicate retriable accuracy/performance failures
|
|
||||||
RETRIABLE_PATTERNS = [
|
|
||||||
r"AssertionError:.*not greater than",
|
|
||||||
r"AssertionError:.*not less than",
|
|
||||||
r"AssertionError:.*not equal to",
|
|
||||||
r"AssertionError:.*!=.*expected",
|
|
||||||
r"accuracy",
|
|
||||||
r"score",
|
|
||||||
r"latency",
|
|
||||||
r"throughput",
|
|
||||||
]
|
|
||||||
|
|
||||||
# Patterns that indicate non-retriable failures (real code errors)
|
|
||||||
NON_RETRIABLE_PATTERNS = [
|
|
||||||
r"SyntaxError",
|
|
||||||
r"ImportError",
|
|
||||||
r"ModuleNotFoundError",
|
|
||||||
r"NameError",
|
|
||||||
r"TypeError",
|
|
||||||
r"AttributeError",
|
|
||||||
r"RuntimeError",
|
|
||||||
r"CUDA out of memory",
|
|
||||||
r"OOM",
|
|
||||||
r"Segmentation fault",
|
|
||||||
r"core dumped",
|
|
||||||
r"ConnectionRefusedError",
|
|
||||||
r"FileNotFoundError",
|
|
||||||
]
|
|
||||||
|
|
||||||
|
|
||||||
def is_retriable_failure(output: str) -> tuple[bool, str]:
|
|
||||||
"""
|
|
||||||
Determine if a test failure is retriable based on output patterns.
|
|
||||||
|
|
||||||
Returns:
|
|
||||||
tuple: (is_retriable, reason)
|
|
||||||
"""
|
|
||||||
# Check for non-retriable patterns first
|
|
||||||
for pattern in NON_RETRIABLE_PATTERNS:
|
|
||||||
if re.search(pattern, output, re.IGNORECASE):
|
|
||||||
return False, f"non-retriable error: {pattern}"
|
|
||||||
|
|
||||||
# Check for retriable patterns
|
|
||||||
for pattern in RETRIABLE_PATTERNS:
|
|
||||||
if re.search(pattern, output, re.IGNORECASE):
|
|
||||||
return True, f"retriable pattern: {pattern}"
|
|
||||||
|
|
||||||
# If we have an AssertionError but didn't match non-retriable, assume retriable
|
|
||||||
if re.search(r"AssertionError", output):
|
|
||||||
return True, "AssertionError (assuming retriable)"
|
|
||||||
|
|
||||||
# Default: not retriable
|
|
||||||
return False, "unknown failure type"
|
|
||||||
|
|
||||||
|
|
||||||
def run_with_timeout(
|
def run_with_timeout(
|
||||||
func: Callable,
|
func: Callable,
|
||||||
args: tuple = (),
|
args: tuple = (),
|
||||||
@@ -97,21 +38,8 @@ def run_with_timeout(
|
|||||||
return ret_value[0]
|
return ret_value[0]
|
||||||
|
|
||||||
|
|
||||||
def write_github_step_summary(content: str):
|
|
||||||
"""Write content to GitHub Step Summary if available."""
|
|
||||||
summary_file = os.environ.get("GITHUB_STEP_SUMMARY")
|
|
||||||
if summary_file:
|
|
||||||
with open(summary_file, "a") as f:
|
|
||||||
f.write(content)
|
|
||||||
|
|
||||||
|
|
||||||
def run_unittest_files(
|
def run_unittest_files(
|
||||||
files: List[TestFile],
|
files: List[TestFile], timeout_per_file: float, continue_on_error: bool = False
|
||||||
timeout_per_file: float,
|
|
||||||
continue_on_error: bool = False,
|
|
||||||
enable_retry: bool = False,
|
|
||||||
max_attempts: int = 2,
|
|
||||||
retry_wait_seconds: int = 60,
|
|
||||||
):
|
):
|
||||||
"""
|
"""
|
||||||
Run a list of test files.
|
Run a list of test files.
|
||||||
@@ -121,166 +49,86 @@ def run_unittest_files(
|
|||||||
timeout_per_file: Timeout in seconds for each test file
|
timeout_per_file: Timeout in seconds for each test file
|
||||||
continue_on_error: If True, continue running remaining tests even if one fails.
|
continue_on_error: If True, continue running remaining tests even if one fails.
|
||||||
If False, stop at first failure (default behavior for PR tests).
|
If False, stop at first failure (default behavior for PR tests).
|
||||||
enable_retry: If True, retry failed tests that appear to be accuracy/performance
|
|
||||||
assertion failures (not code errors).
|
|
||||||
max_attempts: Maximum number of attempts per file including initial run (default: 2).
|
|
||||||
retry_wait_seconds: Seconds to wait between retries (default: 60).
|
|
||||||
"""
|
"""
|
||||||
tic = time.perf_counter()
|
tic = time.perf_counter()
|
||||||
success = True
|
success = True
|
||||||
passed_tests = []
|
passed_tests = []
|
||||||
failed_tests = []
|
failed_tests = []
|
||||||
retried_tests = [] # Track which tests were retried
|
|
||||||
|
|
||||||
for i, file in enumerate(files):
|
for i, file in enumerate(files):
|
||||||
filename, estimated_time = file.name, file.estimated_time
|
filename, estimated_time = file.name, file.estimated_time
|
||||||
process = None
|
process = None
|
||||||
output_lines = []
|
|
||||||
|
|
||||||
def run_one_file(filename, capture_output=False):
|
def run_one_file(filename):
|
||||||
nonlocal process, output_lines
|
nonlocal process
|
||||||
|
|
||||||
full_path = os.path.join(os.getcwd(), filename)
|
filename = os.path.join(os.getcwd(), filename)
|
||||||
logger.info(
|
print(
|
||||||
f".\n.\nBegin ({i}/{len(files) - 1}):\npython3 {full_path}\n.\n.\n"
|
f".\n.\nBegin ({i}/{len(files) - 1}):\npython3 {filename}\n.\n.\n",
|
||||||
|
flush=True,
|
||||||
)
|
)
|
||||||
file_tic = time.perf_counter()
|
tic = time.perf_counter()
|
||||||
|
|
||||||
if capture_output:
|
|
||||||
# Capture output for retry decision
|
|
||||||
process = subprocess.Popen(
|
process = subprocess.Popen(
|
||||||
["python3", full_path],
|
["python3", filename], stdout=None, stderr=None, env=os.environ
|
||||||
stdout=subprocess.PIPE,
|
|
||||||
stderr=subprocess.STDOUT,
|
|
||||||
env=os.environ,
|
|
||||||
text=True,
|
|
||||||
)
|
|
||||||
output_lines = []
|
|
||||||
for line in process.stdout:
|
|
||||||
logger.info(line.rstrip())
|
|
||||||
output_lines.append(line)
|
|
||||||
process.wait()
|
|
||||||
else:
|
|
||||||
process = subprocess.Popen(
|
|
||||||
["python3", full_path], stdout=None, stderr=None, env=os.environ
|
|
||||||
)
|
)
|
||||||
process.wait()
|
process.wait()
|
||||||
|
elapsed = time.perf_counter() - tic
|
||||||
|
|
||||||
elapsed = time.perf_counter() - file_tic
|
print(
|
||||||
|
f".\n.\nEnd ({i}/{len(files) - 1}):\n{filename=}, {elapsed=:.0f}, {estimated_time=}\n.\n.\n",
|
||||||
logger.info(
|
flush=True,
|
||||||
f".\n.\nEnd ({i}/{len(files) - 1}):\n{filename=}, {elapsed=:.0f}, {estimated_time=}\n.\n.\n"
|
|
||||||
)
|
)
|
||||||
return process.returncode
|
return process.returncode
|
||||||
|
|
||||||
# Retry loop for each file
|
|
||||||
attempt = 1
|
|
||||||
file_passed = False
|
|
||||||
was_retried = False
|
|
||||||
|
|
||||||
while attempt <= (max_attempts if enable_retry else 1):
|
|
||||||
if attempt > 1:
|
|
||||||
logger.info(
|
|
||||||
f"\n[CI Retry] Attempt {attempt}/{max_attempts} for {filename}\n"
|
|
||||||
)
|
|
||||||
was_retried = True
|
|
||||||
|
|
||||||
try:
|
try:
|
||||||
ret_code = run_with_timeout(
|
ret_code = run_with_timeout(
|
||||||
run_one_file,
|
run_one_file, args=(filename,), timeout=timeout_per_file
|
||||||
args=(filename,),
|
|
||||||
kwargs={"capture_output": enable_retry},
|
|
||||||
timeout=timeout_per_file,
|
|
||||||
)
|
)
|
||||||
|
if ret_code != 0:
|
||||||
if ret_code == 0:
|
print(
|
||||||
file_passed = True
|
f"\n✗ FAILED: {filename} returned exit code {ret_code}\n",
|
||||||
if was_retried:
|
flush=True,
|
||||||
logger.info(
|
|
||||||
f"\n✓ PASSED on retry (attempt {attempt}): {filename}\n"
|
|
||||||
)
|
)
|
||||||
retried_tests.append((filename, attempt, "passed"))
|
success = False
|
||||||
passed_tests.append(filename)
|
|
||||||
break
|
|
||||||
else:
|
|
||||||
# Check if we should retry
|
|
||||||
if enable_retry and attempt < max_attempts:
|
|
||||||
output = "".join(output_lines)
|
|
||||||
is_retriable, reason = is_retriable_failure(output)
|
|
||||||
|
|
||||||
if is_retriable:
|
|
||||||
logger.info(f"\n[CI Retry] {filename} failed with {reason}")
|
|
||||||
logger.info(
|
|
||||||
f"[CI Retry] Waiting {retry_wait_seconds}s before retry...\n"
|
|
||||||
)
|
|
||||||
time.sleep(retry_wait_seconds)
|
|
||||||
attempt += 1
|
|
||||||
continue
|
|
||||||
else:
|
|
||||||
logger.info(
|
|
||||||
f"\n[CI Retry] {filename} failed with {reason} - not retrying\n"
|
|
||||||
)
|
|
||||||
|
|
||||||
# No retry or not retriable
|
|
||||||
logger.info(
|
|
||||||
f"\n✗ FAILED: {filename} returned exit code {ret_code}\n"
|
|
||||||
)
|
|
||||||
if was_retried:
|
|
||||||
retried_tests.append((filename, attempt, "failed"))
|
|
||||||
failed_tests.append((filename, f"exit code {ret_code}"))
|
failed_tests.append((filename, f"exit code {ret_code}"))
|
||||||
|
if not continue_on_error:
|
||||||
|
# Stop at first failure for PR tests
|
||||||
break
|
break
|
||||||
|
# Otherwise continue to next test for nightly tests
|
||||||
|
else:
|
||||||
|
passed_tests.append(filename)
|
||||||
except TimeoutError:
|
except TimeoutError:
|
||||||
kill_process_tree(process.pid)
|
kill_process_tree(process.pid)
|
||||||
time.sleep(5)
|
time.sleep(5)
|
||||||
logger.info(
|
print(
|
||||||
f"\n✗ TIMEOUT: {filename} after {timeout_per_file} seconds\n"
|
f"\n✗ TIMEOUT: {filename} after {timeout_per_file} seconds\n",
|
||||||
|
flush=True,
|
||||||
)
|
)
|
||||||
if was_retried:
|
|
||||||
retried_tests.append((filename, attempt, "timeout"))
|
|
||||||
failed_tests.append((filename, f"timeout after {timeout_per_file}s"))
|
|
||||||
break
|
|
||||||
|
|
||||||
if not file_passed:
|
|
||||||
success = False
|
success = False
|
||||||
|
failed_tests.append((filename, f"timeout after {timeout_per_file}s"))
|
||||||
if not continue_on_error:
|
if not continue_on_error:
|
||||||
|
# Stop at first timeout for PR tests
|
||||||
break
|
break
|
||||||
|
# Otherwise continue to next test for nightly tests
|
||||||
elapsed_total = time.perf_counter() - tic
|
|
||||||
|
|
||||||
if success:
|
if success:
|
||||||
logger.info(f"Success. Time elapsed: {elapsed_total:.2f}s")
|
print(f"Success. Time elapsed: {time.perf_counter() - tic:.2f}s", flush=True)
|
||||||
else:
|
else:
|
||||||
logger.info(f"Fail. Time elapsed: {elapsed_total:.2f}s")
|
print(f"Fail. Time elapsed: {time.perf_counter() - tic:.2f}s", flush=True)
|
||||||
|
|
||||||
# Print summary
|
# Print summary
|
||||||
logger.info(f"\n{'='*60}")
|
print(f"\n{'='*60}", flush=True)
|
||||||
logger.info(f"Test Summary: {len(passed_tests)}/{len(files)} passed")
|
print(f"Test Summary: {len(passed_tests)}/{len(files)} passed", flush=True)
|
||||||
if enable_retry and retried_tests:
|
print(f"{'='*60}", flush=True)
|
||||||
logger.info(f"Retries: {len(retried_tests)} test(s) were retried")
|
|
||||||
logger.info(f"{'='*60}")
|
|
||||||
if passed_tests:
|
if passed_tests:
|
||||||
logger.info("✓ PASSED:")
|
print("✓ PASSED:", flush=True)
|
||||||
for test in passed_tests:
|
for test in passed_tests:
|
||||||
logger.info(f" {test}")
|
print(f" {test}", flush=True)
|
||||||
if failed_tests:
|
if failed_tests:
|
||||||
logger.info("\n✗ FAILED:")
|
print("\n✗ FAILED:", flush=True)
|
||||||
for test, reason in failed_tests:
|
for test, reason in failed_tests:
|
||||||
logger.info(f" {test} ({reason})")
|
print(f" {test} ({reason})", flush=True)
|
||||||
if retried_tests:
|
print(f"{'='*60}\n", flush=True)
|
||||||
logger.info("\n↻ RETRIED:")
|
|
||||||
for test, attempts, result in retried_tests:
|
|
||||||
logger.info(f" {test} ({attempts} attempts, {result})")
|
|
||||||
logger.info(f"{'='*60}\n")
|
|
||||||
|
|
||||||
# Write GitHub Step Summary
|
|
||||||
if retried_tests:
|
|
||||||
summary = "\n### CI Retry Summary\n\n"
|
|
||||||
summary += "| Test File | Attempts | Result |\n"
|
|
||||||
summary += "|-----------|----------|--------|\n"
|
|
||||||
for test, attempts, result in retried_tests:
|
|
||||||
summary += f"| `{test}` | {attempts} | {result} |\n"
|
|
||||||
summary += "\n"
|
|
||||||
write_github_step_summary(summary)
|
|
||||||
|
|
||||||
return 0 if success else -1
|
return 0 if success else -1
|
||||||
|
|||||||
@@ -136,9 +136,6 @@ def run_a_suite(args):
|
|||||||
test_files,
|
test_files,
|
||||||
timeout_per_file=args.timeout_per_file,
|
timeout_per_file=args.timeout_per_file,
|
||||||
continue_on_error=args.continue_on_error,
|
continue_on_error=args.continue_on_error,
|
||||||
enable_retry=args.enable_retry,
|
|
||||||
max_attempts=args.max_attempts,
|
|
||||||
retry_wait_seconds=args.retry_wait_seconds,
|
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
@@ -181,24 +178,6 @@ def main():
|
|||||||
type=int,
|
type=int,
|
||||||
help="Use auto load balancing. The number of parts.",
|
help="Use auto load balancing. The number of parts.",
|
||||||
)
|
)
|
||||||
parser.add_argument(
|
|
||||||
"--enable-retry",
|
|
||||||
action="store_true",
|
|
||||||
default=False,
|
|
||||||
help="Enable smart retry for accuracy/performance assertion failures (not code errors)",
|
|
||||||
)
|
|
||||||
parser.add_argument(
|
|
||||||
"--max-attempts",
|
|
||||||
type=int,
|
|
||||||
default=2,
|
|
||||||
help="Maximum number of attempts per file including initial run (default: 2)",
|
|
||||||
)
|
|
||||||
parser.add_argument(
|
|
||||||
"--retry-wait-seconds",
|
|
||||||
type=int,
|
|
||||||
default=60,
|
|
||||||
help="Seconds to wait between retries (default: 60)",
|
|
||||||
)
|
|
||||||
args = parser.parse_args()
|
args = parser.parse_args()
|
||||||
|
|
||||||
# Validate auto-partition arguments
|
# Validate auto-partition arguments
|
||||||
|
|||||||
+1
-26
@@ -511,24 +511,6 @@ def main():
|
|||||||
default=False,
|
default=False,
|
||||||
help="Continue running remaining tests even if one fails (useful for nightly tests)",
|
help="Continue running remaining tests even if one fails (useful for nightly tests)",
|
||||||
)
|
)
|
||||||
arg_parser.add_argument(
|
|
||||||
"--enable-retry",
|
|
||||||
action="store_true",
|
|
||||||
default=False,
|
|
||||||
help="Enable smart retry for accuracy/performance assertion failures (not code errors)",
|
|
||||||
)
|
|
||||||
arg_parser.add_argument(
|
|
||||||
"--max-attempts",
|
|
||||||
type=int,
|
|
||||||
default=2,
|
|
||||||
help="Maximum number of attempts per file including initial run (default: 2)",
|
|
||||||
)
|
|
||||||
arg_parser.add_argument(
|
|
||||||
"--retry-wait-seconds",
|
|
||||||
type=int,
|
|
||||||
default=60,
|
|
||||||
help="Seconds to wait between retries (default: 60)",
|
|
||||||
)
|
|
||||||
args = arg_parser.parse_args()
|
args = arg_parser.parse_args()
|
||||||
print(f"{args=}")
|
print(f"{args=}")
|
||||||
|
|
||||||
@@ -544,14 +526,7 @@ def main():
|
|||||||
|
|
||||||
print("The running tests are ", [f.name for f in files])
|
print("The running tests are ", [f.name for f in files])
|
||||||
|
|
||||||
exit_code = run_unittest_files(
|
exit_code = run_unittest_files(files, args.timeout_per_file, args.continue_on_error)
|
||||||
files,
|
|
||||||
args.timeout_per_file,
|
|
||||||
args.continue_on_error,
|
|
||||||
enable_retry=args.enable_retry,
|
|
||||||
max_attempts=args.max_attempts,
|
|
||||||
retry_wait_seconds=args.retry_wait_seconds,
|
|
||||||
)
|
|
||||||
exit(exit_code)
|
exit(exit_code)
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user