ci: use rerun_failed_jobs for skipped workflows in /rerun-failed-ci (#23008)

Co-authored-by: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
This commit is contained in:
Jia Guo
2026-04-21 23:59:15 -07:00
committed by GitHub
co-authored by Claude Opus 4.6
parent 88c9bab830
commit 286fba2073
+50 -41
View File
@@ -192,62 +192,71 @@ def handle_rerun_failed_ci(gh_repo, pr, comment, user_perms, react_on_success=Tr
head_sha = pr.head.sha head_sha = pr.head.sha
print(f"Checking workflows for commit: {head_sha}") print(f"Checking workflows for commit: {head_sha}")
# If PR has sgl-kernel changes, check whether the wheel build already # If PR has sgl-kernel changes, check whether ALL wheel builds already
# succeeded for this commit. If so, we can skip the full rerun and just # succeeded for this commit (CUDA + ARM). If so, we can use
# rerun failed jobs — avoids retriggering all tests (including flaky ones). # rerun_failed_jobs and avoid retriggering all tests. If any wheel
# build is pending/failed, a dependent job could fail for missing
# artifacts, so fall back to full rerun.
# Check-runs display names: "Build Wheel (<python>, <cuda>)" (CUDA) and
# "Build Wheel Arm (<python>, <cuda>)" (ARM). The YAML job ids
# sgl-kernel-build-wheels{,-arm} are NOT what the check-runs API
# returns — it returns the job's `name:` field.
kernel_wheel_built = False kernel_wheel_built = False
if sgl_kernel_changes: if sgl_kernel_changes:
try: try:
check_runs = gh_repo.get_commit(head_sha).get_check_runs() wheel_builds = [
for cr in check_runs: cr
if "sgl-kernel-build-wheels" in cr.name and cr.conclusion == "success": for cr in gh_repo.get_commit(head_sha).get_check_runs()
kernel_wheel_built = True if cr.name.startswith("Build Wheel")
print( ]
f"sgl-kernel-build-wheels already passed (check run {cr.id})" kernel_wheel_built = bool(wheel_builds) and all(
" - using rerun_failed_jobs" cr.conclusion == "success" for cr in wheel_builds
) )
break print(
if not kernel_wheel_built: f"All {len(wheel_builds)} kernel wheel build(s) passed - using rerun_failed_jobs"
print( if kernel_wheel_built
"sgl-kernel-build-wheels has not passed yet" else f"Kernel wheel not fully built "
" - will use full rerun" f"({sum(1 for c in wheel_builds if c.conclusion == 'success')}"
) f"/{len(wheel_builds)} success) - will use full rerun"
)
except Exception as e: except Exception as e:
print( print(
f"Failed to check sgl-kernel-build-wheels status: {e}" f"Failed to check kernel wheel status: {e} - falling back to full rerun"
" - falling back to full rerun"
) )
# List all workflow runs for this commit # Rerun workflows with conclusion=failure or conclusion=skipped.
#
# - failure: use rerun_failed_jobs() which reruns failed jobs *and their
# dependent jobs* (GitHub API). Fast-fail cascades call
# core.setFailed(...) so their conclusion is "failure" and are covered.
# - skipped: the entire run was skipped (no jobs ran), so there are no
# failed jobs for rerun_failed_jobs() to target. Use run.rerun().
# - kernel wheel escape: if the PR touches sgl-kernel and not all wheel
# builds are success yet, full-rerun failure runs too — Build Wheel
# lives in pr-test-sgl-kernel.yml, consumers in pr-test.yml, and
# rerun_failed_jobs() is scoped to a single workflow run.
runs = gh_repo.get_workflow_runs(head_sha=head_sha) runs = gh_repo.get_workflow_runs(head_sha=head_sha)
rerun_count = 0 rerun_count = 0
for run in runs: for run in runs:
if run.status != "completed": if run.status != "completed":
continue continue
if run.conclusion not in ("failure", "skipped"):
continue
if run.conclusion == "failure": print(f"Processing {run.conclusion} workflow: {run.name} (ID: {run.id})")
print(f"Rerunning failed workflow: {run.name} (ID: {run.id})") try:
try: if run.conclusion == "skipped" or (
if sgl_kernel_changes and not kernel_wheel_built: sgl_kernel_changes and not kernel_wheel_built
# Full rerun to ensure sgl-kernel-build-wheels runs ):
# and produces fresh artifacts for dependent jobs print(" Full rerun")
run.rerun()
else:
# Use rerun_failed_jobs for efficiency on failures
run.rerun_failed_jobs()
rerun_count += 1
except Exception as e:
print(f"Failed to rerun workflow {run.id}: {e}")
elif run.conclusion == "skipped":
print(f"Rerunning skipped workflow: {run.name} (ID: {run.id})")
try:
# Skipped workflows don't have 'failed jobs', so we use full rerun()
run.rerun() run.rerun()
rerun_count += 1 else:
except Exception as e: print(" rerun_failed_jobs")
print(f"Failed to rerun workflow {run.id}: {e}") run.rerun_failed_jobs()
rerun_count += 1
except Exception as e:
print(f"Failed to rerun workflow {run.id}: {e}")
if rerun_count > 0: if rerun_count > 0:
print(f"Triggered rerun for {rerun_count} workflows.") print(f"Triggered rerun for {rerun_count} workflows.")