fix: improving format and design (#15791)

This commit is contained in:
Douglas Yang
2025-12-25 13:13:27 -08:00
committed by GitHub
parent caa95c7eb4
commit f2ccc44240
2 changed files with 116 additions and 53 deletions
+41 -20
View File
@@ -699,10 +699,8 @@ class SGLangFailuresAnalyzer:
# Build final results # Build final results
job_streak_data = {} job_streak_data = {}
for job_name in job_current_streak.keys(): for job_name in job_current_streak.keys():
# Get last 5 runs (most recent first) # Get last 5 runs (oldest to latest, chronological order)
recent_runs = job_recent_runs.get(job_name, [])[-5:][ recent_runs = job_recent_runs.get(job_name, [])[-5:]
::-1
] # Last 5, reversed
job_streak_data[job_name] = { job_streak_data[job_name] = {
"current_streak": job_current_streak[job_name], "current_streak": job_current_streak[job_name],
@@ -832,8 +830,10 @@ class SGLangFailuresAnalyzer:
if broken: if broken:
print(f"## {title} ({len(broken)} jobs with active streaks)") print(f"## {title} ({len(broken)} jobs with active streaks)")
print("=" * 130) print("=" * 130)
# Add sub-header to clarify this shows ongoing failure streaks
print("\n🔥 Consecutive failures (>=2) & currently failing")
print( print(
f"\n{'Job Name':<40} {'Current':<8} {'Max':<6} {'Runs':<6} {'First':<13} {'Last':<13} {'Recent History':<30}" f"\n{'Job Name':<40} {'Current':<8} {'Max':<6} {'Runs':<6} {'First':<13} {'Last':<13} {'Recent Runs (oldest → latest)':<30}"
) )
print("-" * 130) print("-" * 130)
for job_name, d in broken[:15]: for job_name, d in broken[:15]:
@@ -856,7 +856,7 @@ class SGLangFailuresAnalyzer:
# Recent history (last 5 runs as emoji) # Recent history (last 5 runs as emoji)
recent_runs = d.get("recent_runs", []) recent_runs = d.get("recent_runs", [])
history_str = ( history_str = (
" ".join([r["status"] for r in recent_runs]) "… " + " ".join([r["status"] for r in recent_runs])
if recent_runs if recent_runs
else "N/A" else "N/A"
) )
@@ -864,11 +864,11 @@ class SGLangFailuresAnalyzer:
# Color red if color_failures is True (for critical sections) # Color red if color_failures is True (for critical sections)
if color_failures: if color_failures:
print( print(
f"\033[91m{display_name:<40}\033[0m {d['current_streak']:<8} {d['max_streak']:<6} {d['total_runs']:<6} {first_str:<13} {last_str:<13} {history_str:<30}" f"\033[91m{display_name:<40}\033[0m {d['current_streak']:<8} {d['max_streak']:<6} {d['total_runs']:<6} {first_str:<13} {last_str:<13} {history_str:<32}"
) )
else: else:
print( print(
f"{display_name:<40} {d['current_streak']:<8} {d['max_streak']:<6} {d['total_runs']:<6} {first_str:<13} {last_str:<13} {history_str:<30}" f"{display_name:<40} {d['current_streak']:<8} {d['max_streak']:<6} {d['total_runs']:<6} {first_str:<13} {last_str:<13} {history_str:<32}"
) )
else: else:
print(f"## {title}") print(f"## {title}")
@@ -877,11 +877,17 @@ class SGLangFailuresAnalyzer:
# Show recently failed jobs in a collapsed section (terminal doesn't support collapse, so just show as separate section) # Show recently failed jobs in a collapsed section (terminal doesn't support collapse, so just show as separate section)
if recently_failed: if recently_failed:
# Extract just the workflow name without the run count for cleaner display
short_title = title.split("(")[0].strip()
if short_title and short_title[0].isdigit():
short_title = short_title.split(".", 1)[-1].strip()
# Get the max total_runs from recently_failed jobs to show the analysis window
max_total_runs = max(d["total_runs"] for _, d in recently_failed)
print( print(
f"\n Recently failed jobs (no active streak): {len(recently_failed)} jobs" f"\n 📋 [{short_title}] No current failure streak, but had failures in the past {max_total_runs} runs - {len(recently_failed)} jobs"
) )
print( print(
f" {'Job Name':<38} {'Failures':<12} {'Fail Rate':<12} {'Total Runs':<12} {'Recent History (last 5)':<30}" f" {'Job Name':<38} {'Failures':<12} {'Fail Rate':<12} {'Total Runs':<12} {'Recent Runs (oldest → latest)':<30}"
) )
print(" " + "-" * 120) print(" " + "-" * 120)
for job_name, d in recently_failed[:10]: for job_name, d in recently_failed[:10]:
@@ -890,12 +896,12 @@ class SGLangFailuresAnalyzer:
) )
recent_runs = d.get("recent_runs", []) recent_runs = d.get("recent_runs", [])
history_str = ( history_str = (
" ".join([r["status"] for r in recent_runs]) "… " + " ".join([r["status"] for r in recent_runs])
if recent_runs if recent_runs
else "N/A" else "N/A"
) )
print( print(
f" {display_name:<38} {d['total_failures']:<12} {d['failure_rate']:.1f}%{'':<7} {d['total_runs']:<12} {history_str:<30}" f" {display_name:<38} {d['total_failures']:<12} {d['failure_rate']:.1f}%{'':<7} {d['total_runs']:<12} {history_str:<32}"
) )
# ========== SCHEDULED/MAIN BRANCH RUNS (9 sections) ========== # ========== SCHEDULED/MAIN BRANCH RUNS (9 sections) ==========
@@ -1185,7 +1191,9 @@ class SGLangFailuresAnalyzer:
summary_lines.append( summary_lines.append(
f"**Analysis Timestamp:** {report_data['summary']['analysis_timestamp']}" f"**Analysis Timestamp:** {report_data['summary']['analysis_timestamp']}"
) )
summary_lines.append("_Note: Recent runs are shown left to right_") summary_lines.append(
"_Note: Recent runs are shown oldest → latest (left to right)_"
)
summary_lines.append("") summary_lines.append("")
# Summary stats - COLLAPSIBLE # Summary stats - COLLAPSIBLE
@@ -1267,11 +1275,16 @@ class SGLangFailuresAnalyzer:
summary_lines.append("") summary_lines.append("")
if broken: if broken:
# Add sub-header to clarify this shows ongoing failure streaks
summary_lines.append( summary_lines.append(
"| Job Name | Current | Max | Runs | First | Last | Recent History |" "🔥 **Consecutive failures (≥2) & currently failing**"
)
summary_lines.append("")
summary_lines.append(
"| Job Name | Current | Max | Runs | First | Last | Recent Runs (oldest → latest) |"
) )
summary_lines.append( summary_lines.append(
"|----------|---------|-----|------|-------|------|----------------|" "|----------|---------|-----|------|-------|------|-------------------------------|"
) )
for job_name, d in broken[:15]: for job_name, d in broken[:15]:
display_name = ( display_name = (
@@ -1295,7 +1308,7 @@ class SGLangFailuresAnalyzer:
# Recent history (last 5 runs as clickable emoji) # Recent history (last 5 runs as clickable emoji)
recent_runs = d.get("recent_runs", []) recent_runs = d.get("recent_runs", [])
if recent_runs: if recent_runs:
history_links = " ".join( history_links = "… " + " ".join(
[ [
f"[{r['status']}]({r['job_url']})" f"[{r['status']}]({r['job_url']})"
for r in recent_runs for r in recent_runs
@@ -1324,16 +1337,24 @@ class SGLangFailuresAnalyzer:
# Show recently failed jobs in a collapsible section # Show recently failed jobs in a collapsible section
if recently_failed: if recently_failed:
# Extract just the workflow name without the run count for cleaner display
# e.g., "1. PR Test NVIDIA - Scheduled (latest 12 runs)" -> "PR Test NVIDIA - Scheduled"
short_title = title.split("(")[0].strip()
# Remove the leading number and period if present
if short_title and short_title[0].isdigit():
short_title = short_title.split(".", 1)[-1].strip()
# Get the max total_runs from recently_failed jobs to show the analysis window
max_total_runs = max(d["total_runs"] for _, d in recently_failed)
summary_lines.append("<details>") summary_lines.append("<details>")
summary_lines.append( summary_lines.append(
f"<summary>Recently failed jobs (no active streak) - {len(recently_failed)} jobs</summary>" f"<summary>📋 [{short_title}] No current failure streak, but had failures in the past {max_total_runs} runs - {len(recently_failed)} jobs</summary>"
) )
summary_lines.append("") summary_lines.append("")
summary_lines.append( summary_lines.append(
"| Job Name | Failures | Fail Rate | Total Runs | Recent History (last 5) |" "| Job Name | Failures | Fail Rate | Total Runs | Recent Runs (oldest → latest) |"
) )
summary_lines.append( summary_lines.append(
"|----------|----------|-----------|------------|-------------------------|" "|----------|----------|-----------|------------|-------------------------------|"
) )
for job_name, d in recently_failed[:15]: for job_name, d in recently_failed[:15]:
display_name = ( display_name = (
@@ -1341,7 +1362,7 @@ class SGLangFailuresAnalyzer:
) )
recent_runs = d.get("recent_runs", []) recent_runs = d.get("recent_runs", [])
if recent_runs: if recent_runs:
history_links = " ".join( history_links = "… " + " ".join(
[ [
f"[{r['status']}]({r['job_url']})" f"[{r['status']}]({r['job_url']})"
for r in recent_runs for r in recent_runs
+75 -33
View File
@@ -55,21 +55,29 @@ def post_ci_failures_to_slack(report_file: str) -> bool:
critical_failures = [] critical_failures = []
# Map workflow data keys to display names # Map workflow data keys to display names and hardware category
workflow_name_map = { # Format: (display_name, hardware, test_type_order)
# PR Tests - Scheduled (5 workflows) # test_type_order: 0 = PR Test, 1 = Nightly (so PR Test comes first)
"pr_test_nvidia_scheduled_data": "PR Test (Nvidia, scheduled)", workflow_info_map = {
"pr_test_amd_scheduled_data": "PR Test (AMD, scheduled)", # Nvidia
"pr_test_xeon_scheduled_data": "PR Test (Xeon, scheduled)", "pr_test_nvidia_scheduled_data": ("PR Test", "Nvidia", 0),
"pr_test_xpu_scheduled_data": "PR Test (XPU, scheduled)", "nightly_nvidia_scheduled_data": ("Nightly", "Nvidia", 1),
"pr_test_npu_scheduled_data": "PR Test (NPU, scheduled)", # AMD
# Nightly Tests - Scheduled (4 workflows) "pr_test_amd_scheduled_data": ("PR Test", "AMD", 0),
"nightly_nvidia_scheduled_data": "Nightly Test (Nvidia, scheduled)", "nightly_amd_scheduled_data": ("Nightly", "AMD", 1),
"nightly_amd_scheduled_data": "Nightly Test (AMD, scheduled)", # Intel/Xeon
"nightly_intel_scheduled_data": "Nightly Test (Intel, scheduled)", "pr_test_xeon_scheduled_data": ("PR Test", "Intel", 0),
"nightly_npu_scheduled_data": "Nightly Test (NPU, scheduled)", "nightly_intel_scheduled_data": ("Nightly", "Intel", 1),
# XPU
"pr_test_xpu_scheduled_data": ("PR Test", "XPU", 0),
# NPU
"pr_test_npu_scheduled_data": ("PR Test", "NPU", 0),
"nightly_npu_scheduled_data": ("Nightly", "NPU", 1),
} }
# Hardware priority order (Nvidia first)
hardware_order = ["Nvidia", "AMD", "Intel", "XPU", "NPU"]
# Iterate through each workflow section # Iterate through each workflow section
for workflow_key, workflow_data in report_data.items(): for workflow_key, workflow_data in report_data.items():
# Skip non-workflow keys (summary, limits, etc.) # Skip non-workflow keys (summary, limits, etc.)
@@ -79,13 +87,12 @@ def post_ci_failures_to_slack(report_file: str) -> bool:
): ):
continue continue
# Get workflow display name # Only process scheduled workflows that are in our map
workflow_name = workflow_name_map.get(workflow_key, workflow_key) if workflow_key not in workflow_info_map:
# Only process scheduled workflows
if "scheduled" not in workflow_key.lower():
continue continue
test_type, hardware, test_order = workflow_info_map[workflow_key]
# Check each job in this workflow # Check each job in this workflow
for job_name, job_data in workflow_data.items(): for job_name, job_data in workflow_data.items():
if not isinstance(job_data, dict): if not isinstance(job_data, dict):
@@ -100,7 +107,9 @@ def post_ci_failures_to_slack(report_file: str) -> bool:
critical_failures.append( critical_failures.append(
{ {
"workflow_name": workflow_name, "hardware": hardware,
"test_type": test_type,
"test_order": test_order,
"job_name": job_name, "job_name": job_name,
"consecutive_failures": current_streak, "consecutive_failures": current_streak,
"first_failed_at": ( "first_failed_at": (
@@ -124,14 +133,18 @@ def post_ci_failures_to_slack(report_file: str) -> bool:
} }
) )
# Group by workflow # Group by hardware, then by test type
workflow_jobs = {} # Structure: {hardware: {test_type: [job_names]}}
hardware_jobs = {}
for job in critical_failures: for job in critical_failures:
workflow = job.get("workflow_name", "Unknown") hardware = job.get("hardware", "Unknown")
test_type = job.get("test_type", "Unknown")
job_name = job.get("job_name", "unknown") job_name = job.get("job_name", "unknown")
if workflow not in workflow_jobs: if hardware not in hardware_jobs:
workflow_jobs[workflow] = [] hardware_jobs[hardware] = {}
workflow_jobs[workflow].append(job_name) if test_type not in hardware_jobs[hardware]:
hardware_jobs[hardware][test_type] = []
hardware_jobs[hardware][test_type].append(job_name)
# Create summary message # Create summary message
workflow_url = "" workflow_url = ""
@@ -140,7 +153,7 @@ def post_ci_failures_to_slack(report_file: str) -> bool:
f"https://github.com/sgl-project/sglang/actions/runs/{run_id}" f"https://github.com/sgl-project/sglang/actions/runs/{run_id}"
) )
if not workflow_jobs: if not hardware_jobs:
summary = "✅ No critical failures detected in scheduled runs" summary = "✅ No critical failures detected in scheduled runs"
if workflow_url: if workflow_url:
summary += f"\n<{workflow_url}|View CI Monitor Run>" summary += f"\n<{workflow_url}|View CI Monitor Run>"
@@ -149,9 +162,20 @@ def post_ci_failures_to_slack(report_file: str) -> bool:
# Ping relevant people when there are failures # Ping relevant people when there are failures
mentions = "<@U09RR5TNC94> <@U09ABMCKQPM>" mentions = "<@U09RR5TNC94> <@U09ABMCKQPM>"
summary_lines = [f"{mentions} 🚨 *CI Critical Failures (Scheduled Runs)*"] summary_lines = [f"{mentions} 🚨 *CI Critical Failures (Scheduled Runs)*"]
for workflow, jobs in sorted(workflow_jobs.items()):
job_list = ", ".join(jobs) # Iterate in hardware priority order, with PR Test before Nightly
summary_lines.append(f"• *{workflow}*: {job_list}") test_type_order = ["PR Test", "Nightly"]
for hardware in hardware_order:
if hardware not in hardware_jobs:
continue
summary_lines.append(f"\n*{hardware}:*")
for test_type in test_type_order:
if test_type not in hardware_jobs[hardware]:
continue
jobs = hardware_jobs[hardware][test_type]
job_list = ", ".join(jobs)
summary_lines.append(f" • {test_type}: {job_list}")
if workflow_url: if workflow_url:
summary_lines.append(f"\n<{workflow_url}|View Full CI Monitor Report>") summary_lines.append(f"\n<{workflow_url}|View Full CI Monitor Report>")
summary = "\n".join(summary_lines) summary = "\n".join(summary_lines)
@@ -174,11 +198,24 @@ def post_ci_failures_to_slack(report_file: str) -> bool:
thread_ts = response["ts"] thread_ts = response["ts"]
# If there are failures, post detailed breakdown in thread # If there are failures, post detailed breakdown in thread
if workflow_jobs: if hardware_jobs:
details_lines = ["*Detailed Failure Breakdown*\n"] details_lines = ["*Detailed Failure Breakdown*\n"]
for job in critical_failures: # Sort critical_failures by hardware order, then test_order
workflow = job.get("workflow_name", "Unknown") hardware_order_map = {hw: i for i, hw in enumerate(hardware_order)}
sorted_failures = sorted(
critical_failures,
key=lambda x: (
hardware_order_map.get(x.get("hardware", ""), 99),
x.get("test_order", 99),
x.get("job_name", ""),
),
)
current_hardware = None
for job in sorted_failures:
hardware = job.get("hardware", "Unknown")
test_type = job.get("test_type", "Unknown")
job_name = job.get("job_name", "unknown") job_name = job.get("job_name", "unknown")
consecutive = job.get("consecutive_failures", 0) consecutive = job.get("consecutive_failures", 0)
first_url = job.get("first_failed_url", "") first_url = job.get("first_failed_url", "")
@@ -186,8 +223,13 @@ def post_ci_failures_to_slack(report_file: str) -> bool:
last_url = job.get("last_failed_url", "") last_url = job.get("last_failed_url", "")
last_at = job.get("last_failed_at", "unknown") last_at = job.get("last_failed_at", "unknown")
# Add hardware section header
if hardware != current_hardware:
details_lines.append(f"\n*━━━ {hardware} ━━━*")
current_hardware = hardware
details_lines.append( details_lines.append(
f"• *{workflow}* → `{job_name}`\n" f"• *{test_type}* → `{job_name}`\n"
f" Consecutive failures: {consecutive}\n" f" Consecutive failures: {consecutive}\n"
f" First failed: <{first_url}|{first_at}>\n" f" First failed: <{first_url}|{first_at}>\n"
f" Last failed: <{last_url}|{last_at}>\n" f" Last failed: <{last_url}|{last_at}>\n"