ci: fix always-failing coverage job, add by-GPU-count view (#39697)

This commit is contained in:
Alison Shao
2026-09-16 01:13:08 -07:00
committed by GitHub
parent 9c8d4641ff
commit ad28b91fae
2 changed files with 184 additions and 5 deletions
+29 -4
View File
@@ -63,10 +63,28 @@ jobs:
run: |
python scripts/ci/utils/ci_coverage_report.py --section by-suite
by-gpu-count:
name: Tests by GPU Count
runs-on: ubuntu-latest
steps:
- name: Checkout code
uses: actions/checkout@v4
- name: Set up Python
uses: actions/setup-python@v5
with:
python-version: '3.10'
- name: Generate Tests by GPU Count Report
run: |
python scripts/ci/utils/ci_coverage_report.py --section by-gpu-count
unit-test-coverage:
name: Unit Test Code Coverage
runs-on: 1-gpu-h100
timeout-minutes: 30
# ~6 min install + ~22 min of tests, and the suite keeps growing
# (7.7k collected items in Aug, 11k now).
timeout-minutes: 60
steps:
- name: Checkout code
uses: actions/checkout@v4
@@ -88,13 +106,19 @@ jobs:
pip install -e "python/[test]"
- name: Run unit tests with coverage
timeout-minutes: 10
# Non-gating on purpose: running the whole tree in one process
# cross-contaminates tests that CI proper runs one file per
# process. `| tee` used to hide that by accident; pipefail plus an
# explicit `|| echo` makes it deliberate and reports the counts.
timeout-minutes: 45
run: |
set -o pipefail
pytest test/registered/unit/ \
--cov --cov-config=.coveragerc \
--cov-report=term-missing:skip-covered \
--continue-on-collection-errors \
-v | tee coverage_output.txt
-v | tee coverage_output.txt \
|| echo "pytest exited $? -- reported below, not gating this job"
- name: Write coverage to summary
if: always()
@@ -106,7 +130,8 @@ jobs:
# Test result line (e.g., "== 42 passed, 1 failed in 23.5s ==")
echo '```' >> $GITHUB_STEP_SUMMARY
grep -E '^=+.*passed' coverage_output.txt >> $GITHUB_STEP_SUMMARY || true
grep -E '^=+.*passed' coverage_output.txt >> $GITHUB_STEP_SUMMARY \
|| echo "pytest did not reach its summary line (step killed or crashed)" >> $GITHUB_STEP_SUMMARY
echo "" >> $GITHUB_STEP_SUMMARY
# Coverage total, reformatted
awk '/^TOTAL / { for(i=1;i<=NF;i++) if($i~/^[0-9]+$/ || $i~/^[0-9]+%$/) a[++n]=$i; if(n>=3) printf "TOTAL Stmts: %s Miss: %s Cover: %s\n", a[1], a[2], a[3] }' coverage_output.txt >> $GITHUB_STEP_SUMMARY || true
+155 -1
View File
@@ -13,9 +13,11 @@ import argparse
import glob
import json
import os
import re
import sys
from collections import defaultdict
from pathlib import Path
from typing import Optional
# Add the ci_register module path directly to avoid heavy sglang imports
sys.path.insert(
@@ -240,15 +242,43 @@ def get_test_basename(filename: str) -> str:
return Path(filename).name
# Suite names carry the runner shape as a `<n>-gpu` / `<n>-npu` token, e.g.
# `base-b-test-1-gpu-small`, `base-c-test-acc-16-npu-a3`. The token is the
# only machine-readable record of how many accelerators a test asks for, so
# the GPU-count grouping parses it back out. Anchored on a dash (or the
# string start/end) so a trailing `-tp4` or a model name like `qwen3-235b`
# can't be mistaken for a runner size.
_SUITE_ACCEL_COUNT_RE = re.compile(r"(?:^|-)(\d+)-(?:gpu|npu)(?:-|$)")
def get_accel_count(test: CIRegistry) -> Optional[int]:
"""Number of GPUs/NPUs a test's suite runs on, or None if unencoded.
Returns None for suites whose name has no size token -- `stress`,
`nightly-amd-vlm`, the CPU suites, and the synthesized `mm-gen-*`
suites (multimodal_gen spreads one suite across 1-gpu and 2-gpu
runners, so there is no single answer to report).
"""
match = _SUITE_ACCEL_COUNT_RE.search(test.effective_suite or "")
return int(match.group(1)) if match else None
def accel_count_label(count: Optional[int]) -> str:
"""Row label for an accelerator count."""
return f"{count}-GPU" if count is not None else "Unsized"
def organize_test_data(tests: list[CIRegistry]) -> dict:
"""Organize tests into various groupings."""
by_backend = defaultdict(list)
by_folder = defaultdict(list)
by_accel = defaultdict(list)
disabled_tests = []
for t in tests:
by_backend[t.backend.name].append(t)
by_folder[get_folder_name(t.filename)].append(t)
by_accel[get_accel_count(t)].append(t)
if t.disabled:
disabled_tests.append(t)
@@ -266,10 +296,16 @@ def organize_test_data(tests: list[CIRegistry]) -> dict:
"disabled_unique_files": len(unique_disabled_files),
"by_backend": by_backend,
"by_folder": by_folder,
"by_accel": by_accel,
"disabled_tests": disabled_tests,
}
def sorted_accel_counts(by_accel: dict) -> list:
"""Accelerator counts in ascending order, with `None` (unsized) last."""
return sorted(by_accel.keys(), key=lambda c: (c is None, c if c is not None else 0))
def generate_summary_section(data: dict) -> str:
"""Generate the summary/overview section."""
lines = []
@@ -331,6 +367,36 @@ def generate_summary_section(data: dict) -> str:
lines.append("\n</details>\n")
# GPU count summary (collapsible). Answers "how many 1-GPU tests do we
# have?" without expanding every suite -- the bulk of the fleet is
# single-GPU runners, so this is the row that decides capacity.
lines.append("<details>")
lines.append("<summary><h2>GPU Count Summary</h2></summary>\n")
lines.append(
"*Enabled registrations, grouped by the `<n>-gpu` / `<n>-npu` size in "
"the suite name. `Unsized` covers suites with no size token: the CPU "
"suites, `stress`, and `mm-gen-*` (multimodal_gen splits one suite "
"across 1-gpu and 2-gpu runners).*\n"
)
header_cells = ["GPUs", *active_backends, "Total", "Disabled"]
lines.append("| " + " | ".join(header_cells) + " |")
lines.append("|" + "|".join(["-" * max(len(c), 3) for c in header_cells]) + "|")
by_accel = data["by_accel"]
for count in sorted_accel_counts(by_accel):
accel_tests = by_accel[count]
enabled = [t for t in accel_tests if not t.disabled]
backend_counts = {b.name: 0 for b in HWBackend}
for t in enabled:
backend_counts[t.backend.name] += 1
row = [accel_count_label(count)]
row += [str(backend_counts[b]) for b in active_backends]
row.append(str(len(enabled)))
row.append(str(len(accel_tests) - len(enabled)))
lines.append("| " + " | ".join(row) + " |")
lines.append("\n</details>\n")
# Disabled tests section (collapsible)
if disabled_tests:
lines.append("<details>")
@@ -394,6 +460,69 @@ def generate_by_folder_section(data: dict) -> str:
return "\n".join(lines)
def generate_by_gpu_count_section(data: dict) -> str:
"""Generate the 'All Tests by GPU Count' section.
Same registrations as the by-suite section, pivoted on runner size
instead of suite name, so per-GPU-count fleet demand is readable at a
glance (which backend leans on 1-GPU boxes, where the 8-GPU load sits).
"""
lines = []
by_accel = data["by_accel"]
lines.append("# All Tests by GPU Count\n")
for count in sorted_accel_counts(by_accel):
accel_tests = by_accel[count]
a_disabled = sum(1 for t in accel_tests if t.disabled)
a_enabled = len(accel_tests) - a_disabled
lines.append("<details>")
lines.append(
f"<summary><h2>{accel_count_label(count)} "
f"({a_enabled} enabled, {a_disabled} disabled)</h2></summary>\n"
)
accel_by_backend = defaultdict(list)
for t in accel_tests:
accel_by_backend[t.backend.name].append(t)
for backend in BACKEND_DISPLAY_ORDER:
backend_tests = accel_by_backend.get(backend, [])
if not backend_tests:
continue
b_disabled = sum(1 for t in backend_tests if t.disabled)
b_enabled = len(backend_tests) - b_disabled
lines.append(
f"### {backend} ({b_enabled} enabled, {b_disabled} disabled)\n"
)
lines.append("| Suite | Enabled | Disabled | Est. Time | Type |")
lines.append("|-------|---------|----------|-----------|------|")
backend_suites = defaultdict(list)
for t in backend_tests:
backend_suites[t.effective_suite].append(t)
for suite in sorted(backend_suites.keys()):
suite_tests = backend_suites[suite]
s_disabled = sum(1 for t in suite_tests if t.disabled)
s_enabled = len(suite_tests) - s_disabled
s_est_time = sum(t.est_time for t in suite_tests if not t.disabled)
is_nightly = any(t.nightly for t in suite_tests if not t.disabled)
suite_type = "Nightly" if is_nightly else "Per-Commit"
lines.append(
f"| {suite} | {s_enabled} | {s_disabled} | "
f"{s_est_time:.0f}s | {suite_type} |"
)
lines.append("")
lines.append("</details>\n")
return "\n".join(lines)
def generate_by_suite_section(data: dict) -> str:
"""Generate the 'All Tests by Test Suite' section."""
lines = []
@@ -469,6 +598,8 @@ def generate_markdown_report(tests: list[CIRegistry], section: str = "all") -> s
return generate_by_folder_section(data)
elif section == "by-suite":
return generate_by_suite_section(data)
elif section == "by-gpu-count":
return generate_by_gpu_count_section(data)
else: # "all"
parts = [
generate_summary_section(data),
@@ -476,6 +607,8 @@ def generate_markdown_report(tests: list[CIRegistry], section: str = "all") -> s
generate_by_folder_section(data),
"---",
generate_by_suite_section(data),
"---",
generate_by_gpu_count_section(data),
]
return "\n".join(parts)
@@ -502,6 +635,7 @@ def generate_json_report(tests: list[CIRegistry]) -> str:
"tests_by_suite": {},
"backend_summary": {},
"folder_summary": {},
"gpu_count_summary": {},
"disabled_tests": [],
}
@@ -606,6 +740,26 @@ def generate_json_report(tests: list[CIRegistry]) -> str:
"total": len(folder_tests),
}
# GPU count summary -- enabled registrations per runner size, keyed by
# the same labels the markdown table uses ("1-GPU", ..., "Unsized").
by_accel = defaultdict(list)
for t in tests:
by_accel[get_accel_count(t)].append(t)
for count in sorted_accel_counts(by_accel):
accel_tests = by_accel[count]
enabled = [t for t in accel_tests if not t.disabled]
backend_counts = {b: 0 for b in BACKEND_DISPLAY_ORDER}
for t in enabled:
backend_counts[t.backend.name] += 1
data["gpu_count_summary"][accel_count_label(count)] = {
**backend_counts,
"gpus": count,
"enabled": len(enabled),
"disabled": len(accel_tests) - len(enabled),
"suites": sorted({t.effective_suite for t in accel_tests}),
}
# Disabled tests
for t in sorted(disabled_tests, key=lambda x: (x.backend.name, x.filename)):
data["disabled_tests"].append(
@@ -630,7 +784,7 @@ def main():
)
parser.add_argument(
"--section",
choices=["all", "summary", "by-folder", "by-suite"],
choices=["all", "summary", "by-folder", "by-suite", "by-gpu-count"],
default="all",
help="Which section to output (default: all). Only applies to markdown format.",
)