ci: fix always-failing coverage job, add by-GPU-count view (#39697)

This commit is contained in:
Alison Shao
2026-09-16 01:13:08 -07:00
committed by GitHub
parent 9c8d4641ff
commit ad28b91fae
2 changed files with 184 additions and 5 deletions
+29 -4
View File
@@ -63,10 +63,28 @@ jobs:
run: | run: |
python scripts/ci/utils/ci_coverage_report.py --section by-suite python scripts/ci/utils/ci_coverage_report.py --section by-suite
by-gpu-count:
name: Tests by GPU Count
runs-on: ubuntu-latest
steps:
- name: Checkout code
uses: actions/checkout@v4
- name: Set up Python
uses: actions/setup-python@v5
with:
python-version: '3.10'
- name: Generate Tests by GPU Count Report
run: |
python scripts/ci/utils/ci_coverage_report.py --section by-gpu-count
unit-test-coverage: unit-test-coverage:
name: Unit Test Code Coverage name: Unit Test Code Coverage
runs-on: 1-gpu-h100 runs-on: 1-gpu-h100
timeout-minutes: 30 # ~6 min install + ~22 min of tests, and the suite keeps growing
# (7.7k collected items in Aug, 11k now).
timeout-minutes: 60
steps: steps:
- name: Checkout code - name: Checkout code
uses: actions/checkout@v4 uses: actions/checkout@v4
@@ -88,13 +106,19 @@ jobs:
pip install -e "python/[test]" pip install -e "python/[test]"
- name: Run unit tests with coverage - name: Run unit tests with coverage
timeout-minutes: 10 # Non-gating on purpose: running the whole tree in one process
# cross-contaminates tests that CI proper runs one file per
# process. `| tee` used to hide that by accident; pipefail plus an
# explicit `|| echo` makes it deliberate and reports the counts.
timeout-minutes: 45
run: | run: |
set -o pipefail
pytest test/registered/unit/ \ pytest test/registered/unit/ \
--cov --cov-config=.coveragerc \ --cov --cov-config=.coveragerc \
--cov-report=term-missing:skip-covered \ --cov-report=term-missing:skip-covered \
--continue-on-collection-errors \ --continue-on-collection-errors \
-v | tee coverage_output.txt -v | tee coverage_output.txt \
|| echo "pytest exited $? -- reported below, not gating this job"
- name: Write coverage to summary - name: Write coverage to summary
if: always() if: always()
@@ -106,7 +130,8 @@ jobs:
# Test result line (e.g., "== 42 passed, 1 failed in 23.5s ==") # Test result line (e.g., "== 42 passed, 1 failed in 23.5s ==")
echo '```' >> $GITHUB_STEP_SUMMARY echo '```' >> $GITHUB_STEP_SUMMARY
grep -E '^=+.*passed' coverage_output.txt >> $GITHUB_STEP_SUMMARY || true grep -E '^=+.*passed' coverage_output.txt >> $GITHUB_STEP_SUMMARY \
|| echo "pytest did not reach its summary line (step killed or crashed)" >> $GITHUB_STEP_SUMMARY
echo "" >> $GITHUB_STEP_SUMMARY echo "" >> $GITHUB_STEP_SUMMARY
# Coverage total, reformatted # Coverage total, reformatted
awk '/^TOTAL / { for(i=1;i<=NF;i++) if($i~/^[0-9]+$/ || $i~/^[0-9]+%$/) a[++n]=$i; if(n>=3) printf "TOTAL Stmts: %s Miss: %s Cover: %s\n", a[1], a[2], a[3] }' coverage_output.txt >> $GITHUB_STEP_SUMMARY || true awk '/^TOTAL / { for(i=1;i<=NF;i++) if($i~/^[0-9]+$/ || $i~/^[0-9]+%$/) a[++n]=$i; if(n>=3) printf "TOTAL Stmts: %s Miss: %s Cover: %s\n", a[1], a[2], a[3] }' coverage_output.txt >> $GITHUB_STEP_SUMMARY || true
+155 -1
View File
@@ -13,9 +13,11 @@ import argparse
import glob import glob
import json import json
import os import os
import re
import sys import sys
from collections import defaultdict from collections import defaultdict
from pathlib import Path from pathlib import Path
from typing import Optional
# Add the ci_register module path directly to avoid heavy sglang imports # Add the ci_register module path directly to avoid heavy sglang imports
sys.path.insert( sys.path.insert(
@@ -240,15 +242,43 @@ def get_test_basename(filename: str) -> str:
return Path(filename).name return Path(filename).name
# Suite names carry the runner shape as a `<n>-gpu` / `<n>-npu` token, e.g.
# `base-b-test-1-gpu-small`, `base-c-test-acc-16-npu-a3`. The token is the
# only machine-readable record of how many accelerators a test asks for, so
# the GPU-count grouping parses it back out. Anchored on a dash (or the
# string start/end) so a trailing `-tp4` or a model name like `qwen3-235b`
# can't be mistaken for a runner size.
_SUITE_ACCEL_COUNT_RE = re.compile(r"(?:^|-)(\d+)-(?:gpu|npu)(?:-|$)")
def get_accel_count(test: CIRegistry) -> Optional[int]:
"""Number of GPUs/NPUs a test's suite runs on, or None if unencoded.
Returns None for suites whose name has no size token -- `stress`,
`nightly-amd-vlm`, the CPU suites, and the synthesized `mm-gen-*`
suites (multimodal_gen spreads one suite across 1-gpu and 2-gpu
runners, so there is no single answer to report).
"""
match = _SUITE_ACCEL_COUNT_RE.search(test.effective_suite or "")
return int(match.group(1)) if match else None
def accel_count_label(count: Optional[int]) -> str:
"""Row label for an accelerator count."""
return f"{count}-GPU" if count is not None else "Unsized"
def organize_test_data(tests: list[CIRegistry]) -> dict: def organize_test_data(tests: list[CIRegistry]) -> dict:
"""Organize tests into various groupings.""" """Organize tests into various groupings."""
by_backend = defaultdict(list) by_backend = defaultdict(list)
by_folder = defaultdict(list) by_folder = defaultdict(list)
by_accel = defaultdict(list)
disabled_tests = [] disabled_tests = []
for t in tests: for t in tests:
by_backend[t.backend.name].append(t) by_backend[t.backend.name].append(t)
by_folder[get_folder_name(t.filename)].append(t) by_folder[get_folder_name(t.filename)].append(t)
by_accel[get_accel_count(t)].append(t)
if t.disabled: if t.disabled:
disabled_tests.append(t) disabled_tests.append(t)
@@ -266,10 +296,16 @@ def organize_test_data(tests: list[CIRegistry]) -> dict:
"disabled_unique_files": len(unique_disabled_files), "disabled_unique_files": len(unique_disabled_files),
"by_backend": by_backend, "by_backend": by_backend,
"by_folder": by_folder, "by_folder": by_folder,
"by_accel": by_accel,
"disabled_tests": disabled_tests, "disabled_tests": disabled_tests,
} }
def sorted_accel_counts(by_accel: dict) -> list:
"""Accelerator counts in ascending order, with `None` (unsized) last."""
return sorted(by_accel.keys(), key=lambda c: (c is None, c if c is not None else 0))
def generate_summary_section(data: dict) -> str: def generate_summary_section(data: dict) -> str:
"""Generate the summary/overview section.""" """Generate the summary/overview section."""
lines = [] lines = []
@@ -331,6 +367,36 @@ def generate_summary_section(data: dict) -> str:
lines.append("\n</details>\n") lines.append("\n</details>\n")
# GPU count summary (collapsible). Answers "how many 1-GPU tests do we
# have?" without expanding every suite -- the bulk of the fleet is
# single-GPU runners, so this is the row that decides capacity.
lines.append("<details>")
lines.append("<summary><h2>GPU Count Summary</h2></summary>\n")
lines.append(
"*Enabled registrations, grouped by the `<n>-gpu` / `<n>-npu` size in "
"the suite name. `Unsized` covers suites with no size token: the CPU "
"suites, `stress`, and `mm-gen-*` (multimodal_gen splits one suite "
"across 1-gpu and 2-gpu runners).*\n"
)
header_cells = ["GPUs", *active_backends, "Total", "Disabled"]
lines.append("| " + " | ".join(header_cells) + " |")
lines.append("|" + "|".join(["-" * max(len(c), 3) for c in header_cells]) + "|")
by_accel = data["by_accel"]
for count in sorted_accel_counts(by_accel):
accel_tests = by_accel[count]
enabled = [t for t in accel_tests if not t.disabled]
backend_counts = {b.name: 0 for b in HWBackend}
for t in enabled:
backend_counts[t.backend.name] += 1
row = [accel_count_label(count)]
row += [str(backend_counts[b]) for b in active_backends]
row.append(str(len(enabled)))
row.append(str(len(accel_tests) - len(enabled)))
lines.append("| " + " | ".join(row) + " |")
lines.append("\n</details>\n")
# Disabled tests section (collapsible) # Disabled tests section (collapsible)
if disabled_tests: if disabled_tests:
lines.append("<details>") lines.append("<details>")
@@ -394,6 +460,69 @@ def generate_by_folder_section(data: dict) -> str:
return "\n".join(lines) return "\n".join(lines)
def generate_by_gpu_count_section(data: dict) -> str:
"""Generate the 'All Tests by GPU Count' section.
Same registrations as the by-suite section, pivoted on runner size
instead of suite name, so per-GPU-count fleet demand is readable at a
glance (which backend leans on 1-GPU boxes, where the 8-GPU load sits).
"""
lines = []
by_accel = data["by_accel"]
lines.append("# All Tests by GPU Count\n")
for count in sorted_accel_counts(by_accel):
accel_tests = by_accel[count]
a_disabled = sum(1 for t in accel_tests if t.disabled)
a_enabled = len(accel_tests) - a_disabled
lines.append("<details>")
lines.append(
f"<summary><h2>{accel_count_label(count)} "
f"({a_enabled} enabled, {a_disabled} disabled)</h2></summary>\n"
)
accel_by_backend = defaultdict(list)
for t in accel_tests:
accel_by_backend[t.backend.name].append(t)
for backend in BACKEND_DISPLAY_ORDER:
backend_tests = accel_by_backend.get(backend, [])
if not backend_tests:
continue
b_disabled = sum(1 for t in backend_tests if t.disabled)
b_enabled = len(backend_tests) - b_disabled
lines.append(
f"### {backend} ({b_enabled} enabled, {b_disabled} disabled)\n"
)
lines.append("| Suite | Enabled | Disabled | Est. Time | Type |")
lines.append("|-------|---------|----------|-----------|------|")
backend_suites = defaultdict(list)
for t in backend_tests:
backend_suites[t.effective_suite].append(t)
for suite in sorted(backend_suites.keys()):
suite_tests = backend_suites[suite]
s_disabled = sum(1 for t in suite_tests if t.disabled)
s_enabled = len(suite_tests) - s_disabled
s_est_time = sum(t.est_time for t in suite_tests if not t.disabled)
is_nightly = any(t.nightly for t in suite_tests if not t.disabled)
suite_type = "Nightly" if is_nightly else "Per-Commit"
lines.append(
f"| {suite} | {s_enabled} | {s_disabled} | "
f"{s_est_time:.0f}s | {suite_type} |"
)
lines.append("")
lines.append("</details>\n")
return "\n".join(lines)
def generate_by_suite_section(data: dict) -> str: def generate_by_suite_section(data: dict) -> str:
"""Generate the 'All Tests by Test Suite' section.""" """Generate the 'All Tests by Test Suite' section."""
lines = [] lines = []
@@ -469,6 +598,8 @@ def generate_markdown_report(tests: list[CIRegistry], section: str = "all") -> s
return generate_by_folder_section(data) return generate_by_folder_section(data)
elif section == "by-suite": elif section == "by-suite":
return generate_by_suite_section(data) return generate_by_suite_section(data)
elif section == "by-gpu-count":
return generate_by_gpu_count_section(data)
else: # "all" else: # "all"
parts = [ parts = [
generate_summary_section(data), generate_summary_section(data),
@@ -476,6 +607,8 @@ def generate_markdown_report(tests: list[CIRegistry], section: str = "all") -> s
generate_by_folder_section(data), generate_by_folder_section(data),
"---", "---",
generate_by_suite_section(data), generate_by_suite_section(data),
"---",
generate_by_gpu_count_section(data),
] ]
return "\n".join(parts) return "\n".join(parts)
@@ -502,6 +635,7 @@ def generate_json_report(tests: list[CIRegistry]) -> str:
"tests_by_suite": {}, "tests_by_suite": {},
"backend_summary": {}, "backend_summary": {},
"folder_summary": {}, "folder_summary": {},
"gpu_count_summary": {},
"disabled_tests": [], "disabled_tests": [],
} }
@@ -606,6 +740,26 @@ def generate_json_report(tests: list[CIRegistry]) -> str:
"total": len(folder_tests), "total": len(folder_tests),
} }
# GPU count summary -- enabled registrations per runner size, keyed by
# the same labels the markdown table uses ("1-GPU", ..., "Unsized").
by_accel = defaultdict(list)
for t in tests:
by_accel[get_accel_count(t)].append(t)
for count in sorted_accel_counts(by_accel):
accel_tests = by_accel[count]
enabled = [t for t in accel_tests if not t.disabled]
backend_counts = {b: 0 for b in BACKEND_DISPLAY_ORDER}
for t in enabled:
backend_counts[t.backend.name] += 1
data["gpu_count_summary"][accel_count_label(count)] = {
**backend_counts,
"gpus": count,
"enabled": len(enabled),
"disabled": len(accel_tests) - len(enabled),
"suites": sorted({t.effective_suite for t in accel_tests}),
}
# Disabled tests # Disabled tests
for t in sorted(disabled_tests, key=lambda x: (x.backend.name, x.filename)): for t in sorted(disabled_tests, key=lambda x: (x.backend.name, x.filename)):
data["disabled_tests"].append( data["disabled_tests"].append(
@@ -630,7 +784,7 @@ def main():
) )
parser.add_argument( parser.add_argument(
"--section", "--section",
choices=["all", "summary", "by-folder", "by-suite"], choices=["all", "summary", "by-folder", "by-suite", "by-gpu-count"],
default="all", default="all",
help="Which section to output (default: all). Only applies to markdown format.", help="Which section to output (default: all). Only applies to markdown format.",
) )