From ad28b91faec29cd9e51ecc5b76b06327ed93389e Mon Sep 17 00:00:00 2001 From: Alison Shao <54658187+alisonshao@users.noreply.github.com> Date: Wed, 16 Sep 2026 01:13:08 -0700 Subject: [PATCH] ci: fix always-failing coverage job, add by-GPU-count view (#39697) --- .github/workflows/ci-coverage-overview.yml | 33 ++++- scripts/ci/utils/ci_coverage_report.py | 156 ++++++++++++++++++++- 2 files changed, 184 insertions(+), 5 deletions(-) diff --git a/.github/workflows/ci-coverage-overview.yml b/.github/workflows/ci-coverage-overview.yml index 121a68764..96c4ce572 100644 --- a/.github/workflows/ci-coverage-overview.yml +++ b/.github/workflows/ci-coverage-overview.yml @@ -63,10 +63,28 @@ jobs: run: | python scripts/ci/utils/ci_coverage_report.py --section by-suite + by-gpu-count: + name: Tests by GPU Count + runs-on: ubuntu-latest + steps: + - name: Checkout code + uses: actions/checkout@v4 + + - name: Set up Python + uses: actions/setup-python@v5 + with: + python-version: '3.10' + + - name: Generate Tests by GPU Count Report + run: | + python scripts/ci/utils/ci_coverage_report.py --section by-gpu-count + unit-test-coverage: name: Unit Test Code Coverage runs-on: 1-gpu-h100 - timeout-minutes: 30 + # ~6 min install + ~22 min of tests, and the suite keeps growing + # (7.7k collected items in Aug, 11k now). + timeout-minutes: 60 steps: - name: Checkout code uses: actions/checkout@v4 @@ -88,13 +106,19 @@ jobs: pip install -e "python/[test]" - name: Run unit tests with coverage - timeout-minutes: 10 + # Non-gating on purpose: running the whole tree in one process + # cross-contaminates tests that CI proper runs one file per + # process. `| tee` used to hide that by accident; pipefail plus an + # explicit `|| echo` makes it deliberate and reports the counts. + timeout-minutes: 45 run: | + set -o pipefail pytest test/registered/unit/ \ --cov --cov-config=.coveragerc \ --cov-report=term-missing:skip-covered \ --continue-on-collection-errors \ - -v | tee coverage_output.txt + -v | tee coverage_output.txt \ + || echo "pytest exited $? -- reported below, not gating this job" - name: Write coverage to summary if: always() @@ -106,7 +130,8 @@ jobs: # Test result line (e.g., "== 42 passed, 1 failed in 23.5s ==") echo '```' >> $GITHUB_STEP_SUMMARY - grep -E '^=+.*passed' coverage_output.txt >> $GITHUB_STEP_SUMMARY || true + grep -E '^=+.*passed' coverage_output.txt >> $GITHUB_STEP_SUMMARY \ + || echo "pytest did not reach its summary line (step killed or crashed)" >> $GITHUB_STEP_SUMMARY echo "" >> $GITHUB_STEP_SUMMARY # Coverage total, reformatted awk '/^TOTAL / { for(i=1;i<=NF;i++) if($i~/^[0-9]+$/ || $i~/^[0-9]+%$/) a[++n]=$i; if(n>=3) printf "TOTAL Stmts: %s Miss: %s Cover: %s\n", a[1], a[2], a[3] }' coverage_output.txt >> $GITHUB_STEP_SUMMARY || true diff --git a/scripts/ci/utils/ci_coverage_report.py b/scripts/ci/utils/ci_coverage_report.py index 3829e0465..0862f048c 100755 --- a/scripts/ci/utils/ci_coverage_report.py +++ b/scripts/ci/utils/ci_coverage_report.py @@ -13,9 +13,11 @@ import argparse import glob import json import os +import re import sys from collections import defaultdict from pathlib import Path +from typing import Optional # Add the ci_register module path directly to avoid heavy sglang imports sys.path.insert( @@ -240,15 +242,43 @@ def get_test_basename(filename: str) -> str: return Path(filename).name +# Suite names carry the runner shape as a `-gpu` / `-npu` token, e.g. +# `base-b-test-1-gpu-small`, `base-c-test-acc-16-npu-a3`. The token is the +# only machine-readable record of how many accelerators a test asks for, so +# the GPU-count grouping parses it back out. Anchored on a dash (or the +# string start/end) so a trailing `-tp4` or a model name like `qwen3-235b` +# can't be mistaken for a runner size. +_SUITE_ACCEL_COUNT_RE = re.compile(r"(?:^|-)(\d+)-(?:gpu|npu)(?:-|$)") + + +def get_accel_count(test: CIRegistry) -> Optional[int]: + """Number of GPUs/NPUs a test's suite runs on, or None if unencoded. + + Returns None for suites whose name has no size token -- `stress`, + `nightly-amd-vlm`, the CPU suites, and the synthesized `mm-gen-*` + suites (multimodal_gen spreads one suite across 1-gpu and 2-gpu + runners, so there is no single answer to report). + """ + match = _SUITE_ACCEL_COUNT_RE.search(test.effective_suite or "") + return int(match.group(1)) if match else None + + +def accel_count_label(count: Optional[int]) -> str: + """Row label for an accelerator count.""" + return f"{count}-GPU" if count is not None else "Unsized" + + def organize_test_data(tests: list[CIRegistry]) -> dict: """Organize tests into various groupings.""" by_backend = defaultdict(list) by_folder = defaultdict(list) + by_accel = defaultdict(list) disabled_tests = [] for t in tests: by_backend[t.backend.name].append(t) by_folder[get_folder_name(t.filename)].append(t) + by_accel[get_accel_count(t)].append(t) if t.disabled: disabled_tests.append(t) @@ -266,10 +296,16 @@ def organize_test_data(tests: list[CIRegistry]) -> dict: "disabled_unique_files": len(unique_disabled_files), "by_backend": by_backend, "by_folder": by_folder, + "by_accel": by_accel, "disabled_tests": disabled_tests, } +def sorted_accel_counts(by_accel: dict) -> list: + """Accelerator counts in ascending order, with `None` (unsized) last.""" + return sorted(by_accel.keys(), key=lambda c: (c is None, c if c is not None else 0)) + + def generate_summary_section(data: dict) -> str: """Generate the summary/overview section.""" lines = [] @@ -331,6 +367,36 @@ def generate_summary_section(data: dict) -> str: lines.append("\n\n") + # GPU count summary (collapsible). Answers "how many 1-GPU tests do we + # have?" without expanding every suite -- the bulk of the fleet is + # single-GPU runners, so this is the row that decides capacity. + lines.append("
") + lines.append("

GPU Count Summary

\n") + lines.append( + "*Enabled registrations, grouped by the `-gpu` / `-npu` size in " + "the suite name. `Unsized` covers suites with no size token: the CPU " + "suites, `stress`, and `mm-gen-*` (multimodal_gen splits one suite " + "across 1-gpu and 2-gpu runners).*\n" + ) + header_cells = ["GPUs", *active_backends, "Total", "Disabled"] + lines.append("| " + " | ".join(header_cells) + " |") + lines.append("|" + "|".join(["-" * max(len(c), 3) for c in header_cells]) + "|") + + by_accel = data["by_accel"] + for count in sorted_accel_counts(by_accel): + accel_tests = by_accel[count] + enabled = [t for t in accel_tests if not t.disabled] + backend_counts = {b.name: 0 for b in HWBackend} + for t in enabled: + backend_counts[t.backend.name] += 1 + row = [accel_count_label(count)] + row += [str(backend_counts[b]) for b in active_backends] + row.append(str(len(enabled))) + row.append(str(len(accel_tests) - len(enabled))) + lines.append("| " + " | ".join(row) + " |") + + lines.append("\n
\n") + # Disabled tests section (collapsible) if disabled_tests: lines.append("
") @@ -394,6 +460,69 @@ def generate_by_folder_section(data: dict) -> str: return "\n".join(lines) +def generate_by_gpu_count_section(data: dict) -> str: + """Generate the 'All Tests by GPU Count' section. + + Same registrations as the by-suite section, pivoted on runner size + instead of suite name, so per-GPU-count fleet demand is readable at a + glance (which backend leans on 1-GPU boxes, where the 8-GPU load sits). + """ + lines = [] + by_accel = data["by_accel"] + + lines.append("# All Tests by GPU Count\n") + + for count in sorted_accel_counts(by_accel): + accel_tests = by_accel[count] + a_disabled = sum(1 for t in accel_tests if t.disabled) + a_enabled = len(accel_tests) - a_disabled + + lines.append("
") + lines.append( + f"

{accel_count_label(count)} " + f"({a_enabled} enabled, {a_disabled} disabled)

\n" + ) + + accel_by_backend = defaultdict(list) + for t in accel_tests: + accel_by_backend[t.backend.name].append(t) + + for backend in BACKEND_DISPLAY_ORDER: + backend_tests = accel_by_backend.get(backend, []) + if not backend_tests: + continue + + b_disabled = sum(1 for t in backend_tests if t.disabled) + b_enabled = len(backend_tests) - b_disabled + lines.append( + f"### {backend} ({b_enabled} enabled, {b_disabled} disabled)\n" + ) + lines.append("| Suite | Enabled | Disabled | Est. Time | Type |") + lines.append("|-------|---------|----------|-----------|------|") + + backend_suites = defaultdict(list) + for t in backend_tests: + backend_suites[t.effective_suite].append(t) + + for suite in sorted(backend_suites.keys()): + suite_tests = backend_suites[suite] + s_disabled = sum(1 for t in suite_tests if t.disabled) + s_enabled = len(suite_tests) - s_disabled + s_est_time = sum(t.est_time for t in suite_tests if not t.disabled) + is_nightly = any(t.nightly for t in suite_tests if not t.disabled) + suite_type = "Nightly" if is_nightly else "Per-Commit" + lines.append( + f"| {suite} | {s_enabled} | {s_disabled} | " + f"{s_est_time:.0f}s | {suite_type} |" + ) + + lines.append("") + + lines.append("
\n") + + return "\n".join(lines) + + def generate_by_suite_section(data: dict) -> str: """Generate the 'All Tests by Test Suite' section.""" lines = [] @@ -469,6 +598,8 @@ def generate_markdown_report(tests: list[CIRegistry], section: str = "all") -> s return generate_by_folder_section(data) elif section == "by-suite": return generate_by_suite_section(data) + elif section == "by-gpu-count": + return generate_by_gpu_count_section(data) else: # "all" parts = [ generate_summary_section(data), @@ -476,6 +607,8 @@ def generate_markdown_report(tests: list[CIRegistry], section: str = "all") -> s generate_by_folder_section(data), "---", generate_by_suite_section(data), + "---", + generate_by_gpu_count_section(data), ] return "\n".join(parts) @@ -502,6 +635,7 @@ def generate_json_report(tests: list[CIRegistry]) -> str: "tests_by_suite": {}, "backend_summary": {}, "folder_summary": {}, + "gpu_count_summary": {}, "disabled_tests": [], } @@ -606,6 +740,26 @@ def generate_json_report(tests: list[CIRegistry]) -> str: "total": len(folder_tests), } + # GPU count summary -- enabled registrations per runner size, keyed by + # the same labels the markdown table uses ("1-GPU", ..., "Unsized"). + by_accel = defaultdict(list) + for t in tests: + by_accel[get_accel_count(t)].append(t) + + for count in sorted_accel_counts(by_accel): + accel_tests = by_accel[count] + enabled = [t for t in accel_tests if not t.disabled] + backend_counts = {b: 0 for b in BACKEND_DISPLAY_ORDER} + for t in enabled: + backend_counts[t.backend.name] += 1 + data["gpu_count_summary"][accel_count_label(count)] = { + **backend_counts, + "gpus": count, + "enabled": len(enabled), + "disabled": len(accel_tests) - len(enabled), + "suites": sorted({t.effective_suite for t in accel_tests}), + } + # Disabled tests for t in sorted(disabled_tests, key=lambda x: (x.backend.name, x.filename)): data["disabled_tests"].append( @@ -630,7 +784,7 @@ def main(): ) parser.add_argument( "--section", - choices=["all", "summary", "by-folder", "by-suite"], + choices=["all", "summary", "by-folder", "by-suite", "by-gpu-count"], default="all", help="Which section to output (default: all). Only applies to markdown format.", )