Files
sglang/.github/workflows/nightly-72-gpu-gb200.yml
T

207 lines
7.7 KiB
YAML

name: Nightly Test (GB200 72GPU)
# NOTE: Nightly (schedule) runs require no approval.
# Manual (workflow_dispatch) runs are gated by the gb200-ci environment
# to prevent individuals from queuing arbitrary jobs on the shared GB200 cluster.
on:
schedule:
- cron: '0 2 * * *' # 2 AM UTC daily (offset from other nightly runs)
workflow_dispatch: # allow manual trigger; gated by gb200-ci environment
inputs:
image:
description: 'SGLang Docker image to benchmark'
required: false
default: 'lmsysorg/sglang:dev-cu13'
concurrency:
group: nightly-test-gb200
cancel-in-progress: false
env:
SGLANG_IS_IN_CI: true
SRT_SLURM_BRANCH: sglang-nightly-regression
SLURM_PARTITION: batch
SLURM_ACCOUNT: sglang
jobs:
# ---------------------------------------------------------------------------
# Reads scripts/ci/slurm/nightly-configs.yaml and generates one matrix entry
# per recipe YAML. Each job runs the full concurrency sweep defined in the
# recipe as a single Slurm job.
# To add/remove configs, edit nightly-configs.yaml only.
# ---------------------------------------------------------------------------
setup:
if: github.repository == 'sgl-project/sglang'
runs-on: ubuntu-latest
outputs:
matrix: ${{ steps.generate.outputs.matrix }}
steps:
- name: Checkout code
uses: actions/checkout@v4
- name: Generate benchmark matrix
id: generate
run: |
pip install pyyaml -q
MATRIX=$(python3 scripts/ci/slurm/generate_matrix.py scripts/ci/slurm/nightly-configs.yaml --runner gb200)
echo "matrix=$MATRIX" >> $GITHUB_OUTPUT
# ---------------------------------------------------------------------------
# Import Docker images to Lustre squash files once before all benchmark jobs.
# This avoids parallel jobs racing to enroot import the same image.
# ---------------------------------------------------------------------------
prepare-image:
needs: setup
if: github.repository == 'sgl-project/sglang'
environment: ${{ github.event_name == 'workflow_dispatch' && 'gb200-ci' || '' }}
runs-on: 72-gpu-gb200
outputs:
squash_file: ${{ steps.import.outputs.squash_file }}
nginx_squash_file: ${{ steps.import.outputs.nginx_squash_file }}
env:
IMAGE: ${{ inputs.image || 'lmsysorg/sglang:dev-cu13' }}
NGINX_IMAGE: nginx:1.27.4
steps:
- name: Import Docker images to Lustre
id: import
run: |
SQUASH_FILE="/mnt/lustre01/users-public/sglang-ci/$(echo "$IMAGE" | sed 's/[\/:@#]/_/g')_$(date +%Y%m%d).sqsh"
NGINX_SQUASH_FILE="/mnt/lustre01/users-public/sglang-ci/$(echo "$NGINX_IMAGE" | sed 's/[\/:@#]/_/g').sqsh"
if [ -f "$SQUASH_FILE" ]; then
echo "Squash file already exists, skipping import: $SQUASH_FILE"
else
enroot import -o "$SQUASH_FILE" "docker://$IMAGE"
fi
if [ -f "$NGINX_SQUASH_FILE" ]; then
echo "Nginx squash file already exists, skipping import: $NGINX_SQUASH_FILE"
else
enroot import -o "$NGINX_SQUASH_FILE" "docker://$NGINX_IMAGE"
fi
echo "squash_file=$SQUASH_FILE" >> $GITHUB_OUTPUT
echo "nginx_squash_file=$NGINX_SQUASH_FILE" >> $GITHUB_OUTPUT
nightly-gb200-benchmark:
needs: [setup, prepare-image]
if: github.repository == 'sgl-project/sglang'
runs-on: 72-gpu-gb200
strategy:
fail-fast: false
matrix:
config: ${{ fromJson(needs.setup.outputs.matrix) }}
env:
FRAMEWORK: dynamo-sglang
MODEL: ${{ matrix.config.model }}
MODEL_PREFIX: ${{ matrix.config.model_prefix }}
PRECISION: ${{ matrix.config.precision }}
ISL: ${{ matrix.config.isl }}
OSL: ${{ matrix.config.osl }}
CONFIG_FILE: ${{ matrix.config.config_file }}
RESULT_FILENAME: gb200-${{ matrix.config.name }}
SQUASH_FILE: ${{ needs.prepare-image.outputs.squash_file }}
NGINX_SQUASH_FILE: ${{ needs.prepare-image.outputs.nginx_squash_file }}
steps:
- name: Checkout code
uses: actions/checkout@v4
- name: Clean up prior Slurm jobs from this runner
continue-on-error: true
env:
RUNNER_NAME: ${{ runner.name }}
run: |
STALE_JOBS=$(squeue --noheader --format="%i %j" | grep "${RUNNER_NAME}" | awk '{print $1}')
if [ -n "$STALE_JOBS" ]; then
echo "Cancelling stale jobs: $STALE_JOBS"
scancel $STALE_JOBS
fi
- name: Launch GB200 benchmark via srt-slurm
timeout-minutes: 360
env:
RUNNER_NAME: ${{ runner.name }}
run: bash scripts/ci/slurm/launch_gb200.sh
- name: Process results
if: always()
env:
RUNNER_NAME: ${{ runner.name }}
run: |
pip install tabulate pyyaml -q
SRT_REPO_DIR="/mnt/lustre01/users-public/sglang-ci/workspace/${RUNNER_NAME}/srt-slurm"
for result_file in ${{ github.workspace }}/${RESULT_FILENAME}_*.json; do
[ -f "$result_file" ] || continue
basename_file=$(basename "$result_file")
ctx=$(echo "$basename_file" | sed -n 's/.*_ctx_\([0-9]*\)_gen.*/\1/p')
gen=$(echo "$basename_file" | sed -n 's/.*_gen_\([0-9]*\)\.json/\1/p')
[ -n "$ctx" ] && [ -n "$gen" ] || continue
RESULT_FILENAME="${result_file%.json}" PREFILL_GPUS="$ctx" DECODE_GPUS="$gen" \
RECIPE_FILE="$SRT_REPO_DIR/$CONFIG_FILE" \
python3 scripts/ci/slurm/process_result.py
done
- name: Upload results
if: always()
uses: actions/upload-artifact@v4
with:
name: gb200-${{ matrix.config.name }}-${{ github.run_id }}
path: |
${{ github.workspace }}/*.json
${{ github.workspace }}/multinode_server_logs.tar.gz
retention-days: 30
if-no-files-found: warn
- name: Analyze logs with AI on failure
if: failure()
continue-on-error: true
env:
MODAL_TOKEN_ID: ${{ secrets.NV_MODAL_TOKEN_ID }}
MODAL_TOKEN_SECRET: ${{ secrets.NV_MODAL_TOKEN_SECRET }}
run: |
TARBALL="${{ github.workspace }}/multinode_server_logs.tar.gz"
if [ -f "$TARBALL" ]; then
uv run --with modal python scripts/ci/slurm/analyze_logs_with_modal.py \
--tarball "$TARBALL" \
--job-id "${{ matrix.config.name }}-${{ github.run_id }}" \
--output "${{ github.workspace }}/ai_analysis.md"
if [ -f "${{ github.workspace }}/ai_analysis.md" ]; then
echo "## AI Log Analysis" >> $GITHUB_STEP_SUMMARY
cat "${{ github.workspace }}/ai_analysis.md" >> $GITHUB_STEP_SUMMARY
fi
else
echo "No log tarball found, skipping analysis"
fi
- name: Clean up Slurm jobs on failure/cancel
if: failure() || cancelled()
continue-on-error: true
env:
RUNNER_NAME: ${{ runner.name }}
run: |
ACTIVE_JOBS=$(squeue --noheader --format="%i %j" | grep "${RUNNER_NAME}" | awk '{print $1}')
if [ -n "$ACTIVE_JOBS" ]; then
echo "Cancelling jobs: $ACTIVE_JOBS"
scancel $ACTIVE_JOBS
fi
collect-results:
needs: nightly-gb200-benchmark
if: github.repository == 'sgl-project/sglang' && always()
runs-on: ubuntu-latest
steps:
- name: Checkout code
uses: actions/checkout@v4
- name: Download artifacts
uses: actions/download-artifact@v4
with:
path: results/
pattern: gb200-*
- name: Print summary
run: |
pip install tabulate -q
python3 scripts/ci/slurm/summarize.py results/ >> $GITHUB_STEP_SUMMARY