From 12de907bc2cd41312de057c5b277d45bf36d8699 Mon Sep 17 00:00:00 2001 From: Stella-17 <75434074+Stella-17@users.noreply.github.com> Date: Mon, 8 Jun 2026 23:45:16 +0800 Subject: [PATCH] [MUSA][23/N] CI: Fix torchada preflight lock cleanup and add LLM server smoke test (#27242) Co-authored-by: xinyue.fan --- .github/workflows/nightly-test-musa.yml | 57 ++++++++- .github/workflows/pr-test-musa.yml | 8 +- .github/workflows/release-whl-kernel.yml | 12 ++ python/sglang/test/ci/ci_register.py | 15 +++ scripts/ci/musa/musa_install_dependency.sh | 19 +++ .../musa/test_llm_server_smoke_musa.py | 111 ++++++++++++++++++ test/run_suite.py | 12 +- 7 files changed, 225 insertions(+), 9 deletions(-) create mode 100644 test/registered/musa/test_llm_server_smoke_musa.py diff --git a/.github/workflows/nightly-test-musa.yml b/.github/workflows/nightly-test-musa.yml index 9216800dd..a916c2bb3 100644 --- a/.github/workflows/nightly-test-musa.yml +++ b/.github/workflows/nightly-test-musa.yml @@ -14,6 +14,7 @@ on: - 'all' - 'nightly-test-musa-general-kernel' - 'nightly-test-musa-general-multimodal-layer' + - 'nightly-test-llm-server-1-gpu-musa' - 'nightly-test-multimodal-server-1-gpu-musa' - 'nightly-test-multimodal-server-2-gpu-musa' workflow_call: @@ -65,7 +66,7 @@ jobs: - name: Install dependencies if: steps.gate.outputs.run_job == 'true' - timeout-minutes: 10 + timeout-minutes: 15 run: | bash scripts/ci/musa/musa_install_dependency.sh @@ -111,7 +112,7 @@ jobs: ref: ${{ inputs.ref || github.ref }} - name: Install dependencies - timeout-minutes: 10 + timeout-minutes: 15 if: steps.gate.outputs.run_job == 'true' run: | bash scripts/ci/musa/musa_install_dependency.sh @@ -123,6 +124,53 @@ jobs: pytest python/sglang/multimodal_gen/test/layers/test_musa_rmsnorm.py pytest python/sglang/multimodal_gen/test/layers/test_musa_silu_and_mul.py + # ==================== LLM server: 1-GPU ==================== + nightly-test-llm-server-1-gpu-musa: + if: github.event_name == 'schedule' || github.event_name == 'workflow_dispatch' || github.event_name == 'workflow_call' + runs-on: s5000-1-gpu-runner + timeout-minutes: 240 + env: + SGLANG_USE_MODELSCOPE: false + SGLANG_IS_IN_CI: true + TORCHADA_ENABLE_CPP_OPS: 1 + HF_HOME: /hf-cache + HF_HUB_CACHE: /hf-cache/hub + HF_HUB_OFFLINE: 1 + steps: + - name: Gate by job_filter + id: gate + run: | + filter="${{ inputs.job_filter || 'all' }}" + if [[ -z "$filter" || "$filter" == "all" || "$filter" == "nightly-test-llm-server-1-gpu-musa" ]]; then + echo "run_job=true" >> "$GITHUB_OUTPUT" + else + echo "run_job=false" >> "$GITHUB_OUTPUT" + fi + + - name: Checkout code + timeout-minutes: 10 + if: steps.gate.outputs.run_job == 'true' + uses: actions/checkout@v4 + with: + ref: ${{ inputs.ref || github.ref }} + + - name: Install dependencies + timeout-minutes: 15 + if: steps.gate.outputs.run_job == 'true' + run: | + bash scripts/ci/musa/musa_install_dependency.sh + + - name: Run DeepSeek-V2-Lite-Chat LLM server smoke test + if: steps.gate.outputs.run_job == 'true' + timeout-minutes: 120 + run: | + cd test + python3 run_suite.py \ + --hw musa \ + --suite nightly-musa-1-gpu \ + --nightly \ + --timeout-per-file 7200 + # ==================== Multimodal: 1-GPU (split) ==================== nightly-test-multimodal-server-1-gpu-musa: if: github.event_name == 'schedule' || github.event_name == 'workflow_dispatch' || github.event_name == 'workflow_call' @@ -159,7 +207,7 @@ jobs: ref: ${{ inputs.ref || github.ref }} - name: Install dependencies - timeout-minutes: 10 + timeout-minutes: 15 if: steps.gate.outputs.run_job == 'true' run: | bash scripts/ci/musa/musa_install_dependency.sh @@ -208,7 +256,7 @@ jobs: ref: ${{ inputs.ref || github.ref }} - name: Install dependencies - timeout-minutes: 10 + timeout-minutes: 15 if: steps.gate.outputs.run_job == 'true' run: | bash scripts/ci/musa/musa_install_dependency.sh @@ -230,6 +278,7 @@ jobs: [ nightly-test-musa-general-kernel, nightly-test-musa-general-multimodal-layer, + nightly-test-llm-server-1-gpu-musa, nightly-test-multimodal-server-1-gpu-musa, nightly-test-multimodal-server-2-gpu-musa, ] diff --git a/.github/workflows/pr-test-musa.yml b/.github/workflows/pr-test-musa.yml index 5d6ae58da..4f26b5766 100644 --- a/.github/workflows/pr-test-musa.yml +++ b/.github/workflows/pr-test-musa.yml @@ -117,7 +117,7 @@ jobs: ref: ${{ inputs.ref || github.ref }} - name: Install dependencies - timeout-minutes: 10 + timeout-minutes: 15 run: | bash scripts/ci/musa/musa_install_dependency.sh @@ -152,7 +152,7 @@ jobs: ref: ${{ inputs.ref || github.ref }} - name: Install dependencies - timeout-minutes: 10 + timeout-minutes: 15 run: | bash scripts/ci/musa/musa_install_dependency.sh @@ -183,7 +183,7 @@ jobs: uses: actions/checkout@v4 - name: Install dependencies - timeout-minutes: 10 + timeout-minutes: 15 run: | bash scripts/ci/musa/musa_install_dependency.sh @@ -210,7 +210,7 @@ jobs: ref: ${{ inputs.ref || github.ref }} - name: Install dependencies - timeout-minutes: 10 + timeout-minutes: 15 run: | bash scripts/ci/musa/musa_install_dependency.sh diff --git a/.github/workflows/release-whl-kernel.yml b/.github/workflows/release-whl-kernel.yml index 775fafacc..98a89df32 100644 --- a/.github/workflows/release-whl-kernel.yml +++ b/.github/workflows/release-whl-kernel.yml @@ -437,6 +437,18 @@ jobs: with: submodules: "recursive" + - name: Clean stale torchada extension locks + run: | + if ps -eo comm=,args= | grep -E '(^|[[:space:]])(mcc|ninja)([[:space:]]|$)|torchada_cpp_ops' | grep -v grep; then + echo "::error::Active torchada extension build detected; refusing to remove lock files" + exit 1 + fi + find "${HOME}/.cache/torch_extensions" \ + -path '*/torchada_cpp_ops/lock' \ + -type f \ + -print \ + -delete 2>/dev/null || true + - name: Build wheels run: | cd sgl-kernel diff --git a/python/sglang/test/ci/ci_register.py b/python/sglang/test/ci/ci_register.py index d42c310eb..4eed1ee4c 100644 --- a/python/sglang/test/ci/ci_register.py +++ b/python/sglang/test/ci/ci_register.py @@ -12,6 +12,7 @@ __all__ = [ "register_cpu_ci", "register_cuda_ci", "register_amd_ci", + "register_musa_ci", "register_npu_ci", "register_xpu_ci", "register_musa_ci", @@ -95,6 +96,19 @@ def register_amd_ci( return None +def register_musa_ci( + est_time: float, + suite: Optional[str] = None, + nightly: bool = False, + disabled: Optional[str] = None, + *, + stage: Optional[str] = None, + runner_config: Optional[str] = None, +): + """Marker for MUSA CI registration (parsed via AST; runtime no-op).""" + return None + + def register_npu_ci( est_time: float, suite: Optional[str] = None, @@ -138,6 +152,7 @@ REGISTER_MAPPING = { "register_cpu_ci": HWBackend.CPU, "register_cuda_ci": HWBackend.CUDA, "register_amd_ci": HWBackend.AMD, + "register_musa_ci": HWBackend.MUSA, "register_npu_ci": HWBackend.NPU, "register_xpu_ci": HWBackend.XPU, "register_musa_ci": HWBackend.MUSA, diff --git a/scripts/ci/musa/musa_install_dependency.sh b/scripts/ci/musa/musa_install_dependency.sh index 2cb90a324..0171a4440 100755 --- a/scripts/ci/musa/musa_install_dependency.sh +++ b/scripts/ci/musa/musa_install_dependency.sh @@ -24,6 +24,25 @@ done PIP_INSTALL="python3 -m pip install --no-cache-dir" ${PIP_INSTALL} --upgrade pip setuptools torchada --user +echo "Checking stale torchada extension locks..." +active_torchada_builds="$( + pgrep -af '(^|[[:space:]/])(mcc|ninja)([[:space:]]|$)|torchada_cpp_ops' 2>/dev/null \ + | awk -v self="$$" '$1 != self' +)" || true +if [ -n "$active_torchada_builds" ]; then + echo "$active_torchada_builds" + echo "::error::Active torchada extension build detected; refusing to remove lock files" + exit 1 +fi +torch_extensions_dir="${HOME}/.cache/torch_extensions" +if [ -d "$torch_extensions_dir" ]; then + find "$torch_extensions_dir" \ + -path '*/torchada_cpp_ops/lock' \ + -type f \ + -print \ + -delete +fi + WHL_DIR="/sglang-checkout/whl" if [ -d "$WHL_DIR" ] && compgen -G "${WHL_DIR}"/*.whl > /dev/null; then echo "Uninstall old packages based on wheel METADATA..." diff --git a/test/registered/musa/test_llm_server_smoke_musa.py b/test/registered/musa/test_llm_server_smoke_musa.py new file mode 100644 index 000000000..b5968d2dd --- /dev/null +++ b/test/registered/musa/test_llm_server_smoke_musa.py @@ -0,0 +1,111 @@ +import os +import unittest + +import requests +import torch + +from sglang.test.ci.ci_register import register_musa_ci +from sglang.test.server_fixtures.default_fixture import DefaultServerBase + +register_musa_ci(est_time=1200, suite="nightly-musa-1-gpu", nightly=True) + + +_REQUEST_TIMEOUT = 60 + + +@unittest.skipIf( + not (hasattr(torch, "musa") and torch.musa.is_available()), + "MUSA device not available", +) +class TestMusaDeepSeekV2LiteChatServerSmoke(DefaultServerBase): + """MUSA LLM server smoke test: launch, health check, and non-empty generation.""" + + model = os.getenv("SGLANG_MUSA_LLM_MODEL", "deepseek-ai/DeepSeek-V2-Lite-Chat") + served_model_name = "deepseek-v2-lite-chat" + other_args = [ + "--trust-remote-code", + "--served-model-name", + served_model_name, + "--attention-backend", + "fa3", + "--cuda-graph-max-bs", + "32", + "--tp-size", + "1", + "--chunked-prefill-size", + "-1", + "--disable-piecewise-cuda-graph", + "--context-length", + "4096", + "--max-total-tokens", + "8192", + "--max-running-requests", + "4", + ] + + def test_health(self): + resp = requests.get(self.base_url + "/health", timeout=10) + self.assertEqual(resp.status_code, 200, resp.text) + + def test_health_generate(self): + resp = requests.get( + self.base_url + "/health_generate", timeout=_REQUEST_TIMEOUT + ) + self.assertEqual(resp.status_code, 200, resp.text) + + def test_send_receive_chat_message_contains_beijing(self): + resp = requests.post( + self.base_url + "/v1/chat/completions", + json={ + "model": self.served_model_name, + "messages": [ + { + "role": "user", + "content": ( + "What is the capital of China? Answer in one word." + ), + }, + ], + "temperature": 0.0, + "max_tokens": 16, + }, + timeout=_REQUEST_TIMEOUT, + ) + self.assertEqual(resp.status_code, 200, resp.text) + body = resp.json() + self.assertIn("choices", body) + self.assertGreater(len(body["choices"]), 0) + content = body["choices"][0]["message"]["content"] + print( + f"[MUSA Chat Completion] prompt='What is the capital of China? Answer in one word.' response={content!r}", + flush=True, + ) + self.assertIsInstance(content, str) + self.assertGreater(len(content.strip()), 0) + self.assertIn("Beijing", content.lower()) + + def test_generate(self): + resp = requests.post( + self.base_url + "/generate", + json={ + "text": "The capital of France is", + "sampling_params": { + "temperature": 0.0, + "max_new_tokens": 16, + }, + "stream": False, + }, + timeout=_REQUEST_TIMEOUT, + ) + self.assertEqual(resp.status_code, 200, resp.text) + body = resp.json() + if isinstance(body, list): + self.assertGreater(len(body), 0) + body = body[0] + self.assertIn("text", body) + self.assertIsInstance(body["text"], str) + self.assertGreater(len(body["text"].strip()), 0) + + +if __name__ == "__main__": + unittest.main() diff --git a/test/run_suite.py b/test/run_suite.py index 0fef1b85c..1f4bfc952 100644 --- a/test/run_suite.py +++ b/test/run_suite.py @@ -19,6 +19,7 @@ HW_MAPPING = { "cpu": HWBackend.CPU, "cuda": HWBackend.CUDA, "amd": HWBackend.AMD, + "musa": HWBackend.MUSA, "npu": HWBackend.NPU, "xpu": HWBackend.XPU, } @@ -42,6 +43,7 @@ PER_COMMIT_SUITES = { "stage-c-test-large-8-gpu-amd", "stage-c-test-large-8-gpu-amd-mi35x", ], + HWBackend.MUSA: [], HWBackend.CUDA: [ "base-a-test-1-gpu-small", "base-b-test-1-gpu-small", @@ -122,6 +124,9 @@ NIGHTLY_SUITES = { # MI35x 8-GPU suite (different model configs) "nightly-amd-8-gpu-mi35x", ], + HWBackend.MUSA: [ + "nightly-musa-1-gpu", + ], HWBackend.CPU: [], HWBackend.NPU: [ "nightly-1-npu-a3", @@ -150,7 +155,12 @@ OTHER_SUITES = { } -_SUITE_CHECKED_BACKENDS = {HWBackend.CUDA, HWBackend.CPU, HWBackend.XPU} +_SUITE_CHECKED_BACKENDS = { + HWBackend.CUDA, + HWBackend.CPU, + HWBackend.MUSA, + HWBackend.XPU, +} def _valid_suites_by_backend() -> dict: