[Intel XPU] Initially add nightly GSM8K accuracy tests for Llama-3.1-8B (TP=2) and Qwen3-32B (TP=4) (#28908)
Co-authored-by: Singhal, Shubham <shubham.singhal@intel.com>
This commit is contained in:
co-authored by
Singhal, Shubham
parent
548f505cc5
commit
5134dcdcab
+1
-1
@@ -1,3 +1,3 @@
|
|||||||
[codespell]
|
[codespell]
|
||||||
ignore-words-list = ans, als, hel, boostrap, childs, te, vas, hsa, ment, cann, thi, makro, wil, rouge, PRIS, ather, MIS, medias, allready, inout, nd, fo, visibles, nothink, renderD, ond, tbe, CopyIn, notin, subtile, subtiles
|
ignore-words-list = ans, als, hel, boostrap, childs, te, vas, hsa, ment, cann, thi, makro, wil, rouge, PRIS, ather, MIS, medias, allready, inout, nd, fo, visibles, nothink, renderD, ond, tbe, CopyIn, notin, subtile, subtiles, IST
|
||||||
skip = *.json, *.jsonl, *.patch, *.txt, *.lock
|
skip = *.json, *.jsonl, *.patch, *.txt, *.lock
|
||||||
|
|||||||
@@ -2,13 +2,20 @@ name: Nightly Test (Intel)
|
|||||||
|
|
||||||
on:
|
on:
|
||||||
schedule:
|
schedule:
|
||||||
- cron: '0 0 * * *'
|
# 22:15 IST (UTC+5:30) → 16:45 UTC daily.
|
||||||
|
- cron: '45 16 * * *'
|
||||||
push:
|
push:
|
||||||
branches:
|
branches:
|
||||||
- main
|
- main
|
||||||
paths:
|
paths:
|
||||||
- "python/sglang/version.py"
|
- "python/sglang/version.py"
|
||||||
workflow_dispatch:
|
workflow_dispatch:
|
||||||
|
inputs:
|
||||||
|
continue_on_error:
|
||||||
|
description: 'Continue on error (do not fail the workflow on test failures)'
|
||||||
|
required: false
|
||||||
|
type: boolean
|
||||||
|
default: true
|
||||||
workflow_call:
|
workflow_call:
|
||||||
inputs:
|
inputs:
|
||||||
ref:
|
ref:
|
||||||
@@ -16,18 +23,159 @@ on:
|
|||||||
required: false
|
required: false
|
||||||
type: string
|
type: string
|
||||||
default: ""
|
default: ""
|
||||||
|
continue_on_error:
|
||||||
|
description: 'Continue on error (do not fail the workflow on test failures)'
|
||||||
|
required: false
|
||||||
|
type: boolean
|
||||||
|
default: true
|
||||||
|
|
||||||
concurrency:
|
concurrency:
|
||||||
group: nightly-test-intel-${{ inputs.ref || github.ref }}
|
group: nightly-test-intel-${{ github.event_name == 'workflow_dispatch' && format('manual-{0}', github.run_id) || inputs.ref && format('caller-{0}', github.run_id) || github.ref }}
|
||||||
cancel-in-progress: ${{ github.event_name != 'workflow_call' }}
|
cancel-in-progress: ${{ !inputs.ref && github.event_name != 'workflow_call' && github.event_name != 'workflow_dispatch' }}
|
||||||
|
|
||||||
jobs:
|
jobs:
|
||||||
# Placeholder for Intel GPU tests
|
nightly-xpu-1-gpu:
|
||||||
# Add Intel-specific nightly test workflows here when available
|
# Placeholder: no models currently registered to nightly-xpu-1-gpu run on
|
||||||
|
# intel-bmg-nightly. Add a model test to this suite to re-enable.
|
||||||
placeholder:
|
if: false
|
||||||
if: github.repository == 'sgl-project/sglang'
|
runs-on: intel-bmg-nightly
|
||||||
runs-on: ubuntu-latest
|
|
||||||
steps:
|
steps:
|
||||||
- name: Placeholder
|
- name: Placeholder
|
||||||
run: echo "Intel nightly tests will be added here"
|
run: echo "nightly-xpu-1-gpu has no validated models; skipping."
|
||||||
|
|
||||||
|
nightly-xpu-2-gpu:
|
||||||
|
if: github.repository == 'sgl-project/sglang'
|
||||||
|
runs-on: intel-bmg-nightly
|
||||||
|
env:
|
||||||
|
DOCKERHUB_INTEL_USERNAME: ${{ secrets.DOCKERHUB_INTEL_USERNAME }}
|
||||||
|
DOCKERHUB_INTEL_TOKEN: ${{ secrets.DOCKERHUB_INTEL_TOKEN }}
|
||||||
|
steps:
|
||||||
|
- name: Reset workspace ownership
|
||||||
|
run: |
|
||||||
|
docker run --rm -v "${{ github.workspace }}:/w" busybox:latest \
|
||||||
|
chown -R "$(id -u):$(id -g)" /w || true
|
||||||
|
|
||||||
|
- name: Checkout code
|
||||||
|
uses: actions/checkout@v4
|
||||||
|
with:
|
||||||
|
fetch-depth: 0
|
||||||
|
ref: ${{ inputs.ref || github.sha }}
|
||||||
|
|
||||||
|
- name: Start CI container (pull intel/sglang-dev:latest)
|
||||||
|
run: |
|
||||||
|
export HF_TOKEN="$(cat ~/huggingface_token.txt)"
|
||||||
|
bash scripts/ci/xpu/xpu_ci_start_container.sh
|
||||||
|
env:
|
||||||
|
GITHUB_WORKSPACE: ${{ github.workspace }}
|
||||||
|
|
||||||
|
- name: HF login + install run_suite extras
|
||||||
|
timeout-minutes: 10
|
||||||
|
run: |
|
||||||
|
docker exec ci_sglang_xpu /opt/venv/bin/python3 -m pip install --no-cache-dir tabulate
|
||||||
|
docker exec ci_sglang_xpu /opt/venv/bin/python3 -m pip install --no-cache-dir --no-deps xgrammar==0.1.33
|
||||||
|
docker exec ci_sglang_xpu /opt/venv/bin/python3 -m pip install --no-cache-dir "lm-eval==0.4.9"
|
||||||
|
docker exec ci_sglang_xpu /bin/bash -c '/opt/venv/bin/hf auth login --token ${HF_TOKEN}'
|
||||||
|
|
||||||
|
- name: Nightly Test (2-GPU XPU)
|
||||||
|
timeout-minutes: 60
|
||||||
|
run: |
|
||||||
|
touch github_summary.md
|
||||||
|
docker exec ci_sglang_xpu bash -c "
|
||||||
|
source /opt/venv/bin/activate &&
|
||||||
|
cd /sglang-checkout/test &&
|
||||||
|
GITHUB_STEP_SUMMARY=/sglang-checkout/github_summary.md \
|
||||||
|
python3 run_suite.py --hw xpu --suite nightly-xpu-2-gpu --nightly --timeout-per-file 1800 ${{ (github.event_name == 'schedule' || inputs.continue_on_error) && '--continue-on-error' || '' }}
|
||||||
|
" || TEST_EXIT_CODE=$?
|
||||||
|
echo "$(<github_summary.md)" >> $GITHUB_STEP_SUMMARY || true
|
||||||
|
exit ${TEST_EXIT_CODE:-0}
|
||||||
|
|
||||||
|
- name: Cleanup container
|
||||||
|
if: always()
|
||||||
|
run: |
|
||||||
|
docker run --rm -v "${{ github.workspace }}:/w" busybox:latest \
|
||||||
|
chown -R "$(id -u):$(id -g)" /w || true
|
||||||
|
rm -rf test/result.jsonl test/results test/.pytest_cache .pytest_cache || true
|
||||||
|
find . -type d -name "__pycache__" -prune -exec rm -rf {} + || true
|
||||||
|
find . -type f -name "*.pyc" -delete || true
|
||||||
|
docker rm -f ci_sglang_xpu || true
|
||||||
|
if [[ -n "${CI_SGLANG_XPU_IMAGE:-}" ]]; then
|
||||||
|
docker rmi -f "${CI_SGLANG_XPU_IMAGE}" || true
|
||||||
|
fi
|
||||||
|
|
||||||
|
nightly-xpu-4-gpu:
|
||||||
|
if: github.repository == 'sgl-project/sglang'
|
||||||
|
runs-on: intel-bmg-nightly
|
||||||
|
env:
|
||||||
|
DOCKERHUB_INTEL_USERNAME: ${{ secrets.DOCKERHUB_INTEL_USERNAME }}
|
||||||
|
DOCKERHUB_INTEL_TOKEN: ${{ secrets.DOCKERHUB_INTEL_TOKEN }}
|
||||||
|
steps:
|
||||||
|
- name: Reset workspace ownership
|
||||||
|
run: |
|
||||||
|
docker run --rm -v "${{ github.workspace }}:/w" busybox:latest \
|
||||||
|
chown -R "$(id -u):$(id -g)" /w || true
|
||||||
|
|
||||||
|
- name: Checkout code
|
||||||
|
uses: actions/checkout@v4
|
||||||
|
with:
|
||||||
|
fetch-depth: 0
|
||||||
|
ref: ${{ inputs.ref || github.sha }}
|
||||||
|
|
||||||
|
- name: Start CI container (pull intel/sglang-dev:latest)
|
||||||
|
run: |
|
||||||
|
export HF_TOKEN="$(cat ~/huggingface_token.txt)"
|
||||||
|
bash scripts/ci/xpu/xpu_ci_start_container.sh
|
||||||
|
env:
|
||||||
|
GITHUB_WORKSPACE: ${{ github.workspace }}
|
||||||
|
|
||||||
|
- name: HF login + install run_suite extras
|
||||||
|
timeout-minutes: 10
|
||||||
|
run: |
|
||||||
|
docker exec ci_sglang_xpu /opt/venv/bin/python3 -m pip install --no-cache-dir tabulate
|
||||||
|
docker exec ci_sglang_xpu /opt/venv/bin/python3 -m pip install --no-cache-dir --no-deps xgrammar==0.1.33
|
||||||
|
docker exec ci_sglang_xpu /opt/venv/bin/python3 -m pip install --no-cache-dir "lm-eval==0.4.9"
|
||||||
|
docker exec ci_sglang_xpu /bin/bash -c '/opt/venv/bin/hf auth login --token ${HF_TOKEN}'
|
||||||
|
|
||||||
|
- name: Nightly Test (4-GPU XPU)
|
||||||
|
timeout-minutes: 120
|
||||||
|
run: |
|
||||||
|
touch github_summary.md
|
||||||
|
docker exec ci_sglang_xpu bash -c "
|
||||||
|
source /opt/venv/bin/activate &&
|
||||||
|
cd /sglang-checkout/test &&
|
||||||
|
GITHUB_STEP_SUMMARY=/sglang-checkout/github_summary.md \
|
||||||
|
python3 run_suite.py --hw xpu --suite nightly-xpu-4-gpu --nightly --timeout-per-file 2400 ${{ (github.event_name == 'schedule' || inputs.continue_on_error) && '--continue-on-error' || '' }}
|
||||||
|
" || TEST_EXIT_CODE=$?
|
||||||
|
echo "$(<github_summary.md)" >> $GITHUB_STEP_SUMMARY || true
|
||||||
|
exit ${TEST_EXIT_CODE:-0}
|
||||||
|
|
||||||
|
- name: Cleanup container
|
||||||
|
if: always()
|
||||||
|
run: |
|
||||||
|
docker run --rm -v "${{ github.workspace }}:/w" busybox:latest \
|
||||||
|
chown -R "$(id -u):$(id -g)" /w || true
|
||||||
|
rm -rf test/result.jsonl test/results test/.pytest_cache .pytest_cache || true
|
||||||
|
find . -type d -name "__pycache__" -prune -exec rm -rf {} + || true
|
||||||
|
find . -type f -name "*.pyc" -delete || true
|
||||||
|
docker rm -f ci_sglang_xpu || true
|
||||||
|
if [[ -n "${CI_SGLANG_XPU_IMAGE:-}" ]]; then
|
||||||
|
docker rmi -f "${CI_SGLANG_XPU_IMAGE}" || true
|
||||||
|
fi
|
||||||
|
|
||||||
|
check-all-jobs:
|
||||||
|
if: always() && (github.repository == 'sgl-project/sglang' || github.event_name == 'workflow_dispatch')
|
||||||
|
needs:
|
||||||
|
- nightly-xpu-2-gpu
|
||||||
|
- nightly-xpu-4-gpu
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
steps:
|
||||||
|
- name: Check if any job failed
|
||||||
|
run: |
|
||||||
|
if [[ "${{ contains(needs.*.result, 'failure') }}" == "true" ]]; then
|
||||||
|
echo "One or more nightly test jobs failed"
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
if [[ "${{ contains(needs.*.result, 'cancelled') }}" == "true" ]]; then
|
||||||
|
echo "One or more nightly test jobs were cancelled"
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
echo "All nightly test jobs passed"
|
||||||
|
|||||||
@@ -0,0 +1,129 @@
|
|||||||
|
"""simple-evals GSM8K accuracy mixin for Intel XPU nightly tests.
|
||||||
|
|
||||||
|
Mirrors the AMD/NVIDIA nightly flow (``test_gsm8k_eval_amd.py`` /
|
||||||
|
``test_text_models_gsm8k_eval.py``): launch an SGLang server with XPU
|
||||||
|
flags, then call ``sglang.test.run_eval`` with ``eval_name="gsm8k"`` so
|
||||||
|
the same ``simple_eval_gsm8k.GSM8KEval`` evaluator scores every backend.
|
||||||
|
|
||||||
|
Subclasses set ``model``, ``tp_size``, ``accuracy``, and may override
|
||||||
|
``other_args`` / ``env`` / ``num_examples`` / ``num_threads``.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import os
|
||||||
|
import subprocess
|
||||||
|
from abc import ABC
|
||||||
|
from types import SimpleNamespace
|
||||||
|
|
||||||
|
from sglang.srt.utils import kill_process_tree
|
||||||
|
from sglang.test.run_eval import run_eval
|
||||||
|
from sglang.test.test_utils import (
|
||||||
|
DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH,
|
||||||
|
DEFAULT_URL_FOR_TEST,
|
||||||
|
popen_launch_server,
|
||||||
|
write_github_step_summary,
|
||||||
|
)
|
||||||
|
from sglang.test.xpu.test_xpu_utils import write_results_to_github_step_summary
|
||||||
|
|
||||||
|
|
||||||
|
class SimpleEvalGSM8KXPUMixin(ABC):
|
||||||
|
model: str = ""
|
||||||
|
tp_size: int = 1
|
||||||
|
|
||||||
|
timeout_for_server_launch = DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH
|
||||||
|
other_args: list[str] = [
|
||||||
|
"--device",
|
||||||
|
"xpu",
|
||||||
|
"--attention-backend",
|
||||||
|
"intel_xpu",
|
||||||
|
"--dtype",
|
||||||
|
"bfloat16",
|
||||||
|
"--trust-remote-code",
|
||||||
|
"--disable-overlap-schedule",
|
||||||
|
"--disable-radix-cache",
|
||||||
|
]
|
||||||
|
env: dict | None = None
|
||||||
|
|
||||||
|
server_cmd: str = ""
|
||||||
|
# 200 questions matches the limit used by the XPU 70B lm-eval YAML and
|
||||||
|
# fits inside run_suite's per-file timeout when num_threads=1 keeps
|
||||||
|
# throughput low. Subclasses on cheaper-per-token hardware (TP=1, no
|
||||||
|
# Level Zero wedge) can raise this or set None for the full 1319-question
|
||||||
|
# GSM8K test set, matching the AMD/NVIDIA nightly defaults.
|
||||||
|
num_examples: int | None = 200
|
||||||
|
# Single-stream eval: intel_xpu attention at TP>=2 wedges the Level Zero
|
||||||
|
# driver in ur_command_list_manager::appendUSMMemcpy on concurrent prefill.
|
||||||
|
# Subclasses on hardware that handles parallel prefill cleanly may bump.
|
||||||
|
num_threads: int = 1
|
||||||
|
# Short generations reduce the rate of prefill->decode->prefill handoffs,
|
||||||
|
# which is what trips the same Level Zero wedge on TP>=2 (observed at the
|
||||||
|
# default 2048; 512 matches the original few_shot_gsm8k limit and is still
|
||||||
|
# enough for GSM8K CoT answers).
|
||||||
|
max_tokens: int = 512
|
||||||
|
|
||||||
|
@classmethod
|
||||||
|
def setUpClass(cls):
|
||||||
|
cls.base_url = DEFAULT_URL_FOR_TEST
|
||||||
|
env = {**os.environ, **(cls.env or {})}
|
||||||
|
args = list(cls.other_args) + ["--tp-size", str(cls.tp_size)]
|
||||||
|
try:
|
||||||
|
cls.process = popen_launch_server(
|
||||||
|
cls.model,
|
||||||
|
cls.base_url,
|
||||||
|
timeout=cls.timeout_for_server_launch,
|
||||||
|
other_args=args,
|
||||||
|
env=env,
|
||||||
|
)
|
||||||
|
cls.server_cmd = subprocess.list2cmdline(cls.process.args)
|
||||||
|
except Exception as e:
|
||||||
|
write_github_step_summary(f"Failed to launch server for {cls.model}: {e}")
|
||||||
|
raise AssertionError(f"Test failed for {cls.model}: {e}")
|
||||||
|
|
||||||
|
@classmethod
|
||||||
|
def tearDownClass(cls):
|
||||||
|
kill_process_tree(cls.process.pid)
|
||||||
|
|
||||||
|
def test_gsm8k(self):
|
||||||
|
accuracy_threshold = getattr(self, "accuracy", 0.0)
|
||||||
|
output_throughput_threshold = getattr(self, "output_throughput", 0.0)
|
||||||
|
|
||||||
|
model_metrics = {
|
||||||
|
"server": self.server_cmd,
|
||||||
|
"client": "simple_eval_gsm8k",
|
||||||
|
"accuracy_threshold": getattr(self, "accuracy", "N/A"),
|
||||||
|
"output_throughput_threshold": getattr(self, "output_throughput", "N/A"),
|
||||||
|
}
|
||||||
|
|
||||||
|
try:
|
||||||
|
args = SimpleNamespace(
|
||||||
|
base_url=self.base_url,
|
||||||
|
model=self.model,
|
||||||
|
eval_name="gsm8k",
|
||||||
|
num_examples=self.num_examples,
|
||||||
|
num_threads=self.num_threads,
|
||||||
|
max_tokens=self.max_tokens,
|
||||||
|
)
|
||||||
|
metrics = run_eval(args)
|
||||||
|
model_metrics["accuracy"] = metrics["score"]
|
||||||
|
model_metrics["output_throughput"] = metrics.get("output_throughput")
|
||||||
|
model_metrics["latency"] = metrics["latency"]
|
||||||
|
self.assertGreaterEqual(
|
||||||
|
metrics["score"],
|
||||||
|
accuracy_threshold,
|
||||||
|
f'Accuracy of {self.model} is {metrics["score"]}, '
|
||||||
|
f"is lower than {accuracy_threshold}",
|
||||||
|
)
|
||||||
|
if "output_throughput" in metrics:
|
||||||
|
self.assertGreaterEqual(
|
||||||
|
metrics["output_throughput"],
|
||||||
|
output_throughput_threshold,
|
||||||
|
f"Output throughput of {self.model} is "
|
||||||
|
f'{metrics["output_throughput"]}, is lower than '
|
||||||
|
f"{output_throughput_threshold}",
|
||||||
|
)
|
||||||
|
except Exception as e:
|
||||||
|
model_metrics["error"] = str(e)
|
||||||
|
self.fail(f"Test failed for {self.model}: {e}")
|
||||||
|
finally:
|
||||||
|
write_results_to_github_step_summary({self.model: model_metrics})
|
||||||
@@ -0,0 +1,50 @@
|
|||||||
|
"""Helpers shared by XPU nightly model tests.
|
||||||
|
|
||||||
|
The summary writer mirrors `python/sglang/test/ascend/test_ascend_utils.py`
|
||||||
|
so XPU and Ascend nightly runs render the same Markdown table in
|
||||||
|
`$GITHUB_STEP_SUMMARY`.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from sglang.test.test_utils import is_in_ci, write_github_step_summary
|
||||||
|
|
||||||
|
HEADER = """
|
||||||
|
| Model | Server | Client | Output Throughput | Expected Output Throughput | Accuracy | Expected Accuracy | Status |
|
||||||
|
| ----- | ------ | ------ | ----------------- | -------------------------- | -------- | ----------------- | ------ |
|
||||||
|
"""
|
||||||
|
|
||||||
|
_HEADER_WRITTEN = False
|
||||||
|
|
||||||
|
|
||||||
|
def _write_header_once():
|
||||||
|
global _HEADER_WRITTEN
|
||||||
|
if not _HEADER_WRITTEN:
|
||||||
|
write_github_step_summary(HEADER)
|
||||||
|
_HEADER_WRITTEN = True
|
||||||
|
|
||||||
|
|
||||||
|
def write_results_to_github_step_summary(results: dict):
|
||||||
|
if not is_in_ci():
|
||||||
|
return
|
||||||
|
|
||||||
|
_write_header_once()
|
||||||
|
|
||||||
|
def fmt(metrics, key, precision):
|
||||||
|
v = metrics.get(key, "-")
|
||||||
|
return f"{v:.{precision}f}" if isinstance(v, (int, float)) else v
|
||||||
|
|
||||||
|
summary = ""
|
||||||
|
for model, metrics in results.items():
|
||||||
|
accuracy = fmt(metrics, "accuracy", 4)
|
||||||
|
accuracy_threshold = metrics.get("accuracy_threshold", "N/A")
|
||||||
|
output_throughput = fmt(metrics, "output_throughput", 2)
|
||||||
|
output_throughput_threshold = metrics.get("output_throughput_threshold", "N/A")
|
||||||
|
server = metrics.get("server", "N/A")
|
||||||
|
client = metrics.get("client", "N/A")
|
||||||
|
error = metrics.get("error", "")
|
||||||
|
status = "PASS" if error == "" else f"FAIL: {error}"
|
||||||
|
summary += (
|
||||||
|
f"| {model} | {server} | {client} | {output_throughput} "
|
||||||
|
f"| {output_throughput_threshold} | {accuracy} "
|
||||||
|
f"| {accuracy_threshold} | {status} |\n"
|
||||||
|
)
|
||||||
|
write_github_step_summary(summary)
|
||||||
@@ -0,0 +1,39 @@
|
|||||||
|
"""Llama-3.1-8B-Instruct GSM8K accuracy on Intel XPU (TP=2).
|
||||||
|
|
||||||
|
TP=4 wedges the Level Zero driver during the first prefill batch on Arc/BMG;
|
||||||
|
TP=2 runs cleanly with the same model and serves at ~18 tok/s.
|
||||||
|
|
||||||
|
Scored by ``simple_eval_gsm8k.GSM8KEval`` (the same evaluator AMD and
|
||||||
|
NVIDIA nightlies use); threshold mirrors theirs.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import unittest
|
||||||
|
|
||||||
|
import torch
|
||||||
|
|
||||||
|
from sglang.test.ci.ci_register import register_xpu_ci
|
||||||
|
from sglang.test.test_utils import CustomTestCase
|
||||||
|
from sglang.test.xpu.simple_eval_gsm8k_xpu_mixin import SimpleEvalGSM8KXPUMixin
|
||||||
|
|
||||||
|
register_xpu_ci(est_time=1200, suite="nightly-xpu-2-gpu", nightly=True)
|
||||||
|
|
||||||
|
|
||||||
|
@unittest.skipUnless(
|
||||||
|
torch.xpu.is_available(),
|
||||||
|
"Intel XPU not available (torch.xpu.is_available() returned False)",
|
||||||
|
)
|
||||||
|
class TestLlama31_8BInstructXPU(SimpleEvalGSM8KXPUMixin, CustomTestCase):
|
||||||
|
model = "meta-llama/Llama-3.1-8B-Instruct"
|
||||||
|
tp_size = 2
|
||||||
|
accuracy = 0.80
|
||||||
|
|
||||||
|
other_args = SimpleEvalGSM8KXPUMixin.other_args + [
|
||||||
|
"--max-total-tokens",
|
||||||
|
"65536",
|
||||||
|
"--mem-fraction-static",
|
||||||
|
"0.8",
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
unittest.main()
|
||||||
@@ -0,0 +1,40 @@
|
|||||||
|
"""Qwen3-32B GSM8K accuracy on Intel XPU (TP=4).
|
||||||
|
|
||||||
|
Scored by ``simple_eval_gsm8k.GSM8KEval`` (the same evaluator AMD and
|
||||||
|
NVIDIA nightlies use).
|
||||||
|
"""
|
||||||
|
|
||||||
|
import unittest
|
||||||
|
|
||||||
|
import torch
|
||||||
|
|
||||||
|
from sglang.test.ci.ci_register import register_xpu_ci
|
||||||
|
from sglang.test.test_utils import CustomTestCase
|
||||||
|
from sglang.test.xpu.simple_eval_gsm8k_xpu_mixin import SimpleEvalGSM8KXPUMixin
|
||||||
|
|
||||||
|
register_xpu_ci(est_time=1800, suite="nightly-xpu-4-gpu", nightly=True)
|
||||||
|
|
||||||
|
|
||||||
|
@unittest.skipUnless(
|
||||||
|
torch.xpu.is_available(),
|
||||||
|
"Intel XPU not available (torch.xpu.is_available() returned False)",
|
||||||
|
)
|
||||||
|
class TestQwen3_32BXPU(SimpleEvalGSM8KXPUMixin, CustomTestCase):
|
||||||
|
model = "Qwen/Qwen3-32B"
|
||||||
|
tp_size = 4
|
||||||
|
accuracy = 0.85
|
||||||
|
# 64GB BF16 weights split across 4 ranks take ~9 min to load on Intel
|
||||||
|
# Arc Pro B60; the default 600s timeout fires mid-startup. Mirror the
|
||||||
|
# XPU 70B test's 1-hour budget.
|
||||||
|
timeout_for_server_launch = 3600
|
||||||
|
|
||||||
|
other_args = SimpleEvalGSM8KXPUMixin.other_args + [
|
||||||
|
"--max-total-tokens",
|
||||||
|
"65536",
|
||||||
|
"--mem-fraction-static",
|
||||||
|
"0.8",
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
unittest.main()
|
||||||
+5
-1
@@ -169,7 +169,11 @@ NIGHTLY_SUITES = {
|
|||||||
"full-8-npu-a3",
|
"full-8-npu-a3",
|
||||||
"full-16-npu-a3",
|
"full-16-npu-a3",
|
||||||
],
|
],
|
||||||
HWBackend.XPU: [],
|
HWBackend.XPU: [
|
||||||
|
"nightly-xpu-1-gpu",
|
||||||
|
"nightly-xpu-2-gpu",
|
||||||
|
"nightly-xpu-4-gpu",
|
||||||
|
],
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user