ci(xpu): harden nightly + PR XPU CI (HF login / tag fetch / docker push retries) (#32390)
This commit is contained in:
@@ -69,7 +69,18 @@ jobs:
|
|||||||
# Diffusion extras (needed by test_xpu_flux2_dev, test_xpu_zimage_turbo).
|
# Diffusion extras (needed by test_xpu_flux2_dev, test_xpu_zimage_turbo).
|
||||||
# Pins mirror python/pyproject_xpu.toml [project.optional-dependencies.diffusion].
|
# Pins mirror python/pyproject_xpu.toml [project.optional-dependencies.diffusion].
|
||||||
docker exec ci_sglang_xpu /opt/venv/bin/python3 -m pip install --no-cache-dir "sglang[diffusion]"
|
docker exec ci_sglang_xpu /opt/venv/bin/python3 -m pip install --no-cache-dir "sglang[diffusion]"
|
||||||
docker exec ci_sglang_xpu /bin/bash -c '/opt/venv/bin/hf auth login --token ${HF_TOKEN}'
|
# HF /whoami-v2 is aggressively rate-limited and the three matrix
|
||||||
|
# jobs (1/2/4-GPU) all hit it at the same second under a scheduled
|
||||||
|
# run. Retry with linear backoff + jitter to desynchronize them.
|
||||||
|
docker exec ci_sglang_xpu /bin/bash -c '
|
||||||
|
for i in 1 2 3 4 5; do
|
||||||
|
/opt/venv/bin/hf auth login --token "${HF_TOKEN}" && exit 0
|
||||||
|
sleep_secs=$((i * 15 + RANDOM % 15))
|
||||||
|
echo "hf auth login failed (attempt $i), retrying in ${sleep_secs}s..."
|
||||||
|
sleep "${sleep_secs}"
|
||||||
|
done
|
||||||
|
exit 1
|
||||||
|
'
|
||||||
|
|
||||||
- name: Download olmOCR-bench dataset (for test_deepseek_ocr_2_olmbench)
|
- name: Download olmOCR-bench dataset (for test_deepseek_ocr_2_olmbench)
|
||||||
timeout-minutes: 30
|
timeout-minutes: 30
|
||||||
@@ -143,7 +154,18 @@ jobs:
|
|||||||
# Diffusion extras (needed by test_xpu_flux2_dev, test_xpu_zimage_turbo).
|
# Diffusion extras (needed by test_xpu_flux2_dev, test_xpu_zimage_turbo).
|
||||||
# Pins mirror python/pyproject_xpu.toml [project.optional-dependencies.diffusion].
|
# Pins mirror python/pyproject_xpu.toml [project.optional-dependencies.diffusion].
|
||||||
docker exec ci_sglang_xpu /opt/venv/bin/python3 -m pip install --no-cache-dir "sglang[diffusion]"
|
docker exec ci_sglang_xpu /opt/venv/bin/python3 -m pip install --no-cache-dir "sglang[diffusion]"
|
||||||
docker exec ci_sglang_xpu /bin/bash -c '/opt/venv/bin/hf auth login --token ${HF_TOKEN}'
|
# HF /whoami-v2 is aggressively rate-limited and the three matrix
|
||||||
|
# jobs (1/2/4-GPU) all hit it at the same second under a scheduled
|
||||||
|
# run. Retry with linear backoff + jitter to desynchronize them.
|
||||||
|
docker exec ci_sglang_xpu /bin/bash -c '
|
||||||
|
for i in 1 2 3 4 5; do
|
||||||
|
/opt/venv/bin/hf auth login --token "${HF_TOKEN}" && exit 0
|
||||||
|
sleep_secs=$((i * 15 + RANDOM % 15))
|
||||||
|
echo "hf auth login failed (attempt $i), retrying in ${sleep_secs}s..."
|
||||||
|
sleep "${sleep_secs}"
|
||||||
|
done
|
||||||
|
exit 1
|
||||||
|
'
|
||||||
|
|
||||||
- name: Nightly Test (2-GPU XPU)
|
- name: Nightly Test (2-GPU XPU)
|
||||||
timeout-minutes: 240
|
timeout-minutes: 240
|
||||||
@@ -206,7 +228,18 @@ jobs:
|
|||||||
# Diffusion extras (needed by test_xpu_flux2_dev, test_xpu_zimage_turbo).
|
# Diffusion extras (needed by test_xpu_flux2_dev, test_xpu_zimage_turbo).
|
||||||
# Pins mirror python/pyproject_xpu.toml [project.optional-dependencies.diffusion].
|
# Pins mirror python/pyproject_xpu.toml [project.optional-dependencies.diffusion].
|
||||||
docker exec ci_sglang_xpu /opt/venv/bin/python3 -m pip install --no-cache-dir "sglang[diffusion]"
|
docker exec ci_sglang_xpu /opt/venv/bin/python3 -m pip install --no-cache-dir "sglang[diffusion]"
|
||||||
docker exec ci_sglang_xpu /bin/bash -c '/opt/venv/bin/hf auth login --token ${HF_TOKEN}'
|
# HF /whoami-v2 is aggressively rate-limited and the three matrix
|
||||||
|
# jobs (1/2/4-GPU) all hit it at the same second under a scheduled
|
||||||
|
# run. Retry with linear backoff + jitter to desynchronize them.
|
||||||
|
docker exec ci_sglang_xpu /bin/bash -c '
|
||||||
|
for i in 1 2 3 4 5; do
|
||||||
|
/opt/venv/bin/hf auth login --token "${HF_TOKEN}" && exit 0
|
||||||
|
sleep_secs=$((i * 15 + RANDOM % 15))
|
||||||
|
echo "hf auth login failed (attempt $i), retrying in ${sleep_secs}s..."
|
||||||
|
sleep "${sleep_secs}"
|
||||||
|
done
|
||||||
|
exit 1
|
||||||
|
'
|
||||||
|
|
||||||
- name: Nightly Test (4-GPU XPU)
|
- name: Nightly Test (4-GPU XPU)
|
||||||
timeout-minutes: 480
|
timeout-minutes: 480
|
||||||
|
|||||||
@@ -106,6 +106,9 @@ jobs:
|
|||||||
docker exec ci_sglang_xpu /opt/venv/bin/python3 -m pip install pytest expecttest ray huggingface_hub tabulate "lmcache>=0.3.9"
|
docker exec ci_sglang_xpu /opt/venv/bin/python3 -m pip install pytest expecttest ray huggingface_hub tabulate "lmcache>=0.3.9"
|
||||||
docker exec ci_sglang_xpu /opt/venv/bin/python3 -m pip uninstall -y flashinfer-python sgl-kernel sglang
|
docker exec ci_sglang_xpu /opt/venv/bin/python3 -m pip uninstall -y flashinfer-python sgl-kernel sglang
|
||||||
docker exec ci_sglang_xpu cp /sglang-checkout/python/pyproject_xpu.toml /sglang-checkout/python/pyproject.toml
|
docker exec ci_sglang_xpu cp /sglang-checkout/python/pyproject_xpu.toml /sglang-checkout/python/pyproject.toml
|
||||||
|
# Fetch tags so setuptools_scm resolves a real version instead of
|
||||||
|
# falling back to 0.0.0 on a shallow/tag-less checkout.
|
||||||
|
docker exec -w /sglang-checkout ci_sglang_xpu git fetch origin '+refs/tags/*:refs/tags/*' --force
|
||||||
docker exec -w /sglang-checkout/python ci_sglang_xpu /opt/venv/bin/python3 -m pip install --no-cache-dir . --extra-index-url https://download.pytorch.org/whl/xpu
|
docker exec -w /sglang-checkout/python ci_sglang_xpu /opt/venv/bin/python3 -m pip install --no-cache-dir . --extra-index-url https://download.pytorch.org/whl/xpu
|
||||||
docker exec ci_sglang_xpu /opt/venv/bin/python3 -m pip install --no-cache-dir --no-deps xgrammar==0.1.33
|
docker exec ci_sglang_xpu /opt/venv/bin/python3 -m pip install --no-cache-dir --no-deps xgrammar==0.1.33
|
||||||
docker exec ci_sglang_xpu /bin/bash -c '/opt/venv/bin/hf auth login --token ${HF_TOKEN}'
|
docker exec ci_sglang_xpu /bin/bash -c '/opt/venv/bin/hf auth login --token ${HF_TOKEN}'
|
||||||
@@ -195,6 +198,9 @@ jobs:
|
|||||||
docker exec ci_sglang_xpu /opt/venv/bin/python3 -m pip install pytest expecttest ray huggingface_hub tabulate "lmcache>=0.3.9"
|
docker exec ci_sglang_xpu /opt/venv/bin/python3 -m pip install pytest expecttest ray huggingface_hub tabulate "lmcache>=0.3.9"
|
||||||
docker exec ci_sglang_xpu /opt/venv/bin/python3 -m pip uninstall -y flashinfer-python sgl-kernel sglang
|
docker exec ci_sglang_xpu /opt/venv/bin/python3 -m pip uninstall -y flashinfer-python sgl-kernel sglang
|
||||||
docker exec ci_sglang_xpu cp /sglang-checkout/python/pyproject_xpu.toml /sglang-checkout/python/pyproject.toml
|
docker exec ci_sglang_xpu cp /sglang-checkout/python/pyproject_xpu.toml /sglang-checkout/python/pyproject.toml
|
||||||
|
# Fetch tags so setuptools_scm resolves a real version instead of
|
||||||
|
# falling back to 0.0.0 on a shallow/tag-less checkout.
|
||||||
|
docker exec -w /sglang-checkout ci_sglang_xpu git fetch origin '+refs/tags/*:refs/tags/*' --force
|
||||||
docker exec -w /sglang-checkout/python ci_sglang_xpu /opt/venv/bin/python3 -m pip install --no-cache-dir . --extra-index-url https://download.pytorch.org/whl/xpu
|
docker exec -w /sglang-checkout/python ci_sglang_xpu /opt/venv/bin/python3 -m pip install --no-cache-dir . --extra-index-url https://download.pytorch.org/whl/xpu
|
||||||
docker exec ci_sglang_xpu /opt/venv/bin/python3 -m pip install --no-cache-dir --no-deps xgrammar==0.1.33
|
docker exec ci_sglang_xpu /opt/venv/bin/python3 -m pip install --no-cache-dir --no-deps xgrammar==0.1.33
|
||||||
docker exec ci_sglang_xpu /bin/bash -c '/opt/venv/bin/hf auth login --token ${HF_TOKEN}'
|
docker exec ci_sglang_xpu /bin/bash -c '/opt/venv/bin/hf auth login --token ${HF_TOKEN}'
|
||||||
|
|||||||
@@ -55,5 +55,13 @@ jobs:
|
|||||||
|
|
||||||
- name: Push intel/sglang-dev
|
- name: Push intel/sglang-dev
|
||||||
run: |
|
run: |
|
||||||
docker push "intel/sglang-dev:${{ env.IMAGE_TAG }}"
|
push_with_retry() {
|
||||||
docker push "intel/sglang-dev:latest"
|
for i in 1 2 3 4 5; do
|
||||||
|
docker push "$1" && return 0
|
||||||
|
echo "push failed (attempt $i), retrying in $((i*15))s..."
|
||||||
|
sleep $((i*15))
|
||||||
|
done
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
push_with_retry "intel/sglang-dev:${{ env.IMAGE_TAG }}"
|
||||||
|
push_with_retry "intel/sglang-dev:latest"
|
||||||
|
|||||||
Reference in New Issue
Block a user