[AMD] Test DeepSeek V4 FlashMLA backend variants nightly (#28290)
This commit is contained in:
@@ -1250,15 +1250,28 @@ jobs:
|
|||||||
bash scripts/ci/amd/amd_ci_install_dependency.sh --skip-test-time-deps
|
bash scripts/ci/amd/amd_ci_install_dependency.sh --skip-test-time-deps
|
||||||
bash scripts/ci/amd/amd_ci_exec.sh pip install tabulate
|
bash scripts/ci/amd/amd_ci_exec.sh pip install tabulate
|
||||||
|
|
||||||
- name: Accuracy + Performance Test MI35x ROCm 7.2 (8-GPU DeepSeek-V4-Flash FP8 + FP4)
|
- name: Accuracy + Performance Test MI35x ROCm 7.2 (8-GPU DeepSeek-V4-Flash FP8 + FP4, unified_kv_triton)
|
||||||
timeout-minutes: 300
|
timeout-minutes: 300
|
||||||
run: |
|
run: |
|
||||||
> github_summary.md # Clear summary file
|
> github_summary.md # Clear summary file
|
||||||
|
echo "## SGLANG_HACK_FLASHMLA_BACKEND=unified_kv_triton" >> github_summary.md
|
||||||
bash scripts/ci/amd/amd_ci_exec.sh -w /sglang-checkout/test \
|
bash scripts/ci/amd/amd_ci_exec.sh -w /sglang-checkout/test \
|
||||||
|
-e SGLANG_HACK_FLASHMLA_BACKEND=unified_kv_triton \
|
||||||
-e GITHUB_STEP_SUMMARY="/sglang-checkout/github_summary.md" \
|
-e GITHUB_STEP_SUMMARY="/sglang-checkout/github_summary.md" \
|
||||||
python3 run_suite.py --hw amd --suite nightly-amd-8-gpu-mi35x-deepseek-v4-flash --nightly --timeout-per-file 7200 ${{ (github.event_name == 'schedule' || inputs.continue_on_error) && '--continue-on-error' || '' }} || TEST_EXIT_CODE=$?
|
python3 run_suite.py --hw amd --suite nightly-amd-8-gpu-mi35x-deepseek-v4-flash --nightly --timeout-per-file 7200 ${{ (github.event_name == 'schedule' || inputs.continue_on_error) && '--continue-on-error' || '' }}
|
||||||
|
echo "$(<github_summary.md )" >> $GITHUB_STEP_SUMMARY || true
|
||||||
|
|
||||||
|
- name: Accuracy + Performance Test MI35x ROCm 7.2 (8-GPU DeepSeek-V4-Flash FP8 + FP4, triton)
|
||||||
|
if: ${{ !cancelled() }}
|
||||||
|
timeout-minutes: 300
|
||||||
|
run: |
|
||||||
|
> github_summary.md # Clear summary file
|
||||||
|
echo "## SGLANG_HACK_FLASHMLA_BACKEND=triton" >> github_summary.md
|
||||||
|
bash scripts/ci/amd/amd_ci_exec.sh -w /sglang-checkout/test \
|
||||||
|
-e SGLANG_HACK_FLASHMLA_BACKEND=triton \
|
||||||
|
-e GITHUB_STEP_SUMMARY="/sglang-checkout/github_summary.md" \
|
||||||
|
python3 run_suite.py --hw amd --suite nightly-amd-8-gpu-mi35x-deepseek-v4-flash --nightly --timeout-per-file 7200 ${{ (github.event_name == 'schedule' || inputs.continue_on_error) && '--continue-on-error' || '' }}
|
||||||
echo "$(<github_summary.md )" >> $GITHUB_STEP_SUMMARY || true
|
echo "$(<github_summary.md )" >> $GITHUB_STEP_SUMMARY || true
|
||||||
exit ${TEST_EXIT_CODE:-0}
|
|
||||||
|
|
||||||
nightly-8-gpu-mi35x-deepseek-v4-pro-rocm720:
|
nightly-8-gpu-mi35x-deepseek-v4-pro-rocm720:
|
||||||
if: (github.repository == 'sgl-project/sglang' || github.event_name == 'pull_request') && (!(inputs.job_filter || inputs.job_select) || (inputs.job_filter || inputs.job_select) == 'all' || contains(format(',{0},', inputs.job_filter || inputs.job_select), ',nightly-8-gpu-mi35x-deepseek-v4-pro-rocm720,'))
|
if: (github.repository == 'sgl-project/sglang' || github.event_name == 'pull_request') && (!(inputs.job_filter || inputs.job_select) || (inputs.job_filter || inputs.job_select) == 'all' || contains(format(',{0},', inputs.job_filter || inputs.job_select), ',nightly-8-gpu-mi35x-deepseek-v4-pro-rocm720,'))
|
||||||
@@ -1285,15 +1298,28 @@ jobs:
|
|||||||
bash scripts/ci/amd/amd_ci_install_dependency.sh --skip-test-time-deps
|
bash scripts/ci/amd/amd_ci_install_dependency.sh --skip-test-time-deps
|
||||||
bash scripts/ci/amd/amd_ci_exec.sh pip install tabulate
|
bash scripts/ci/amd/amd_ci_exec.sh pip install tabulate
|
||||||
|
|
||||||
- name: Accuracy + Performance Test MI35x ROCm 7.2 (8-GPU DeepSeek-V4-Pro FP8 + FP4)
|
- name: Accuracy + Performance Test MI35x ROCm 7.2 (8-GPU DeepSeek-V4-Pro FP8 + FP4, unified_kv_triton)
|
||||||
timeout-minutes: 480
|
timeout-minutes: 480
|
||||||
run: |
|
run: |
|
||||||
> github_summary.md # Clear summary file
|
> github_summary.md # Clear summary file
|
||||||
|
echo "## SGLANG_HACK_FLASHMLA_BACKEND=unified_kv_triton" >> github_summary.md
|
||||||
bash scripts/ci/amd/amd_ci_exec.sh -w /sglang-checkout/test \
|
bash scripts/ci/amd/amd_ci_exec.sh -w /sglang-checkout/test \
|
||||||
|
-e SGLANG_HACK_FLASHMLA_BACKEND=unified_kv_triton \
|
||||||
-e GITHUB_STEP_SUMMARY="/sglang-checkout/github_summary.md" \
|
-e GITHUB_STEP_SUMMARY="/sglang-checkout/github_summary.md" \
|
||||||
python3 run_suite.py --hw amd --suite nightly-amd-8-gpu-mi35x-deepseek-v4-pro --nightly --timeout-per-file 14400 ${{ (github.event_name == 'schedule' || inputs.continue_on_error) && '--continue-on-error' || '' }} || TEST_EXIT_CODE=$?
|
python3 run_suite.py --hw amd --suite nightly-amd-8-gpu-mi35x-deepseek-v4-pro --nightly --timeout-per-file 14400 ${{ (github.event_name == 'schedule' || inputs.continue_on_error) && '--continue-on-error' || '' }}
|
||||||
|
echo "$(<github_summary.md )" >> $GITHUB_STEP_SUMMARY || true
|
||||||
|
|
||||||
|
- name: Accuracy + Performance Test MI35x ROCm 7.2 (8-GPU DeepSeek-V4-Pro FP8 + FP4, triton)
|
||||||
|
if: ${{ !cancelled() }}
|
||||||
|
timeout-minutes: 480
|
||||||
|
run: |
|
||||||
|
> github_summary.md # Clear summary file
|
||||||
|
echo "## SGLANG_HACK_FLASHMLA_BACKEND=triton" >> github_summary.md
|
||||||
|
bash scripts/ci/amd/amd_ci_exec.sh -w /sglang-checkout/test \
|
||||||
|
-e SGLANG_HACK_FLASHMLA_BACKEND=triton \
|
||||||
|
-e GITHUB_STEP_SUMMARY="/sglang-checkout/github_summary.md" \
|
||||||
|
python3 run_suite.py --hw amd --suite nightly-amd-8-gpu-mi35x-deepseek-v4-pro --nightly --timeout-per-file 14400 ${{ (github.event_name == 'schedule' || inputs.continue_on_error) && '--continue-on-error' || '' }}
|
||||||
echo "$(<github_summary.md )" >> $GITHUB_STEP_SUMMARY || true
|
echo "$(<github_summary.md )" >> $GITHUB_STEP_SUMMARY || true
|
||||||
exit ${TEST_EXIT_CODE:-0}
|
|
||||||
|
|
||||||
# ==============================================================================
|
# ==============================================================================
|
||||||
# 8-GPU Kimi-K2.6 (MI30x + MI35x)
|
# 8-GPU Kimi-K2.6 (MI30x + MI35x)
|
||||||
|
|||||||
@@ -34,6 +34,7 @@ DEEPSEEK_V4_FP4_MODEL_PATH = os.environ.get(
|
|||||||
"DEEPSEEK_V4_FP4_MODEL_PATH", "deepseek-ai/DeepSeek-V4-Flash"
|
"DEEPSEEK_V4_FP4_MODEL_PATH", "deepseek-ai/DeepSeek-V4-Flash"
|
||||||
)
|
)
|
||||||
SERVER_LAUNCH_TIMEOUT = 3600
|
SERVER_LAUNCH_TIMEOUT = 3600
|
||||||
|
FLASHMLA_BACKEND = os.environ.get("SGLANG_HACK_FLASHMLA_BACKEND", "unified_kv_triton")
|
||||||
|
|
||||||
# Common DeepSeek-V4 env vars (AMD ROCm 7.2 path: AITER indexer + triton attn + ROCm700A).
|
# Common DeepSeek-V4 env vars (AMD ROCm 7.2 path: AITER indexer + triton attn + ROCm700A).
|
||||||
COMMON_ENV_VARS = {
|
COMMON_ENV_VARS = {
|
||||||
@@ -44,7 +45,7 @@ COMMON_ENV_VARS = {
|
|||||||
"SGLANG_USE_ROCM700A": "1",
|
"SGLANG_USE_ROCM700A": "1",
|
||||||
"SGLANG_OPT_USE_FUSED_COMPRESS": "true",
|
"SGLANG_OPT_USE_FUSED_COMPRESS": "true",
|
||||||
"SGLANG_OPT_USE_FUSED_COMPRESS_TRITON": "true",
|
"SGLANG_OPT_USE_FUSED_COMPRESS_TRITON": "true",
|
||||||
"SGLANG_HACK_FLASHMLA_BACKEND": "unified_kv_triton",
|
"SGLANG_HACK_FLASHMLA_BACKEND": FLASHMLA_BACKEND,
|
||||||
"SGLANG_OPT_FP8_WO_A_GEMM": "false",
|
"SGLANG_OPT_FP8_WO_A_GEMM": "false",
|
||||||
"SGLANG_OPT_USE_JIT_INDEXER_METADATA": "false",
|
"SGLANG_OPT_USE_JIT_INDEXER_METADATA": "false",
|
||||||
"SGLANG_OPT_USE_TOPK_V2": "false",
|
"SGLANG_OPT_USE_TOPK_V2": "false",
|
||||||
@@ -126,7 +127,7 @@ class TestDeepseekV4Fp4(CustomTestCase):
|
|||||||
|
|
||||||
if is_in_ci():
|
if is_in_ci():
|
||||||
write_github_step_summary(
|
write_github_step_summary(
|
||||||
f"### test_gsm8k (deepseek-v4-flash-fp4)\n"
|
f"### test_gsm8k (deepseek-v4-flash-fp4, {FLASHMLA_BACKEND})\n"
|
||||||
f'{metrics["accuracy"]=:.3f}\n'
|
f'{metrics["accuracy"]=:.3f}\n'
|
||||||
)
|
)
|
||||||
self.assertGreater(metrics["accuracy"], 0.91)
|
self.assertGreater(metrics["accuracy"], 0.91)
|
||||||
@@ -185,7 +186,7 @@ class TestDeepseekV4Fp4(CustomTestCase):
|
|||||||
report_results = results_data
|
report_results = results_data
|
||||||
|
|
||||||
summary_lines = [
|
summary_lines = [
|
||||||
"### test_perf_8k_1k (deepseek-v4-flash-fp4)",
|
f"### test_perf_8k_1k (deepseek-v4-flash-fp4, {FLASHMLA_BACKEND})",
|
||||||
"input_len=8192 output_len=1024",
|
"input_len=8192 output_len=1024",
|
||||||
"",
|
"",
|
||||||
"| batch size | latency (s) | input throughput (tok/s) | output throughput (tok/s) | ITL (ms) |",
|
"| batch size | latency (s) | input throughput (tok/s) | output throughput (tok/s) | ITL (ms) |",
|
||||||
|
|||||||
@@ -34,6 +34,7 @@ DEEPSEEK_V4_FP8_MODEL_PATH = os.environ.get(
|
|||||||
"DEEPSEEK_V4_FP8_MODEL_PATH", "sgl-project/DeepSeek-V4-Flash-FP8"
|
"DEEPSEEK_V4_FP8_MODEL_PATH", "sgl-project/DeepSeek-V4-Flash-FP8"
|
||||||
)
|
)
|
||||||
SERVER_LAUNCH_TIMEOUT = 3600
|
SERVER_LAUNCH_TIMEOUT = 3600
|
||||||
|
FLASHMLA_BACKEND = os.environ.get("SGLANG_HACK_FLASHMLA_BACKEND", "unified_kv_triton")
|
||||||
|
|
||||||
# Common DeepSeek-V4 env vars (AMD ROCm 7.2 path: AITER indexer + triton attn + ROCm700A).
|
# Common DeepSeek-V4 env vars (AMD ROCm 7.2 path: AITER indexer + triton attn + ROCm700A).
|
||||||
COMMON_ENV_VARS = {
|
COMMON_ENV_VARS = {
|
||||||
@@ -44,7 +45,7 @@ COMMON_ENV_VARS = {
|
|||||||
"SGLANG_USE_ROCM700A": "1",
|
"SGLANG_USE_ROCM700A": "1",
|
||||||
"SGLANG_OPT_USE_FUSED_COMPRESS": "true",
|
"SGLANG_OPT_USE_FUSED_COMPRESS": "true",
|
||||||
"SGLANG_OPT_USE_FUSED_COMPRESS_TRITON": "true",
|
"SGLANG_OPT_USE_FUSED_COMPRESS_TRITON": "true",
|
||||||
"SGLANG_HACK_FLASHMLA_BACKEND": "unified_kv_triton",
|
"SGLANG_HACK_FLASHMLA_BACKEND": FLASHMLA_BACKEND,
|
||||||
"SGLANG_OPT_FP8_WO_A_GEMM": "false",
|
"SGLANG_OPT_FP8_WO_A_GEMM": "false",
|
||||||
"SGLANG_OPT_USE_JIT_INDEXER_METADATA": "false",
|
"SGLANG_OPT_USE_JIT_INDEXER_METADATA": "false",
|
||||||
"SGLANG_OPT_USE_TOPK_V2": "false",
|
"SGLANG_OPT_USE_TOPK_V2": "false",
|
||||||
@@ -126,7 +127,7 @@ class TestDeepseekV4Fp8(CustomTestCase):
|
|||||||
|
|
||||||
if is_in_ci():
|
if is_in_ci():
|
||||||
write_github_step_summary(
|
write_github_step_summary(
|
||||||
f"### test_gsm8k (deepseek-v4-flash-fp8)\n"
|
f"### test_gsm8k (deepseek-v4-flash-fp8, {FLASHMLA_BACKEND})\n"
|
||||||
f'{metrics["accuracy"]=:.3f}\n'
|
f'{metrics["accuracy"]=:.3f}\n'
|
||||||
)
|
)
|
||||||
self.assertGreater(metrics["accuracy"], 0.91)
|
self.assertGreater(metrics["accuracy"], 0.91)
|
||||||
@@ -185,7 +186,7 @@ class TestDeepseekV4Fp8(CustomTestCase):
|
|||||||
report_results = results_data
|
report_results = results_data
|
||||||
|
|
||||||
summary_lines = [
|
summary_lines = [
|
||||||
"### test_perf_8k_1k (deepseek-v4-flash-fp8)",
|
f"### test_perf_8k_1k (deepseek-v4-flash-fp8, {FLASHMLA_BACKEND})",
|
||||||
"input_len=8192 output_len=1024",
|
"input_len=8192 output_len=1024",
|
||||||
"",
|
"",
|
||||||
"| batch size | latency (s) | input throughput (tok/s) | output throughput (tok/s) | ITL (ms) |",
|
"| batch size | latency (s) | input throughput (tok/s) | output throughput (tok/s) | ITL (ms) |",
|
||||||
|
|||||||
@@ -36,6 +36,7 @@ DEEPSEEK_V4_PRO_FP4_MODEL_PATH = os.environ.get(
|
|||||||
)
|
)
|
||||||
# Pro is 1.6T; weight load + warmup is much longer than Flash 285B.
|
# Pro is 1.6T; weight load + warmup is much longer than Flash 285B.
|
||||||
SERVER_LAUNCH_TIMEOUT = 5400
|
SERVER_LAUNCH_TIMEOUT = 5400
|
||||||
|
FLASHMLA_BACKEND = os.environ.get("SGLANG_HACK_FLASHMLA_BACKEND", "unified_kv_triton")
|
||||||
|
|
||||||
# Common DeepSeek-V4 env vars (AMD ROCm 7.2 path: AITER indexer + triton attn + ROCm700A).
|
# Common DeepSeek-V4 env vars (AMD ROCm 7.2 path: AITER indexer + triton attn + ROCm700A).
|
||||||
COMMON_ENV_VARS = {
|
COMMON_ENV_VARS = {
|
||||||
@@ -46,7 +47,7 @@ COMMON_ENV_VARS = {
|
|||||||
"SGLANG_USE_ROCM700A": "1",
|
"SGLANG_USE_ROCM700A": "1",
|
||||||
"SGLANG_OPT_USE_FUSED_COMPRESS": "true",
|
"SGLANG_OPT_USE_FUSED_COMPRESS": "true",
|
||||||
"SGLANG_OPT_USE_FUSED_COMPRESS_TRITON": "true",
|
"SGLANG_OPT_USE_FUSED_COMPRESS_TRITON": "true",
|
||||||
"SGLANG_HACK_FLASHMLA_BACKEND": "unified_kv_triton",
|
"SGLANG_HACK_FLASHMLA_BACKEND": FLASHMLA_BACKEND,
|
||||||
"SGLANG_OPT_FP8_WO_A_GEMM": "false",
|
"SGLANG_OPT_FP8_WO_A_GEMM": "false",
|
||||||
"SGLANG_OPT_USE_JIT_INDEXER_METADATA": "false",
|
"SGLANG_OPT_USE_JIT_INDEXER_METADATA": "false",
|
||||||
"SGLANG_OPT_USE_TOPK_V2": "false",
|
"SGLANG_OPT_USE_TOPK_V2": "false",
|
||||||
@@ -128,7 +129,7 @@ class TestDeepseekV4ProFp4(CustomTestCase):
|
|||||||
|
|
||||||
if is_in_ci():
|
if is_in_ci():
|
||||||
write_github_step_summary(
|
write_github_step_summary(
|
||||||
f"### test_gsm8k (deepseek-v4-pro-fp4)\n"
|
f"### test_gsm8k (deepseek-v4-pro-fp4, {FLASHMLA_BACKEND})\n"
|
||||||
f'{metrics["accuracy"]=:.3f}\n'
|
f'{metrics["accuracy"]=:.3f}\n'
|
||||||
)
|
)
|
||||||
self.assertGreater(metrics["accuracy"], 0.92)
|
self.assertGreater(metrics["accuracy"], 0.92)
|
||||||
@@ -187,7 +188,7 @@ class TestDeepseekV4ProFp4(CustomTestCase):
|
|||||||
report_results = results_data
|
report_results = results_data
|
||||||
|
|
||||||
summary_lines = [
|
summary_lines = [
|
||||||
"### test_perf_8k_1k (deepseek-v4-pro-fp4)",
|
f"### test_perf_8k_1k (deepseek-v4-pro-fp4, {FLASHMLA_BACKEND})",
|
||||||
"input_len=8192 output_len=1024",
|
"input_len=8192 output_len=1024",
|
||||||
"",
|
"",
|
||||||
"| batch size | latency (s) | input throughput (tok/s) | output throughput (tok/s) | ITL (ms) |",
|
"| batch size | latency (s) | input throughput (tok/s) | output throughput (tok/s) | ITL (ms) |",
|
||||||
|
|||||||
@@ -36,6 +36,7 @@ DEEPSEEK_V4_PRO_FP8_MODEL_PATH = os.environ.get(
|
|||||||
)
|
)
|
||||||
# Pro is 1.6T; weight load + warmup is much longer than Flash 285B.
|
# Pro is 1.6T; weight load + warmup is much longer than Flash 285B.
|
||||||
SERVER_LAUNCH_TIMEOUT = 5400
|
SERVER_LAUNCH_TIMEOUT = 5400
|
||||||
|
FLASHMLA_BACKEND = os.environ.get("SGLANG_HACK_FLASHMLA_BACKEND", "unified_kv_triton")
|
||||||
|
|
||||||
# Common DeepSeek-V4 env vars (AMD ROCm 7.2 path: AITER indexer + triton attn + ROCm700A).
|
# Common DeepSeek-V4 env vars (AMD ROCm 7.2 path: AITER indexer + triton attn + ROCm700A).
|
||||||
COMMON_ENV_VARS = {
|
COMMON_ENV_VARS = {
|
||||||
@@ -46,7 +47,7 @@ COMMON_ENV_VARS = {
|
|||||||
"SGLANG_USE_ROCM700A": "1",
|
"SGLANG_USE_ROCM700A": "1",
|
||||||
"SGLANG_OPT_USE_FUSED_COMPRESS": "true",
|
"SGLANG_OPT_USE_FUSED_COMPRESS": "true",
|
||||||
"SGLANG_OPT_USE_FUSED_COMPRESS_TRITON": "true",
|
"SGLANG_OPT_USE_FUSED_COMPRESS_TRITON": "true",
|
||||||
"SGLANG_HACK_FLASHMLA_BACKEND": "unified_kv_triton",
|
"SGLANG_HACK_FLASHMLA_BACKEND": FLASHMLA_BACKEND,
|
||||||
"SGLANG_OPT_FP8_WO_A_GEMM": "false",
|
"SGLANG_OPT_FP8_WO_A_GEMM": "false",
|
||||||
"SGLANG_OPT_USE_JIT_INDEXER_METADATA": "false",
|
"SGLANG_OPT_USE_JIT_INDEXER_METADATA": "false",
|
||||||
"SGLANG_OPT_USE_TOPK_V2": "false",
|
"SGLANG_OPT_USE_TOPK_V2": "false",
|
||||||
@@ -128,7 +129,7 @@ class TestDeepseekV4ProFp8(CustomTestCase):
|
|||||||
|
|
||||||
if is_in_ci():
|
if is_in_ci():
|
||||||
write_github_step_summary(
|
write_github_step_summary(
|
||||||
f"### test_gsm8k (deepseek-v4-pro-fp8)\n"
|
f"### test_gsm8k (deepseek-v4-pro-fp8, {FLASHMLA_BACKEND})\n"
|
||||||
f'{metrics["accuracy"]=:.3f}\n'
|
f'{metrics["accuracy"]=:.3f}\n'
|
||||||
)
|
)
|
||||||
self.assertGreater(metrics["accuracy"], 0.91)
|
self.assertGreater(metrics["accuracy"], 0.91)
|
||||||
@@ -187,7 +188,7 @@ class TestDeepseekV4ProFp8(CustomTestCase):
|
|||||||
report_results = results_data
|
report_results = results_data
|
||||||
|
|
||||||
summary_lines = [
|
summary_lines = [
|
||||||
"### test_perf_8k_1k (deepseek-v4-pro-fp8)",
|
f"### test_perf_8k_1k (deepseek-v4-pro-fp8, {FLASHMLA_BACKEND})",
|
||||||
"input_len=8192 output_len=1024",
|
"input_len=8192 output_len=1024",
|
||||||
"",
|
"",
|
||||||
"| batch size | latency (s) | input throughput (tok/s) | output throughput (tok/s) | ITL (ms) |",
|
"| batch size | latency (s) | input throughput (tok/s) | output throughput (tok/s) | ITL (ms) |",
|
||||||
|
|||||||
Reference in New Issue
Block a user