[AMD] [CI] Remove hardcoded model/cache paths from MI35x nightly tests (#27001)

This commit is contained in:
Bingxu Chen
2026-06-03 02:14:32 -07:00
committed by GitHub
parent e67810bea7
commit d7013b6537
27 changed files with 11 additions and 471 deletions
@@ -8,11 +8,6 @@ Registry: nightly-amd-accuracy-8-gpu-mi35x-deepseek-r1 suite
import ast
import os
# Set HF cache for MI35x
os.environ.setdefault("HF_HOME", "/data2/models/huggingface")
os.environ.setdefault("HF_HUB_CACHE", "/data2/models/huggingface/hub")
import re
import time
import unittest
@@ -8,11 +8,6 @@ Registry: nightly-amd-8-gpu-mi35x-deepseek-r1-mxfp4-ar-fusion suite
import ast
import os
# Set HF cache for MI35x
os.environ.setdefault("HF_HOME", "/data2/models/huggingface")
os.environ.setdefault("HF_HUB_CACHE", "/data2/models/huggingface/hub")
import re
import time
import unittest
@@ -41,21 +36,6 @@ register_amd_ci(
INVALID = -9999999
# Model path configuration for MI35x DeepSeek-R1-MXFP4
# Priority: 1) env var, 2) local path, 3) HuggingFace model ID
DEEPSEEK_R1_MXFP4_LOCAL_PATH = "/data2/models/amd-DeepSeek-R1-MXFP4-Preview"
DEEPSEEK_R1_MXFP4_HF_MODEL_ID = "amd/DeepSeek-R1-MXFP4-Preview"
def get_model_path() -> str:
"""Get effective model path: env var > local path > HF model ID."""
env_path = os.environ.get("DEEPSEEK_R1_MXFP4_MODEL_PATH")
if env_path:
return env_path
if os.path.exists(DEEPSEEK_R1_MXFP4_LOCAL_PATH):
return DEEPSEEK_R1_MXFP4_LOCAL_PATH
return DEEPSEEK_R1_MXFP4_HF_MODEL_ID
@dataclass
class ModelConfig:
@@ -83,10 +63,9 @@ class ModelConfig:
def get_mxfp4_models() -> List[ModelConfig]:
"""Get DeepSeek-R1-MXFP4 model configurations for MI35x with AllReduce Fusion."""
model_path = get_model_path()
return [
ModelConfig(
model_path=model_path,
model_path="amd/DeepSeek-R1-MXFP4-Preview",
tp_size=8,
accuracy_threshold=0.93,
timeout=3600,
@@ -193,19 +172,6 @@ class TestDeepSeekR1MXFP4ArFusionEvalMI35x(unittest.TestCase):
def test_deepseek_r1_mxfp4_ar_fusion_accuracy(self):
"""Test DeepSeek-R1-MXFP4 models with AllReduce Fusion on GSM8K."""
# Check if model exists
model_path = get_model_path()
is_local_path = model_path.startswith("/")
if is_local_path and not os.path.exists(model_path):
print(f"\n⏭️ SKIPPING: Local model not found at {model_path}")
self.skipTest(f"Local model not found at {model_path}")
return
if is_local_path:
print(f"📁 Using local model: {model_path}")
else:
print(f"📥 Using HuggingFace model: {model_path}")
all_results = []
summary = "### DeepSeek-R1-MXFP4 AllReduce Fusion Models (MI35x)\n\n"
summary += "| Model | Variant | TP | Accuracy | Threshold | Status |\n"
@@ -8,11 +8,6 @@ Registry: nightly-amd-8-gpu-mi35x-deepseek-r1-mxfp4 suite
import ast
import os
# Set HF cache for MI35x
os.environ.setdefault("HF_HOME", "/data2/models/huggingface")
os.environ.setdefault("HF_HUB_CACHE", "/data2/models/huggingface/hub")
import re
import time
import unittest
@@ -39,21 +34,6 @@ register_amd_ci(
INVALID = -9999999
# Model path configuration for MI35x DeepSeek-R1-MXFP4
# Priority: 1) env var, 2) local path, 3) HuggingFace model ID
DEEPSEEK_R1_MXFP4_LOCAL_PATH = "/data2/models/amd-DeepSeek-R1-MXFP4-Preview"
DEEPSEEK_R1_MXFP4_HF_MODEL_ID = "amd/DeepSeek-R1-MXFP4-Preview"
def get_model_path() -> str:
"""Get effective model path: env var > local path > HF model ID."""
env_path = os.environ.get("DEEPSEEK_R1_MXFP4_MODEL_PATH")
if env_path:
return env_path
if os.path.exists(DEEPSEEK_R1_MXFP4_LOCAL_PATH):
return DEEPSEEK_R1_MXFP4_LOCAL_PATH
return DEEPSEEK_R1_MXFP4_HF_MODEL_ID
@dataclass
class ModelConfig:
@@ -81,11 +61,10 @@ class ModelConfig:
def get_mxfp4_models() -> List[ModelConfig]:
"""Get DeepSeek-R1-MXFP4 model configurations for MI35x."""
model_path = get_model_path()
return [
# DeepSeek-R1-MXFP4 basic only (MTP tested in perf job)
ModelConfig(
model_path=model_path,
model_path="amd/DeepSeek-R1-MXFP4-Preview",
tp_size=8,
accuracy_threshold=0.93,
timeout=3600,
@@ -191,19 +170,6 @@ class TestDeepSeekR1MXFP4EvalMI35x(unittest.TestCase):
def test_deepseek_r1_mxfp4_accuracy(self):
"""Test DeepSeek-R1-MXFP4 models with GSM8K completion benchmark."""
# Check if model exists
model_path = get_model_path()
is_local_path = model_path.startswith("/")
if is_local_path and not os.path.exists(model_path):
print(f"\n⏭️ SKIPPING: Local model not found at {model_path}")
self.skipTest(f"Local model not found at {model_path}")
return
if is_local_path:
print(f"📁 Using local model: {model_path}")
else:
print(f"📥 Using HuggingFace model: {model_path}")
all_results = []
summary = "### DeepSeek-R1-MXFP4 Models (MI35x)\n\n"
summary += "| Model | Variant | TP | Accuracy | Threshold | Status |\n"
@@ -8,11 +8,6 @@ Registry: nightly-amd-8-gpu-mi35x-deepseek-r1-mxfp4-kv-fp8 suite
import ast
import os
# Set HF cache for MI35x
os.environ.setdefault("HF_HOME", "/data2/models/huggingface")
os.environ.setdefault("HF_HUB_CACHE", "/data2/models/huggingface/hub")
import re
import time
import unittest
@@ -41,21 +36,6 @@ register_amd_ci(
INVALID = -9999999
# Model path configuration for MI35x DeepSeek-R1-MXFP4
# Priority: 1) env var, 2) local path, 3) HuggingFace model ID
DEEPSEEK_R1_MXFP4_LOCAL_PATH = "/data2/models/amd-DeepSeek-R1-MXFP4-Preview"
DEEPSEEK_R1_MXFP4_HF_MODEL_ID = "amd/DeepSeek-R1-MXFP4-Preview"
def get_model_path() -> str:
"""Get effective model path: env var > local path > HF model ID."""
env_path = os.environ.get("DEEPSEEK_R1_MXFP4_MODEL_PATH")
if env_path:
return env_path
if os.path.exists(DEEPSEEK_R1_MXFP4_LOCAL_PATH):
return DEEPSEEK_R1_MXFP4_LOCAL_PATH
return DEEPSEEK_R1_MXFP4_HF_MODEL_ID
@dataclass
class ModelConfig:
@@ -83,10 +63,9 @@ class ModelConfig:
def get_mxfp4_models() -> List[ModelConfig]:
"""Get DeepSeek-R1-MXFP4 model configurations for MI35x with KV cache FP8."""
model_path = get_model_path()
return [
ModelConfig(
model_path=model_path,
model_path="amd/DeepSeek-R1-MXFP4-Preview",
tp_size=8,
accuracy_threshold=0.93,
timeout=3600,
@@ -194,19 +173,6 @@ class TestDeepSeekR1MXFP4KvFp8EvalMI35x(unittest.TestCase):
def test_deepseek_r1_mxfp4_kv_fp8_accuracy(self):
"""Test DeepSeek-R1-MXFP4 models with KV cache FP8 on GSM8K."""
# Check if model exists
model_path = get_model_path()
is_local_path = model_path.startswith("/")
if is_local_path and not os.path.exists(model_path):
print(f"\n⏭️ SKIPPING: Local model not found at {model_path}")
self.skipTest(f"Local model not found at {model_path}")
return
if is_local_path:
print(f"📁 Using local model: {model_path}")
else:
print(f"📥 Using HuggingFace model: {model_path}")
all_results = []
summary = "### DeepSeek-R1-MXFP4 KV FP8 Models (MI35x)\n\n"
summary += "| Model | Variant | TP | Accuracy | Threshold | Status |\n"
@@ -6,12 +6,6 @@ completion benchmark on MI35x.
Registry: nightly-amd-accuracy-8-gpu-mi35x-deepseek-v32-dp suite
"""
import os
# Set HF cache for MI35x
os.environ.setdefault("HF_HOME", "/data2/models/huggingface")
os.environ.setdefault("HF_HUB_CACHE", "/data2/models/huggingface/hub")
import unittest
from types import SimpleNamespace
@@ -8,11 +8,6 @@ Registry: nightly-amd-accuracy-8-gpu-mi35x-deepseek-v32 suite
import ast
import os
# Set HF cache for MI35x
os.environ.setdefault("HF_HOME", "/data2/models/huggingface")
os.environ.setdefault("HF_HUB_CACHE", "/data2/models/huggingface/hub")
import re
import time
import unittest
@@ -6,12 +6,6 @@ completion benchmark on MI35x.
Registry: nightly-amd-accuracy-8-gpu-mi35x-deepseek-v32-mtp suite
"""
import os
# Set HF cache for MI35x
os.environ.setdefault("HF_HOME", "/data2/models/huggingface")
os.environ.setdefault("HF_HUB_CACHE", "/data2/models/huggingface/hub")
import unittest
from types import SimpleNamespace
@@ -5,12 +5,6 @@ Tests GLM-4.7-FP8 accuracy using GSM8K benchmark on MI35x.
Registry: nightly-amd-8-gpu-mi35x-glm47-fp8 suite
"""
import os
# Set HF cache for MI35x
os.environ.setdefault("HF_HOME", "/data2/models/huggingface")
os.environ.setdefault("HF_HUB_CACHE", "/data2/models/huggingface/hub")
import unittest
from sglang.test.accuracy_test_runner import AccuracyTestParams
@@ -8,10 +8,6 @@ Registry: nightly-amd-8-gpu-mi35x-glm51 suite
import ast
import os
os.environ.setdefault("HF_HOME", "/data2/models/huggingface")
os.environ.setdefault("HF_HUB_CACHE", "/data2/models/huggingface/hub")
import re
import time
import unittest
@@ -8,11 +8,6 @@ Registry: nightly-amd-8-gpu-mi35x-glm5 suite
import ast
import os
# Set HF cache for MI35x
os.environ.setdefault("HF_HOME", "/data2/models/huggingface")
os.environ.setdefault("HF_HUB_CACHE", "/data2/models/huggingface/hub")
import re
import time
import unittest
@@ -11,10 +11,6 @@ Registry: nightly-amd-8-gpu-mi35x-glm5-mxfp4 suite
import ast
import os
os.environ.setdefault("HF_HOME", "/data2/models/huggingface")
os.environ.setdefault("HF_HUB_CACHE", "/data2/models/huggingface/hub")
import re
import time
import unittest
@@ -42,19 +38,6 @@ register_amd_ci(
INVALID = -9999999
GLM5_MXFP4_LOCAL_PATH = "/data2/models/amd-GLM-5-MXFP4"
GLM5_MXFP4_HF_MODEL_ID = "amd/GLM-5-MXFP4"
def get_model_path() -> str:
"""Get effective model path: env var > local path > HF model ID."""
env_path = os.environ.get("GLM5_MXFP4_MODEL_PATH")
if env_path:
return env_path
if os.path.exists(GLM5_MXFP4_LOCAL_PATH):
return GLM5_MXFP4_LOCAL_PATH
return GLM5_MXFP4_HF_MODEL_ID
@dataclass
class ModelConfig:
@@ -82,10 +65,9 @@ class ModelConfig:
def get_glm5_mxfp4_models() -> List[ModelConfig]:
"""Get GLM-5-MXFP4 model configurations for MI35x."""
model_path = get_model_path()
return [
ModelConfig(
model_path=model_path,
model_path="amd/GLM-5-MXFP4",
tp_size=8,
accuracy_threshold=0.90,
timeout=5400,
@@ -195,18 +177,6 @@ class TestGLM5MXFP4EvalMI35x(unittest.TestCase):
def test_glm5_mxfp4_accuracy(self):
"""Test GLM-5-MXFP4 with GSM8K completion benchmark."""
model_path = get_model_path()
is_local_path = model_path.startswith("/")
if is_local_path and not os.path.exists(model_path):
print(f"\nSKIPPING: Local model not found at {model_path}")
self.skipTest(f"Local model not found at {model_path}")
return
if is_local_path:
print(f"Using local model: {model_path}")
else:
print(f"Using HuggingFace model: {model_path}")
all_results = []
summary = "### GLM-5-MXFP4 Models (MI35x)\n\n"
summary += "| Model | Variant | TP | Accuracy | Threshold | Status |\n"
@@ -10,11 +10,6 @@ Registry: nightly-amd-8-gpu-mi35x suite
import ast
import os
# Set HF cache for MI35x
os.environ.setdefault("HF_HOME", "/data2/models/huggingface")
os.environ.setdefault("HF_HUB_CACHE", "/data2/models/huggingface/hub")
import re
import time
import unittest
@@ -18,10 +18,6 @@ Registry: nightly-amd-8-gpu-mi35x-kimi-k25-aiter-mla suite
"""
import os
os.environ.setdefault("HF_HOME", "/data2/models/huggingface")
os.environ.setdefault("HF_HUB_CACHE", "/data2/models/huggingface/hub")
import unittest
from dataclasses import dataclass
from typing import List, Optional
@@ -40,19 +36,6 @@ register_amd_ci(
est_time=7200, suite="nightly-amd-8-gpu-mi35x-kimi-k25-aiter-mla", nightly=True
)
KIMI_K25_LOCAL_PATH = "/data/models/amd/Kimi-K2.5"
KIMI_K25_HF_MODEL_ID = "moonshotai/Kimi-K2.5"
def get_model_path() -> str:
"""Get effective model path: env var > local path > HF model ID."""
env_path = os.environ.get("KIMI_K25_MODEL_PATH")
if env_path:
return env_path
if os.path.exists(KIMI_K25_LOCAL_PATH):
return KIMI_K25_LOCAL_PATH
return KIMI_K25_HF_MODEL_ID
@dataclass
class ModelConfig:
@@ -80,9 +63,8 @@ class ModelConfig:
def get_kimi_k25_models() -> List[ModelConfig]:
"""Get Kimi-K2.5 model configurations for MI35x."""
model_path = get_model_path()
common_kwargs = {
"model_path": model_path,
"model_path": "moonshotai/Kimi-K2.5",
# TP=4 required: Kimi-K2.5 has 64 attn heads; aiter ASM MLA needs
# heads_per_gpu % 16 == 0 → 64/4=16 works, 64/8=8 does not.
"tp_size": 4,
@@ -138,18 +120,6 @@ class TestKimiK25AiterMlaEvalMI35x(unittest.TestCase):
def test_kimi_k25_accuracy(self):
"""Test Kimi-K2.5 with GSM8K completion benchmark (default & fp8kv)."""
model_path = get_model_path()
is_local_path = model_path.startswith("/")
if is_local_path and not os.path.exists(model_path):
print(f"\nSKIPPING: Local model not found at {model_path}")
self.skipTest(f"Local model not found at {model_path}")
return
if is_local_path:
print(f"Using local model: {model_path}")
else:
print(f"Using HuggingFace model: {model_path}")
from types import SimpleNamespace
from sglang.test.few_shot_gsm8k import run_eval as run_eval_few_shot_gsm8k
@@ -11,10 +11,6 @@ Registry: nightly-amd-8-gpu-mi35x-kimi-k25-mxfp4-aiter-mla suite
"""
import os
os.environ.setdefault("HF_HOME", "/data2/models/huggingface")
os.environ.setdefault("HF_HUB_CACHE", "/data2/models/huggingface/hub")
import unittest
from dataclasses import dataclass
from typing import List, Optional
@@ -35,19 +31,6 @@ register_amd_ci(
nightly=True,
)
KIMI_K25_MXFP4_LOCAL_PATH = "/data/models/amd/Kimi-K2.5-MXFP4"
KIMI_K25_MXFP4_HF_MODEL_ID = "moonshotai/Kimi-K2.5-MXFP4"
def get_model_path() -> str:
"""Get effective model path: env var > local path > HF model ID."""
env_path = os.environ.get("KIMI_K25_MXFP4_MODEL_PATH")
if env_path:
return env_path
if os.path.exists(KIMI_K25_MXFP4_LOCAL_PATH):
return KIMI_K25_MXFP4_LOCAL_PATH
return KIMI_K25_MXFP4_HF_MODEL_ID
@dataclass
class ModelConfig:
@@ -75,9 +58,8 @@ class ModelConfig:
def get_kimi_k25_mxfp4_models() -> List[ModelConfig]:
"""Get Kimi-K2.5-MXFP4 model configurations for MI35x."""
model_path = get_model_path()
common_kwargs = {
"model_path": model_path,
"model_path": "moonshotai/Kimi-K2.5-MXFP4",
"tp_size": 8,
"accuracy_threshold": 0.92,
"timeout": 3600,
@@ -131,18 +113,6 @@ class TestKimiK25MXFP4AiterMlaEvalMI35x(unittest.TestCase):
def test_kimi_k25_mxfp4_accuracy(self):
"""Test Kimi-K2.5-MXFP4 with GSM8K completion benchmark (default & fp8kv)."""
model_path = get_model_path()
is_local_path = model_path.startswith("/")
if is_local_path and not os.path.exists(model_path):
print(f"\nSKIPPING: Local model not found at {model_path}")
self.skipTest(f"Local model not found at {model_path}")
return
if is_local_path:
print(f"Using local model: {model_path}")
else:
print(f"Using HuggingFace model: {model_path}")
from types import SimpleNamespace
from sglang.test.few_shot_gsm8k import run_eval as run_eval_few_shot_gsm8k
@@ -8,10 +8,6 @@ Registry: nightly-amd-8-gpu-mi35x-minimax-m25 suite
import ast
import os
os.environ.setdefault("HF_HOME", "/data2/models/huggingface")
os.environ.setdefault("HF_HUB_CACHE", "/data2/models/huggingface/hub")
import re
import time
import unittest
@@ -8,10 +8,6 @@ Registry: nightly-amd-8-gpu-mi35x-minimax-m27 suite
import ast
import os
os.environ.setdefault("HF_HOME", "/data2/models/huggingface")
os.environ.setdefault("HF_HUB_CACHE", "/data2/models/huggingface/hub")
import re
import time
import unittest
@@ -8,11 +8,6 @@ Registry: nightly-amd-8-gpu-mi35x-qwen3-coder-next suite
import ast
import os
# Set HF cache for MI35x
os.environ.setdefault("HF_HOME", "/data2/models/huggingface")
os.environ.setdefault("HF_HUB_CACHE", "/data2/models/huggingface/hub")
import re
import time
import unittest
@@ -37,21 +32,6 @@ register_amd_ci(est_time=3600, suite="nightly-amd-8-gpu-mi35x", nightly=True)
INVALID = -9999999
# Model path configuration for MI35x Qwen3-Coder-Next
# Priority: 1) env var, 2) local path
QWEN3_CODER_NEXT_LOCAL_PATH = "/data/Qwen/Qwen3-Coder-Next/"
QWEN3_CODER_NEXT_HF_MODEL_ID = "Qwen/Qwen3-Coder-Next"
def get_model_path() -> str:
"""Get effective model path: env var > local path > HF model ID."""
env_path = os.environ.get("QWEN3_CODER_NEXT_MODEL_PATH")
if env_path:
return env_path
if os.path.exists(QWEN3_CODER_NEXT_LOCAL_PATH):
return QWEN3_CODER_NEXT_LOCAL_PATH
return QWEN3_CODER_NEXT_HF_MODEL_ID
@dataclass
class ModelConfig:
@@ -79,9 +59,8 @@ class ModelConfig:
def get_qwen3_coder_next_models() -> List[ModelConfig]:
"""Get Qwen3-Coder-Next model configurations for MI35x."""
model_path = get_model_path()
common_kwargs = {
"model_path": model_path,
"model_path": "Qwen/Qwen3-Coder-Next",
"tp_size": 8,
"accuracy_threshold": 0.90,
"timeout": 3600,
@@ -215,19 +194,6 @@ class TestQwen3CoderNextEvalMI35x(unittest.TestCase):
def test_qwen3_coder_next_accuracy(self):
"""Test Qwen3-Coder-Next models with GSM8K completion benchmark."""
# Check if model exists
model_path = get_model_path()
is_local_path = model_path.startswith("/")
if is_local_path and not os.path.exists(model_path):
print(f"\nSKIPPING: Local model not found at {model_path}")
self.skipTest(f"Local model not found at {model_path}")
return
if is_local_path:
print(f"Using local model: {model_path}")
else:
print(f"Using HuggingFace model: {model_path}")
all_results = []
summary = "### Qwen3-Coder-Next Models (MI35x)\n\n"
summary += "| Model | Variant | TP | Accuracy | Threshold | Status |\n"
@@ -3,19 +3,10 @@
This test benchmarks the DeepSeek-R1-MXFP4 quantized model on MI35x with 8 GPUs
using --enable-aiter-allreduce-fusion.
The model path can be configured via DEEPSEEK_R1_MXFP4_MODEL_PATH environment variable.
Registry: nightly-perf-8-gpu-mi35x-deepseek-r1-mxfp4-ar-fusion suite
Example usage:
DEEPSEEK_R1_MXFP4_MODEL_PATH=/data2/models/amd-DeepSeek-R1-MXFP4-Preview python -m pytest test_deepseek_r1_mxfp4_ar_fusion_perf_mi35x.py -v
"""
import os
# Set HF cache to /data2/models/ for MI35x so HF models download there
os.environ.setdefault("HF_HOME", "/data2/models/huggingface")
os.environ.setdefault("HF_HUB_CACHE", "/data2/models/huggingface/hub")
import unittest
from typing import List
@@ -63,26 +54,9 @@ def generate_simple_markdown_report(results: List[BenchmarkResult]) -> str:
return summary
# Model path configuration for MI35x DeepSeek-R1-MXFP4
# Priority: 1) env var, 2) local path, 3) HuggingFace model ID
DEEPSEEK_R1_MXFP4_LOCAL_PATH = "/data2/models/amd-DeepSeek-R1-MXFP4-Preview"
DEEPSEEK_R1_MXFP4_HF_MODEL_ID = "amd/DeepSeek-R1-MXFP4-Preview"
PROFILE_DIR = "performance_profiles_deepseek_r1_mxfp4_ar_fusion_mi35x"
def get_model_path() -> str:
"""Get effective model path: env var > local path > HF model ID."""
# Check env var first
env_path = os.environ.get("DEEPSEEK_R1_MXFP4_MODEL_PATH")
if env_path:
return env_path
# Check local path
if os.path.exists(DEEPSEEK_R1_MXFP4_LOCAL_PATH):
return DEEPSEEK_R1_MXFP4_LOCAL_PATH
# Fall back to HF model ID
return DEEPSEEK_R1_MXFP4_HF_MODEL_ID
class TestDeepseekR1MXFP4ArFusionPerfMI35x(unittest.TestCase):
"""MI35x Nightly performance benchmark for DeepSeek-R1-MXFP4 with AllReduce Fusion.
@@ -92,7 +66,7 @@ class TestDeepseekR1MXFP4ArFusionPerfMI35x(unittest.TestCase):
@classmethod
def setUpClass(cls):
cls.model = get_model_path()
cls.model = "amd/DeepSeek-R1-MXFP4-Preview"
print(f"Using model path: {cls.model}")
cls.base_url = DEFAULT_URL_FOR_TEST
cls.batch_sizes = [1, 8, 16, 64]
@@ -124,22 +98,6 @@ class TestDeepseekR1MXFP4ArFusionPerfMI35x(unittest.TestCase):
"""Run benchmark across all configured variants."""
failed_variants = []
is_local_path = self.model.startswith("/")
if is_local_path and not os.path.exists(self.model):
print(f"\n⏭️ SKIPPING: Local model not found at {self.model}")
self.runner.full_report += (
f"\n⏭️ Test skipped: Local model not found at {self.model}\n"
)
self.runner.write_final_report()
return
if is_local_path:
print(f"📁 Using local model: {self.model}")
else:
print(
f"📥 Using HuggingFace model: {self.model} (will download if not cached)"
)
try:
for variant_config in self.variants:
with self.subTest(variant=variant_config["name"]):
@@ -3,19 +3,10 @@
This test benchmarks the DeepSeek-R1-MXFP4 quantized model on MI35x with 8 GPUs
using --kv-cache-dtype fp8_e4m3.
The model path can be configured via DEEPSEEK_R1_MXFP4_MODEL_PATH environment variable.
Registry: nightly-perf-8-gpu-mi35x-deepseek-r1-mxfp4-kv-fp8 suite
Example usage:
DEEPSEEK_R1_MXFP4_MODEL_PATH=/data2/models/amd-DeepSeek-R1-MXFP4-Preview python -m pytest test_deepseek_r1_mxfp4_kv_fp8_perf_mi35x.py -v
"""
import os
# Set HF cache to /data2/models/ for MI35x so HF models download there
os.environ.setdefault("HF_HOME", "/data2/models/huggingface")
os.environ.setdefault("HF_HUB_CACHE", "/data2/models/huggingface/hub")
import unittest
from typing import List
@@ -63,26 +54,9 @@ def generate_simple_markdown_report(results: List[BenchmarkResult]) -> str:
return summary
# Model path configuration for MI35x DeepSeek-R1-MXFP4
# Priority: 1) env var, 2) local path, 3) HuggingFace model ID
DEEPSEEK_R1_MXFP4_LOCAL_PATH = "/data2/models/amd-DeepSeek-R1-MXFP4-Preview"
DEEPSEEK_R1_MXFP4_HF_MODEL_ID = "amd/DeepSeek-R1-MXFP4-Preview"
PROFILE_DIR = "performance_profiles_deepseek_r1_mxfp4_kv_fp8_mi35x"
def get_model_path() -> str:
"""Get effective model path: env var > local path > HF model ID."""
# Check env var first
env_path = os.environ.get("DEEPSEEK_R1_MXFP4_MODEL_PATH")
if env_path:
return env_path
# Check local path
if os.path.exists(DEEPSEEK_R1_MXFP4_LOCAL_PATH):
return DEEPSEEK_R1_MXFP4_LOCAL_PATH
# Fall back to HF model ID
return DEEPSEEK_R1_MXFP4_HF_MODEL_ID
class TestDeepseekR1MXFP4KvFp8PerfMI35x(unittest.TestCase):
"""MI35x Nightly performance benchmark for DeepSeek-R1-MXFP4 with KV Cache FP8.
@@ -92,7 +66,7 @@ class TestDeepseekR1MXFP4KvFp8PerfMI35x(unittest.TestCase):
@classmethod
def setUpClass(cls):
cls.model = get_model_path()
cls.model = "amd/DeepSeek-R1-MXFP4-Preview"
print(f"Using model path: {cls.model}")
cls.base_url = DEFAULT_URL_FOR_TEST
cls.batch_sizes = [1, 8, 16, 64]
@@ -125,22 +99,6 @@ class TestDeepseekR1MXFP4KvFp8PerfMI35x(unittest.TestCase):
"""Run benchmark across all configured variants."""
failed_variants = []
is_local_path = self.model.startswith("/")
if is_local_path and not os.path.exists(self.model):
print(f"\n⏭️ SKIPPING: Local model not found at {self.model}")
self.runner.full_report += (
f"\n⏭️ Test skipped: Local model not found at {self.model}\n"
)
self.runner.write_final_report()
return
if is_local_path:
print(f"📁 Using local model: {self.model}")
else:
print(
f"📥 Using HuggingFace model: {self.model} (will download if not cached)"
)
try:
for variant_config in self.variants:
with self.subTest(variant=variant_config["name"]):
@@ -2,19 +2,10 @@
This test benchmarks the DeepSeek-R1-MXFP4 quantized model on MI35x with 8 GPUs.
The model path can be configured via DEEPSEEK_R1_MXFP4_MODEL_PATH environment variable.
Registry: nightly-perf-8-gpu-mi35x-deepseek-r1-mxfp4 suite
Example usage:
DEEPSEEK_R1_MXFP4_MODEL_PATH=/data2/models/amd-DeepSeek-R1-MXFP4-Preview python -m pytest test_deepseek_r1_mxfp4_perf_mi35x.py -v
"""
import os
# Set HF cache to /data2/models/ for MI35x so HF models download there
os.environ.setdefault("HF_HOME", "/data2/models/huggingface")
os.environ.setdefault("HF_HUB_CACHE", "/data2/models/huggingface/hub")
import unittest
from typing import List
@@ -60,26 +51,9 @@ def generate_simple_markdown_report(results: List[BenchmarkResult]) -> str:
return summary
# Model path configuration for MI35x DeepSeek-R1-MXFP4
# Priority: 1) env var, 2) local path, 3) HuggingFace model ID
DEEPSEEK_R1_MXFP4_LOCAL_PATH = "/data2/models/amd-DeepSeek-R1-MXFP4-Preview"
DEEPSEEK_R1_MXFP4_HF_MODEL_ID = "amd/DeepSeek-R1-MXFP4-Preview"
PROFILE_DIR = "performance_profiles_deepseek_r1_mxfp4_mi35x"
def get_model_path() -> str:
"""Get effective model path: env var > local path > HF model ID."""
# Check env var first
env_path = os.environ.get("DEEPSEEK_R1_MXFP4_MODEL_PATH")
if env_path:
return env_path
# Check local path
if os.path.exists(DEEPSEEK_R1_MXFP4_LOCAL_PATH):
return DEEPSEEK_R1_MXFP4_LOCAL_PATH
# Fall back to HF model ID
return DEEPSEEK_R1_MXFP4_HF_MODEL_ID
class TestDeepseekR1MXFP4PerfMI35x(unittest.TestCase):
"""MI35x Nightly performance benchmark for DeepSeek-R1-MXFP4 model.
@@ -89,7 +63,7 @@ class TestDeepseekR1MXFP4PerfMI35x(unittest.TestCase):
@classmethod
def setUpClass(cls):
cls.model = get_model_path()
cls.model = "amd/DeepSeek-R1-MXFP4-Preview"
print(f"Using model path: {cls.model}")
cls.base_url = DEFAULT_URL_FOR_TEST
cls.batch_sizes = [1, 8, 16, 64]
@@ -123,24 +97,6 @@ class TestDeepseekR1MXFP4PerfMI35x(unittest.TestCase):
"""Run benchmark across all configured variants."""
failed_variants = []
# For local paths, check if exists. HF model IDs will download automatically.
is_local_path = self.model.startswith("/")
if is_local_path and not os.path.exists(self.model):
print(f"\n⏭️ SKIPPING: Local model not found at {self.model}")
self.runner.full_report += (
f"\n⏭️ Test skipped: Local model not found at {self.model}\n"
)
self.runner.write_final_report()
return
# Log model source
if is_local_path:
print(f"📁 Using local model: {self.model}")
else:
print(
f"📥 Using HuggingFace model: {self.model} (will download if not cached)"
)
try:
for variant_config in self.variants:
with self.subTest(variant=variant_config["name"]):
@@ -7,10 +7,6 @@ Registry: nightly-perf-8-gpu-mi35x-glm51 suite
"""
import os
os.environ.setdefault("HF_HOME", "/data2/models/huggingface")
os.environ.setdefault("HF_HUB_CACHE", "/data2/models/huggingface/hub")
import unittest
from typing import List
@@ -9,10 +9,6 @@ Registry: nightly-perf-8-gpu-mi35x-glm5-mxfp4 suite
"""
import os
os.environ.setdefault("HF_HOME", "/data2/models/huggingface")
os.environ.setdefault("HF_HUB_CACHE", "/data2/models/huggingface/hub")
import unittest
from typing import List
@@ -62,27 +58,15 @@ def generate_simple_markdown_report(results: List[BenchmarkResult]) -> str:
return summary
GLM5_MXFP4_LOCAL_PATH = "/data2/models/amd-GLM-5-MXFP4"
GLM5_MXFP4_HF_MODEL_ID = "amd/GLM-5-MXFP4"
PROFILE_DIR = "performance_profiles_glm5_mxfp4_mi35x"
def get_model_path() -> str:
"""Get effective model path: env var > local path > HF model ID."""
env_path = os.environ.get("GLM5_MXFP4_MODEL_PATH")
if env_path:
return env_path
if os.path.exists(GLM5_MXFP4_LOCAL_PATH):
return GLM5_MXFP4_LOCAL_PATH
return GLM5_MXFP4_HF_MODEL_ID
class TestGLM5MXFP4PerfMI35x(unittest.TestCase):
"""MI35x Nightly performance benchmark for GLM-5-MXFP4 model."""
@classmethod
def setUpClass(cls):
cls.model = get_model_path()
cls.model = "amd/GLM-5-MXFP4"
print(f"Using model path: {cls.model}")
cls.base_url = DEFAULT_URL_FOR_TEST
cls.batch_sizes = [1, 8, 16, 64]
@@ -123,22 +107,6 @@ class TestGLM5MXFP4PerfMI35x(unittest.TestCase):
"""Run benchmark across all configured variants."""
failed_variants = []
is_local_path = self.model.startswith("/")
if is_local_path and not os.path.exists(self.model):
print(f"\nSKIPPING: Local model not found at {self.model}")
self.runner.full_report += (
f"\nTest skipped: Local model not found at {self.model}\n"
)
self.runner.write_final_report()
return
if is_local_path:
print(f"Using local model: {self.model}")
else:
print(
f"Using HuggingFace model: {self.model} (will download if not cached)"
)
old_env = {}
env_vars = {"SGLANG_USE_AITER": "1"}
for key, value in env_vars.items():
@@ -6,10 +6,6 @@ Registry: nightly-perf-8-gpu-mi35x-glm5 suite
"""
import os
os.environ.setdefault("HF_HOME", "/data2/models/huggingface")
os.environ.setdefault("HF_HUB_CACHE", "/data2/models/huggingface/hub")
import unittest
from typing import List
@@ -15,10 +15,6 @@ Example usage:
"""
import os
os.environ.setdefault("HF_HOME", "/data2/models/huggingface")
os.environ.setdefault("HF_HUB_CACHE", "/data2/models/huggingface/hub")
import unittest
from typing import List
@@ -11,10 +11,6 @@ Example usage:
"""
import os
os.environ.setdefault("HF_HOME", "/data2/models/huggingface")
os.environ.setdefault("HF_HUB_CACHE", "/data2/models/huggingface/hub")
import unittest
from typing import List
@@ -11,10 +11,6 @@ Example usage:
"""
import os
os.environ.setdefault("HF_HOME", "/data2/models/huggingface")
os.environ.setdefault("HF_HUB_CACHE", "/data2/models/huggingface/hub")
import unittest
from typing import List
@@ -7,10 +7,6 @@ Registry: nightly-perf-8-gpu-mi35x-qwen35-fp8 suite
"""
import os
os.environ.setdefault("HF_HOME", "/data2/models/huggingface")
os.environ.setdefault("HF_HUB_CACHE", "/data2/models/huggingface/hub")
import unittest
from typing import List