[Test] Replace NVFP4 MoE runner backend e2e matrix with a layer-level unit test (#33611)

This commit is contained in:
Liangsheng Yin
2026-08-04 16:03:29 -07:00
committed by GitHub
parent 0d99d91e49
commit 76dc89f5aa
4 changed files with 233 additions and 234 deletions
@@ -1,18 +1,7 @@
"""Backend tests for CuteDSL MoE (FusedMoE + moe_runner, moe_a2a=none).
Exercises the CuteDSL moe_runner path with ModelOpt FP4 by launching a
server with --moe-runner-backend flashinfer_cutedsl.
Two configurations are tested:
- EP=1, TP=4: each GPU holds all experts with TP-sharded intermediate dim
- EP=4, TP=4: each GPU holds 1/4 of experts at full intermediate width,
partial results combined via all-reduce (no A2A dispatch)
Requires 4 GPUs. Run from repo root with:
python -m pytest test/registered/backends/test_deepseek_v3_fp4_cutedsl_moe.py -v -s
Or via the nightly suite:
python test/run_suite.py --hw cuda --suite nightly-4-gpu-b200 --nightly
"""
"""CuteDSL MoE e2e with EP=TP=4 (moe_a2a=none): each GPU holds 1/4 of the
experts at full intermediate width, partial results combined via all-reduce.
Kept for the distributed-EP dimension; single-GPU backend numerics live in
unit/layers/quantization/test_nvfp4_moe_backends.py."""
import unittest
from types import SimpleNamespace
@@ -28,68 +17,13 @@ from sglang.test.test_utils import (
write_github_step_summary,
)
register_cuda_ci(est_time=900, suite="nightly-4-gpu-b200", nightly=True)
register_cuda_ci(est_time=450, suite="nightly-4-gpu-b200", nightly=True)
FULL_DEEPSEEK_V3_FP4_MODEL_PATH = "nvidia/DeepSeek-V3-0324-FP4"
SERVER_LAUNCH_TIMEOUT = 1000
GSM8K_ACCURACY_THRESHOLD = 0.935
class TestDeepseekV3FP4CuteDSLMoE(CustomTestCase):
"""CuteDSL standard moe_runner path: flashinfer_cutedsl + modelopt_fp4, EP=1."""
@classmethod
def setUpClass(cls):
cls.model = FULL_DEEPSEEK_V3_FP4_MODEL_PATH
cls.base_url = DEFAULT_URL_FOR_TEST
other_args = [
"--tp",
"4",
"--ep",
"1",
"--mem-fraction-static",
"0.75",
"--attention-backend",
"trtllm_mla",
"--moe-runner-backend",
"flashinfer_cutedsl",
"--quantization",
"modelopt_fp4",
"--model-loader-extra-config",
'{"enable_multithread_load": true}',
]
cls.process = popen_launch_server(
cls.model,
cls.base_url,
timeout=SERVER_LAUNCH_TIMEOUT,
other_args=other_args,
)
@classmethod
def tearDownClass(cls):
kill_process_tree(cls.process.pid)
def test_a_gsm8k(
self,
): # Append an "a" to make this test run first (alphabetically) to warm up the server
args = SimpleNamespace(
num_shots=8,
data_path=None,
num_questions=1319,
parallel=1319,
max_new_tokens=512,
host="http://127.0.0.1",
port=int(self.base_url.split(":")[-1]),
)
metrics = run_eval_few_shot_gsm8k(args)
if is_in_ci():
write_github_step_summary(
f"### test_gsm8k (deepseek-v3-fp4-cutedsl-moe)\n"
f'{metrics["accuracy"]=:.3f}\n'
)
self.assertGreater(metrics["accuracy"], GSM8K_ACCURACY_THRESHOLD)
class TestDeepseekV3FP4CuteDSLMoEEP4(CustomTestCase):
"""CuteDSL standard moe_runner path: flashinfer_cutedsl + modelopt_fp4, EP=TP=4."""
@@ -1,76 +0,0 @@
import unittest
from types import SimpleNamespace
from sglang.srt.utils import kill_process_tree
from sglang.test.ci.ci_register import register_cuda_ci
from sglang.test.run_eval import run_eval
from sglang.test.test_utils import (
DEFAULT_URL_FOR_TEST,
CustomTestCase,
is_in_ci,
popen_launch_server,
write_github_step_summary,
)
register_cuda_ci(est_time=900, suite="nightly-4-gpu-b200", nightly=True)
FULL_DEEPSEEK_V3_FP4_MODEL_PATH = "nvidia/DeepSeek-V3-0324-FP4"
SERVER_LAUNCH_TIMEOUT = 1000
class TestDeepseekV3FP4CutlassMoE(CustomTestCase):
@classmethod
def setUpClass(cls):
cls.model = FULL_DEEPSEEK_V3_FP4_MODEL_PATH
cls.base_url = DEFAULT_URL_FOR_TEST
other_args = [
"--tp",
"4",
"--ep",
"4",
"--attention-backend",
"trtllm_mla",
"--moe-runner-backend",
"flashinfer_cutlass",
"--quantization",
"modelopt_fp4",
"--model-loader-extra-config",
'{"enable_multithread_load": true}',
]
cls.process = popen_launch_server(
cls.model,
cls.base_url,
timeout=SERVER_LAUNCH_TIMEOUT,
other_args=other_args,
)
@classmethod
def tearDownClass(cls):
kill_process_tree(cls.process.pid)
def test_a_gsm8k(
self,
): # Append an "a" to make this test run first (alphabetically) to warm up the server
args = SimpleNamespace(
base_url=self.base_url,
model=self.model,
eval_name="gsm8k",
api="completion",
max_tokens=512,
num_examples=1319,
num_threads=1319,
num_shots=8,
)
metrics = run_eval(args)
print(f"{metrics=}")
if is_in_ci():
write_github_step_summary(
f"### test_gsm8k (deepseek-v3-fp4-cutlass-moe)\n"
f'{metrics["score"]=:.3f}\n'
)
self.assertGreater(metrics["score"], 0.935)
if __name__ == "__main__":
unittest.main()