Feat/add w4a16 moe support to nemotron (#25655)

This commit is contained in:
Shaun Kotek
2026-06-02 22:42:26 -07:00
committed by GitHub
parent 512bfbb1e1
commit b8d7351a74
19 changed files with 999 additions and 61 deletions
@@ -7,6 +7,7 @@ Run with `python3 test/manual/models/test_nvidia_nemotron_3_nano_archived.py`.
import unittest
from sglang.srt.utils import is_sm80_supported, is_sm90_supported
from sglang.test.kits.lm_eval_kit import LMEvalMixin
from sglang.test.server_fixtures.default_fixture import DefaultServerBase
@@ -43,5 +44,27 @@ class TestNvidiaNemotron3Nano30BBF16FlashInfer(LMEvalMixin, DefaultServerBase):
] + NEMOTRON_3_NANO_THINKING_ARGS
@unittest.skip("Skip, test pass locally but compiling takes too long in CI")
@unittest.skipIf(
not (is_sm80_supported() or is_sm90_supported()),
"NVFP4 Marlin fallback test requires CUDA SM8X/SM9X",
)
class TestNvidiaNemotron3Nano30BNVFP4Marlin(LMEvalMixin, DefaultServerBase):
"""Test Nemotron-3-Nano-30B NVFP4 model with the Marlin path."""
model = "nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-NVFP4"
model_config_name = "lm_eval_configs/NVIDIA-Nemotron-3-Nano-30B-A3B-NVFP4.yaml"
other_args = [
"--tp-size",
"1",
"--quantization",
"modelopt_fp4",
"--fp4-gemm-backend",
"marlin",
"--moe-runner-backend",
"marlin",
] + NEMOTRON_3_NANO_THINKING_ARGS
if __name__ == "__main__":
unittest.main()