test: extend NVFP4 Marlin tests to SM120 (#34327)
This commit is contained in:
@@ -1,5 +1,4 @@
|
|||||||
import sys
|
import sys
|
||||||
from types import SimpleNamespace
|
|
||||||
|
|
||||||
import pytest
|
import pytest
|
||||||
import torch
|
import torch
|
||||||
@@ -7,19 +6,11 @@ from sgl_kernel.scalar_type import scalar_types
|
|||||||
|
|
||||||
from sglang.kernels.ops.quantization.gptq_marlin import gptq_marlin_gemm
|
from sglang.kernels.ops.quantization.gptq_marlin import gptq_marlin_gemm
|
||||||
from sglang.srt.layers.quantization.marlin_utils import (
|
from sglang.srt.layers.quantization.marlin_utils import (
|
||||||
check_marlin_supported,
|
|
||||||
marlin_make_workspace,
|
marlin_make_workspace,
|
||||||
)
|
)
|
||||||
from sglang.srt.layers.quantization.marlin_utils_fp4 import (
|
|
||||||
apply_fp4_marlin_linear,
|
|
||||||
nvfp4_marlin_process_global_scale,
|
|
||||||
prepare_nvfp4_layer_for_marlin,
|
|
||||||
)
|
|
||||||
from sglang.srt.utils.common import is_sm80_supported, is_sm90_supported
|
|
||||||
from sglang.test.ci.ci_register import register_cuda_ci
|
from sglang.test.ci.ci_register import register_cuda_ci
|
||||||
from sglang.test.test_marlin_utils import (
|
from sglang.test.test_marlin_utils import (
|
||||||
awq_marlin_quantize,
|
awq_marlin_quantize,
|
||||||
make_nvfp4_weight_and_ref,
|
|
||||||
marlin_quantize,
|
marlin_quantize,
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -114,78 +105,5 @@ def test_gptq_marlin_gemm(
|
|||||||
assert max_diff < 0.04
|
assert max_diff < 0.04
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.skipif(
|
|
||||||
not (is_sm80_supported() or is_sm90_supported()),
|
|
||||||
reason="NVFP4 Marlin fallback tests require CUDA SM8X/SM9X",
|
|
||||||
)
|
|
||||||
@pytest.mark.parametrize("dtype", [torch.float16, torch.bfloat16])
|
|
||||||
def test_nvfp4_marlin_support_and_scale_transforms_sm80_sm90(dtype):
|
|
||||||
major, minor = torch.cuda.get_device_capability()
|
|
||||||
capability = major * 10 + minor
|
|
||||||
assert check_marlin_supported(
|
|
||||||
scalar_types.float4_e2m1f,
|
|
||||||
group_size=16,
|
|
||||||
has_zp=False,
|
|
||||||
device_capability=capability,
|
|
||||||
)
|
|
||||||
|
|
||||||
global_scale = torch.tensor(1.0, dtype=dtype, device="cuda")
|
|
||||||
actual_global_scale = nvfp4_marlin_process_global_scale(global_scale)
|
|
||||||
assert actual_global_scale.is_cuda
|
|
||||||
assert actual_global_scale.ndim == 1
|
|
||||||
assert actual_global_scale.numel() == 1
|
|
||||||
if dtype == torch.float16:
|
|
||||||
assert actual_global_scale.item() == 128.0
|
|
||||||
else:
|
|
||||||
assert actual_global_scale.item() == 2.0**119
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.skipif(
|
|
||||||
not (is_sm80_supported() or is_sm90_supported()),
|
|
||||||
reason="NVFP4 Marlin dense numeric test requires CUDA SM80, SM86, or SM90",
|
|
||||||
)
|
|
||||||
@pytest.mark.parametrize("dtype", [torch.float16, torch.bfloat16])
|
|
||||||
def test_nvfp4_marlin_dense_matches_dequant_reference(dtype):
|
|
||||||
torch.manual_seed(0)
|
|
||||||
|
|
||||||
size_m = 17
|
|
||||||
size_k = 256
|
|
||||||
size_n = 192
|
|
||||||
group_size = 16
|
|
||||||
|
|
||||||
a_input = torch.randn((size_m, size_k), dtype=dtype, device="cuda") / 10
|
|
||||||
fp4_weight, scales, global_scale, weight_ref = make_nvfp4_weight_and_ref(
|
|
||||||
size_n, size_k, dtype, group_size=group_size
|
|
||||||
)
|
|
||||||
|
|
||||||
layer = torch.nn.Module()
|
|
||||||
layer.quant_config = SimpleNamespace(group_size=group_size)
|
|
||||||
layer.output_size_per_partition = size_n
|
|
||||||
layer.input_size_per_partition = size_k
|
|
||||||
layer.params_dtype = dtype
|
|
||||||
layer.weight = torch.nn.Parameter(fp4_weight, requires_grad=False)
|
|
||||||
layer.weight_scale = torch.nn.Parameter(scales, requires_grad=False)
|
|
||||||
layer.weight_global_scale = torch.nn.Parameter(
|
|
||||||
global_scale.reshape(1), requires_grad=False
|
|
||||||
)
|
|
||||||
prepare_nvfp4_layer_for_marlin(layer)
|
|
||||||
|
|
||||||
output = apply_fp4_marlin_linear(
|
|
||||||
a_input,
|
|
||||||
layer.weight,
|
|
||||||
layer.weight_scale,
|
|
||||||
layer.weight_global_scale,
|
|
||||||
layer.workspace,
|
|
||||||
size_n,
|
|
||||||
size_k,
|
|
||||||
use_fp32_reduce=True,
|
|
||||||
)
|
|
||||||
|
|
||||||
output_ref = torch.matmul(a_input, weight_ref.T)
|
|
||||||
torch.cuda.synchronize()
|
|
||||||
|
|
||||||
torch.testing.assert_close(output, output_ref, rtol=0.04, atol=0.04)
|
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
sys.exit(pytest.main([__file__, "-v", "-s"]))
|
sys.exit(pytest.main([__file__, "-v", "-s"]))
|
||||||
|
|||||||
@@ -0,0 +1,100 @@
|
|||||||
|
import sys
|
||||||
|
from types import SimpleNamespace
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
import torch
|
||||||
|
from sgl_kernel.scalar_type import scalar_types
|
||||||
|
|
||||||
|
from sglang.srt.layers.quantization.marlin_utils import check_marlin_supported
|
||||||
|
from sglang.srt.layers.quantization.marlin_utils_fp4 import (
|
||||||
|
apply_fp4_marlin_linear,
|
||||||
|
nvfp4_marlin_process_global_scale,
|
||||||
|
prepare_nvfp4_layer_for_marlin,
|
||||||
|
)
|
||||||
|
from sglang.srt.utils.common import (
|
||||||
|
is_sm80_supported,
|
||||||
|
is_sm90_supported,
|
||||||
|
is_sm120_supported,
|
||||||
|
)
|
||||||
|
from sglang.test.ci.ci_register import register_cuda_ci
|
||||||
|
from sglang.test.test_marlin_utils import make_nvfp4_weight_and_ref
|
||||||
|
|
||||||
|
register_cuda_ci(est_time=6, stage="base-b", runner_config="1-gpu-large")
|
||||||
|
register_cuda_ci(est_time=6, stage="base-b", runner_config="1-gpu-small")
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.skipif(
|
||||||
|
not (is_sm80_supported() or is_sm90_supported() or is_sm120_supported()),
|
||||||
|
reason="NVFP4 Marlin fallback tests require CUDA SM8X/SM9X/SM120",
|
||||||
|
)
|
||||||
|
@pytest.mark.parametrize("dtype", [torch.float16, torch.bfloat16])
|
||||||
|
def test_nvfp4_marlin_support_and_scale_transforms_sm80_sm90_sm120(dtype):
|
||||||
|
major, minor = torch.cuda.get_device_capability()
|
||||||
|
capability = major * 10 + minor
|
||||||
|
assert check_marlin_supported(
|
||||||
|
scalar_types.float4_e2m1f,
|
||||||
|
group_size=16,
|
||||||
|
has_zp=False,
|
||||||
|
device_capability=capability,
|
||||||
|
)
|
||||||
|
|
||||||
|
global_scale = torch.tensor(1.0, dtype=dtype, device="cuda")
|
||||||
|
actual_global_scale = nvfp4_marlin_process_global_scale(global_scale)
|
||||||
|
assert actual_global_scale.is_cuda
|
||||||
|
assert actual_global_scale.ndim == 1
|
||||||
|
assert actual_global_scale.numel() == 1
|
||||||
|
if dtype == torch.float16:
|
||||||
|
assert actual_global_scale.item() == 128.0
|
||||||
|
else:
|
||||||
|
assert actual_global_scale.item() == 2.0**119
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.skipif(
|
||||||
|
not (is_sm80_supported() or is_sm90_supported() or is_sm120_supported()),
|
||||||
|
reason="NVFP4 Marlin dense numeric test requires CUDA SM80, SM86, SM90 or SM120",
|
||||||
|
)
|
||||||
|
@pytest.mark.parametrize("dtype", [torch.float16, torch.bfloat16])
|
||||||
|
def test_nvfp4_marlin_dense_matches_dequant_reference(dtype):
|
||||||
|
torch.manual_seed(0)
|
||||||
|
|
||||||
|
size_m = 17
|
||||||
|
size_k = 256
|
||||||
|
size_n = 192
|
||||||
|
group_size = 16
|
||||||
|
|
||||||
|
a_input = torch.randn((size_m, size_k), dtype=dtype, device="cuda") / 10
|
||||||
|
fp4_weight, scales, global_scale, weight_ref = make_nvfp4_weight_and_ref(
|
||||||
|
size_n, size_k, dtype, group_size=group_size
|
||||||
|
)
|
||||||
|
|
||||||
|
layer = torch.nn.Module()
|
||||||
|
layer.quant_config = SimpleNamespace(group_size=group_size)
|
||||||
|
layer.output_size_per_partition = size_n
|
||||||
|
layer.input_size_per_partition = size_k
|
||||||
|
layer.params_dtype = dtype
|
||||||
|
layer.weight = torch.nn.Parameter(fp4_weight, requires_grad=False)
|
||||||
|
layer.weight_scale = torch.nn.Parameter(scales, requires_grad=False)
|
||||||
|
layer.weight_global_scale = torch.nn.Parameter(
|
||||||
|
global_scale.reshape(1), requires_grad=False
|
||||||
|
)
|
||||||
|
prepare_nvfp4_layer_for_marlin(layer)
|
||||||
|
|
||||||
|
output = apply_fp4_marlin_linear(
|
||||||
|
a_input,
|
||||||
|
layer.weight,
|
||||||
|
layer.weight_scale,
|
||||||
|
layer.weight_global_scale,
|
||||||
|
layer.workspace,
|
||||||
|
size_n,
|
||||||
|
size_k,
|
||||||
|
use_fp32_reduce=True,
|
||||||
|
)
|
||||||
|
|
||||||
|
output_ref = torch.matmul(a_input, weight_ref.T)
|
||||||
|
torch.cuda.synchronize()
|
||||||
|
|
||||||
|
torch.testing.assert_close(output, output_ref, rtol=0.04, atol=0.04)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
sys.exit(pytest.main([__file__, "-v", "-s"]))
|
||||||
Reference in New Issue
Block a user