[CI] Fix stale per-token group quant callers (#32047)
This commit is contained in:
@@ -1,16 +1,10 @@
|
|||||||
import itertools
|
import itertools
|
||||||
import os
|
import os
|
||||||
import time
|
|
||||||
from functools import partial
|
|
||||||
from pathlib import Path
|
|
||||||
|
|
||||||
import torch
|
import torch
|
||||||
import triton
|
import triton
|
||||||
from sgl_kernel.test_utils import create_per_token_group_quant_test_data
|
from sgl_kernel.test_utils import create_per_token_group_quant_test_data
|
||||||
|
|
||||||
from sglang.kernels.ops.quantization.fp8_kernel import (
|
|
||||||
create_per_token_group_quant_fp8_output_scale,
|
|
||||||
)
|
|
||||||
from sglang.kernels.ops.quantization.fp8_kernel import (
|
from sglang.kernels.ops.quantization.fp8_kernel import (
|
||||||
per_token_group_quant_8bit as triton_per_token_group_quant_8bit,
|
per_token_group_quant_8bit as triton_per_token_group_quant_8bit,
|
||||||
)
|
)
|
||||||
@@ -223,17 +217,19 @@ def benchmark(
|
|||||||
"_per_token_group_quant_8bit|_silu_and_mul_post_quant_kernel",
|
"_per_token_group_quant_8bit|_silu_and_mul_post_quant_kernel",
|
||||||
),
|
),
|
||||||
"sglang": (
|
"sglang": (
|
||||||
partial(sglang_per_token_group_quant_8bit, enable_v2=True),
|
sglang_per_token_group_quant_8bit,
|
||||||
"per_token_group_quant_8bit_kernel",
|
"per_token_group_quant_8bit_kernel",
|
||||||
),
|
),
|
||||||
}[provider]
|
}[provider]
|
||||||
bench_fn = lambda: fn(
|
|
||||||
x=x,
|
def bench_fn():
|
||||||
masked_m=masked_m,
|
return fn(
|
||||||
group_size=group_size,
|
x=x,
|
||||||
dst_dtype=dst_dtype,
|
masked_m=masked_m,
|
||||||
**{k: v for k, v in flags.items() if k not in ["masked_layout_mode"]},
|
group_size=group_size,
|
||||||
)
|
dst_dtype=dst_dtype,
|
||||||
|
**{k: v for k, v in flags.items() if k not in ["masked_layout_mode"]},
|
||||||
|
)
|
||||||
|
|
||||||
time_s = bench_kineto(
|
time_s = bench_kineto(
|
||||||
bench_fn, kernel_names=kernel_names, num_tests=300 if mode_concentrated else 30
|
bench_fn, kernel_names=kernel_names, num_tests=300 if mode_concentrated else 30
|
||||||
|
|||||||
@@ -1,8 +1,5 @@
|
|||||||
import itertools
|
import itertools
|
||||||
import os
|
|
||||||
import sys
|
import sys
|
||||||
import time
|
|
||||||
from pathlib import Path
|
|
||||||
|
|
||||||
import pytest
|
import pytest
|
||||||
import torch
|
import torch
|
||||||
@@ -17,7 +14,7 @@ from sglang.kernels.ops.quantization.fp8_kernel import (
|
|||||||
from sglang.kernels.ops.quantization.fp8_kernel import (
|
from sglang.kernels.ops.quantization.fp8_kernel import (
|
||||||
sglang_per_token_group_quant_8bit,
|
sglang_per_token_group_quant_8bit,
|
||||||
)
|
)
|
||||||
from sglang.srt.utils import get_bool_env_var, is_hip
|
from sglang.srt.utils import is_hip
|
||||||
|
|
||||||
_is_hip = is_hip()
|
_is_hip = is_hip()
|
||||||
fp8_type_ = torch.float8_e4m3fnuz if _is_hip else torch.float8_e4m3fn
|
fp8_type_ = torch.float8_e4m3fnuz if _is_hip else torch.float8_e4m3fn
|
||||||
@@ -156,7 +153,7 @@ def test_per_token_group_quant_with_column_major(
|
|||||||
*triton_per_token_group_quant_8bit(**execute_kwargs)
|
*triton_per_token_group_quant_8bit(**execute_kwargs)
|
||||||
)
|
)
|
||||||
x_q_sglang, x_s_sglang = _postprocess(
|
x_q_sglang, x_s_sglang = _postprocess(
|
||||||
*sglang_per_token_group_quant_8bit(**execute_kwargs, enable_v2=True)
|
*sglang_per_token_group_quant_8bit(**execute_kwargs)
|
||||||
)
|
)
|
||||||
|
|
||||||
try:
|
try:
|
||||||
|
|||||||
Reference in New Issue
Block a user