Fix gpt-oss-20b with mxfp4 support for Xeon (#27782)
This commit is contained in:
@@ -55,6 +55,7 @@ from sglang.srt.layers.quantization.w8a8_int8 import W8A8Int8Config
|
|||||||
from sglang.srt.platforms import current_platform
|
from sglang.srt.platforms import current_platform
|
||||||
from sglang.srt.utils import (
|
from sglang.srt.utils import (
|
||||||
cpu_has_amx_support,
|
cpu_has_amx_support,
|
||||||
|
is_cpu,
|
||||||
is_cuda,
|
is_cuda,
|
||||||
is_hip,
|
is_hip,
|
||||||
is_mps,
|
is_mps,
|
||||||
@@ -99,7 +100,7 @@ BASE_QUANTIZATION_METHODS: Dict[str, Type[QuantizationConfig]] = {
|
|||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
if is_cuda() or (_is_mxfp_supported and is_hip()):
|
if is_cpu() or is_cuda() or (_is_mxfp_supported and is_hip()):
|
||||||
BASE_QUANTIZATION_METHODS.update(
|
BASE_QUANTIZATION_METHODS.update(
|
||||||
{
|
{
|
||||||
"mxfp4": Mxfp4Config,
|
"mxfp4": Mxfp4Config,
|
||||||
@@ -130,6 +131,7 @@ CPU_QUANTIZATION_METHODS = {
|
|||||||
"compressed-tensors": CompressedTensorsConfig,
|
"compressed-tensors": CompressedTensorsConfig,
|
||||||
"awq": AWQCPUConfig,
|
"awq": AWQCPUConfig,
|
||||||
"gptq": CPUGPTQConfig,
|
"gptq": CPUGPTQConfig,
|
||||||
|
"mxfp4": Mxfp4Config,
|
||||||
}
|
}
|
||||||
|
|
||||||
QUANTIZATION_METHODS = {**BASE_QUANTIZATION_METHODS}
|
QUANTIZATION_METHODS = {**BASE_QUANTIZATION_METHODS}
|
||||||
|
|||||||
Reference in New Issue
Block a user