Use the correct wrapper for fp4_quantize (#27956)
Co-authored-by: Brayden Zhong <brayden@radixark.ai>
This commit is contained in:
co-authored by
Brayden Zhong
parent
40894be3c3
commit
8bfcc0c39c
@@ -286,8 +286,7 @@ class DeepseekV2MLP(nn.Module):
|
|||||||
and self.swiglu_limit is None
|
and self.swiglu_limit is None
|
||||||
and not isinstance(x, tuple)
|
and not isinstance(x, tuple)
|
||||||
):
|
):
|
||||||
from flashinfer import fp4_quantize
|
from sglang.srt.layers.quantization.fp4_utils import fp4_quantize
|
||||||
|
|
||||||
from sglang.srt.layers.quantization.nvfp4_gemm_swiglu_nvfp4_quant import (
|
from sglang.srt.layers.quantization.nvfp4_gemm_swiglu_nvfp4_quant import (
|
||||||
nvfp4_gemm_swiglu_nvfp4_quant,
|
nvfp4_gemm_swiglu_nvfp4_quant,
|
||||||
)
|
)
|
||||||
|
|||||||
Reference in New Issue
Block a user