[Intel GPU] DeepSeek V4 12/N: use sgl-kernel implementation of silu_and_mul_clamp to run on XPU (#28428)
Signed-off-by: P V R K Jyothendra Varma <polisetty.v.r.k.jyothendra.varma@intel.com>
This commit is contained in:
@@ -9,9 +9,12 @@ from sglang.jit_kernel.utils import (
|
|||||||
load_jit,
|
load_jit,
|
||||||
make_cpp_args,
|
make_cpp_args,
|
||||||
)
|
)
|
||||||
|
from sglang.srt.utils import is_xpu
|
||||||
|
|
||||||
from .utils import make_name
|
from .utils import make_name
|
||||||
|
|
||||||
|
_is_xpu = is_xpu()
|
||||||
|
|
||||||
|
|
||||||
@cache_once
|
@cache_once
|
||||||
def _jit_mask_topk_module():
|
def _jit_mask_topk_module():
|
||||||
@@ -175,8 +178,13 @@ def silu_and_mul_clamp(
|
|||||||
output: torch.Tensor,
|
output: torch.Tensor,
|
||||||
swiglu_limit: float,
|
swiglu_limit: float,
|
||||||
) -> None:
|
) -> None:
|
||||||
module = _jit_silu_and_mul_clamp_module(input.dtype)
|
if _is_xpu:
|
||||||
module.run(input, output, float(swiglu_limit))
|
from sgl_kernel import silu_and_mul_clamp
|
||||||
|
|
||||||
|
silu_and_mul_clamp(input, output, float(swiglu_limit))
|
||||||
|
else:
|
||||||
|
module = _jit_silu_and_mul_clamp_module(input.dtype)
|
||||||
|
module.run(input, output, float(swiglu_limit))
|
||||||
|
|
||||||
|
|
||||||
def silu_and_mul_masked_post_quant(
|
def silu_and_mul_masked_post_quant(
|
||||||
|
|||||||
Reference in New Issue
Block a user