llama4 npu adapt (#17123)
Co-authored-by: cy <chenyang08056032@163.com> Co-authored-by: sglang-npu-bot <sglangnpu@163.com>
This commit is contained in:
co-authored by
cy
sglang-npu-bot
parent
f0c2089597
commit
bc9190435b
@@ -56,11 +56,13 @@ from sglang.srt.utils import (
|
|||||||
fast_topk,
|
fast_topk,
|
||||||
get_compiler_backend,
|
get_compiler_backend,
|
||||||
is_cuda,
|
is_cuda,
|
||||||
|
is_npu,
|
||||||
make_layers,
|
make_layers,
|
||||||
)
|
)
|
||||||
from sglang.srt.utils.common import get_current_device_stream_fast
|
from sglang.srt.utils.common import get_current_device_stream_fast
|
||||||
|
|
||||||
_is_cuda = is_cuda()
|
_is_cuda = is_cuda()
|
||||||
|
_is_npu = is_npu()
|
||||||
|
|
||||||
logger = logging.getLogger(__name__)
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
@@ -329,6 +331,8 @@ class Llama4Attention(nn.Module):
|
|||||||
if self.rotary_emb is not None:
|
if self.rotary_emb is not None:
|
||||||
q_view, k_view = qk.split([self.q_size, self.kv_size], dim=-1)
|
q_view, k_view = qk.split([self.q_size, self.kv_size], dim=-1)
|
||||||
q_out_unused, k_out_unused = self.rotary_emb(positions, q_view, k_view)
|
q_out_unused, k_out_unused = self.rotary_emb(positions, q_view, k_view)
|
||||||
|
if _is_npu:
|
||||||
|
qk = torch.cat([q_out_unused, k_out_unused], dim=-1)
|
||||||
del q_view, k_view, q_out_unused, k_out_unused
|
del q_view, k_view, q_out_unused, k_out_unused
|
||||||
|
|
||||||
if self.qk_norm is not None:
|
if self.qk_norm is not None:
|
||||||
|
|||||||
@@ -1480,9 +1480,10 @@ class ServerArgs:
|
|||||||
"fa3",
|
"fa3",
|
||||||
"aiter",
|
"aiter",
|
||||||
"triton",
|
"triton",
|
||||||
|
"ascend",
|
||||||
"trtllm_mha",
|
"trtllm_mha",
|
||||||
"intel_xpu",
|
"intel_xpu",
|
||||||
}, f"fa3, aiter, triton, trtllm_mha or intel_xpu is required for Llama4 model but got {self.attention_backend}"
|
}, f"fa3, aiter, triton, ascend, trtllm_mha or intel_xpu is required for Llama4 model but got {self.attention_backend}"
|
||||||
if is_sm100_supported() and self.moe_runner_backend == "auto":
|
if is_sm100_supported() and self.moe_runner_backend == "auto":
|
||||||
if self.quantization in {"fp8", "modelopt_fp8"}:
|
if self.quantization in {"fp8", "modelopt_fp8"}:
|
||||||
self.moe_runner_backend = "flashinfer_trtllm"
|
self.moe_runner_backend = "flashinfer_trtllm"
|
||||||
|
|||||||
@@ -0,0 +1,31 @@
|
|||||||
|
import unittest
|
||||||
|
|
||||||
|
from sglang.test.ascend.gsm8k_ascend_mixin import GSM8KAscendMixin
|
||||||
|
from sglang.test.ci.ci_register import register_npu_ci
|
||||||
|
from sglang.test.test_utils import CustomTestCase
|
||||||
|
|
||||||
|
register_npu_ci(est_time=400, suite="nightly-4-npu-a3", nightly=True)
|
||||||
|
|
||||||
|
|
||||||
|
class TestLlama4(GSM8KAscendMixin, CustomTestCase):
|
||||||
|
model = (
|
||||||
|
"/root/.cache/modelscope/hub/models/meta-llama/Llama-4-Scout-17B-16E-Instruct"
|
||||||
|
)
|
||||||
|
accuracy = 0.9
|
||||||
|
other_args = [
|
||||||
|
"--chat-template",
|
||||||
|
"llama-4",
|
||||||
|
"--tp-size",
|
||||||
|
4,
|
||||||
|
"--mem-fraction-static",
|
||||||
|
"0.9",
|
||||||
|
"--context-length",
|
||||||
|
"8192",
|
||||||
|
"--attention-backend",
|
||||||
|
"ascend",
|
||||||
|
"--disable-cuda-graph",
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
unittest.main()
|
||||||
Reference in New Issue
Block a user