[sgl-kernel/cpu]: exclude amx gemm source from arm build (#29286)

This commit is contained in:
Yibo Cai
2026-06-26 08:23:43 +08:00
committed by GitHub
parent ed71fb8f95
commit 4b0b0ea55f
3 changed files with 18 additions and 2 deletions
+1
View File
@@ -93,6 +93,7 @@ endif()
# Keep them out of Arm64 bootstrap builds until native Arm paths land.
set(SGLANG_CPU_X86_ONLY_SOURCES
${CMAKE_CURRENT_SOURCE_DIR}/gemm_int4.cpp
${CMAKE_CURRENT_SOURCE_DIR}/gemm_int8.cpp
${CMAKE_CURRENT_SOURCE_DIR}/moe.cpp
${CMAKE_CURRENT_SOURCE_DIR}/moe_fp8.cpp
${CMAKE_CURRENT_SOURCE_DIR}/moe_int4.cpp
+17
View File
@@ -56,6 +56,23 @@ void int8_scaled_mm_impl<at::BFloat16>(
} // anonymous namespace
std::tuple<at::Tensor, at::Tensor> per_token_quant_int8_cpu(at::Tensor& /*A*/) {
TORCH_CHECK(false, "not implemented yet");
return {at::Tensor(), at::Tensor()};
}
at::Tensor int8_scaled_mm_cpu(
at::Tensor& /*mat1*/,
at::Tensor& /*mat2*/,
at::Tensor& /*scales1*/,
at::Tensor& /*scales2*/,
const std::optional<at::Tensor>& /*bias*/,
at::ScalarType /*out_dtype*/,
bool /*is_vnni*/) {
TORCH_CHECK(false, "not implemented yet");
return at::Tensor();
}
// weight : static, per-channel, symmetric
// activation : dynamic, per-token, symmetric
//
-2
View File
@@ -473,7 +473,6 @@ at::Tensor int8_scaled_mm_cpu(
return out;
}
#ifndef __aarch64__
// fused `per_token_quant_int8_cpu` and `int8_scaled_mm_cpu`
at::Tensor int8_scaled_mm_with_quant(
at::Tensor& mat1,
@@ -540,4 +539,3 @@ at::Tensor int8_scaled_mm_with_quant(
});
return out;
}
#endif // #ifndef __aarch64__