[sgl-kernel/cpu] support w8a8 int8 model for arm cpu (#16045)
skip gpu test as this one is not related to gpu backend.
This commit is contained in:
@@ -473,6 +473,7 @@ at::Tensor int8_scaled_mm_cpu(
|
||||
return out;
|
||||
}
|
||||
|
||||
#ifndef __aarch64__
|
||||
// fused `per_token_quant_int8_cpu` and `int8_scaled_mm_cpu`
|
||||
at::Tensor int8_scaled_mm_with_quant(
|
||||
at::Tensor& mat1,
|
||||
@@ -539,3 +540,4 @@ at::Tensor int8_scaled_mm_with_quant(
|
||||
});
|
||||
return out;
|
||||
}
|
||||
#endif // #ifndef __aarch64__
|
||||
|
||||
Reference in New Issue
Block a user