ggml-blas: default hadamard mul_mat to cpu routine (llama/25710)

Signed-off-by: Aaron Teo <aaron.teo1@ibm.com>
This commit is contained in:
Aaron Teo 2026-07-17 16:39:33 +08:00 committed by Georgi Gerganov
parent ca31169a53
commit 6c8bac6ef8
No known key found for this signature in database
GPG Key ID: 449E073F9DC10735
1 changed files with 7 additions and 0 deletions

View File

@ -1,3 +1,4 @@
#include "ggml.h"
#include "ggml-impl.h"
#include "ggml-blas.h"
#include "ggml-backend-impl.h"
@ -415,6 +416,12 @@ static bool ggml_backend_blas_device_supports_op(ggml_backend_dev_t dev, const s
// TODO: find the optimal value
const int64_t min_batch = 32;
// default back to CPU fast path
// see: https://github.com/ggml-org/llama.cpp/issues/25565
if (ggml_get_op_params_i32(op, 1) == GGML_HINT_SRC0_IS_HADAMARD) {
return false;
}
return ggml_is_contiguous(src0) &&
ggml_is_contiguous(src1) &&
src1->type == GGML_TYPE_F32 &&