diff --git a/ggml/src/ggml-openvino/ggml-openvino.cpp b/ggml/src/ggml-openvino/ggml-openvino.cpp index b2d81df97d..d57eb55382 100644 --- a/ggml/src/ggml-openvino/ggml-openvino.cpp +++ b/ggml/src/ggml-openvino/ggml-openvino.cpp @@ -1211,6 +1211,10 @@ static ggml_openvino_op_support is_op_supported_case(const ggml_tensor * op) { if (op->src[1]->op == GGML_OP_PERMUTE) { return {false, "ADD/MUL/SUB with PERMUTE src1 is not supported"}; } + if (op->src[0]->type != op->src[1]->type && + (op->src[0]->type == GGML_TYPE_BF16 || op->src[1]->type == GGML_TYPE_BF16)) { + return {false, "ADD/MUL/SUB with BF16 and a different src1 type is not supported"}; + } // >8-expert MoE ReduceSum drifts past the 1e-7 tolerance (f32 order vs CPU); intermittent. if (op->op == GGML_OP_ADD && is_moe_expert_sum_add(op) && op->src[1]->src[0]->ne[1] > 8) { return {false, "MoE expert-plane sum with more than 8 experts is not supported"}; @@ -1224,6 +1228,12 @@ static ggml_openvino_op_support is_op_supported_case(const ggml_tensor * op) { } break; } + case GGML_OP_SCALE: { + if (op->type == GGML_TYPE_BF16) { + return {false, "SCALE with BF16 type is not supported"}; + } + break; + } case GGML_OP_ADD_ID: { // Keep support aligned with the CPU backend implementation, which only handles f32 inputs/output and i32 ids. if (op->type != GGML_TYPE_F32 || op->src[0]->type != GGML_TYPE_F32 || op->src[1]->type != GGML_TYPE_F32 ||