ggml-openvino : reject BF16 SCALE and mixed-type BF16 ADD/MUL/SUB

This commit is contained in:
Aman Gupta
2026-09-30 10:56:54 +08:00
parent f575038ea4
commit 71d72e1472
+10
View File
@@ -1211,6 +1211,10 @@ static ggml_openvino_op_support is_op_supported_case(const ggml_tensor * op) {
if (op->src[1]->op == GGML_OP_PERMUTE) {
return {false, "ADD/MUL/SUB with PERMUTE src1 is not supported"};
}
if (op->src[0]->type != op->src[1]->type &&
(op->src[0]->type == GGML_TYPE_BF16 || op->src[1]->type == GGML_TYPE_BF16)) {
return {false, "ADD/MUL/SUB with BF16 and a different src1 type is not supported"};
}
// >8-expert MoE ReduceSum drifts past the 1e-7 tolerance (f32 order vs CPU); intermittent.
if (op->op == GGML_OP_ADD && is_moe_expert_sum_add(op) && op->src[1]->src[0]->ne[1] > 8) {
return {false, "MoE expert-plane sum with more than 8 experts is not supported"};
@@ -1224,6 +1228,12 @@ static ggml_openvino_op_support is_op_supported_case(const ggml_tensor * op) {
}
break;
}
case GGML_OP_SCALE: {
if (op->type == GGML_TYPE_BF16) {
return {false, "SCALE with BF16 type is not supported"};
}
break;
}
case GGML_OP_ADD_ID: {
// Keep support aligned with the CPU backend implementation, which only handles f32 inputs/output and i32 ids.
if (op->type != GGML_TYPE_F32 || op->src[0]->type != GGML_TYPE_F32 || op->src[1]->type != GGML_TYPE_F32 ||