mirror of
https://github.com/ggml-org/llama.cpp.git
synced 2026-10-01 18:37:28 -05:00
ggml-openvino : reject BF16 SCALE and mixed-type BF16 ADD/MUL/SUB
This commit is contained in:
@@ -1211,6 +1211,10 @@ static ggml_openvino_op_support is_op_supported_case(const ggml_tensor * op) {
|
||||
if (op->src[1]->op == GGML_OP_PERMUTE) {
|
||||
return {false, "ADD/MUL/SUB with PERMUTE src1 is not supported"};
|
||||
}
|
||||
if (op->src[0]->type != op->src[1]->type &&
|
||||
(op->src[0]->type == GGML_TYPE_BF16 || op->src[1]->type == GGML_TYPE_BF16)) {
|
||||
return {false, "ADD/MUL/SUB with BF16 and a different src1 type is not supported"};
|
||||
}
|
||||
// >8-expert MoE ReduceSum drifts past the 1e-7 tolerance (f32 order vs CPU); intermittent.
|
||||
if (op->op == GGML_OP_ADD && is_moe_expert_sum_add(op) && op->src[1]->src[0]->ne[1] > 8) {
|
||||
return {false, "MoE expert-plane sum with more than 8 experts is not supported"};
|
||||
@@ -1224,6 +1228,12 @@ static ggml_openvino_op_support is_op_supported_case(const ggml_tensor * op) {
|
||||
}
|
||||
break;
|
||||
}
|
||||
case GGML_OP_SCALE: {
|
||||
if (op->type == GGML_TYPE_BF16) {
|
||||
return {false, "SCALE with BF16 type is not supported"};
|
||||
}
|
||||
break;
|
||||
}
|
||||
case GGML_OP_ADD_ID: {
|
||||
// Keep support aligned with the CPU backend implementation, which only handles f32 inputs/output and i32 ids.
|
||||
if (op->type != GGML_TYPE_F32 || op->src[0]->type != GGML_TYPE_F32 || op->src[1]->type != GGML_TYPE_F32 ||
|
||||
|
||||
Reference in New Issue
Block a user