diff --git a/ggml/include/ggml.h b/ggml/include/ggml.h index 224bdef927..d8dd2b0c4a 100644 --- a/ggml/include/ggml.h +++ b/ggml/include/ggml.h @@ -442,7 +442,6 @@ extern "C" { // the precision parameters are stored as ggml_tensor.op_params to the respective ops enum ggml_prec { GGML_PREC_UNDEFINED = 0, - GGML_PREC_DEFAULT = 0, // note: deprecated, use GGML_PREC_UNDEFINED GGML_PREC_F32 = 10, GGML_PREC_BF16 = 15, GGML_PREC_F16 = 20, @@ -495,7 +494,6 @@ extern "C" { GGML_OP_DUP, GGML_OP_ADD, GGML_OP_ADD_ID, - GGML_OP_ADD1, GGML_OP_ACC, GGML_OP_SUB, GGML_OP_MUL, @@ -760,10 +758,6 @@ extern "C" { GGML_API size_t ggml_type_size(enum ggml_type type); // size in bytes for all elements in a block GGML_API size_t ggml_row_size (enum ggml_type type, int64_t ne); // size in bytes for all elements in a row - GGML_DEPRECATED( - GGML_API double ggml_type_sizef(enum ggml_type type), // ggml_type_size()/ggml_blck_size() as float - "use ggml_row_size() instead"); - GGML_API const char * ggml_type_name(enum ggml_type type); GGML_API const char * ggml_op_name (enum ggml_op op); GGML_API const char * ggml_op_symbol(enum ggml_op op); @@ -931,18 +925,6 @@ extern "C" { struct ggml_tensor * b, struct ggml_tensor * ids); - GGML_DEPRECATED(GGML_API struct ggml_tensor * ggml_add1( - struct ggml_context * ctx, - struct ggml_tensor * a, - struct ggml_tensor * b), - "use ggml_add instead"); - - GGML_DEPRECATED(GGML_API struct ggml_tensor * ggml_add1_inplace( - struct ggml_context * ctx, - struct ggml_tensor * a, - struct ggml_tensor * b), - "use ggml_add_inplace instead"); - // dst = a // view(dst, nb1, nb2, nb3, offset) += b // return dst @@ -1484,13 +1466,6 @@ extern "C" { struct ggml_tensor * a, struct ggml_tensor * b); - // change the precision of a matrix multiplication - // set to GGML_PREC_F32 for higher precision (useful for phi-2) - GGML_DEPRECATED(GGML_API void ggml_mul_mat_set_prec( - struct ggml_tensor * a, - enum ggml_prec prec), - "use ggml_prec_set_acc() instead"); - // change the hint of a matrix multiplication GGML_API void ggml_mul_mat_set_hint( struct ggml_tensor * a, @@ -1982,36 +1957,6 @@ extern "C" { float beta_fast, float beta_slow); - GGML_DEPRECATED(GGML_API struct ggml_tensor * ggml_rope_custom( - struct ggml_context * ctx, - struct ggml_tensor * a, - struct ggml_tensor * b, - int n_dims, - int mode, - int n_ctx_orig, - float freq_base, - float freq_scale, - float ext_factor, - float attn_factor, - float beta_fast, - float beta_slow), - "use ggml_rope_ext instead"); - - GGML_DEPRECATED(GGML_API struct ggml_tensor * ggml_rope_custom_inplace( - struct ggml_context * ctx, - struct ggml_tensor * a, - struct ggml_tensor * b, - int n_dims, - int mode, - int n_ctx_orig, - float freq_base, - float freq_scale, - float ext_factor, - float attn_factor, - float beta_fast, - float beta_slow), - "use ggml_rope_ext_inplace instead"); - // compute correction dims for YaRN RoPE scaling GGML_API void ggml_rope_yarn_corr_dims( int n_dims, int n_ctx_orig, float freq_base, float beta_fast, float beta_slow, float dims[2]); @@ -2316,7 +2261,7 @@ extern "C" { GGML_SCALE_MODE_BILINEAR = 1, GGML_SCALE_MODE_BICUBIC = 2, - GGML_SCALE_MODE_COUNT + GGML_SCALE_MODE_COUNT = 3 }; enum ggml_scale_flag { @@ -2332,18 +2277,6 @@ extern "C" { int scale_factor, enum ggml_scale_mode mode); - // interpolate - // interpolate scale to specified dimensions - GGML_DEPRECATED(GGML_API struct ggml_tensor * ggml_upscale_ext( - struct ggml_context * ctx, - struct ggml_tensor * a, - int ne0, - int ne1, - int ne2, - int ne3, - enum ggml_scale_mode mode), - "use ggml_interpolate instead"); - // Up- or downsamples the input to the specified size. // 2D scale modes (eg. bilinear) are applied to the first two dimensions. GGML_API struct ggml_tensor * ggml_interpolate( @@ -2494,11 +2427,6 @@ extern "C" { float max_bias, float logit_softcap); - GGML_DEPRECATED(GGML_API void ggml_flash_attn_ext_set_prec( - struct ggml_tensor * a, - enum ggml_prec prec), - "use ggml_prec_set_acc() instead"); - GGML_API enum ggml_prec ggml_flash_attn_ext_get_prec( const struct ggml_tensor * a); diff --git a/ggml/src/ggml-alloc.c b/ggml/src/ggml-alloc.c index a71838eafc..fcef0afee7 100644 --- a/ggml/src/ggml-alloc.c +++ b/ggml/src/ggml-alloc.c @@ -27,7 +27,6 @@ bool ggml_op_can_inplace(enum ggml_op op) { case GGML_OP_DIAG_MASK_INF: case GGML_OP_ADD: case GGML_OP_ADD_ID: - case GGML_OP_ADD1: case GGML_OP_SUB: case GGML_OP_MUL: case GGML_OP_DIV: diff --git a/ggml/src/ggml-backend-meta.cpp b/ggml/src/ggml-backend-meta.cpp index 39c5e478cd..581215cbe8 100644 --- a/ggml/src/ggml-backend-meta.cpp +++ b/ggml/src/ggml-backend-meta.cpp @@ -874,7 +874,6 @@ static struct ggml_backend_meta_split_state ggml_backend_meta_get_split_state( case GGML_OP_ADD_ID: { split_state = handle_bin_bcast(src_ss); } break; - case GGML_OP_ADD1: case GGML_OP_ACC: { split_state = handle_generic(src_ss, /*scalar_only =*/ true); } break; diff --git a/ggml/src/ggml-cann/ggml-cann.cpp b/ggml/src/ggml-cann/ggml-cann.cpp index c2745014a1..a4dc8b7153 100644 --- a/ggml/src/ggml-cann/ggml-cann.cpp +++ b/ggml/src/ggml-cann/ggml-cann.cpp @@ -1785,7 +1785,6 @@ static bool ggml_cann_compute_forward(ggml_backend_cann_context & ctx, struct gg ggml_cann_dup(ctx, dst); break; case GGML_OP_ADD: - case GGML_OP_ADD1: ggml_cann_binary_op(ctx, dst); break; case GGML_OP_SUB: @@ -2602,7 +2601,6 @@ static bool ggml_backend_cann_supports_op(ggml_backend_dev_t dev, const ggml_ten case GGML_OP_TRANSPOSE: case GGML_OP_NORM: case GGML_OP_ADD: - case GGML_OP_ADD1: case GGML_OP_SUB: case GGML_OP_MUL: case GGML_OP_DIV: diff --git a/ggml/src/ggml-cpu/ggml-cpu.c b/ggml/src/ggml-cpu/ggml-cpu.c index 8620c6c7a6..eb1e9ab4ea 100644 --- a/ggml/src/ggml-cpu/ggml-cpu.c +++ b/ggml/src/ggml-cpu/ggml-cpu.c @@ -1762,10 +1762,6 @@ static void ggml_compute_forward(struct ggml_compute_params * params, struct ggm { ggml_compute_forward_add_id(params, tensor); } break; - case GGML_OP_ADD1: - { - ggml_compute_forward_add1(params, tensor); - } break; case GGML_OP_ACC: { ggml_compute_forward_acc(params, tensor); @@ -2261,7 +2257,6 @@ static int ggml_get_n_tasks(struct ggml_tensor * node, int n_threads) { case GGML_OP_CONT: case GGML_OP_ADD: case GGML_OP_ADD_ID: - case GGML_OP_ADD1: case GGML_OP_ACC: case GGML_OP_CUMSUM: case GGML_OP_TRI: @@ -2864,7 +2859,6 @@ struct ggml_cplan ggml_graph_plan( } break; case GGML_OP_ADD: case GGML_OP_ADD_ID: - case GGML_OP_ADD1: { if (ggml_is_quantized(node->src[0]->type)) { cur = ggml_type_size(GGML_TYPE_F32) * node->src[0]->ne[0] * n_tasks; diff --git a/ggml/src/ggml-cpu/ops.cpp b/ggml/src/ggml-cpu/ops.cpp index 3e377298f3..e2795c8a45 100644 --- a/ggml/src/ggml-cpu/ops.cpp +++ b/ggml/src/ggml-cpu/ops.cpp @@ -9578,7 +9578,7 @@ void ggml_compute_forward_flash_attn_ext( const ggml_compute_params * params, ggml_tensor * dst) { switch (dst->op_params[3]) { - case GGML_PREC_DEFAULT: + case GGML_PREC_UNDEFINED: case GGML_PREC_F32: { // uses F32 accumulators diff --git a/ggml/src/ggml-cpu/spacemit/ime.cpp b/ggml/src/ggml-cpu/spacemit/ime.cpp index 29d683270e..d9ddb9afd5 100644 --- a/ggml/src/ggml-cpu/spacemit/ime.cpp +++ b/ggml/src/ggml-cpu/spacemit/ime.cpp @@ -1165,7 +1165,7 @@ class tensor_traits_common : public tensor_traits_base { const int64_t DK = nek0; const int64_t DV = nev0; - const bool supported_prec = (dst->op_params[3] == GGML_PREC_F32 || dst->op_params[3] == GGML_PREC_DEFAULT); + const bool supported_prec = (dst->op_params[3] == GGML_PREC_F32 || dst->op_params[3] == GGML_PREC_UNDEFINED); const bool supported_types = (q->type == GGML_TYPE_F32 && k->type == GGML_TYPE_F16 && v->type == GGML_TYPE_F16); const bool supported_shape = (DK > 0 && DK <= 128 && DV > 0 && DV <= 128); const bool supported_vlen = (__riscv_vlenb() == 128); diff --git a/ggml/src/ggml-cuda/ggml-cuda.cu b/ggml/src/ggml-cuda/ggml-cuda.cu index f9fb46c2a3..5b9cab017e 100644 --- a/ggml/src/ggml-cuda/ggml-cuda.cu +++ b/ggml/src/ggml-cuda/ggml-cuda.cu @@ -2099,7 +2099,6 @@ static bool ggml_cuda_compute_forward(ggml_backend_cuda_context & ctx, struct gg ggml_cuda_dup(ctx, dst); break; case GGML_OP_ADD: - case GGML_OP_ADD1: // TODO: more efficient implementation ggml_cuda_op_add(ctx, dst); break; case GGML_OP_ADD_ID: @@ -5480,7 +5479,6 @@ static bool ggml_backend_cuda_device_supports_op(ggml_backend_dev_t dev, const g case GGML_OP_PERMUTE: case GGML_OP_TRANSPOSE: case GGML_OP_ADD_ID: - case GGML_OP_ADD1: case GGML_OP_SQR: case GGML_OP_SQRT: case GGML_OP_SIN: diff --git a/ggml/src/ggml-cuda/mmvf.cu b/ggml/src/ggml-cuda/mmvf.cu index bd5c5d421a..b39697441a 100644 --- a/ggml/src/ggml-cuda/mmvf.cu +++ b/ggml/src/ggml-cuda/mmvf.cu @@ -617,7 +617,7 @@ static void mul_mat_vec_f_cuda( const int64_t ids_stride, enum ggml_prec prec, cudaStream_t stream) { if constexpr(std::is_same_v) { - if (prec == GGML_PREC_DEFAULT) { + if (prec == GGML_PREC_UNDEFINED) { mul_mat_vec_f_cuda_switch_ncols_dst (x, y, ids, fusion, dst, ncols, nrows, ncols_dst, stride_row, stride_col_y, stride_col_dst, nchannels_x, nchannels_y, nchannels_dst, stride_channel_x, stride_channel_y, diff --git a/ggml/src/ggml-et/ggml-et-ops.cpp b/ggml/src/ggml-et/ggml-et-ops.cpp index 8765138672..7ba3e7f2c3 100644 --- a/ggml/src/ggml-et/ggml-et-ops.cpp +++ b/ggml/src/ggml-et/ggml-et-ops.cpp @@ -1523,7 +1523,7 @@ bool ggml_et_op_flash_attn_ext(ggml_backend_et_device_context * dev_ctx, const g } const ggml_prec prec = ggml_flash_attn_ext_get_prec(node); - if (prec != GGML_PREC_F32 && prec != GGML_PREC_DEFAULT) { + if (prec != GGML_PREC_F32 && prec != GGML_PREC_UNDEFINED) { GGML_LOG_ERROR("ET: FLASH_ATTN_EXT baseline kernel only supports F32 precision\n"); return false; } diff --git a/ggml/src/ggml-et/ggml-et.cpp b/ggml/src/ggml-et/ggml-et.cpp index 755077fad5..328526309d 100644 --- a/ggml/src/ggml-et/ggml-et.cpp +++ b/ggml/src/ggml-et/ggml-et.cpp @@ -1294,7 +1294,7 @@ static bool ggml_backend_et_device_supports_op(ggml_backend_dev_t dev, const ggm const bool me_eligible = op->src[1]->type == GGML_TYPE_F16 && op->src[2]->type == GGML_TYPE_F16 && (op->src[0]->ne[0] % 32) == 0; - supported = me_eligible && mask_ok && (prec == GGML_PREC_F32 || prec == GGML_PREC_DEFAULT) && + supported = me_eligible && mask_ok && (prec == GGML_PREC_F32 || prec == GGML_PREC_UNDEFINED) && max_bias == 0.0f && logit_softcap == 0.0f && op->src[0]->nb[0] == sizeof(float) && op->src[1]->nb[0] == k_elem && op->src[2]->nb[0] == v_elem && op->nb[0] == sizeof(float) && op->src[0]->ne[0] == op->src[1]->ne[0] && // dk matches diff --git a/ggml/src/ggml-openvino/openvino/op_table.cpp b/ggml/src/ggml-openvino/openvino/op_table.cpp index f249a06bb8..f315d9e967 100644 --- a/ggml/src/ggml-openvino/openvino/op_table.cpp +++ b/ggml/src/ggml-openvino/openvino/op_table.cpp @@ -24,7 +24,6 @@ std::unordered_map get_supported_ops() { using namespace ov::op; return { {"GGML_OP_ADD", op::translate_add }, - {"GGML_OP_ADD1", op::translate_1to1_match_2_inputs }, {"GGML_OP_ADD_ID", op::translate_add_id }, {"GGML_OP_CONCAT", op::translate_concat }, {"GGML_OP_CONT", op::translate_cont }, diff --git a/ggml/src/ggml-sycl/ggml-sycl.cpp b/ggml/src/ggml-sycl/ggml-sycl.cpp index 2bc2aaa326..565a0e121f 100644 --- a/ggml/src/ggml-sycl/ggml-sycl.cpp +++ b/ggml/src/ggml-sycl/ggml-sycl.cpp @@ -2981,7 +2981,7 @@ inline void ggml_sycl_op_mul_mat_sycl( #endif if ((src0->type == GGML_TYPE_F16 || ggml_is_quantized(src0->type)) && use_fp16 && ggml_is_contiguous(src0) && - row_diff == src0->ne[1] && dst->op_params[0] == GGML_PREC_DEFAULT) { + row_diff == src0->ne[1] && dst->op_params[0] == GGML_PREC_UNDEFINED) { ggml_sycl_pool_alloc src0_as_f16(ctx.pool()); if (src0->type != GGML_TYPE_F16) { scope_op_debug_print scope_dbg_print(__func__, "/to_fp16_sycl", dst, /*num_src=*/2, @@ -5524,7 +5524,6 @@ static bool ggml_sycl_compute_forward(ggml_backend_sycl_context & ctx, struct gg ggml_sycl_dup(ctx, dst); break; case GGML_OP_ADD: - case GGML_OP_ADD1: // TODO: more efficient implementation ggml_sycl_add(ctx, dst); break; case GGML_OP_ADD_ID: @@ -6700,7 +6699,6 @@ static bool do_ggml_backend_sycl_device_supports_op(ggml_backend_dev_t dev, cons case GGML_OP_PERMUTE: case GGML_OP_TRANSPOSE: case GGML_OP_ADD: - case GGML_OP_ADD1: case GGML_OP_ADD_ID: case GGML_OP_SUB: case GGML_OP_COUNT_EQUAL: diff --git a/ggml/src/ggml-vulkan/ggml-vulkan-debug.cpp b/ggml/src/ggml-vulkan/ggml-vulkan-debug.cpp index 15abd54681..cf6505ce32 100644 --- a/ggml/src/ggml-vulkan/ggml-vulkan-debug.cpp +++ b/ggml/src/ggml-vulkan/ggml-vulkan-debug.cpp @@ -980,8 +980,6 @@ static void ggml_vk_check_results_0(ggml_backend_vk_context * ctx, ggml_cgraph * } else if (tensor->op == GGML_OP_SCALE) { const float * params = (const float *)tensor->op_params; tensor_clone = ggml_scale_bias(ggml_ctx, src_clone[0], params[0], params[1]); - } else if (tensor->op == GGML_OP_ADD1) { - tensor_clone = ggml_add1(ggml_ctx, src_clone[0], src_clone[1]); } else if (tensor->op == GGML_OP_ARANGE) { const float start = ggml_get_op_params_f32(tensor, 0); const float stop = ggml_get_op_params_f32(tensor, 1); diff --git a/ggml/src/ggml-vulkan/ggml-vulkan.cpp b/ggml/src/ggml-vulkan/ggml-vulkan.cpp index cbd3697ef9..d6fc5eabcb 100644 --- a/ggml/src/ggml-vulkan/ggml-vulkan.cpp +++ b/ggml/src/ggml-vulkan/ggml-vulkan.cpp @@ -3455,10 +3455,6 @@ void ggml_vk_load_shaders(vk_device& device, vk_pipeline requested) { CREATE_UNARY_MUL(softplus, 3) #undef CREATE_UNARY_MUL - ggml_vk_create_pipeline(device, device->pipeline_add1_f16_f16, "add1_f16_f16", add1_f16_f16_len, add1_f16_f16_data, "main", 3, sizeof(vk_op_binary_push_constants), {512, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_add1_f16_f32, "add1_f16_f32", add1_f16_f32_len, add1_f16_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {512, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_add1_f32_f32, "add1_f32_f32", add1_f32_f32_len, add1_f32_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {512, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_arange_f32, "arange_f32", arange_f32_len, arange_f32_data, "main", 1, sizeof(vk_op_push_constants), {512, 1, 1}, {}, 1); ggml_vk_create_pipeline(device, device->pipeline_fill_f32, "fill_f32", fill_f32_len, fill_f32_data, "main", 1, sizeof(vk_op_push_constants), {512, 1, 1}, {}, 1); @@ -5944,16 +5940,16 @@ static bool ggml_vk_get_mul_mat_mat_f16acc(ggml_backend_vk_context * ctx, ggml_t if (src0_type == GGML_TYPE_F32 || src0_type == GGML_TYPE_BF16) return false; if (src1_type == GGML_TYPE_Q8_1) return false; if (src0_type == GGML_TYPE_F16) { - return prec == GGML_PREC_DEFAULT && ctx->device->fp16 && !(ctx->device->coopmat_support && !ctx->device->coopmat_acc_f16_support); + return prec == GGML_PREC_UNDEFINED && ctx->device->fp16 && !(ctx->device->coopmat_support && !ctx->device->coopmat_acc_f16_support); } // quant types if (ctx->device->coopmat2) { - return prec == GGML_PREC_DEFAULT; + return prec == GGML_PREC_UNDEFINED; } if (ctx->device->coopmat_support) { - return ctx->device->fp16 && ctx->device->coopmat_acc_f16_support && prec == GGML_PREC_DEFAULT; + return ctx->device->fp16 && ctx->device->coopmat_acc_f16_support && prec == GGML_PREC_UNDEFINED; } - return ctx->device->fp16 && prec == GGML_PREC_DEFAULT; + return ctx->device->fp16 && prec == GGML_PREC_UNDEFINED; } static const std::vector* ggml_vk_get_mul_mat_mat_pipeline_map( @@ -9297,17 +9293,6 @@ static vk_pipeline ggml_vk_op_get_pipeline(ggml_backend_vk_context * ctx, const return pipeline; } return nullptr; - case GGML_OP_ADD1: - if (src0->type == GGML_TYPE_F16 && src1->type == GGML_TYPE_F16 && dst->type == GGML_TYPE_F16) { - return ctx->device->pipeline_add1_f16_f16; - } - if (src0->type == GGML_TYPE_F16 && src1->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F16) { - return ctx->device->pipeline_add1_f16_f32; - } - if (src0->type == GGML_TYPE_F32 && src1->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { - return ctx->device->pipeline_add1_f32_f32; - } - return nullptr; case GGML_OP_ARANGE: if (dst->type == GGML_TYPE_F32) { return ctx->device->pipeline_arange_f32; @@ -9574,7 +9559,6 @@ static void ggml_vk_op_f32(ggml_backend_vk_context * ctx, vk_context& subctx, co case GGML_OP_SUB: case GGML_OP_DIV: case GGML_OP_MUL: - case GGML_OP_ADD1: case GGML_OP_OUT_PROD: case GGML_OP_ARANGE: case GGML_OP_FILL: @@ -10459,21 +10443,6 @@ void ggml_vk_sqrt(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_ ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_SQRT, vk_op_unary_push_constants_init(src0, dst)); } -void ggml_vk_add1(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { - const uint32_t src0_type_size = ggml_type_size(src0->type); - const uint32_t src1_type_size = ggml_type_size(src1->type); - const uint32_t dst_type_size = ggml_type_size(dst->type); - - ggml_vk_op_f32(ctx, subctx, src0, src1, nullptr, nullptr, dst, GGML_OP_ADD1, { - (uint32_t)ggml_nelements(src0), - (uint32_t)src0->ne[0], (uint32_t)src0->ne[1], (uint32_t)src0->ne[2],(uint32_t)src0->ne[3], (uint32_t)src0->nb[0] / src0_type_size, (uint32_t)src0->nb[1] / src0_type_size, (uint32_t)src0->nb[2] / src0_type_size, (uint32_t)src0->nb[3] / src0_type_size, - (uint32_t)src1->ne[0], (uint32_t)src1->ne[1], (uint32_t)src1->ne[2],(uint32_t)src1->ne[3], (uint32_t)src1->nb[0] / src1_type_size, (uint32_t)src1->nb[1] / src1_type_size, (uint32_t)src1->nb[2] / src1_type_size, (uint32_t)src1->nb[3] / src1_type_size, - (uint32_t) dst->ne[0], (uint32_t) dst->ne[1], (uint32_t) dst->ne[2],(uint32_t) dst->ne[3], (uint32_t) dst->nb[0] / dst_type_size, (uint32_t) dst->nb[1] / dst_type_size, (uint32_t) dst->nb[2] / dst_type_size, (uint32_t) dst->nb[3] / dst_type_size, - 0, - 0.0f, 0.0f, 0, - }); -} - void ggml_vk_arange(ggml_backend_vk_context * ctx, vk_context& subctx, ggml_tensor * dst) { VK_LOG_DEBUG("ggml_vk_arange(dst=" << dst << ", ne=" << ggml_nelements(dst) << ")"); @@ -12361,10 +12330,6 @@ bool ggml_vk_build_graph(ggml_backend_vk_context * ctx, ggml_cgraph * cgraph, in case GGML_OP_UPSCALE: ggml_vk_upscale(ctx, compute_ctx, src0, node); - break; - case GGML_OP_ADD1: - ggml_vk_add1(ctx, compute_ctx, src0, src1, node); - break; case GGML_OP_ARANGE: ggml_vk_arange(ctx, compute_ctx, node); @@ -15716,10 +15681,6 @@ static bool ggml_backend_vk_device_supports_op(ggml_backend_dev_t dev, const ggm case GGML_OP_CONCAT: { return ggml_vk_concat_supported(op->src[0], op->src[1], op); } - case GGML_OP_ADD1: - return (op->src[0]->type == GGML_TYPE_F32 && op->src[1]->type == GGML_TYPE_F32) - || (op->src[0]->type == GGML_TYPE_F16 && op->src[1]->type == GGML_TYPE_F32) - || (op->src[0]->type == GGML_TYPE_F16 && op->src[1]->type == GGML_TYPE_F16); case GGML_OP_ARANGE: return op->type == GGML_TYPE_F32; case GGML_OP_FILL: diff --git a/ggml/src/ggml.c b/ggml/src/ggml.c index 9bcabdbd34..3841a594cf 100644 --- a/ggml/src/ggml.c +++ b/ggml/src/ggml.c @@ -707,13 +707,13 @@ static const struct ggml_type_traits type_traits[GGML_TYPE_COUNT] = { .from_float_ref = (ggml_from_float_t) quantize_row_q4_1_ref, }, [4] = { // GGML_TYPE_Q4_2 - .type_name = "DEPRECATED", + .type_name = "REMOVED", .blck_size = 0, .type_size = 0, .is_quantized = false, }, [5] = { // GGML_TYPE_Q4_3 - .type_name = "DEPRECATED", + .type_name = "REMOVED", .blck_size = 0, .type_size = 0, .is_quantized = false, @@ -994,7 +994,6 @@ static const char * GGML_OP_NAME[GGML_OP_COUNT] = { "DUP", "ADD", "ADD_ID", - "ADD1", "ACC", "SUB", "MUL", @@ -1101,7 +1100,7 @@ static const char * GGML_OP_NAME[GGML_OP_COUNT] = { "GLU", }; -static_assert(GGML_OP_COUNT == 101, "GGML_OP_COUNT != 101"); +static_assert(GGML_OP_COUNT == 100, "GGML_OP_COUNT != 101"); static const char * GGML_OP_SYMBOL[GGML_OP_COUNT] = { "none", @@ -1109,7 +1108,6 @@ static const char * GGML_OP_SYMBOL[GGML_OP_COUNT] = { "x", "x+y", "x[i]+y", - "x+y", "view(x,nb,offset)+=y->x", "x-y", "x*y", @@ -1216,7 +1214,7 @@ static const char * GGML_OP_SYMBOL[GGML_OP_COUNT] = { "glu(x)", }; -static_assert(GGML_OP_COUNT == 101, "GGML_OP_COUNT != 101"); +static_assert(GGML_OP_COUNT == 100, "GGML_OP_COUNT != 101"); static_assert(GGML_OP_POOL_COUNT == 2, "GGML_OP_POOL_COUNT != 2"); @@ -1343,12 +1341,6 @@ size_t ggml_row_size(enum ggml_type type, int64_t ne) { return ggml_type_size(type)*ne/ggml_blck_size(type); } -double ggml_type_sizef(enum ggml_type type) { - assert(type >= 0); - assert(type < GGML_TYPE_COUNT); - return ((double)(type_traits[type].type_size))/type_traits[type].blck_size; -} - const char * ggml_type_name(enum ggml_type type) { assert(type >= 0); assert(type < GGML_TYPE_COUNT); @@ -2148,39 +2140,6 @@ struct ggml_tensor * ggml_add_id( return result; } -// ggml_add1 - -static struct ggml_tensor * ggml_add1_impl( - struct ggml_context * ctx, - struct ggml_tensor * a, - struct ggml_tensor * b, - bool inplace) { - GGML_ASSERT(ggml_is_scalar(b)); - GGML_ASSERT(ggml_is_padded_1d(a)); - - struct ggml_tensor * result = inplace ? ggml_view_tensor(ctx, a) : ggml_dup_tensor(ctx, a); - - result->op = GGML_OP_ADD1; - result->src[0] = a; - result->src[1] = b; - - return result; -} - -struct ggml_tensor * ggml_add1( - struct ggml_context * ctx, - struct ggml_tensor * a, - struct ggml_tensor * b) { - return ggml_add1_impl(ctx, a, b, false); -} - -struct ggml_tensor * ggml_add1_inplace( - struct ggml_context * ctx, - struct ggml_tensor * a, - struct ggml_tensor * b) { - return ggml_add1_impl(ctx, a, b, true); -} - // ggml_acc static struct ggml_tensor * ggml_acc_impl( @@ -3364,16 +3323,6 @@ struct ggml_tensor * ggml_mul_mat( return result; } -void ggml_mul_mat_set_prec( - struct ggml_tensor * a, - enum ggml_prec prec) { - GGML_ASSERT(a->op == GGML_OP_MUL_MAT); - - const int32_t prec_i32 = (int32_t) prec; - - ggml_set_op_params_i32(a, 0, prec_i32); -} - void ggml_mul_mat_set_hint( struct ggml_tensor * a, enum ggml_op_hint hint) { @@ -4435,44 +4384,6 @@ struct ggml_tensor * ggml_rope_ext_inplace( ); } -struct ggml_tensor * ggml_rope_custom( - struct ggml_context * ctx, - struct ggml_tensor * a, - struct ggml_tensor * b, - int n_dims, - int mode, - int n_ctx_orig, - float freq_base, - float freq_scale, - float ext_factor, - float attn_factor, - float beta_fast, - float beta_slow) { - return ggml_rope_impl( - ctx, a, b, NULL, n_dims, NULL, mode, n_ctx_orig, freq_base, freq_scale, - ext_factor, attn_factor, beta_fast, beta_slow, false - ); -} - -struct ggml_tensor * ggml_rope_custom_inplace( - struct ggml_context * ctx, - struct ggml_tensor * a, - struct ggml_tensor * b, - int n_dims, - int mode, - int n_ctx_orig, - float freq_base, - float freq_scale, - float ext_factor, - float attn_factor, - float beta_fast, - float beta_slow) { - return ggml_rope_impl( - ctx, a, b, NULL, n_dims, NULL, mode, n_ctx_orig, freq_base, freq_scale, - ext_factor, attn_factor, beta_fast, beta_slow, true - ); -} - // Apparently solving `n_rot = 2pi * x * base^((2 * max_pos_emb) / n_dims)` for x, we get // `corr_dim(n_rot) = n_dims * log(max_pos_emb / (n_rot * 2pi)) / (2 * log(base))` static float ggml_rope_yarn_corr_dim(int n_dims, int n_ctx_orig, float n_rot, float base) { @@ -5194,17 +5105,6 @@ struct ggml_tensor * ggml_upscale( return ggml_interpolate_impl(ctx, a, a->ne[0] * scale_factor, a->ne[1] * scale_factor, a->ne[2], a->ne[3], mode); } -struct ggml_tensor * ggml_upscale_ext( - struct ggml_context * ctx, - struct ggml_tensor * a, - int ne0, - int ne1, - int ne2, - int ne3, - enum ggml_scale_mode mode) { - return ggml_interpolate_impl(ctx, a, ne0, ne1, ne2, ne3, mode); -} - struct ggml_tensor * ggml_interpolate( struct ggml_context * ctx, struct ggml_tensor * a, @@ -5548,16 +5448,6 @@ struct ggml_tensor * ggml_flash_attn_ext( } -void ggml_flash_attn_ext_set_prec( - struct ggml_tensor * a, - enum ggml_prec prec) { - GGML_ASSERT(a->op == GGML_OP_FLASH_ATTN_EXT); - - const int32_t prec_i32 = (int32_t) prec; - - ggml_set_op_params_i32(a, 3, prec_i32); // scale is on first pos, max_bias on second -} - enum ggml_prec ggml_flash_attn_ext_get_prec( const struct ggml_tensor * a) { GGML_ASSERT(a->op == GGML_OP_FLASH_ATTN_EXT); @@ -6725,22 +6615,6 @@ static void ggml_acc_or_set( ggml_build_forward_expand(cgraph, cgraph->grads[isrc]); } -static void ggml_add1_or_set( - struct ggml_context * ctx, - struct ggml_cgraph * cgraph, - size_t isrc, - struct ggml_tensor * tensor) { - struct ggml_tensor * src = cgraph->visited_hash_set.keys[isrc]; - GGML_ASSERT(src); - if (cgraph->grads[isrc]) { - cgraph->grads[isrc] = ggml_add1_impl(ctx, cgraph->grads[isrc], tensor, cgraph->grad_accs[isrc]); - } else { - cgraph->grads[isrc] = ggml_repeat(ctx, tensor, src); - } - ggml_format_name(cgraph->grads[isrc], "grad for %s", src->name); - ggml_build_forward_expand(cgraph, cgraph->grads[isrc]); -} - static void ggml_sub_or_set( struct ggml_context * ctx, struct ggml_cgraph * cgraph, @@ -6795,14 +6669,6 @@ static void ggml_compute_backward( ggml_add_or_set(ctx, cgraph, isrc1, tmp); } } break; - case GGML_OP_ADD1: { - if (src0_needs_grads) { - ggml_add_or_set(ctx, cgraph, isrc0, grad); - } - if (src1_needs_grads) { - ggml_add_or_set(ctx, cgraph, isrc1, ggml_mean(ctx, grad)); // TODO: should probably be sum instead of mean - } - } break; case GGML_OP_ACC: { if (src0_needs_grads) { ggml_add_or_set(ctx, cgraph, isrc0, grad); @@ -6875,7 +6741,7 @@ static void ggml_compute_backward( } break; case GGML_OP_SUM: { if (src0_needs_grads) { - ggml_add1_or_set(ctx, cgraph, isrc0, grad); + ggml_add_or_set(ctx, cgraph, isrc0, grad); } } break; case GGML_OP_SUM_ROWS: { @@ -6885,7 +6751,7 @@ static void ggml_compute_backward( } break; case GGML_OP_MEAN: { if (src0_needs_grads) { - ggml_add1_or_set(ctx, cgraph, isrc0, ggml_scale_impl(ctx, grad, 1.0f/src0->ne[0], 0.0, false)); + ggml_add_or_set(ctx, cgraph, isrc0, ggml_scale_impl(ctx, grad, 1.0f/src0->ne[0], 0.0, false)); } } break; case GGML_OP_REPEAT: { diff --git a/tests/test-backend-ops.cpp b/tests/test-backend-ops.cpp index 081fb23ff1..2a1f939f11 100644 --- a/tests/test-backend-ops.cpp +++ b/tests/test-backend-ops.cpp @@ -11060,8 +11060,8 @@ static std::vector> make_test_cases_eval() { for (int kv : { 113, 512, 1024, }) { if (nr2 != 1 && kv != 512) continue; for (int nb : { 1, 3, 32, 75, }) { - for (ggml_prec prec : {GGML_PREC_F32, GGML_PREC_DEFAULT}) { - if (hsk != 128 && prec == GGML_PREC_DEFAULT) continue; + for (ggml_prec prec : {GGML_PREC_F32, GGML_PREC_UNDEFINED}) { + if (hsk != 128 && prec == GGML_PREC_UNDEFINED) continue; for (ggml_type type_KV : {GGML_TYPE_F32, GGML_TYPE_F16, GGML_TYPE_BF16, GGML_TYPE_Q8_0, GGML_TYPE_Q5_1, GGML_TYPE_Q5_0, GGML_TYPE_Q4_1, GGML_TYPE_Q4_0, GGML_TYPE_IQ4_NL}) { if (type_KV != GGML_TYPE_F16 && hsk != 64 && hsk != 72) continue; // DeepSeek MLA: the V cache is a sub-view of the K cache