diff --git a/src/llama-ext.h b/src/llama-ext.h index 348bbae957..e1d8034b75 100644 --- a/src/llama-ext.h +++ b/src/llama-ext.h @@ -124,3 +124,8 @@ LLAMA_API llama_context * llama_get_ctx_other(struct llama_context * ctx); LLAMA_API const int32_t * llama_model_target_layer_ids (const struct llama_model * model); // returns the number of extracted layers from target model LLAMA_API uint32_t llama_model_target_layer_ids_n(const struct llama_model * model); + +// retrieves the whole token embedding matrix in F32 format (n_embd x n_vocab) +// returns total number of elements (usually n_embd * n_vocab) or 0 on error +// if out is nullptr, returns the number of tokens without writing to out +LLAMA_API uint32_t llama_model_get_tok_embd(const struct llama_model * model, float * out); diff --git a/src/llama-model.cpp b/src/llama-model.cpp index 68c4cbafd4..301a73a3fd 100644 --- a/src/llama-model.cpp +++ b/src/llama-model.cpp @@ -2847,3 +2847,38 @@ const int32_t * llama_model_target_layer_ids(const struct llama_model * model) { uint32_t llama_model_target_layer_ids_n(const struct llama_model * model) { return (uint32_t) model->target_layer_ids.size(); } + +uint32_t llama_model_get_tok_embd(const struct llama_model * model, float * out) { + if (model->vocab.n_tokens() == 0 || model->tok_embd == nullptr) { + return 0; + } + + const ggml_tensor * tensor = model->tok_embd; + const size_t nelements = ggml_nelements(tensor); + GGML_ASSERT(nelements <= UINT32_MAX); // for the return type + + if (out == nullptr) { + return (uint32_t) nelements; + } + + if (tensor->type == GGML_TYPE_F32) { + ggml_backend_tensor_get(tensor, out, 0, nelements * sizeof(float)); + return (uint32_t) nelements; + } + + std::vector buf(ggml_nbytes(tensor)); + ggml_backend_tensor_get(tensor, buf.data(), 0, buf.size()); + + const ggml_type_traits * traits = ggml_get_type_traits(tensor->type); + if (tensor->type == GGML_TYPE_F16) { + ggml_fp16_to_fp32_row((const ggml_fp16_t *) buf.data(), out, nelements); + } else if (tensor->type == GGML_TYPE_BF16) { + ggml_bf16_to_fp32_row((const ggml_bf16_t *) buf.data(), out, nelements); + } else if (ggml_is_quantized(tensor->type) && traits->to_float != nullptr) { + traits->to_float(buf.data(), out, nelements); + } else { + GGML_ABORT("unsupported tensor type for dequantization: %s", ggml_type_name(tensor->type)); + } + + return (uint32_t) nelements; +}