mirror of
https://github.com/ggml-org/llama.cpp.git
synced 2026-09-30 18:07:38 -05:00
server : support typed content (vision/audio/video) input for /v1/embeddings endpoint (#29556)
* server : support multimodal input for /v1/embeddings (Qwen3-VL-Embedding)
Accept the OpenAI-style wrapped content array format for multimodal
embedding requests. Each {"content": [...]} object is one input that
produces one embedding; text parts are concatenated and image_url parts
are decoded via handle_media then spliced with process_mtmd_prompt.
The legacy formats (plain string, token arrays, mixed arrays, and the
{prompt_string, multimodal_data} object) continue to work unchanged via
tokenize_input_prompts. Bare content arrays (the unwrapped shape) are
rejected with a migration message.
Also disables KV prefix reuse for stateless embedding/rerank tasks so
that repeated inputs do not incorrectly share cached KV across requests.
Assisted-by: Opencode Qwen3.8 27B
* clean up comments and docs
* refactor
* add tests
* support video and audio inp
---------
Co-authored-by: timothywang21 <timothywang21@users.noreply.github.com>
Co-authored-by: Xuan Son Nguyen <son@huggingface.co>
This commit is contained in:
co-authored by
timothywang21
Xuan Son Nguyen
parent
66e665c427
commit
680a036285
@@ -3193,7 +3193,9 @@ private:
|
||||
return;
|
||||
}
|
||||
|
||||
if (slot.task->params.cache_prompt) {
|
||||
const bool is_stateless_task = slot.task->type == SERVER_TASK_TYPE_EMBEDDING || slot.task->type == SERVER_TASK_TYPE_RERANK;
|
||||
|
||||
if (slot.task->params.cache_prompt && !is_stateless_task) {
|
||||
// reuse any previously computed tokens that are common with the new prompt
|
||||
n_past = slot.prompt.tokens.get_common_prefix(input_tokens);
|
||||
|
||||
@@ -5403,7 +5405,27 @@ std::unique_ptr<server_res_generator> server_routes::handle_embeddings_impl(cons
|
||||
}
|
||||
}
|
||||
|
||||
auto tokenized_prompts = tokenize_input_prompts(ctx_server.vocab, ctx_server.mctx, prompt, true, true, ctx_server.init_opt);
|
||||
// same shapes as tokenize_input_prompts(), plus OAI content: { "content": [ { "type": "text"|"image_url"|"input_audio"|"input_video", ... } ] }
|
||||
auto tokenize_entry = [&](const json & p) {
|
||||
if (p.is_object() && p.contains("content")) {
|
||||
return tokenize_oai_content_array(ctx_server.vocab, ctx_server.mctx, meta->chat_params, p.at("content"), true, true, ctx_server.init_opt);
|
||||
}
|
||||
return tokenize_input_subprompt(ctx_server.vocab, ctx_server.mctx, p, true, true, ctx_server.init_opt);
|
||||
};
|
||||
|
||||
std::vector<server_tokens> tokenized_prompts;
|
||||
if (prompt.is_array() && !json_is_array_and_contains_numbers(prompt)) {
|
||||
for (const auto & p : prompt) {
|
||||
tokenized_prompts.push_back(tokenize_entry(p));
|
||||
}
|
||||
} else {
|
||||
tokenized_prompts.push_back(tokenize_entry(prompt));
|
||||
}
|
||||
if (tokenized_prompts.empty()) {
|
||||
res->error(format_error_response("\"input\" must not be empty", ERROR_TYPE_INVALID_REQUEST));
|
||||
return res;
|
||||
}
|
||||
|
||||
for (const auto & tokens : tokenized_prompts) {
|
||||
// this check is necessary for models that do not add BOS token to the input
|
||||
if (tokens.empty()) {
|
||||
|
||||
Reference in New Issue
Block a user