mirror of
https://github.com/ggml-org/llama.cpp.git
synced 2026-10-02 02:47:26 -05:00
add vision support
This commit is contained in:
@@ -346,6 +346,7 @@ MMPROJ_MODEL_MAP: dict[str, str] = {
|
||||
"Qwen3TTSForConditionalGeneration": "qwen3tts",
|
||||
"Qwen3VLForConditionalGeneration": "qwen3vl",
|
||||
"Qwen3VLMoeForConditionalGeneration": "qwen3vl",
|
||||
"OpenJevModel": "qwen3vl",
|
||||
"Qwen3_5ForConditionalGeneration": "qwen3vl",
|
||||
"Qwen3_5MoeForConditionalGeneration": "qwen3vl",
|
||||
"Qwen4ExpForConditionalGeneration": "qwen4exp",
|
||||
|
||||
+6
-1
@@ -690,9 +690,14 @@ class OpenJevModel(Qwen3_5TextModel):
|
||||
"{% elif o.key == 'true' %}yes: {% if o.description %}" + description + "{% else %}The statement is true.{% endif %}"
|
||||
"{% else %}no: {% if o.description %}" + description + "{% else %}The statement is false.{% endif %}{% endif %}"
|
||||
)
|
||||
# TODO: only the layout with one image is known (image first), the one with several images is not verified
|
||||
images = (
|
||||
"{% for image in images %}{{ image }}{% endfor %}"
|
||||
"{% if images %}{{ 'The screenshot shows the current screen.\\n' }}{% endif %}"
|
||||
)
|
||||
return (
|
||||
"{% set letters = '" + self._LETTERS + "' %}"
|
||||
"<|im_start|>user\nState:\n" + jinja_str_or_json("state") + "\n\nQuestion: " + jinja_str_or_json("instructions")
|
||||
"<|im_start|>user\n" + images + "State:\n" + jinja_str_or_json("state") + "\n\nQuestion: " + jinja_str_or_json("instructions")
|
||||
+ "{% if type == 'score' %} Rate along the ordered levels below (lowest first).{% endif %}"
|
||||
"{{ '\\nOptions:\\n' }}"
|
||||
"{% for o in options %}[{{ letters[loop.index0] }}] " + option + "{{ '\\n' }}{% endfor %}"
|
||||
|
||||
@@ -13,7 +13,7 @@ from .qwen import Qwen3Model, Qwen3MoeModel
|
||||
from .qwenvl import Qwen25AudioModel
|
||||
|
||||
|
||||
@ModelBase.register("Qwen3VLForConditionalGeneration", "Qwen3VLMoeForConditionalGeneration", "Qwen3_5ForConditionalGeneration", "Qwen3_5MoeForConditionalGeneration")
|
||||
@ModelBase.register("Qwen3VLForConditionalGeneration", "Qwen3VLMoeForConditionalGeneration", "Qwen3_5ForConditionalGeneration", "Qwen3_5MoeForConditionalGeneration", "OpenJevModel")
|
||||
@ModelBase.example("Qwen/Qwen3-VL-4B-Instruct", "Qwen/Qwen3-VL-30B-A3B-Instruct", "Qwen/Qwen3.5-9B", "Qwen/Qwen3.5-35B-A3B")
|
||||
class Qwen3VLVisionModel(MmprojModel):
|
||||
def __init__(self, *args, **kwargs):
|
||||
|
||||
+31
-1
@@ -1677,6 +1677,8 @@ Follows the [TypeSafe API](https://docs.typesafe.ai/api), streaming is not suppo
|
||||
|
||||
`state`: The content to evaluate. Can be a string, an object or an array. A value that is not a string is given to the model as JSON text.
|
||||
|
||||
`images`: Optional. An array of up to 8 images, each one is a data URL (`data:image/...;base64,...`). See the image input section below.
|
||||
|
||||
`questions`: An object that maps a question id to a question. Each question has these fields:
|
||||
|
||||
- `type`: One of `choice`, `score`, `noul`.
|
||||
@@ -1690,6 +1692,17 @@ The questions of a request are answered independently, an answer does not depend
|
||||
|
||||
The number of options of a `choice` question is limited by the model, for example: 52 for openjev, 255 for laya. For laya, long questions and options are truncated to the token budget the model was trained with.
|
||||
|
||||
*Image input:*
|
||||
|
||||
Image input needs a model that supports it (for example: openjev) and its multimodal projector, see `--mmproj`.
|
||||
|
||||
Images can be given in two ways, and both can be used in the same request:
|
||||
|
||||
- The `images` field.
|
||||
- A `state` made of chat messages, either an array of messages or an object with a `messages` array. An `image_url` part in the `content` of a message is taken as an image, in the same format as chat completions. Only data URLs are accepted.
|
||||
|
||||
All the images are placed before the state in the prompt, the ones from `images` first. The image parts are removed from the state. A request can have at most 8 images in total.
|
||||
|
||||
*Response:*
|
||||
|
||||
`answers`: An object that maps each question id to its answer. The fields depend on the question type:
|
||||
@@ -1767,7 +1780,24 @@ Response (values are shortened):
|
||||
}
|
||||
```
|
||||
|
||||
An invalid request returns the error `400`. A model that is not a decision model returns the error `501`.
|
||||
Example with an image:
|
||||
|
||||
```shell
|
||||
curl http://127.0.0.1:8080/v1/systemone \
|
||||
-H "Content-Type: application/json" \
|
||||
-d '{
|
||||
"state": "The document was received by the accounting team this morning.",
|
||||
"images": ["data:image/jpeg;base64,/9j/4AAQSkZJRg..."],
|
||||
"questions": {
|
||||
"has_table": {
|
||||
"type": "noul",
|
||||
"instructions": "Does the image contain a table?"
|
||||
}
|
||||
}
|
||||
}' | jq
|
||||
```
|
||||
|
||||
An invalid request returns the error `400`. A model that is not a decision model returns the error `501`. A request with images returns the error `501` if the model does not support image input, or if no multimodal projector is loaded.
|
||||
|
||||
## Server tools
|
||||
|
||||
|
||||
@@ -1076,7 +1076,7 @@ json oaicompat_completion_params_parse(const json & body) {
|
||||
// - file:// for local files (only allowed if media_path is set)
|
||||
// - data: for base64 encoded data with uri scheme (e.g. data:image/png;base64,...)
|
||||
// - raw base64 encoded data
|
||||
static void handle_media(
|
||||
void handle_media(
|
||||
std::vector<raw_buffer> & out_files,
|
||||
const std::string & url,
|
||||
const std::string & media_path) {
|
||||
|
||||
@@ -270,6 +270,12 @@ llama_tokens tokenize_mixed(const llama_vocab * vocab, const json & json_prompt,
|
||||
// if validate_utf8(text) == text.size(), then the whole text is valid utf8
|
||||
size_t validate_utf8(const std::string& text);
|
||||
|
||||
// load a media file from an URL (http, file, data) or from raw base64 data
|
||||
void handle_media(
|
||||
std::vector<raw_buffer> & out_files,
|
||||
const std::string & url,
|
||||
const std::string & media_path);
|
||||
|
||||
// process mtmd prompt, return the server_tokens containing both text tokens and media chunks
|
||||
// if is_placeholder is true, the media chunk will be treated as placeholder for counting tokens; the output tokens are not usable for actual inference (e.g. for submitting a task to server_queue)
|
||||
server_tokens process_mtmd_prompt(
|
||||
|
||||
@@ -5332,6 +5332,13 @@ void server_routes::init_routes() {
|
||||
const json body = json::parse(req.body);
|
||||
const auto questions = decision.parse_questions(body);
|
||||
|
||||
std::vector<raw_buffer> files;
|
||||
const json state = decision.parse_state(body, files);
|
||||
if (!files.empty() && (!decision.can_use_images() || !meta->has_inp_image)) {
|
||||
res->error(format_error_response("This server does not support image input for decisions. For a model that supports it, start it with `--mmproj`", ERROR_TYPE_NOT_SUPPORTED));
|
||||
return res;
|
||||
}
|
||||
|
||||
// one task per question
|
||||
auto & rd = res->rd;
|
||||
{
|
||||
@@ -5340,7 +5347,7 @@ void server_routes::init_routes() {
|
||||
for (const auto & question : questions) {
|
||||
server_task task = server_task(SERVER_TASK_TYPE_DECISION);
|
||||
task.id = rd.get_new_id();
|
||||
decision.fill_task(body.at("state"), question, task);
|
||||
decision.fill_task(state, question, files, ctx_server.mctx, ctx_server.init_opt, task);
|
||||
tasks.push_back(std::move(task));
|
||||
}
|
||||
if (decision.can_share_prompt()) {
|
||||
|
||||
@@ -164,6 +164,68 @@ std::vector<server_decision_question> server_decision_context::parse_questions(c
|
||||
return questions;
|
||||
}
|
||||
|
||||
//
|
||||
// images
|
||||
//
|
||||
|
||||
static const size_t DECISION_MAX_IMAGES = 8;
|
||||
|
||||
static void decision_load_image(const json & url, std::vector<raw_buffer> & files) {
|
||||
if (!url.is_string() || !string_starts_with(url.get<std::string>(), "data:image/")) {
|
||||
throw std::invalid_argument("images must be data URLs (data:image/...;base64,...)");
|
||||
}
|
||||
if (files.size() >= DECISION_MAX_IMAGES) {
|
||||
throw std::invalid_argument(string_format("too many images, the maximum is %zu", DECISION_MAX_IMAGES));
|
||||
}
|
||||
handle_media(files, url.get<std::string>(), "");
|
||||
}
|
||||
|
||||
json server_decision_context::parse_state(const json & body, std::vector<raw_buffer> & files) const {
|
||||
if (body.contains("images") && !body.at("images").is_null()) {
|
||||
if (!body.at("images").is_array()) {
|
||||
throw std::invalid_argument("\"images\" must be an array");
|
||||
}
|
||||
for (const auto & url : body.at("images")) {
|
||||
decision_load_image(url, files);
|
||||
}
|
||||
}
|
||||
|
||||
const json & state = body.at("state");
|
||||
const bool is_wrapped = state.is_object() && state.contains("messages");
|
||||
const json & messages = is_wrapped ? state.at("messages") : state;
|
||||
if (!messages.is_array()) {
|
||||
return state;
|
||||
}
|
||||
|
||||
// chat messages: take the image parts out of the content
|
||||
json messages_out = json::array();
|
||||
for (const auto & msg : messages) {
|
||||
if (!msg.is_object() || !msg.contains("content") || !msg.at("content").is_array()) {
|
||||
messages_out.push_back(msg);
|
||||
continue;
|
||||
}
|
||||
json content = json::array();
|
||||
for (const auto & part : msg.at("content")) {
|
||||
if (part.is_object() && json_value(part, "type", std::string()) == "image_url" && part.contains("image_url")) {
|
||||
const json & image_url = part.at("image_url");
|
||||
decision_load_image(image_url.is_object() && image_url.contains("url") ? image_url.at("url") : image_url, files);
|
||||
} else {
|
||||
content.push_back(part);
|
||||
}
|
||||
}
|
||||
json msg_out = msg;
|
||||
msg_out["content"] = content;
|
||||
messages_out.push_back(msg_out);
|
||||
}
|
||||
|
||||
if (!is_wrapped) {
|
||||
return messages_out;
|
||||
}
|
||||
json state_out = state;
|
||||
state_out["messages"] = messages_out;
|
||||
return state_out;
|
||||
}
|
||||
|
||||
//
|
||||
// prompt
|
||||
//
|
||||
@@ -192,7 +254,7 @@ static json decision_replace_text(const json & val, const std::string & search,
|
||||
return val;
|
||||
}
|
||||
|
||||
std::string server_decision_context::render(const json & state, const server_decision_question & question) const {
|
||||
std::string server_decision_context::render(const json & state, const server_decision_question & question, size_t n_images) const {
|
||||
json options = json::array();
|
||||
for (const auto & opt : question.options) {
|
||||
options.push_back(json{
|
||||
@@ -214,6 +276,16 @@ std::string server_decision_context::render(const json & state, const server_dec
|
||||
inp = decision_replace_text(inp, text_marker, " ");
|
||||
}
|
||||
|
||||
// the template puts one media marker per image
|
||||
json images = json::array();
|
||||
if (n_images > 0) {
|
||||
inp = decision_replace_text(inp, get_media_marker(), " ");
|
||||
for (size_t i = 0; i < n_images; i++) {
|
||||
images.push_back(get_media_marker());
|
||||
}
|
||||
}
|
||||
inp["images"] = images;
|
||||
|
||||
jinja::context ctx(tmpl->source());
|
||||
jinja::global_from_json(ctx, inp, false);
|
||||
jinja::runtime runtime(ctx);
|
||||
@@ -221,15 +293,27 @@ std::string server_decision_context::render(const json & state, const server_dec
|
||||
return jinja::runtime::gather_string_parts(results)->as_string().str();
|
||||
}
|
||||
|
||||
void server_decision_context::fill_task(const json & state, const server_decision_question & question, server_task & task) const {
|
||||
llama_tokens tokens = common_tokenize(vocab, render(state, question), false, true);
|
||||
void server_decision_context::fill_task(
|
||||
const json & state,
|
||||
const server_decision_question & question,
|
||||
const std::vector<raw_buffer> & files,
|
||||
mtmd_context * mctx,
|
||||
const mtmd_helper_init_opt & init_opt,
|
||||
server_task & task) const {
|
||||
const std::string prompt = render(state, question, files.size());
|
||||
|
||||
if (type == COMMON_DECISION_TYPE_OPENJEV) {
|
||||
task.decision.labels.assign(labels.begin(), labels.begin() + question.options.size());
|
||||
} else {
|
||||
fill_task_laya(tokens, question, task);
|
||||
if (!files.empty()) {
|
||||
task.tokens = process_mtmd_prompt(mctx, prompt, files, init_opt);
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
llama_tokens tokens = common_tokenize(vocab, prompt, false, true);
|
||||
if (type == COMMON_DECISION_TYPE_LAYA) {
|
||||
fill_task_laya(tokens, question, task);
|
||||
}
|
||||
task.tokens = server_tokens(tokens, false);
|
||||
}
|
||||
|
||||
|
||||
@@ -36,13 +36,41 @@ struct server_decision_context {
|
||||
void init(const llama_model * model);
|
||||
|
||||
// true if the questions of a request start with the same tokens, and the model can continue from them
|
||||
bool can_share_prompt() const { return type == COMMON_DECISION_TYPE_OPENJEV; }
|
||||
bool can_share_prompt() const {
|
||||
switch (type) {
|
||||
case COMMON_DECISION_TYPE_OPENJEV:
|
||||
return true;
|
||||
default:
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
// true if the prompt of the model has a place for images
|
||||
bool can_use_images() const {
|
||||
switch (type) {
|
||||
case COMMON_DECISION_TYPE_OPENJEV:
|
||||
return true;
|
||||
default:
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
// throw std::invalid_argument on bad input
|
||||
std::vector<server_decision_question> parse_questions(const json & body) const;
|
||||
|
||||
// returns the state without its images, they are appended to files in order
|
||||
// images come from "images" and from the image_url parts of a state made of chat messages
|
||||
json parse_state(const json & body, std::vector<raw_buffer> & files) const;
|
||||
|
||||
// set the prompt of this question, and where to read its result
|
||||
void fill_task(const json & state, const server_decision_question & question, server_task & task) const;
|
||||
// mctx is only used if there are files
|
||||
void fill_task(
|
||||
const json & state,
|
||||
const server_decision_question & question,
|
||||
const std::vector<raw_buffer> & files,
|
||||
mtmd_context * mctx,
|
||||
const mtmd_helper_init_opt & init_opt,
|
||||
server_task & task) const;
|
||||
|
||||
// scores: one raw model output per option
|
||||
json format_answer(const server_decision_question & question, const std::vector<float> & scores) const;
|
||||
@@ -65,7 +93,7 @@ private:
|
||||
size_t max_head_tokens = 0; // question + options
|
||||
size_t max_option_tokens = 48;
|
||||
|
||||
std::string render(const json & state, const server_decision_question & question) const;
|
||||
std::string render(const json & state, const server_decision_question & question, size_t n_images) const;
|
||||
void fill_task_laya(llama_tokens & tokens, const server_decision_question & question, server_task & task) const;
|
||||
|
||||
float get_temperature(const server_decision_question & question) const;
|
||||
|
||||
@@ -110,3 +110,4 @@ def test_systemone_invalid_request(data: dict):
|
||||
# TODO: test the shared prompt prefix, it needs a small model of a type that supports it (e.g. openjev)
|
||||
# it can be checked with GET /metrics: for one request, prompt_tokens_cached_total must grow by
|
||||
# (shared tokens * number of child tasks) and prompt_tokens_total + prompt_tokens_cached_total == usage.input_tokens
|
||||
# TODO: test the image input ("images" and image_url parts of a chat-message state), it needs a small model with a mmproj
|
||||
|
||||
Reference in New Issue
Block a user