diff --git a/conversion/__init__.py b/conversion/__init__.py index 87fbf2ee76..fac5d95a9f 100644 --- a/conversion/__init__.py +++ b/conversion/__init__.py @@ -346,6 +346,7 @@ MMPROJ_MODEL_MAP: dict[str, str] = { "Qwen3TTSForConditionalGeneration": "qwen3tts", "Qwen3VLForConditionalGeneration": "qwen3vl", "Qwen3VLMoeForConditionalGeneration": "qwen3vl", + "OpenJevModel": "qwen3vl", "Qwen3_5ForConditionalGeneration": "qwen3vl", "Qwen3_5MoeForConditionalGeneration": "qwen3vl", "Qwen4ExpForConditionalGeneration": "qwen4exp", diff --git a/conversion/qwen.py b/conversion/qwen.py index edc77e817f..0b39187337 100644 --- a/conversion/qwen.py +++ b/conversion/qwen.py @@ -690,9 +690,14 @@ class OpenJevModel(Qwen3_5TextModel): "{% elif o.key == 'true' %}yes: {% if o.description %}" + description + "{% else %}The statement is true.{% endif %}" "{% else %}no: {% if o.description %}" + description + "{% else %}The statement is false.{% endif %}{% endif %}" ) + # TODO: only the layout with one image is known (image first), the one with several images is not verified + images = ( + "{% for image in images %}{{ image }}{% endfor %}" + "{% if images %}{{ 'The screenshot shows the current screen.\\n' }}{% endif %}" + ) return ( "{% set letters = '" + self._LETTERS + "' %}" - "<|im_start|>user\nState:\n" + jinja_str_or_json("state") + "\n\nQuestion: " + jinja_str_or_json("instructions") + "<|im_start|>user\n" + images + "State:\n" + jinja_str_or_json("state") + "\n\nQuestion: " + jinja_str_or_json("instructions") + "{% if type == 'score' %} Rate along the ordered levels below (lowest first).{% endif %}" "{{ '\\nOptions:\\n' }}" "{% for o in options %}[{{ letters[loop.index0] }}] " + option + "{{ '\\n' }}{% endfor %}" diff --git a/conversion/qwen3vl.py b/conversion/qwen3vl.py index 11ce68515b..638bcbb246 100644 --- a/conversion/qwen3vl.py +++ b/conversion/qwen3vl.py @@ -13,7 +13,7 @@ from .qwen import Qwen3Model, Qwen3MoeModel from .qwenvl import Qwen25AudioModel -@ModelBase.register("Qwen3VLForConditionalGeneration", "Qwen3VLMoeForConditionalGeneration", "Qwen3_5ForConditionalGeneration", "Qwen3_5MoeForConditionalGeneration") +@ModelBase.register("Qwen3VLForConditionalGeneration", "Qwen3VLMoeForConditionalGeneration", "Qwen3_5ForConditionalGeneration", "Qwen3_5MoeForConditionalGeneration", "OpenJevModel") @ModelBase.example("Qwen/Qwen3-VL-4B-Instruct", "Qwen/Qwen3-VL-30B-A3B-Instruct", "Qwen/Qwen3.5-9B", "Qwen/Qwen3.5-35B-A3B") class Qwen3VLVisionModel(MmprojModel): def __init__(self, *args, **kwargs): diff --git a/tools/server/README.md b/tools/server/README.md index e0005247a7..30218ee16a 100644 --- a/tools/server/README.md +++ b/tools/server/README.md @@ -1677,6 +1677,8 @@ Follows the [TypeSafe API](https://docs.typesafe.ai/api), streaming is not suppo `state`: The content to evaluate. Can be a string, an object or an array. A value that is not a string is given to the model as JSON text. +`images`: Optional. An array of up to 8 images, each one is a data URL (`data:image/...;base64,...`). See the image input section below. + `questions`: An object that maps a question id to a question. Each question has these fields: - `type`: One of `choice`, `score`, `noul`. @@ -1690,6 +1692,17 @@ The questions of a request are answered independently, an answer does not depend The number of options of a `choice` question is limited by the model, for example: 52 for openjev, 255 for laya. For laya, long questions and options are truncated to the token budget the model was trained with. +*Image input:* + +Image input needs a model that supports it (for example: openjev) and its multimodal projector, see `--mmproj`. + +Images can be given in two ways, and both can be used in the same request: + +- The `images` field. +- A `state` made of chat messages, either an array of messages or an object with a `messages` array. An `image_url` part in the `content` of a message is taken as an image, in the same format as chat completions. Only data URLs are accepted. + +All the images are placed before the state in the prompt, the ones from `images` first. The image parts are removed from the state. A request can have at most 8 images in total. + *Response:* `answers`: An object that maps each question id to its answer. The fields depend on the question type: @@ -1767,7 +1780,24 @@ Response (values are shortened): } ``` -An invalid request returns the error `400`. A model that is not a decision model returns the error `501`. +Example with an image: + +```shell +curl http://127.0.0.1:8080/v1/systemone \ + -H "Content-Type: application/json" \ + -d '{ + "state": "The document was received by the accounting team this morning.", + "images": ["data:image/jpeg;base64,/9j/4AAQSkZJRg..."], + "questions": { + "has_table": { + "type": "noul", + "instructions": "Does the image contain a table?" + } + } + }' | jq +``` + +An invalid request returns the error `400`. A model that is not a decision model returns the error `501`. A request with images returns the error `501` if the model does not support image input, or if no multimodal projector is loaded. ## Server tools diff --git a/tools/server/server-common.cpp b/tools/server/server-common.cpp index a37826abea..2d90fcad2d 100644 --- a/tools/server/server-common.cpp +++ b/tools/server/server-common.cpp @@ -1076,7 +1076,7 @@ json oaicompat_completion_params_parse(const json & body) { // - file:// for local files (only allowed if media_path is set) // - data: for base64 encoded data with uri scheme (e.g. data:image/png;base64,...) // - raw base64 encoded data -static void handle_media( +void handle_media( std::vector & out_files, const std::string & url, const std::string & media_path) { diff --git a/tools/server/server-common.h b/tools/server/server-common.h index 8cb6b90da6..7e4ebed2a1 100644 --- a/tools/server/server-common.h +++ b/tools/server/server-common.h @@ -270,6 +270,12 @@ llama_tokens tokenize_mixed(const llama_vocab * vocab, const json & json_prompt, // if validate_utf8(text) == text.size(), then the whole text is valid utf8 size_t validate_utf8(const std::string& text); +// load a media file from an URL (http, file, data) or from raw base64 data +void handle_media( + std::vector & out_files, + const std::string & url, + const std::string & media_path); + // process mtmd prompt, return the server_tokens containing both text tokens and media chunks // if is_placeholder is true, the media chunk will be treated as placeholder for counting tokens; the output tokens are not usable for actual inference (e.g. for submitting a task to server_queue) server_tokens process_mtmd_prompt( diff --git a/tools/server/server-context.cpp b/tools/server/server-context.cpp index e51dd19b80..cda543f8b9 100644 --- a/tools/server/server-context.cpp +++ b/tools/server/server-context.cpp @@ -5332,6 +5332,13 @@ void server_routes::init_routes() { const json body = json::parse(req.body); const auto questions = decision.parse_questions(body); + std::vector files; + const json state = decision.parse_state(body, files); + if (!files.empty() && (!decision.can_use_images() || !meta->has_inp_image)) { + res->error(format_error_response("This server does not support image input for decisions. For a model that supports it, start it with `--mmproj`", ERROR_TYPE_NOT_SUPPORTED)); + return res; + } + // one task per question auto & rd = res->rd; { @@ -5340,7 +5347,7 @@ void server_routes::init_routes() { for (const auto & question : questions) { server_task task = server_task(SERVER_TASK_TYPE_DECISION); task.id = rd.get_new_id(); - decision.fill_task(body.at("state"), question, task); + decision.fill_task(state, question, files, ctx_server.mctx, ctx_server.init_opt, task); tasks.push_back(std::move(task)); } if (decision.can_share_prompt()) { diff --git a/tools/server/server-decision.cpp b/tools/server/server-decision.cpp index 469ebabcc7..014104293c 100644 --- a/tools/server/server-decision.cpp +++ b/tools/server/server-decision.cpp @@ -164,6 +164,68 @@ std::vector server_decision_context::parse_questions(c return questions; } +// +// images +// + +static const size_t DECISION_MAX_IMAGES = 8; + +static void decision_load_image(const json & url, std::vector & files) { + if (!url.is_string() || !string_starts_with(url.get(), "data:image/")) { + throw std::invalid_argument("images must be data URLs (data:image/...;base64,...)"); + } + if (files.size() >= DECISION_MAX_IMAGES) { + throw std::invalid_argument(string_format("too many images, the maximum is %zu", DECISION_MAX_IMAGES)); + } + handle_media(files, url.get(), ""); +} + +json server_decision_context::parse_state(const json & body, std::vector & files) const { + if (body.contains("images") && !body.at("images").is_null()) { + if (!body.at("images").is_array()) { + throw std::invalid_argument("\"images\" must be an array"); + } + for (const auto & url : body.at("images")) { + decision_load_image(url, files); + } + } + + const json & state = body.at("state"); + const bool is_wrapped = state.is_object() && state.contains("messages"); + const json & messages = is_wrapped ? state.at("messages") : state; + if (!messages.is_array()) { + return state; + } + + // chat messages: take the image parts out of the content + json messages_out = json::array(); + for (const auto & msg : messages) { + if (!msg.is_object() || !msg.contains("content") || !msg.at("content").is_array()) { + messages_out.push_back(msg); + continue; + } + json content = json::array(); + for (const auto & part : msg.at("content")) { + if (part.is_object() && json_value(part, "type", std::string()) == "image_url" && part.contains("image_url")) { + const json & image_url = part.at("image_url"); + decision_load_image(image_url.is_object() && image_url.contains("url") ? image_url.at("url") : image_url, files); + } else { + content.push_back(part); + } + } + json msg_out = msg; + msg_out["content"] = content; + messages_out.push_back(msg_out); + } + + if (!is_wrapped) { + return messages_out; + } + json state_out = state; + state_out["messages"] = messages_out; + return state_out; +} + // // prompt // @@ -192,7 +254,7 @@ static json decision_replace_text(const json & val, const std::string & search, return val; } -std::string server_decision_context::render(const json & state, const server_decision_question & question) const { +std::string server_decision_context::render(const json & state, const server_decision_question & question, size_t n_images) const { json options = json::array(); for (const auto & opt : question.options) { options.push_back(json{ @@ -214,6 +276,16 @@ std::string server_decision_context::render(const json & state, const server_dec inp = decision_replace_text(inp, text_marker, " "); } + // the template puts one media marker per image + json images = json::array(); + if (n_images > 0) { + inp = decision_replace_text(inp, get_media_marker(), " "); + for (size_t i = 0; i < n_images; i++) { + images.push_back(get_media_marker()); + } + } + inp["images"] = images; + jinja::context ctx(tmpl->source()); jinja::global_from_json(ctx, inp, false); jinja::runtime runtime(ctx); @@ -221,15 +293,27 @@ std::string server_decision_context::render(const json & state, const server_dec return jinja::runtime::gather_string_parts(results)->as_string().str(); } -void server_decision_context::fill_task(const json & state, const server_decision_question & question, server_task & task) const { - llama_tokens tokens = common_tokenize(vocab, render(state, question), false, true); +void server_decision_context::fill_task( + const json & state, + const server_decision_question & question, + const std::vector & files, + mtmd_context * mctx, + const mtmd_helper_init_opt & init_opt, + server_task & task) const { + const std::string prompt = render(state, question, files.size()); if (type == COMMON_DECISION_TYPE_OPENJEV) { task.decision.labels.assign(labels.begin(), labels.begin() + question.options.size()); - } else { - fill_task_laya(tokens, question, task); + if (!files.empty()) { + task.tokens = process_mtmd_prompt(mctx, prompt, files, init_opt); + return; + } } + llama_tokens tokens = common_tokenize(vocab, prompt, false, true); + if (type == COMMON_DECISION_TYPE_LAYA) { + fill_task_laya(tokens, question, task); + } task.tokens = server_tokens(tokens, false); } diff --git a/tools/server/server-decision.h b/tools/server/server-decision.h index e380121004..937a2620ed 100644 --- a/tools/server/server-decision.h +++ b/tools/server/server-decision.h @@ -36,13 +36,41 @@ struct server_decision_context { void init(const llama_model * model); // true if the questions of a request start with the same tokens, and the model can continue from them - bool can_share_prompt() const { return type == COMMON_DECISION_TYPE_OPENJEV; } + bool can_share_prompt() const { + switch (type) { + case COMMON_DECISION_TYPE_OPENJEV: + return true; + default: + return false; + } + } + + // true if the prompt of the model has a place for images + bool can_use_images() const { + switch (type) { + case COMMON_DECISION_TYPE_OPENJEV: + return true; + default: + return false; + } + } // throw std::invalid_argument on bad input std::vector parse_questions(const json & body) const; + // returns the state without its images, they are appended to files in order + // images come from "images" and from the image_url parts of a state made of chat messages + json parse_state(const json & body, std::vector & files) const; + // set the prompt of this question, and where to read its result - void fill_task(const json & state, const server_decision_question & question, server_task & task) const; + // mctx is only used if there are files + void fill_task( + const json & state, + const server_decision_question & question, + const std::vector & files, + mtmd_context * mctx, + const mtmd_helper_init_opt & init_opt, + server_task & task) const; // scores: one raw model output per option json format_answer(const server_decision_question & question, const std::vector & scores) const; @@ -65,7 +93,7 @@ private: size_t max_head_tokens = 0; // question + options size_t max_option_tokens = 48; - std::string render(const json & state, const server_decision_question & question) const; + std::string render(const json & state, const server_decision_question & question, size_t n_images) const; void fill_task_laya(llama_tokens & tokens, const server_decision_question & question, server_task & task) const; float get_temperature(const server_decision_question & question) const; diff --git a/tools/server/tests/unit/test_systemone.py b/tools/server/tests/unit/test_systemone.py index 1a58ada0b8..45749a4b49 100644 --- a/tools/server/tests/unit/test_systemone.py +++ b/tools/server/tests/unit/test_systemone.py @@ -110,3 +110,4 @@ def test_systemone_invalid_request(data: dict): # TODO: test the shared prompt prefix, it needs a small model of a type that supports it (e.g. openjev) # it can be checked with GET /metrics: for one request, prompt_tokens_cached_total must grow by # (shared tokens * number of child tasks) and prompt_tokens_total + prompt_tokens_cached_total == usage.input_tokens +# TODO: test the image input ("images" and image_url parts of a chat-message state), it needs a small model with a mmproj