add vision support

This commit is contained in:
Xuan Son Nguyen
2026-10-01 20:35:43 +02:00
parent 9efdb2f5d7
commit afabc13b56
10 changed files with 175 additions and 13 deletions
+1
View File
@@ -346,6 +346,7 @@ MMPROJ_MODEL_MAP: dict[str, str] = {
"Qwen3TTSForConditionalGeneration": "qwen3tts",
"Qwen3VLForConditionalGeneration": "qwen3vl",
"Qwen3VLMoeForConditionalGeneration": "qwen3vl",
"OpenJevModel": "qwen3vl",
"Qwen3_5ForConditionalGeneration": "qwen3vl",
"Qwen3_5MoeForConditionalGeneration": "qwen3vl",
"Qwen4ExpForConditionalGeneration": "qwen4exp",
+6 -1
View File
@@ -690,9 +690,14 @@ class OpenJevModel(Qwen3_5TextModel):
"{% elif o.key == 'true' %}yes: {% if o.description %}" + description + "{% else %}The statement is true.{% endif %}"
"{% else %}no: {% if o.description %}" + description + "{% else %}The statement is false.{% endif %}{% endif %}"
)
# TODO: only the layout with one image is known (image first), the one with several images is not verified
images = (
"{% for image in images %}{{ image }}{% endfor %}"
"{% if images %}{{ 'The screenshot shows the current screen.\\n' }}{% endif %}"
)
return (
"{% set letters = '" + self._LETTERS + "' %}"
"<|im_start|>user\nState:\n" + jinja_str_or_json("state") + "\n\nQuestion: " + jinja_str_or_json("instructions")
"<|im_start|>user\n" + images + "State:\n" + jinja_str_or_json("state") + "\n\nQuestion: " + jinja_str_or_json("instructions")
+ "{% if type == 'score' %} Rate along the ordered levels below (lowest first).{% endif %}"
"{{ '\\nOptions:\\n' }}"
"{% for o in options %}[{{ letters[loop.index0] }}] " + option + "{{ '\\n' }}{% endfor %}"
+1 -1
View File
@@ -13,7 +13,7 @@ from .qwen import Qwen3Model, Qwen3MoeModel
from .qwenvl import Qwen25AudioModel
@ModelBase.register("Qwen3VLForConditionalGeneration", "Qwen3VLMoeForConditionalGeneration", "Qwen3_5ForConditionalGeneration", "Qwen3_5MoeForConditionalGeneration")
@ModelBase.register("Qwen3VLForConditionalGeneration", "Qwen3VLMoeForConditionalGeneration", "Qwen3_5ForConditionalGeneration", "Qwen3_5MoeForConditionalGeneration", "OpenJevModel")
@ModelBase.example("Qwen/Qwen3-VL-4B-Instruct", "Qwen/Qwen3-VL-30B-A3B-Instruct", "Qwen/Qwen3.5-9B", "Qwen/Qwen3.5-35B-A3B")
class Qwen3VLVisionModel(MmprojModel):
def __init__(self, *args, **kwargs):
+31 -1
View File
@@ -1677,6 +1677,8 @@ Follows the [TypeSafe API](https://docs.typesafe.ai/api), streaming is not suppo
`state`: The content to evaluate. Can be a string, an object or an array. A value that is not a string is given to the model as JSON text.
`images`: Optional. An array of up to 8 images, each one is a data URL (`data:image/...;base64,...`). See the image input section below.
`questions`: An object that maps a question id to a question. Each question has these fields:
- `type`: One of `choice`, `score`, `noul`.
@@ -1690,6 +1692,17 @@ The questions of a request are answered independently, an answer does not depend
The number of options of a `choice` question is limited by the model, for example: 52 for openjev, 255 for laya. For laya, long questions and options are truncated to the token budget the model was trained with.
*Image input:*
Image input needs a model that supports it (for example: openjev) and its multimodal projector, see `--mmproj`.
Images can be given in two ways, and both can be used in the same request:
- The `images` field.
- A `state` made of chat messages, either an array of messages or an object with a `messages` array. An `image_url` part in the `content` of a message is taken as an image, in the same format as chat completions. Only data URLs are accepted.
All the images are placed before the state in the prompt, the ones from `images` first. The image parts are removed from the state. A request can have at most 8 images in total.
*Response:*
`answers`: An object that maps each question id to its answer. The fields depend on the question type:
@@ -1767,7 +1780,24 @@ Response (values are shortened):
}
```
An invalid request returns the error `400`. A model that is not a decision model returns the error `501`.
Example with an image:
```shell
curl http://127.0.0.1:8080/v1/systemone \
-H "Content-Type: application/json" \
-d '{
"state": "The document was received by the accounting team this morning.",
"images": ["data:image/jpeg;base64,/9j/4AAQSkZJRg..."],
"questions": {
"has_table": {
"type": "noul",
"instructions": "Does the image contain a table?"
}
}
}' | jq
```
An invalid request returns the error `400`. A model that is not a decision model returns the error `501`. A request with images returns the error `501` if the model does not support image input, or if no multimodal projector is loaded.
## Server tools
+1 -1
View File
@@ -1076,7 +1076,7 @@ json oaicompat_completion_params_parse(const json & body) {
// - file:// for local files (only allowed if media_path is set)
// - data: for base64 encoded data with uri scheme (e.g. data:image/png;base64,...)
// - raw base64 encoded data
static void handle_media(
void handle_media(
std::vector<raw_buffer> & out_files,
const std::string & url,
const std::string & media_path) {
+6
View File
@@ -270,6 +270,12 @@ llama_tokens tokenize_mixed(const llama_vocab * vocab, const json & json_prompt,
// if validate_utf8(text) == text.size(), then the whole text is valid utf8
size_t validate_utf8(const std::string& text);
// load a media file from an URL (http, file, data) or from raw base64 data
void handle_media(
std::vector<raw_buffer> & out_files,
const std::string & url,
const std::string & media_path);
// process mtmd prompt, return the server_tokens containing both text tokens and media chunks
// if is_placeholder is true, the media chunk will be treated as placeholder for counting tokens; the output tokens are not usable for actual inference (e.g. for submitting a task to server_queue)
server_tokens process_mtmd_prompt(
+8 -1
View File
@@ -5332,6 +5332,13 @@ void server_routes::init_routes() {
const json body = json::parse(req.body);
const auto questions = decision.parse_questions(body);
std::vector<raw_buffer> files;
const json state = decision.parse_state(body, files);
if (!files.empty() && (!decision.can_use_images() || !meta->has_inp_image)) {
res->error(format_error_response("This server does not support image input for decisions. For a model that supports it, start it with `--mmproj`", ERROR_TYPE_NOT_SUPPORTED));
return res;
}
// one task per question
auto & rd = res->rd;
{
@@ -5340,7 +5347,7 @@ void server_routes::init_routes() {
for (const auto & question : questions) {
server_task task = server_task(SERVER_TASK_TYPE_DECISION);
task.id = rd.get_new_id();
decision.fill_task(body.at("state"), question, task);
decision.fill_task(state, question, files, ctx_server.mctx, ctx_server.init_opt, task);
tasks.push_back(std::move(task));
}
if (decision.can_share_prompt()) {
+89 -5
View File
@@ -164,6 +164,68 @@ std::vector<server_decision_question> server_decision_context::parse_questions(c
return questions;
}
//
// images
//
static const size_t DECISION_MAX_IMAGES = 8;
static void decision_load_image(const json & url, std::vector<raw_buffer> & files) {
if (!url.is_string() || !string_starts_with(url.get<std::string>(), "data:image/")) {
throw std::invalid_argument("images must be data URLs (data:image/...;base64,...)");
}
if (files.size() >= DECISION_MAX_IMAGES) {
throw std::invalid_argument(string_format("too many images, the maximum is %zu", DECISION_MAX_IMAGES));
}
handle_media(files, url.get<std::string>(), "");
}
json server_decision_context::parse_state(const json & body, std::vector<raw_buffer> & files) const {
if (body.contains("images") && !body.at("images").is_null()) {
if (!body.at("images").is_array()) {
throw std::invalid_argument("\"images\" must be an array");
}
for (const auto & url : body.at("images")) {
decision_load_image(url, files);
}
}
const json & state = body.at("state");
const bool is_wrapped = state.is_object() && state.contains("messages");
const json & messages = is_wrapped ? state.at("messages") : state;
if (!messages.is_array()) {
return state;
}
// chat messages: take the image parts out of the content
json messages_out = json::array();
for (const auto & msg : messages) {
if (!msg.is_object() || !msg.contains("content") || !msg.at("content").is_array()) {
messages_out.push_back(msg);
continue;
}
json content = json::array();
for (const auto & part : msg.at("content")) {
if (part.is_object() && json_value(part, "type", std::string()) == "image_url" && part.contains("image_url")) {
const json & image_url = part.at("image_url");
decision_load_image(image_url.is_object() && image_url.contains("url") ? image_url.at("url") : image_url, files);
} else {
content.push_back(part);
}
}
json msg_out = msg;
msg_out["content"] = content;
messages_out.push_back(msg_out);
}
if (!is_wrapped) {
return messages_out;
}
json state_out = state;
state_out["messages"] = messages_out;
return state_out;
}
//
// prompt
//
@@ -192,7 +254,7 @@ static json decision_replace_text(const json & val, const std::string & search,
return val;
}
std::string server_decision_context::render(const json & state, const server_decision_question & question) const {
std::string server_decision_context::render(const json & state, const server_decision_question & question, size_t n_images) const {
json options = json::array();
for (const auto & opt : question.options) {
options.push_back(json{
@@ -214,6 +276,16 @@ std::string server_decision_context::render(const json & state, const server_dec
inp = decision_replace_text(inp, text_marker, " ");
}
// the template puts one media marker per image
json images = json::array();
if (n_images > 0) {
inp = decision_replace_text(inp, get_media_marker(), " ");
for (size_t i = 0; i < n_images; i++) {
images.push_back(get_media_marker());
}
}
inp["images"] = images;
jinja::context ctx(tmpl->source());
jinja::global_from_json(ctx, inp, false);
jinja::runtime runtime(ctx);
@@ -221,15 +293,27 @@ std::string server_decision_context::render(const json & state, const server_dec
return jinja::runtime::gather_string_parts(results)->as_string().str();
}
void server_decision_context::fill_task(const json & state, const server_decision_question & question, server_task & task) const {
llama_tokens tokens = common_tokenize(vocab, render(state, question), false, true);
void server_decision_context::fill_task(
const json & state,
const server_decision_question & question,
const std::vector<raw_buffer> & files,
mtmd_context * mctx,
const mtmd_helper_init_opt & init_opt,
server_task & task) const {
const std::string prompt = render(state, question, files.size());
if (type == COMMON_DECISION_TYPE_OPENJEV) {
task.decision.labels.assign(labels.begin(), labels.begin() + question.options.size());
} else {
fill_task_laya(tokens, question, task);
if (!files.empty()) {
task.tokens = process_mtmd_prompt(mctx, prompt, files, init_opt);
return;
}
}
llama_tokens tokens = common_tokenize(vocab, prompt, false, true);
if (type == COMMON_DECISION_TYPE_LAYA) {
fill_task_laya(tokens, question, task);
}
task.tokens = server_tokens(tokens, false);
}
+31 -3
View File
@@ -36,13 +36,41 @@ struct server_decision_context {
void init(const llama_model * model);
// true if the questions of a request start with the same tokens, and the model can continue from them
bool can_share_prompt() const { return type == COMMON_DECISION_TYPE_OPENJEV; }
bool can_share_prompt() const {
switch (type) {
case COMMON_DECISION_TYPE_OPENJEV:
return true;
default:
return false;
}
}
// true if the prompt of the model has a place for images
bool can_use_images() const {
switch (type) {
case COMMON_DECISION_TYPE_OPENJEV:
return true;
default:
return false;
}
}
// throw std::invalid_argument on bad input
std::vector<server_decision_question> parse_questions(const json & body) const;
// returns the state without its images, they are appended to files in order
// images come from "images" and from the image_url parts of a state made of chat messages
json parse_state(const json & body, std::vector<raw_buffer> & files) const;
// set the prompt of this question, and where to read its result
void fill_task(const json & state, const server_decision_question & question, server_task & task) const;
// mctx is only used if there are files
void fill_task(
const json & state,
const server_decision_question & question,
const std::vector<raw_buffer> & files,
mtmd_context * mctx,
const mtmd_helper_init_opt & init_opt,
server_task & task) const;
// scores: one raw model output per option
json format_answer(const server_decision_question & question, const std::vector<float> & scores) const;
@@ -65,7 +93,7 @@ private:
size_t max_head_tokens = 0; // question + options
size_t max_option_tokens = 48;
std::string render(const json & state, const server_decision_question & question) const;
std::string render(const json & state, const server_decision_question & question, size_t n_images) const;
void fill_task_laya(llama_tokens & tokens, const server_decision_question & question, server_task & task) const;
float get_temperature(const server_decision_question & question) const;
@@ -110,3 +110,4 @@ def test_systemone_invalid_request(data: dict):
# TODO: test the shared prompt prefix, it needs a small model of a type that supports it (e.g. openjev)
# it can be checked with GET /metrics: for one request, prompt_tokens_cached_total must grow by
# (shared tokens * number of child tasks) and prompt_tokens_total + prompt_tokens_cached_total == usage.input_tokens
# TODO: test the image input ("images" and image_url parts of a chat-message state), it needs a small model with a mmproj