From 1b89754bb4448042bcf54cfd990f33406c8eefe3 Mon Sep 17 00:00:00 2001 From: Xuan Son Nguyen Date: Thu, 1 Oct 2026 19:32:36 +0200 Subject: [PATCH] add server code --- tools/server/CMakeLists.txt | 2 + tools/server/server-context.cpp | 119 +++++++ tools/server/server-context.h | 1 + tools/server/server-decision.cpp | 391 ++++++++++++++++++++++ tools/server/server-decision.h | 78 +++++ tools/server/server-task.cpp | 11 + tools/server/server-task.h | 22 ++ tools/server/server.cpp | 2 + tools/server/tests/unit/test_systemone.py | 118 +++++++ tools/server/tests/utils.py | 19 ++ 10 files changed, 763 insertions(+) create mode 100644 tools/server/server-decision.cpp create mode 100644 tools/server/server-decision.h create mode 100644 tools/server/tests/unit/test_systemone.py diff --git a/tools/server/CMakeLists.txt b/tools/server/CMakeLists.txt index 4adaaceefd..70b8d7da8e 100644 --- a/tools/server/CMakeLists.txt +++ b/tools/server/CMakeLists.txt @@ -13,6 +13,8 @@ add_library(${TARGET} STATIC server-queue.h server-common.cpp server-common.h + server-decision.cpp + server-decision.h server-context.cpp server-context.h server-stream.cpp diff --git a/tools/server/server-context.cpp b/tools/server/server-context.cpp index 470fbd9775..da719cdf43 100644 --- a/tools/server/server-context.cpp +++ b/tools/server/server-context.cpp @@ -1,6 +1,7 @@ #include "server-context.h" #include "server-chat.h" #include "server-common.h" +#include "server-decision.h" #include "server-http.h" #include "server-task.h" #include "server-queue.h" @@ -825,6 +826,8 @@ public: mtmd_helper_init_opt init_opt = mtmd_helper_init_opt_default(); const llama_vocab * vocab = nullptr; + server_decision_context decision; + server_queue queue_tasks; server_response queue_results; @@ -1102,6 +1105,13 @@ private: vocab = llama_model_get_vocab(model_tgt); + try { + decision.init(model_tgt); + } catch (const std::exception & e) { + SRV_ERR("failed to init decision model: %s\n", e.what()); + return false; + } + n_ctx = llama_n_ctx(ctx_tgt); add_bos_token = llama_vocab_get_add_bos(vocab); @@ -2194,6 +2204,49 @@ private: queue_results.send(std::move(res)); } + void send_decision(const server_slot & slot, const common_batch & batch, int32_t i_batch) { + auto res = std::make_unique(); + res->id = slot.task->id; + res->index = slot.task->index; + res->n_tokens = slot.task->n_tokens(); + + const auto & decision = slot.task->decision; + + if (!decision.labels.empty()) { + const float * logits = llama_get_logits_ith(slot.ctx_tgt, i_batch); + if (logits == nullptr) { + send_error(slot, "failed to get logits", ERROR_TYPE_SERVER); + return; + } + const int32_t n_vocab = llama_vocab_n_tokens(vocab); + for (const llama_token label : decision.labels) { + GGML_ASSERT(label >= 0 && label < n_vocab); + res->scores.push_back(logits[label]); + } + } else { + // the prompt is evaluated in one batch, the n-th output of this slot is the n-th prompt token + std::vector idx; + for (int i = 0; i < batch.size(); ++i) { + if (batch.tokens[i].output && batch.tokens[i].seq_id == slot.id) { + idx.push_back(i); + } + } + GGML_ASSERT(decision.column >= 0 && decision.column < llama_model_n_embd_out(model_tgt)); + for (const int32_t marker : decision.markers) { + const float * embd = marker >= 0 && marker < (int32_t) idx.size() ? llama_get_embeddings_ith(slot.ctx_tgt, idx[marker]) : nullptr; + if (embd == nullptr) { + send_error(slot, "failed to get embeddings", ERROR_TYPE_SERVER); + return; + } + res->scores.push_back(embd[decision.column]); + } + } + + SLT_DBG(slot, "%s", "sending decision result\n"); + + queue_results.send(std::move(res)); + } + void send_rerank(const server_slot & slot, const common_batch & batch) { auto res = std::make_unique(); res->id = slot.task->id; @@ -2384,6 +2437,7 @@ private: case SERVER_TASK_TYPE_INFILL: case SERVER_TASK_TYPE_EMBEDDING: case SERVER_TASK_TYPE_RERANK: + case SERVER_TASK_TYPE_DECISION: { // special case: if input is provided via CLI, tokenize it first // otherwise, no need to tokenize as it's already done inside the HTTP thread @@ -3831,6 +3885,13 @@ private: return; } + if (slot.task->type == SERVER_TASK_TYPE_DECISION) { + send_decision(slot, batch.view, slot.i_batch - off); + slot.release(); + slot.i_batch = -1; + return; + } + GGML_ASSERT(slot.task->need_sampling()); // prompt evaluated for next-token prediction @@ -5224,6 +5285,64 @@ void server_routes::init_routes() { return res; }; + this->post_systemone = [this](const server_http_req & req) { + auto res = create_response(); + const auto & decision = ctx_server.decision; + if (decision.type == SERVER_DECISION_TYPE_NONE) { + res->error(format_error_response("This model is not a decision model", ERROR_TYPE_NOT_SUPPORTED)); + return res; + } + if (decision.need_embd() && (!params.embedding || meta->pooling_type != LLAMA_POOLING_TYPE_NONE)) { + res->error(format_error_response("This decision model requires `--embedding --pooling none`", ERROR_TYPE_NOT_SUPPORTED)); + return res; + } + + const json body = json::parse(req.body); + const auto questions = decision.parse_questions(body); + + // one task per question + auto & rd = res->rd; + { + std::vector tasks; + tasks.reserve(questions.size()); + for (const auto & question : questions) { + server_task task = server_task(SERVER_TASK_TYPE_DECISION); + task.id = rd.get_new_id(); + decision.fill_task(body.at("state"), question, task); + tasks.push_back(std::move(task)); + } + rd.post_tasks(std::move(tasks)); + } + + auto all_results = rd.wait_for_all(req.should_stop); + + if (all_results.is_terminated) { + return res; // connection is closed + } else if (all_results.error) { + res->error(all_results.error->to_json()); + return res; + } + + json answers = json::object(); + int32_t n_tokens = 0; + for (size_t i = 0; i < questions.size(); i++) { + auto * result = dynamic_cast(all_results.results[i].get()); + GGML_ASSERT(result != nullptr); + answers[questions[i].id] = decision.format_answer(questions[i], result->scores); + n_tokens += result->n_tokens; + } + + res->ok(json{ + {"model", meta->model_name}, + {"answers", answers}, + {"usage", { + {"input_tokens", n_tokens}, + {"output_tokens", 0}, + }}, + }); + return res; + }; + this->get_lora_adapters = [this](const server_http_req & req) { auto res = create_response(); diff --git a/tools/server/server-context.h b/tools/server/server-context.h index 7265ccad15..c554bb95b4 100644 --- a/tools/server/server-context.h +++ b/tools/server/server-context.h @@ -152,6 +152,7 @@ struct server_routes { server_http_context::handler_t post_embeddings; server_http_context::handler_t post_embeddings_oai; server_http_context::handler_t post_rerank; + server_http_context::handler_t post_systemone; server_http_context::handler_t get_lora_adapters; server_http_context::handler_t post_lora_adapters; diff --git a/tools/server/server-decision.cpp b/tools/server/server-decision.cpp new file mode 100644 index 0000000000..f113ab7048 --- /dev/null +++ b/tools/server/server-decision.cpp @@ -0,0 +1,391 @@ +#include "server-decision.h" + +#include +#include +#include + +static const char * decision_question_type_name(server_decision_question_type type) { + switch (type) { + case SERVER_DECISION_QUESTION_CHOICE: return "choice"; + case SERVER_DECISION_QUESTION_SCORE: return "score"; + case SERVER_DECISION_QUESTION_NOUL: return "noul"; + } + return ""; +} + +static std::string decision_meta_str(const llama_model * model, const std::string & key) { + char buf[256]; + const int32_t n = llama_model_meta_val_str(model, key.c_str(), buf, sizeof(buf)); + return n < 0 ? "" : std::string(buf); +} + +// +// model-specific setup +// + +void server_decision_context::init(const llama_model * model) { + *this = server_decision_context(); // the model can be reloaded + + const std::string prefix = decision_meta_str(model, "general.architecture") + ".decision."; + const std::string type_name = decision_meta_str(model, prefix + "type"); + if (type_name.empty()) { + return; + } + + vocab = llama_model_get_vocab(model); + + const char * tmpl_src = llama_model_chat_template(model, "systemone"); + if (tmpl_src == nullptr) { + throw std::runtime_error("decision model has no \"systemone\" template"); + } + tmpl = std::make_shared(tmpl_src, "", ""); + + const std::string prefix_temp = prefix + "temperature."; + for (int32_t i = 0; i < llama_model_meta_count(model); i++) { + char key[256]; + char val[64]; + if (llama_model_meta_key_by_index(model, i, key, sizeof(key)) < 0 || !string_starts_with(key, prefix_temp)) { + continue; + } + if (llama_model_meta_val_str_by_index(model, i, val, sizeof(val)) < 0) { + continue; + } + const float temp = std::strtof(val, nullptr); + if (temp <= 0.0f) { + throw std::runtime_error(string_format("invalid decision temperature: %s = %s", key, val)); + } + temperatures[key + prefix_temp.size()] = temp; + } + + if (type_name == "openjev") { + // one letter per option, each must be a single token + const std::string letters = "ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz"; + for (const char c : letters) { + const auto toks = common_tokenize(vocab, std::string(1, c), false, false); + if (toks.size() != 1) { + throw std::runtime_error(string_format("decision label '%c' is not a single token", c)); + } + labels.push_back(toks[0]); + } + n_options_max = labels.size(); + noul_true_first = true; + type = SERVER_DECISION_TYPE_OPENJEV; + } else if (type_name == "laya") { + token_marker = llama_vocab_mask(vocab); + token_sep = llama_vocab_sep(vocab); + if (token_marker == LLAMA_TOKEN_NULL || token_sep == LLAMA_TOKEN_NULL) { + throw std::runtime_error("decision model has no mask or sep token"); + } + text_marker = common_token_to_piece(vocab, token_marker, true); + + const std::string val = decision_meta_str(model, prefix + "max_head_tokens"); + max_head_tokens = std::strtoul(val.c_str(), nullptr, 10); + if (max_head_tokens == 0) { + throw std::runtime_error("decision model has no valid max_head_tokens"); + } + n_options_max = 255; + type = SERVER_DECISION_TYPE_LAYA; + } else { + throw std::runtime_error("unsupported decision model type: " + type_name); + } + + SRV_INF("decision model type: %s\n", type_name.c_str()); +} + +// +// request parsing +// + +std::vector server_decision_context::parse_questions(const json & body) const { + if (!body.contains("state") || body.at("state").is_null()) { + throw std::invalid_argument("\"state\" must be provided"); + } + if (!body.contains("questions") || !body.at("questions").is_object() || body.at("questions").empty()) { + throw std::invalid_argument("\"questions\" must be a non-empty object"); + } + + std::vector questions; + for (const auto & [id, q] : body.at("questions").items()) { + auto err = [&id = id](const std::string & msg) { + return std::invalid_argument("questions." + id + ": " + msg); + }; + if (!q.is_object()) { + throw err("must be an object"); + } + if (!q.contains("instructions") || q.at("instructions").is_null()) { + throw err("\"instructions\" must be provided"); + } + + server_decision_question question; + question.id = id; + question.instructions = q.at("instructions"); + + const std::string type_name = json_value(q, "type", std::string()); + const json criteria = q.contains("criteria") ? q.at("criteria") : json(); + + if (type_name == "choice") { + question.type = SERVER_DECISION_QUESTION_CHOICE; + if (!criteria.is_object() || criteria.empty()) { + throw err("\"criteria\" must be a non-empty object"); + } + for (const auto & [key, description] : criteria.items()) { + question.options.push_back({key, description}); + } + } else if (type_name == "score") { + question.type = SERVER_DECISION_QUESTION_SCORE; + if (!criteria.is_array() || criteria.size() < 2 || criteria.size() > 10) { + throw err("\"criteria\" must be an array of 2 to 10 levels"); + } + for (size_t i = 0; i < criteria.size(); i++) { + question.options.push_back({std::to_string(i), criteria.at(i)}); + } + } else if (type_name == "noul") { + question.type = SERVER_DECISION_QUESTION_NOUL; + if (!criteria.is_null() && !criteria.is_object()) { + throw err("\"criteria\" must be an object"); + } + for (const char * key : {"false", "true"}) { + question.options.push_back({key, criteria.is_object() && criteria.contains(key) ? criteria.at(key) : json()}); + } + if (noul_true_first) { + std::swap(question.options[0], question.options[1]); + } + } else { + throw err("\"type\" must be one of: choice, score, noul"); + } + + if (question.options.size() > n_options_max) { + throw err(string_format("too many options (%zu), this model supports at most %zu", question.options.size(), n_options_max)); + } + + questions.push_back(std::move(question)); + } + return questions; +} + +// +// prompt +// + +// replace text in all strings of a JSON value +static json decision_replace_text(const json & val, const std::string & search, const std::string & replace) { + if (val.is_string()) { + std::string str = val.get(); + string_replace_all(str, search, replace); + return str; + } + if (val.is_array()) { + json out = json::array(); + for (const auto & item : val) { + out.push_back(decision_replace_text(item, search, replace)); + } + return out; + } + if (val.is_object()) { + json out = json::object(); + for (const auto & [key, item] : val.items()) { + out[key] = decision_replace_text(item, search, replace); + } + return out; + } + return val; +} + +std::string server_decision_context::render(const json & state, const server_decision_question & question) const { + json options = json::array(); + for (const auto & opt : question.options) { + options.push_back(json{ + {"key", opt.key}, + {"description", opt.description}, + }); + } + + // the template is given raw JSON values, it serializes the ones that are not strings + json inp = json{ + {"type", decision_question_type_name(question.type)}, + {"instructions", question.instructions}, + {"state", state}, + {"options", options}, + }; + + // the input must not contain the marker of the options + if (!text_marker.empty()) { + inp = decision_replace_text(inp, text_marker, " "); + } + + jinja::context ctx(tmpl->source()); + jinja::global_from_json(ctx, inp, false); + jinja::runtime runtime(ctx); + const jinja::value results = runtime.execute(tmpl->prog); + return jinja::runtime::gather_string_parts(results)->as_string().str(); +} + +void server_decision_context::fill_task(const json & state, const server_decision_question & question, server_task & task) const { + llama_tokens tokens = common_tokenize(vocab, render(state, question), false, true); + + if (type == SERVER_DECISION_TYPE_OPENJEV) { + task.decision.labels.assign(labels.begin(), labels.begin() + question.options.size()); + } else { + fill_task_laya(tokens, question, task); + } + + task.tokens = server_tokens(tokens, false); +} + +// the prompt is: [cls] question [sep] ([marker] option)* [sep] state [sep] +// options and question are cut to fit max_head_tokens, the same way the model was trained +void server_decision_context::fill_task_laya(llama_tokens & tokens, const server_decision_question & question, server_task & task) const { + const size_t n_options = question.options.size(); + + std::vector markers; + for (size_t i = 0; i < tokens.size(); i++) { + if (tokens[i] == token_marker) { + markers.push_back(i); + } + } + const auto invalid = std::runtime_error("unexpected layout of the decision prompt"); + if (markers.size() != n_options || markers[0] < 2 || tokens[markers[0] - 1] != token_sep || tokens.back() != token_sep) { + throw invalid; + } + const size_t head_end = markers[0] - 1; + const size_t opts_end = std::find(tokens.begin() + markers.back(), tokens.end(), token_sep) - tokens.begin(); + if (opts_end + 1 >= tokens.size()) { + throw invalid; + } + + // marker + text of each option + std::vector options; + size_t n_options_tokens = 0; + auto set_max = [&](size_t n_max) { + n_options_tokens = 0; + for (auto & opt : options) { + opt.resize(std::min(opt.size(), n_max)); + n_options_tokens += opt.size(); + } + }; + for (size_t i = 0; i < n_options; i++) { + const size_t end = i + 1 < n_options ? markers[i + 1] : opts_end; + options.emplace_back(tokens.begin() + markers[i], tokens.begin() + end); + } + set_max(max_option_tokens + 1); + if (n_options_tokens + 16 > max_head_tokens) { + // too many or too long options, shrink them evenly + set_max(std::max((size_t) 4, (max_head_tokens - std::min(max_head_tokens, (size_t) 16)) / n_options)); + } + const size_t n_question_max = std::max((size_t) 8, max_head_tokens - std::min(max_head_tokens, n_options_tokens)); + + llama_tokens out; + out.push_back(tokens[0]); + out.insert(out.end(), tokens.begin() + 1, tokens.begin() + std::min(head_end, 1 + n_question_max)); + out.push_back(token_sep); + for (const auto & opt : options) { + task.decision.markers.push_back(out.size()); + out.insert(out.end(), opt.begin(), opt.end()); + } + out.insert(out.end(), tokens.begin() + opts_end, tokens.end()); + tokens = std::move(out); + + // the output has one score per question type + task.decision.column = question.type; +} + +// +// answer +// + +float server_decision_context::get_temperature(const server_decision_question & question) const { + const size_t n = question.options.size(); + const std::string type_name = decision_question_type_name(question.type); + const std::string bucket = n <= 2 ? "2" : n <= 5 ? "3_5" : n <= 10 ? "6_10" : "11"; + + for (const auto & name : {type_name + "." + bucket, type_name}) { + const auto it = temperatures.find(name); + if (it != temperatures.end()) { + return it->second; + } + } + return 1.0f; +} + +// confidence formulas are the ones published by TypeSafe + +static double decision_confidence_choice(const std::vector & probs) { + if (probs.size() < 2) { + return 1.0; + } + const double uniform = 1.0 / probs.size(); + const double p_max = *std::max_element(probs.begin(), probs.end()); + return std::max(0.0, (p_max - uniform) / (1.0 - uniform)); +} + +static double decision_confidence_score(const std::vector & probs) { + if (probs.size() < 2) { + return 1.0; + } + const size_t n = probs.size(); + const size_t mode = std::max_element(probs.begin(), probs.end()) - probs.begin(); + + // mean distance to the mode, relative to the one of a uniform distribution around its center + double dist = 0.0; + double dist_uniform = 0.0; + for (size_t i = 0; i < n; i++) { + dist += probs[i] * std::fabs((double) i - (double) mode); + dist_uniform += std::fabs((double) i - (n - 1) / 2.0) / n; + } + return std::max(0.0, 1.0 - dist / dist_uniform); +} + +json server_decision_context::format_answer(const server_decision_question & question, const std::vector & scores) const { + const size_t n = question.options.size(); + if (scores.size() != n) { + throw std::runtime_error("decision result does not match the number of options"); + } + + // softmax over the options + const float temperature = get_temperature(question); + const float score_max = *std::max_element(scores.begin(), scores.end()); + std::vector probs(n); + double sum = 0.0; + for (size_t i = 0; i < n; i++) { + probs[i] = std::exp((double) (scores[i] - score_max) / temperature); + sum += probs[i]; + } + for (auto & p : probs) { + p /= sum; + } + + json answer = json{{"type", decision_question_type_name(question.type)}}; + + if (question.type == SERVER_DECISION_QUESTION_NOUL) { + for (size_t i = 0; i < n; i++) { + if (question.options[i].key == "true") { + answer["noul"] = probs[i]; + } + } + return answer; + } + + json probabilities = json::object(); + for (size_t i = 0; i < n; i++) { + probabilities[question.options[i].key] = probs[i]; + } + + if (question.type == SERVER_DECISION_QUESTION_CHOICE) { + const size_t best = std::max_element(probs.begin(), probs.end()) - probs.begin(); + answer["choice"] = question.options[best].key; + answer["probabilities"] = probabilities; + answer["confidence"] = decision_confidence_choice(probs); + } else { + double expected = 0.0; + json legend = json::object(); + for (size_t i = 0; i < n; i++) { + expected += i * probs[i]; + legend[question.options[i].key] = question.options[i].description; + } + answer["score"] = expected; + answer["legend"] = legend; + answer["probabilities"] = probabilities; + answer["confidence"] = decision_confidence_score(probs); + } + return answer; +} diff --git a/tools/server/server-decision.h b/tools/server/server-decision.h new file mode 100644 index 0000000000..3dedb053df --- /dev/null +++ b/tools/server/server-decision.h @@ -0,0 +1,78 @@ +#pragma once + +#include "server-common.h" +#include "server-task.h" + +#include +#include +#include +#include + +// typed decision models (TypeSafe /v1/systemone API) +// the model answers each question in one forward pass, no token is generated + +enum server_decision_type { + SERVER_DECISION_TYPE_NONE, // not a decision model + SERVER_DECISION_TYPE_OPENJEV, // logits of one label token per option, read at the last prompt token + SERVER_DECISION_TYPE_LAYA, // score of one marker token per option, read from the embeddings output +}; + +enum server_decision_question_type { + SERVER_DECISION_QUESTION_CHOICE, + SERVER_DECISION_QUESTION_SCORE, + SERVER_DECISION_QUESTION_NOUL, +}; + +struct server_decision_option { + std::string key; + json description; // null if not provided +}; + +struct server_decision_question { + std::string id; + server_decision_question_type type; + json instructions; + std::vector options; // in the order of the model outputs +}; + +struct server_decision_context { + server_decision_type type = SERVER_DECISION_TYPE_NONE; + + // read the ".decision.*" metadata, type stays NONE if the model has none + void init(const llama_model * model); + + // true if the result is read from the embeddings of each token + bool need_embd() const { return type == SERVER_DECISION_TYPE_LAYA; } + + // throw std::invalid_argument on bad input + std::vector parse_questions(const json & body) const; + + // set the prompt of this question, and where to read its result + void fill_task(const json & state, const server_decision_question & question, server_task & task) const; + + // scores: one raw model output per option + json format_answer(const server_decision_question & question, const std::vector & scores) const; + +private: + const llama_vocab * vocab = nullptr; + std::shared_ptr tmpl; // the "systemone" template + + std::map temperatures; // "" or "." + size_t n_options_max = 0; + bool noul_true_first = false; // noul options are [true, false] instead of [false, true] + + // OPENJEV + std::vector labels; + + // LAYA + llama_token token_marker = LLAMA_TOKEN_NULL; + llama_token token_sep = LLAMA_TOKEN_NULL; + std::string text_marker; + size_t max_head_tokens = 0; // question + options + size_t max_option_tokens = 48; + + std::string render(const json & state, const server_decision_question & question) const; + void fill_task_laya(llama_tokens & tokens, const server_decision_question & question, server_task & task) const; + + float get_temperature(const server_decision_question & question) const; +}; diff --git a/tools/server/server-task.cpp b/tools/server/server-task.cpp index 0d3beb313c..a5c33c056b 100644 --- a/tools/server/server-task.cpp +++ b/tools/server/server-task.cpp @@ -1495,6 +1495,17 @@ json server_task_result_rerank::to_json() { }; } +// +// server_task_result_decision +// +json server_task_result_decision::to_json() { + return json { + {"index", index}, + {"scores", scores}, + {"tokens_evaluated", n_tokens}, + }; +} + // // server_task_result_error // diff --git a/tools/server/server-task.h b/tools/server/server-task.h index 9c99143f8e..56a2eec5b6 100644 --- a/tools/server/server-task.h +++ b/tools/server/server-task.h @@ -16,6 +16,7 @@ enum server_task_type { SERVER_TASK_TYPE_COMPLETION, SERVER_TASK_TYPE_EMBEDDING, SERVER_TASK_TYPE_RERANK, + SERVER_TASK_TYPE_DECISION, SERVER_TASK_TYPE_INFILL, SERVER_TASK_TYPE_CANCEL, SERVER_TASK_TYPE_CONTROL, @@ -172,6 +173,15 @@ struct server_task { // used by SERVER_TASK_TYPE_METRICS bool metrics_reset_bucket = false; + // used by SERVER_TASK_TYPE_DECISION + // where to read the model output of each option, exactly one of the two lists is used + struct decision { + std::vector labels; // logits of these tokens, at the last prompt token + std::vector markers; // embeddings[column] at these prompt positions + int32_t column = 0; + }; + decision decision; + // used by SERVER_TASK_TYPE_SET_LORA std::map set_lora; // mapping adapter ID -> scale @@ -188,6 +198,8 @@ struct server_task { case SERVER_TASK_TYPE_EMBEDDING: case SERVER_TASK_TYPE_RERANK: return true; + case SERVER_TASK_TYPE_DECISION: + return !decision.markers.empty(); default: return false; } @@ -198,6 +210,8 @@ struct server_task { case SERVER_TASK_TYPE_COMPLETION: case SERVER_TASK_TYPE_INFILL: return true; + case SERVER_TASK_TYPE_DECISION: + return !decision.labels.empty(); default: return false; } @@ -474,6 +488,14 @@ struct server_task_result_rerank : server_task_result { virtual json to_json() override; }; +struct server_task_result_decision : server_task_result { + std::vector scores; // one raw model output per option + + int32_t n_tokens; + + virtual json to_json() override; +}; + struct server_task_result_error : server_task_result { error_type err_type = ERROR_TYPE_SERVER; std::string err_msg; diff --git a/tools/server/server.cpp b/tools/server/server.cpp index bcf84e1ae9..ad538a6d63 100644 --- a/tools/server/server.cpp +++ b/tools/server/server.cpp @@ -231,6 +231,7 @@ int llama_server(common_params & params, int argc, char ** argv) { routes.post_embeddings = models_routes->proxy_post; routes.post_embeddings_oai = models_routes->proxy_post; routes.post_rerank = models_routes->proxy_post; + routes.post_systemone = models_routes->proxy_post; routes.post_tokenize = models_routes->proxy_post; routes.post_detokenize = models_routes->proxy_post; routes.post_apply_template = models_routes->proxy_post; @@ -278,6 +279,7 @@ int llama_server(common_params & params, int argc, char ** argv) { ctx_http.post("/reranking", ex_wrapper(routes.post_rerank)); ctx_http.post("/v1/rerank", ex_wrapper(routes.post_rerank)); ctx_http.post("/v1/reranking", ex_wrapper(routes.post_rerank)); + ctx_http.post("/v1/systemone", ex_wrapper(routes.post_systemone)); ctx_http.post("/tokenize", ex_wrapper(routes.post_tokenize)); ctx_http.post("/detokenize", ex_wrapper(routes.post_detokenize)); ctx_http.post("/apply-template", ex_wrapper(routes.post_apply_template)); diff --git a/tools/server/tests/unit/test_systemone.py b/tools/server/tests/unit/test_systemone.py new file mode 100644 index 0000000000..cf983cce91 --- /dev/null +++ b/tools/server/tests/unit/test_systemone.py @@ -0,0 +1,118 @@ +import pytest +from utils import * + +server = ServerPreset.tinylaya() + + +@pytest.fixture(autouse=True) +def create_server(): + global server + server = ServerPreset.tinylaya() + + +TEST_STATE = "I was charged twice for my order last week and nobody has replied." + +TEST_QUESTIONS = { + "route": { + "type": "choice", + "instructions": "Which team should handle this?", + "criteria": {"billing": "payments and refunds", "shipping": None, "technical": None}, + }, + "urgency": { + "type": "score", + "instructions": "How urgent is this?", + "criteria": ["can wait", "this week", "today", "right now"], + }, + "angry": { + "type": "noul", + "instructions": "Is the customer angry?", + }, +} + + +def test_systemone(): + global server + server.start() + res = server.make_request("POST", "/v1/systemone", data={ + "state": TEST_STATE, + "questions": TEST_QUESTIONS, + }) + assert res.status_code == 200 + assert res.body["usage"]["input_tokens"] > 0 + assert res.body["usage"]["output_tokens"] == 0 + + answers = res.body["answers"] + assert list(answers.keys()) == ["route", "urgency", "angry"] + + route = answers["route"] + assert route["type"] == "choice" + assert list(route["probabilities"].keys()) == ["billing", "shipping", "technical"] + assert abs(sum(route["probabilities"].values()) - 1.0) < 1e-4 + assert route["choice"] == max(route["probabilities"], key=route["probabilities"].get) + assert 0.0 <= route["confidence"] <= 1.0 + + urgency = answers["urgency"] + assert urgency["type"] == "score" + assert urgency["legend"] == {"0": "can wait", "1": "this week", "2": "today", "3": "right now"} + assert list(urgency["probabilities"].keys()) == ["0", "1", "2", "3"] + assert abs(sum(urgency["probabilities"].values()) - 1.0) < 1e-4 + assert abs(urgency["score"] - sum(i * p for i, p in enumerate(urgency["probabilities"].values()))) < 1e-4 + assert 0.0 <= urgency["confidence"] <= 1.0 + + angry = answers["angry"] + assert angry["type"] == "noul" + assert 0.0 <= angry["noul"] <= 1.0 + + +def test_systemone_json_state(): + global server + server.start() + questions = { + "refund": { + "type": "noul", + "instructions": "Is a refund requested?", + "criteria": {"false": "no refund is asked", "true": "a refund is asked"}, + }, + } + res_obj = server.make_request("POST", "/v1/systemone", data={ + "state": {"ticket": TEST_STATE, "plan": "pro"}, + "questions": questions, + }) + assert res_obj.status_code == 200 + # an object is given to the model as JSON text + res_str = server.make_request("POST", "/v1/systemone", data={ + "state": '{"ticket": "' + TEST_STATE + '", "plan": "pro"}', + "questions": questions, + }) + assert res_str.status_code == 200 + assert res_obj.body["usage"] == res_str.body["usage"] + assert abs(res_obj.body["answers"]["refund"]["noul"] - res_str.body["answers"]["refund"]["noul"]) < 1e-4 + + +@pytest.mark.parametrize("data", [ + {"questions": TEST_QUESTIONS}, + {"state": TEST_STATE}, + {"state": TEST_STATE, "questions": {}}, + {"state": TEST_STATE, "questions": {"q": {"type": "unknown", "instructions": "x"}}}, + {"state": TEST_STATE, "questions": {"q": {"type": "noul"}}}, + {"state": TEST_STATE, "questions": {"q": {"type": "choice", "instructions": "x"}}}, + {"state": TEST_STATE, "questions": {"q": {"type": "choice", "instructions": "x", "criteria": {}}}}, + {"state": TEST_STATE, "questions": {"q": {"type": "score", "instructions": "x", "criteria": ["only one"]}}}, +]) +def test_systemone_invalid_request(data: dict): + global server + server.start() + res = server.make_request("POST", "/v1/systemone", data=data) + assert res.status_code == 400 + assert "error" in res.body + + +def test_systemone_requires_embedding(): + global server + server.server_embeddings = False + server.start() + res = server.make_request("POST", "/v1/systemone", data={ + "state": TEST_STATE, + "questions": TEST_QUESTIONS, + }) + assert res.status_code == 501 diff --git a/tools/server/tests/utils.py b/tools/server/tests/utils.py index 3a50ae5c3f..63983a1aae 100644 --- a/tools/server/tests/utils.py +++ b/tools/server/tests/utils.py @@ -628,6 +628,25 @@ class ServerPreset: server.server_reranking = True return server + @staticmethod + def tinylaya() -> ServerProcess: + server = ServerProcess() + server.offline = True # will be downloaded by load_all() + local_model = os.environ.get("TINYLAYA_LOCAL_MODEL") + server.model_hf_file = None + if local_model: + server.model_file = local_model + server.model_hf_repo = None + else: + server.model_hf_repo = "ggml-org/tinylaya-for-testing-gguf" + server.n_ctx = 1024 + server.n_batch = 512 + server.n_ubatch = 512 + server.n_slots = 2 + server.seed = 42 + server.server_embeddings = True + return server + @staticmethod def tinygemma3() -> ServerProcess: server = ServerProcess()