mirror of
https://github.com/ggml-org/llama.cpp.git
synced 2026-10-02 02:47:26 -05:00
add server code
This commit is contained in:
@@ -13,6 +13,8 @@ add_library(${TARGET} STATIC
|
||||
server-queue.h
|
||||
server-common.cpp
|
||||
server-common.h
|
||||
server-decision.cpp
|
||||
server-decision.h
|
||||
server-context.cpp
|
||||
server-context.h
|
||||
server-stream.cpp
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
#include "server-context.h"
|
||||
#include "server-chat.h"
|
||||
#include "server-common.h"
|
||||
#include "server-decision.h"
|
||||
#include "server-http.h"
|
||||
#include "server-task.h"
|
||||
#include "server-queue.h"
|
||||
@@ -825,6 +826,8 @@ public:
|
||||
mtmd_helper_init_opt init_opt = mtmd_helper_init_opt_default();
|
||||
const llama_vocab * vocab = nullptr;
|
||||
|
||||
server_decision_context decision;
|
||||
|
||||
server_queue queue_tasks;
|
||||
server_response queue_results;
|
||||
|
||||
@@ -1102,6 +1105,13 @@ private:
|
||||
|
||||
vocab = llama_model_get_vocab(model_tgt);
|
||||
|
||||
try {
|
||||
decision.init(model_tgt);
|
||||
} catch (const std::exception & e) {
|
||||
SRV_ERR("failed to init decision model: %s\n", e.what());
|
||||
return false;
|
||||
}
|
||||
|
||||
n_ctx = llama_n_ctx(ctx_tgt);
|
||||
|
||||
add_bos_token = llama_vocab_get_add_bos(vocab);
|
||||
@@ -2194,6 +2204,49 @@ private:
|
||||
queue_results.send(std::move(res));
|
||||
}
|
||||
|
||||
void send_decision(const server_slot & slot, const common_batch & batch, int32_t i_batch) {
|
||||
auto res = std::make_unique<server_task_result_decision>();
|
||||
res->id = slot.task->id;
|
||||
res->index = slot.task->index;
|
||||
res->n_tokens = slot.task->n_tokens();
|
||||
|
||||
const auto & decision = slot.task->decision;
|
||||
|
||||
if (!decision.labels.empty()) {
|
||||
const float * logits = llama_get_logits_ith(slot.ctx_tgt, i_batch);
|
||||
if (logits == nullptr) {
|
||||
send_error(slot, "failed to get logits", ERROR_TYPE_SERVER);
|
||||
return;
|
||||
}
|
||||
const int32_t n_vocab = llama_vocab_n_tokens(vocab);
|
||||
for (const llama_token label : decision.labels) {
|
||||
GGML_ASSERT(label >= 0 && label < n_vocab);
|
||||
res->scores.push_back(logits[label]);
|
||||
}
|
||||
} else {
|
||||
// the prompt is evaluated in one batch, the n-th output of this slot is the n-th prompt token
|
||||
std::vector<int32_t> idx;
|
||||
for (int i = 0; i < batch.size(); ++i) {
|
||||
if (batch.tokens[i].output && batch.tokens[i].seq_id == slot.id) {
|
||||
idx.push_back(i);
|
||||
}
|
||||
}
|
||||
GGML_ASSERT(decision.column >= 0 && decision.column < llama_model_n_embd_out(model_tgt));
|
||||
for (const int32_t marker : decision.markers) {
|
||||
const float * embd = marker >= 0 && marker < (int32_t) idx.size() ? llama_get_embeddings_ith(slot.ctx_tgt, idx[marker]) : nullptr;
|
||||
if (embd == nullptr) {
|
||||
send_error(slot, "failed to get embeddings", ERROR_TYPE_SERVER);
|
||||
return;
|
||||
}
|
||||
res->scores.push_back(embd[decision.column]);
|
||||
}
|
||||
}
|
||||
|
||||
SLT_DBG(slot, "%s", "sending decision result\n");
|
||||
|
||||
queue_results.send(std::move(res));
|
||||
}
|
||||
|
||||
void send_rerank(const server_slot & slot, const common_batch & batch) {
|
||||
auto res = std::make_unique<server_task_result_rerank>();
|
||||
res->id = slot.task->id;
|
||||
@@ -2384,6 +2437,7 @@ private:
|
||||
case SERVER_TASK_TYPE_INFILL:
|
||||
case SERVER_TASK_TYPE_EMBEDDING:
|
||||
case SERVER_TASK_TYPE_RERANK:
|
||||
case SERVER_TASK_TYPE_DECISION:
|
||||
{
|
||||
// special case: if input is provided via CLI, tokenize it first
|
||||
// otherwise, no need to tokenize as it's already done inside the HTTP thread
|
||||
@@ -3831,6 +3885,13 @@ private:
|
||||
return;
|
||||
}
|
||||
|
||||
if (slot.task->type == SERVER_TASK_TYPE_DECISION) {
|
||||
send_decision(slot, batch.view, slot.i_batch - off);
|
||||
slot.release();
|
||||
slot.i_batch = -1;
|
||||
return;
|
||||
}
|
||||
|
||||
GGML_ASSERT(slot.task->need_sampling());
|
||||
|
||||
// prompt evaluated for next-token prediction
|
||||
@@ -5224,6 +5285,64 @@ void server_routes::init_routes() {
|
||||
return res;
|
||||
};
|
||||
|
||||
this->post_systemone = [this](const server_http_req & req) {
|
||||
auto res = create_response();
|
||||
const auto & decision = ctx_server.decision;
|
||||
if (decision.type == SERVER_DECISION_TYPE_NONE) {
|
||||
res->error(format_error_response("This model is not a decision model", ERROR_TYPE_NOT_SUPPORTED));
|
||||
return res;
|
||||
}
|
||||
if (decision.need_embd() && (!params.embedding || meta->pooling_type != LLAMA_POOLING_TYPE_NONE)) {
|
||||
res->error(format_error_response("This decision model requires `--embedding --pooling none`", ERROR_TYPE_NOT_SUPPORTED));
|
||||
return res;
|
||||
}
|
||||
|
||||
const json body = json::parse(req.body);
|
||||
const auto questions = decision.parse_questions(body);
|
||||
|
||||
// one task per question
|
||||
auto & rd = res->rd;
|
||||
{
|
||||
std::vector<server_task> tasks;
|
||||
tasks.reserve(questions.size());
|
||||
for (const auto & question : questions) {
|
||||
server_task task = server_task(SERVER_TASK_TYPE_DECISION);
|
||||
task.id = rd.get_new_id();
|
||||
decision.fill_task(body.at("state"), question, task);
|
||||
tasks.push_back(std::move(task));
|
||||
}
|
||||
rd.post_tasks(std::move(tasks));
|
||||
}
|
||||
|
||||
auto all_results = rd.wait_for_all(req.should_stop);
|
||||
|
||||
if (all_results.is_terminated) {
|
||||
return res; // connection is closed
|
||||
} else if (all_results.error) {
|
||||
res->error(all_results.error->to_json());
|
||||
return res;
|
||||
}
|
||||
|
||||
json answers = json::object();
|
||||
int32_t n_tokens = 0;
|
||||
for (size_t i = 0; i < questions.size(); i++) {
|
||||
auto * result = dynamic_cast<server_task_result_decision *>(all_results.results[i].get());
|
||||
GGML_ASSERT(result != nullptr);
|
||||
answers[questions[i].id] = decision.format_answer(questions[i], result->scores);
|
||||
n_tokens += result->n_tokens;
|
||||
}
|
||||
|
||||
res->ok(json{
|
||||
{"model", meta->model_name},
|
||||
{"answers", answers},
|
||||
{"usage", {
|
||||
{"input_tokens", n_tokens},
|
||||
{"output_tokens", 0},
|
||||
}},
|
||||
});
|
||||
return res;
|
||||
};
|
||||
|
||||
this->get_lora_adapters = [this](const server_http_req & req) {
|
||||
auto res = create_response();
|
||||
|
||||
|
||||
@@ -152,6 +152,7 @@ struct server_routes {
|
||||
server_http_context::handler_t post_embeddings;
|
||||
server_http_context::handler_t post_embeddings_oai;
|
||||
server_http_context::handler_t post_rerank;
|
||||
server_http_context::handler_t post_systemone;
|
||||
server_http_context::handler_t get_lora_adapters;
|
||||
server_http_context::handler_t post_lora_adapters;
|
||||
|
||||
|
||||
@@ -0,0 +1,391 @@
|
||||
#include "server-decision.h"
|
||||
|
||||
#include <algorithm>
|
||||
#include <cmath>
|
||||
#include <stdexcept>
|
||||
|
||||
static const char * decision_question_type_name(server_decision_question_type type) {
|
||||
switch (type) {
|
||||
case SERVER_DECISION_QUESTION_CHOICE: return "choice";
|
||||
case SERVER_DECISION_QUESTION_SCORE: return "score";
|
||||
case SERVER_DECISION_QUESTION_NOUL: return "noul";
|
||||
}
|
||||
return "";
|
||||
}
|
||||
|
||||
static std::string decision_meta_str(const llama_model * model, const std::string & key) {
|
||||
char buf[256];
|
||||
const int32_t n = llama_model_meta_val_str(model, key.c_str(), buf, sizeof(buf));
|
||||
return n < 0 ? "" : std::string(buf);
|
||||
}
|
||||
|
||||
//
|
||||
// model-specific setup
|
||||
//
|
||||
|
||||
void server_decision_context::init(const llama_model * model) {
|
||||
*this = server_decision_context(); // the model can be reloaded
|
||||
|
||||
const std::string prefix = decision_meta_str(model, "general.architecture") + ".decision.";
|
||||
const std::string type_name = decision_meta_str(model, prefix + "type");
|
||||
if (type_name.empty()) {
|
||||
return;
|
||||
}
|
||||
|
||||
vocab = llama_model_get_vocab(model);
|
||||
|
||||
const char * tmpl_src = llama_model_chat_template(model, "systemone");
|
||||
if (tmpl_src == nullptr) {
|
||||
throw std::runtime_error("decision model has no \"systemone\" template");
|
||||
}
|
||||
tmpl = std::make_shared<const common_chat_template>(tmpl_src, "", "");
|
||||
|
||||
const std::string prefix_temp = prefix + "temperature.";
|
||||
for (int32_t i = 0; i < llama_model_meta_count(model); i++) {
|
||||
char key[256];
|
||||
char val[64];
|
||||
if (llama_model_meta_key_by_index(model, i, key, sizeof(key)) < 0 || !string_starts_with(key, prefix_temp)) {
|
||||
continue;
|
||||
}
|
||||
if (llama_model_meta_val_str_by_index(model, i, val, sizeof(val)) < 0) {
|
||||
continue;
|
||||
}
|
||||
const float temp = std::strtof(val, nullptr);
|
||||
if (temp <= 0.0f) {
|
||||
throw std::runtime_error(string_format("invalid decision temperature: %s = %s", key, val));
|
||||
}
|
||||
temperatures[key + prefix_temp.size()] = temp;
|
||||
}
|
||||
|
||||
if (type_name == "openjev") {
|
||||
// one letter per option, each must be a single token
|
||||
const std::string letters = "ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz";
|
||||
for (const char c : letters) {
|
||||
const auto toks = common_tokenize(vocab, std::string(1, c), false, false);
|
||||
if (toks.size() != 1) {
|
||||
throw std::runtime_error(string_format("decision label '%c' is not a single token", c));
|
||||
}
|
||||
labels.push_back(toks[0]);
|
||||
}
|
||||
n_options_max = labels.size();
|
||||
noul_true_first = true;
|
||||
type = SERVER_DECISION_TYPE_OPENJEV;
|
||||
} else if (type_name == "laya") {
|
||||
token_marker = llama_vocab_mask(vocab);
|
||||
token_sep = llama_vocab_sep(vocab);
|
||||
if (token_marker == LLAMA_TOKEN_NULL || token_sep == LLAMA_TOKEN_NULL) {
|
||||
throw std::runtime_error("decision model has no mask or sep token");
|
||||
}
|
||||
text_marker = common_token_to_piece(vocab, token_marker, true);
|
||||
|
||||
const std::string val = decision_meta_str(model, prefix + "max_head_tokens");
|
||||
max_head_tokens = std::strtoul(val.c_str(), nullptr, 10);
|
||||
if (max_head_tokens == 0) {
|
||||
throw std::runtime_error("decision model has no valid max_head_tokens");
|
||||
}
|
||||
n_options_max = 255;
|
||||
type = SERVER_DECISION_TYPE_LAYA;
|
||||
} else {
|
||||
throw std::runtime_error("unsupported decision model type: " + type_name);
|
||||
}
|
||||
|
||||
SRV_INF("decision model type: %s\n", type_name.c_str());
|
||||
}
|
||||
|
||||
//
|
||||
// request parsing
|
||||
//
|
||||
|
||||
std::vector<server_decision_question> server_decision_context::parse_questions(const json & body) const {
|
||||
if (!body.contains("state") || body.at("state").is_null()) {
|
||||
throw std::invalid_argument("\"state\" must be provided");
|
||||
}
|
||||
if (!body.contains("questions") || !body.at("questions").is_object() || body.at("questions").empty()) {
|
||||
throw std::invalid_argument("\"questions\" must be a non-empty object");
|
||||
}
|
||||
|
||||
std::vector<server_decision_question> questions;
|
||||
for (const auto & [id, q] : body.at("questions").items()) {
|
||||
auto err = [&id = id](const std::string & msg) {
|
||||
return std::invalid_argument("questions." + id + ": " + msg);
|
||||
};
|
||||
if (!q.is_object()) {
|
||||
throw err("must be an object");
|
||||
}
|
||||
if (!q.contains("instructions") || q.at("instructions").is_null()) {
|
||||
throw err("\"instructions\" must be provided");
|
||||
}
|
||||
|
||||
server_decision_question question;
|
||||
question.id = id;
|
||||
question.instructions = q.at("instructions");
|
||||
|
||||
const std::string type_name = json_value(q, "type", std::string());
|
||||
const json criteria = q.contains("criteria") ? q.at("criteria") : json();
|
||||
|
||||
if (type_name == "choice") {
|
||||
question.type = SERVER_DECISION_QUESTION_CHOICE;
|
||||
if (!criteria.is_object() || criteria.empty()) {
|
||||
throw err("\"criteria\" must be a non-empty object");
|
||||
}
|
||||
for (const auto & [key, description] : criteria.items()) {
|
||||
question.options.push_back({key, description});
|
||||
}
|
||||
} else if (type_name == "score") {
|
||||
question.type = SERVER_DECISION_QUESTION_SCORE;
|
||||
if (!criteria.is_array() || criteria.size() < 2 || criteria.size() > 10) {
|
||||
throw err("\"criteria\" must be an array of 2 to 10 levels");
|
||||
}
|
||||
for (size_t i = 0; i < criteria.size(); i++) {
|
||||
question.options.push_back({std::to_string(i), criteria.at(i)});
|
||||
}
|
||||
} else if (type_name == "noul") {
|
||||
question.type = SERVER_DECISION_QUESTION_NOUL;
|
||||
if (!criteria.is_null() && !criteria.is_object()) {
|
||||
throw err("\"criteria\" must be an object");
|
||||
}
|
||||
for (const char * key : {"false", "true"}) {
|
||||
question.options.push_back({key, criteria.is_object() && criteria.contains(key) ? criteria.at(key) : json()});
|
||||
}
|
||||
if (noul_true_first) {
|
||||
std::swap(question.options[0], question.options[1]);
|
||||
}
|
||||
} else {
|
||||
throw err("\"type\" must be one of: choice, score, noul");
|
||||
}
|
||||
|
||||
if (question.options.size() > n_options_max) {
|
||||
throw err(string_format("too many options (%zu), this model supports at most %zu", question.options.size(), n_options_max));
|
||||
}
|
||||
|
||||
questions.push_back(std::move(question));
|
||||
}
|
||||
return questions;
|
||||
}
|
||||
|
||||
//
|
||||
// prompt
|
||||
//
|
||||
|
||||
// replace text in all strings of a JSON value
|
||||
static json decision_replace_text(const json & val, const std::string & search, const std::string & replace) {
|
||||
if (val.is_string()) {
|
||||
std::string str = val.get<std::string>();
|
||||
string_replace_all(str, search, replace);
|
||||
return str;
|
||||
}
|
||||
if (val.is_array()) {
|
||||
json out = json::array();
|
||||
for (const auto & item : val) {
|
||||
out.push_back(decision_replace_text(item, search, replace));
|
||||
}
|
||||
return out;
|
||||
}
|
||||
if (val.is_object()) {
|
||||
json out = json::object();
|
||||
for (const auto & [key, item] : val.items()) {
|
||||
out[key] = decision_replace_text(item, search, replace);
|
||||
}
|
||||
return out;
|
||||
}
|
||||
return val;
|
||||
}
|
||||
|
||||
std::string server_decision_context::render(const json & state, const server_decision_question & question) const {
|
||||
json options = json::array();
|
||||
for (const auto & opt : question.options) {
|
||||
options.push_back(json{
|
||||
{"key", opt.key},
|
||||
{"description", opt.description},
|
||||
});
|
||||
}
|
||||
|
||||
// the template is given raw JSON values, it serializes the ones that are not strings
|
||||
json inp = json{
|
||||
{"type", decision_question_type_name(question.type)},
|
||||
{"instructions", question.instructions},
|
||||
{"state", state},
|
||||
{"options", options},
|
||||
};
|
||||
|
||||
// the input must not contain the marker of the options
|
||||
if (!text_marker.empty()) {
|
||||
inp = decision_replace_text(inp, text_marker, " ");
|
||||
}
|
||||
|
||||
jinja::context ctx(tmpl->source());
|
||||
jinja::global_from_json(ctx, inp, false);
|
||||
jinja::runtime runtime(ctx);
|
||||
const jinja::value results = runtime.execute(tmpl->prog);
|
||||
return jinja::runtime::gather_string_parts(results)->as_string().str();
|
||||
}
|
||||
|
||||
void server_decision_context::fill_task(const json & state, const server_decision_question & question, server_task & task) const {
|
||||
llama_tokens tokens = common_tokenize(vocab, render(state, question), false, true);
|
||||
|
||||
if (type == SERVER_DECISION_TYPE_OPENJEV) {
|
||||
task.decision.labels.assign(labels.begin(), labels.begin() + question.options.size());
|
||||
} else {
|
||||
fill_task_laya(tokens, question, task);
|
||||
}
|
||||
|
||||
task.tokens = server_tokens(tokens, false);
|
||||
}
|
||||
|
||||
// the prompt is: [cls] question [sep] ([marker] option)* [sep] state [sep]
|
||||
// options and question are cut to fit max_head_tokens, the same way the model was trained
|
||||
void server_decision_context::fill_task_laya(llama_tokens & tokens, const server_decision_question & question, server_task & task) const {
|
||||
const size_t n_options = question.options.size();
|
||||
|
||||
std::vector<size_t> markers;
|
||||
for (size_t i = 0; i < tokens.size(); i++) {
|
||||
if (tokens[i] == token_marker) {
|
||||
markers.push_back(i);
|
||||
}
|
||||
}
|
||||
const auto invalid = std::runtime_error("unexpected layout of the decision prompt");
|
||||
if (markers.size() != n_options || markers[0] < 2 || tokens[markers[0] - 1] != token_sep || tokens.back() != token_sep) {
|
||||
throw invalid;
|
||||
}
|
||||
const size_t head_end = markers[0] - 1;
|
||||
const size_t opts_end = std::find(tokens.begin() + markers.back(), tokens.end(), token_sep) - tokens.begin();
|
||||
if (opts_end + 1 >= tokens.size()) {
|
||||
throw invalid;
|
||||
}
|
||||
|
||||
// marker + text of each option
|
||||
std::vector<llama_tokens> options;
|
||||
size_t n_options_tokens = 0;
|
||||
auto set_max = [&](size_t n_max) {
|
||||
n_options_tokens = 0;
|
||||
for (auto & opt : options) {
|
||||
opt.resize(std::min(opt.size(), n_max));
|
||||
n_options_tokens += opt.size();
|
||||
}
|
||||
};
|
||||
for (size_t i = 0; i < n_options; i++) {
|
||||
const size_t end = i + 1 < n_options ? markers[i + 1] : opts_end;
|
||||
options.emplace_back(tokens.begin() + markers[i], tokens.begin() + end);
|
||||
}
|
||||
set_max(max_option_tokens + 1);
|
||||
if (n_options_tokens + 16 > max_head_tokens) {
|
||||
// too many or too long options, shrink them evenly
|
||||
set_max(std::max((size_t) 4, (max_head_tokens - std::min(max_head_tokens, (size_t) 16)) / n_options));
|
||||
}
|
||||
const size_t n_question_max = std::max((size_t) 8, max_head_tokens - std::min(max_head_tokens, n_options_tokens));
|
||||
|
||||
llama_tokens out;
|
||||
out.push_back(tokens[0]);
|
||||
out.insert(out.end(), tokens.begin() + 1, tokens.begin() + std::min(head_end, 1 + n_question_max));
|
||||
out.push_back(token_sep);
|
||||
for (const auto & opt : options) {
|
||||
task.decision.markers.push_back(out.size());
|
||||
out.insert(out.end(), opt.begin(), opt.end());
|
||||
}
|
||||
out.insert(out.end(), tokens.begin() + opts_end, tokens.end());
|
||||
tokens = std::move(out);
|
||||
|
||||
// the output has one score per question type
|
||||
task.decision.column = question.type;
|
||||
}
|
||||
|
||||
//
|
||||
// answer
|
||||
//
|
||||
|
||||
float server_decision_context::get_temperature(const server_decision_question & question) const {
|
||||
const size_t n = question.options.size();
|
||||
const std::string type_name = decision_question_type_name(question.type);
|
||||
const std::string bucket = n <= 2 ? "2" : n <= 5 ? "3_5" : n <= 10 ? "6_10" : "11";
|
||||
|
||||
for (const auto & name : {type_name + "." + bucket, type_name}) {
|
||||
const auto it = temperatures.find(name);
|
||||
if (it != temperatures.end()) {
|
||||
return it->second;
|
||||
}
|
||||
}
|
||||
return 1.0f;
|
||||
}
|
||||
|
||||
// confidence formulas are the ones published by TypeSafe
|
||||
|
||||
static double decision_confidence_choice(const std::vector<double> & probs) {
|
||||
if (probs.size() < 2) {
|
||||
return 1.0;
|
||||
}
|
||||
const double uniform = 1.0 / probs.size();
|
||||
const double p_max = *std::max_element(probs.begin(), probs.end());
|
||||
return std::max(0.0, (p_max - uniform) / (1.0 - uniform));
|
||||
}
|
||||
|
||||
static double decision_confidence_score(const std::vector<double> & probs) {
|
||||
if (probs.size() < 2) {
|
||||
return 1.0;
|
||||
}
|
||||
const size_t n = probs.size();
|
||||
const size_t mode = std::max_element(probs.begin(), probs.end()) - probs.begin();
|
||||
|
||||
// mean distance to the mode, relative to the one of a uniform distribution around its center
|
||||
double dist = 0.0;
|
||||
double dist_uniform = 0.0;
|
||||
for (size_t i = 0; i < n; i++) {
|
||||
dist += probs[i] * std::fabs((double) i - (double) mode);
|
||||
dist_uniform += std::fabs((double) i - (n - 1) / 2.0) / n;
|
||||
}
|
||||
return std::max(0.0, 1.0 - dist / dist_uniform);
|
||||
}
|
||||
|
||||
json server_decision_context::format_answer(const server_decision_question & question, const std::vector<float> & scores) const {
|
||||
const size_t n = question.options.size();
|
||||
if (scores.size() != n) {
|
||||
throw std::runtime_error("decision result does not match the number of options");
|
||||
}
|
||||
|
||||
// softmax over the options
|
||||
const float temperature = get_temperature(question);
|
||||
const float score_max = *std::max_element(scores.begin(), scores.end());
|
||||
std::vector<double> probs(n);
|
||||
double sum = 0.0;
|
||||
for (size_t i = 0; i < n; i++) {
|
||||
probs[i] = std::exp((double) (scores[i] - score_max) / temperature);
|
||||
sum += probs[i];
|
||||
}
|
||||
for (auto & p : probs) {
|
||||
p /= sum;
|
||||
}
|
||||
|
||||
json answer = json{{"type", decision_question_type_name(question.type)}};
|
||||
|
||||
if (question.type == SERVER_DECISION_QUESTION_NOUL) {
|
||||
for (size_t i = 0; i < n; i++) {
|
||||
if (question.options[i].key == "true") {
|
||||
answer["noul"] = probs[i];
|
||||
}
|
||||
}
|
||||
return answer;
|
||||
}
|
||||
|
||||
json probabilities = json::object();
|
||||
for (size_t i = 0; i < n; i++) {
|
||||
probabilities[question.options[i].key] = probs[i];
|
||||
}
|
||||
|
||||
if (question.type == SERVER_DECISION_QUESTION_CHOICE) {
|
||||
const size_t best = std::max_element(probs.begin(), probs.end()) - probs.begin();
|
||||
answer["choice"] = question.options[best].key;
|
||||
answer["probabilities"] = probabilities;
|
||||
answer["confidence"] = decision_confidence_choice(probs);
|
||||
} else {
|
||||
double expected = 0.0;
|
||||
json legend = json::object();
|
||||
for (size_t i = 0; i < n; i++) {
|
||||
expected += i * probs[i];
|
||||
legend[question.options[i].key] = question.options[i].description;
|
||||
}
|
||||
answer["score"] = expected;
|
||||
answer["legend"] = legend;
|
||||
answer["probabilities"] = probabilities;
|
||||
answer["confidence"] = decision_confidence_score(probs);
|
||||
}
|
||||
return answer;
|
||||
}
|
||||
@@ -0,0 +1,78 @@
|
||||
#pragma once
|
||||
|
||||
#include "server-common.h"
|
||||
#include "server-task.h"
|
||||
|
||||
#include <map>
|
||||
#include <memory>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
// typed decision models (TypeSafe /v1/systemone API)
|
||||
// the model answers each question in one forward pass, no token is generated
|
||||
|
||||
enum server_decision_type {
|
||||
SERVER_DECISION_TYPE_NONE, // not a decision model
|
||||
SERVER_DECISION_TYPE_OPENJEV, // logits of one label token per option, read at the last prompt token
|
||||
SERVER_DECISION_TYPE_LAYA, // score of one marker token per option, read from the embeddings output
|
||||
};
|
||||
|
||||
enum server_decision_question_type {
|
||||
SERVER_DECISION_QUESTION_CHOICE,
|
||||
SERVER_DECISION_QUESTION_SCORE,
|
||||
SERVER_DECISION_QUESTION_NOUL,
|
||||
};
|
||||
|
||||
struct server_decision_option {
|
||||
std::string key;
|
||||
json description; // null if not provided
|
||||
};
|
||||
|
||||
struct server_decision_question {
|
||||
std::string id;
|
||||
server_decision_question_type type;
|
||||
json instructions;
|
||||
std::vector<server_decision_option> options; // in the order of the model outputs
|
||||
};
|
||||
|
||||
struct server_decision_context {
|
||||
server_decision_type type = SERVER_DECISION_TYPE_NONE;
|
||||
|
||||
// read the "<arch>.decision.*" metadata, type stays NONE if the model has none
|
||||
void init(const llama_model * model);
|
||||
|
||||
// true if the result is read from the embeddings of each token
|
||||
bool need_embd() const { return type == SERVER_DECISION_TYPE_LAYA; }
|
||||
|
||||
// throw std::invalid_argument on bad input
|
||||
std::vector<server_decision_question> parse_questions(const json & body) const;
|
||||
|
||||
// set the prompt of this question, and where to read its result
|
||||
void fill_task(const json & state, const server_decision_question & question, server_task & task) const;
|
||||
|
||||
// scores: one raw model output per option
|
||||
json format_answer(const server_decision_question & question, const std::vector<float> & scores) const;
|
||||
|
||||
private:
|
||||
const llama_vocab * vocab = nullptr;
|
||||
std::shared_ptr<const common_chat_template> tmpl; // the "systemone" template
|
||||
|
||||
std::map<std::string, float> temperatures; // "<type>" or "<type>.<n_options bucket>"
|
||||
size_t n_options_max = 0;
|
||||
bool noul_true_first = false; // noul options are [true, false] instead of [false, true]
|
||||
|
||||
// OPENJEV
|
||||
std::vector<llama_token> labels;
|
||||
|
||||
// LAYA
|
||||
llama_token token_marker = LLAMA_TOKEN_NULL;
|
||||
llama_token token_sep = LLAMA_TOKEN_NULL;
|
||||
std::string text_marker;
|
||||
size_t max_head_tokens = 0; // question + options
|
||||
size_t max_option_tokens = 48;
|
||||
|
||||
std::string render(const json & state, const server_decision_question & question) const;
|
||||
void fill_task_laya(llama_tokens & tokens, const server_decision_question & question, server_task & task) const;
|
||||
|
||||
float get_temperature(const server_decision_question & question) const;
|
||||
};
|
||||
@@ -1495,6 +1495,17 @@ json server_task_result_rerank::to_json() {
|
||||
};
|
||||
}
|
||||
|
||||
//
|
||||
// server_task_result_decision
|
||||
//
|
||||
json server_task_result_decision::to_json() {
|
||||
return json {
|
||||
{"index", index},
|
||||
{"scores", scores},
|
||||
{"tokens_evaluated", n_tokens},
|
||||
};
|
||||
}
|
||||
|
||||
//
|
||||
// server_task_result_error
|
||||
//
|
||||
|
||||
@@ -16,6 +16,7 @@ enum server_task_type {
|
||||
SERVER_TASK_TYPE_COMPLETION,
|
||||
SERVER_TASK_TYPE_EMBEDDING,
|
||||
SERVER_TASK_TYPE_RERANK,
|
||||
SERVER_TASK_TYPE_DECISION,
|
||||
SERVER_TASK_TYPE_INFILL,
|
||||
SERVER_TASK_TYPE_CANCEL,
|
||||
SERVER_TASK_TYPE_CONTROL,
|
||||
@@ -172,6 +173,15 @@ struct server_task {
|
||||
// used by SERVER_TASK_TYPE_METRICS
|
||||
bool metrics_reset_bucket = false;
|
||||
|
||||
// used by SERVER_TASK_TYPE_DECISION
|
||||
// where to read the model output of each option, exactly one of the two lists is used
|
||||
struct decision {
|
||||
std::vector<llama_token> labels; // logits of these tokens, at the last prompt token
|
||||
std::vector<int32_t> markers; // embeddings[column] at these prompt positions
|
||||
int32_t column = 0;
|
||||
};
|
||||
decision decision;
|
||||
|
||||
// used by SERVER_TASK_TYPE_SET_LORA
|
||||
std::map<int, float> set_lora; // mapping adapter ID -> scale
|
||||
|
||||
@@ -188,6 +198,8 @@ struct server_task {
|
||||
case SERVER_TASK_TYPE_EMBEDDING:
|
||||
case SERVER_TASK_TYPE_RERANK:
|
||||
return true;
|
||||
case SERVER_TASK_TYPE_DECISION:
|
||||
return !decision.markers.empty();
|
||||
default:
|
||||
return false;
|
||||
}
|
||||
@@ -198,6 +210,8 @@ struct server_task {
|
||||
case SERVER_TASK_TYPE_COMPLETION:
|
||||
case SERVER_TASK_TYPE_INFILL:
|
||||
return true;
|
||||
case SERVER_TASK_TYPE_DECISION:
|
||||
return !decision.labels.empty();
|
||||
default:
|
||||
return false;
|
||||
}
|
||||
@@ -474,6 +488,14 @@ struct server_task_result_rerank : server_task_result {
|
||||
virtual json to_json() override;
|
||||
};
|
||||
|
||||
struct server_task_result_decision : server_task_result {
|
||||
std::vector<float> scores; // one raw model output per option
|
||||
|
||||
int32_t n_tokens;
|
||||
|
||||
virtual json to_json() override;
|
||||
};
|
||||
|
||||
struct server_task_result_error : server_task_result {
|
||||
error_type err_type = ERROR_TYPE_SERVER;
|
||||
std::string err_msg;
|
||||
|
||||
@@ -231,6 +231,7 @@ int llama_server(common_params & params, int argc, char ** argv) {
|
||||
routes.post_embeddings = models_routes->proxy_post;
|
||||
routes.post_embeddings_oai = models_routes->proxy_post;
|
||||
routes.post_rerank = models_routes->proxy_post;
|
||||
routes.post_systemone = models_routes->proxy_post;
|
||||
routes.post_tokenize = models_routes->proxy_post;
|
||||
routes.post_detokenize = models_routes->proxy_post;
|
||||
routes.post_apply_template = models_routes->proxy_post;
|
||||
@@ -278,6 +279,7 @@ int llama_server(common_params & params, int argc, char ** argv) {
|
||||
ctx_http.post("/reranking", ex_wrapper(routes.post_rerank));
|
||||
ctx_http.post("/v1/rerank", ex_wrapper(routes.post_rerank));
|
||||
ctx_http.post("/v1/reranking", ex_wrapper(routes.post_rerank));
|
||||
ctx_http.post("/v1/systemone", ex_wrapper(routes.post_systemone));
|
||||
ctx_http.post("/tokenize", ex_wrapper(routes.post_tokenize));
|
||||
ctx_http.post("/detokenize", ex_wrapper(routes.post_detokenize));
|
||||
ctx_http.post("/apply-template", ex_wrapper(routes.post_apply_template));
|
||||
|
||||
@@ -0,0 +1,118 @@
|
||||
import pytest
|
||||
from utils import *
|
||||
|
||||
server = ServerPreset.tinylaya()
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def create_server():
|
||||
global server
|
||||
server = ServerPreset.tinylaya()
|
||||
|
||||
|
||||
TEST_STATE = "I was charged twice for my order last week and nobody has replied."
|
||||
|
||||
TEST_QUESTIONS = {
|
||||
"route": {
|
||||
"type": "choice",
|
||||
"instructions": "Which team should handle this?",
|
||||
"criteria": {"billing": "payments and refunds", "shipping": None, "technical": None},
|
||||
},
|
||||
"urgency": {
|
||||
"type": "score",
|
||||
"instructions": "How urgent is this?",
|
||||
"criteria": ["can wait", "this week", "today", "right now"],
|
||||
},
|
||||
"angry": {
|
||||
"type": "noul",
|
||||
"instructions": "Is the customer angry?",
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def test_systemone():
|
||||
global server
|
||||
server.start()
|
||||
res = server.make_request("POST", "/v1/systemone", data={
|
||||
"state": TEST_STATE,
|
||||
"questions": TEST_QUESTIONS,
|
||||
})
|
||||
assert res.status_code == 200
|
||||
assert res.body["usage"]["input_tokens"] > 0
|
||||
assert res.body["usage"]["output_tokens"] == 0
|
||||
|
||||
answers = res.body["answers"]
|
||||
assert list(answers.keys()) == ["route", "urgency", "angry"]
|
||||
|
||||
route = answers["route"]
|
||||
assert route["type"] == "choice"
|
||||
assert list(route["probabilities"].keys()) == ["billing", "shipping", "technical"]
|
||||
assert abs(sum(route["probabilities"].values()) - 1.0) < 1e-4
|
||||
assert route["choice"] == max(route["probabilities"], key=route["probabilities"].get)
|
||||
assert 0.0 <= route["confidence"] <= 1.0
|
||||
|
||||
urgency = answers["urgency"]
|
||||
assert urgency["type"] == "score"
|
||||
assert urgency["legend"] == {"0": "can wait", "1": "this week", "2": "today", "3": "right now"}
|
||||
assert list(urgency["probabilities"].keys()) == ["0", "1", "2", "3"]
|
||||
assert abs(sum(urgency["probabilities"].values()) - 1.0) < 1e-4
|
||||
assert abs(urgency["score"] - sum(i * p for i, p in enumerate(urgency["probabilities"].values()))) < 1e-4
|
||||
assert 0.0 <= urgency["confidence"] <= 1.0
|
||||
|
||||
angry = answers["angry"]
|
||||
assert angry["type"] == "noul"
|
||||
assert 0.0 <= angry["noul"] <= 1.0
|
||||
|
||||
|
||||
def test_systemone_json_state():
|
||||
global server
|
||||
server.start()
|
||||
questions = {
|
||||
"refund": {
|
||||
"type": "noul",
|
||||
"instructions": "Is a refund requested?",
|
||||
"criteria": {"false": "no refund is asked", "true": "a refund is asked"},
|
||||
},
|
||||
}
|
||||
res_obj = server.make_request("POST", "/v1/systemone", data={
|
||||
"state": {"ticket": TEST_STATE, "plan": "pro"},
|
||||
"questions": questions,
|
||||
})
|
||||
assert res_obj.status_code == 200
|
||||
# an object is given to the model as JSON text
|
||||
res_str = server.make_request("POST", "/v1/systemone", data={
|
||||
"state": '{"ticket": "' + TEST_STATE + '", "plan": "pro"}',
|
||||
"questions": questions,
|
||||
})
|
||||
assert res_str.status_code == 200
|
||||
assert res_obj.body["usage"] == res_str.body["usage"]
|
||||
assert abs(res_obj.body["answers"]["refund"]["noul"] - res_str.body["answers"]["refund"]["noul"]) < 1e-4
|
||||
|
||||
|
||||
@pytest.mark.parametrize("data", [
|
||||
{"questions": TEST_QUESTIONS},
|
||||
{"state": TEST_STATE},
|
||||
{"state": TEST_STATE, "questions": {}},
|
||||
{"state": TEST_STATE, "questions": {"q": {"type": "unknown", "instructions": "x"}}},
|
||||
{"state": TEST_STATE, "questions": {"q": {"type": "noul"}}},
|
||||
{"state": TEST_STATE, "questions": {"q": {"type": "choice", "instructions": "x"}}},
|
||||
{"state": TEST_STATE, "questions": {"q": {"type": "choice", "instructions": "x", "criteria": {}}}},
|
||||
{"state": TEST_STATE, "questions": {"q": {"type": "score", "instructions": "x", "criteria": ["only one"]}}},
|
||||
])
|
||||
def test_systemone_invalid_request(data: dict):
|
||||
global server
|
||||
server.start()
|
||||
res = server.make_request("POST", "/v1/systemone", data=data)
|
||||
assert res.status_code == 400
|
||||
assert "error" in res.body
|
||||
|
||||
|
||||
def test_systemone_requires_embedding():
|
||||
global server
|
||||
server.server_embeddings = False
|
||||
server.start()
|
||||
res = server.make_request("POST", "/v1/systemone", data={
|
||||
"state": TEST_STATE,
|
||||
"questions": TEST_QUESTIONS,
|
||||
})
|
||||
assert res.status_code == 501
|
||||
@@ -628,6 +628,25 @@ class ServerPreset:
|
||||
server.server_reranking = True
|
||||
return server
|
||||
|
||||
@staticmethod
|
||||
def tinylaya() -> ServerProcess:
|
||||
server = ServerProcess()
|
||||
server.offline = True # will be downloaded by load_all()
|
||||
local_model = os.environ.get("TINYLAYA_LOCAL_MODEL")
|
||||
server.model_hf_file = None
|
||||
if local_model:
|
||||
server.model_file = local_model
|
||||
server.model_hf_repo = None
|
||||
else:
|
||||
server.model_hf_repo = "ggml-org/tinylaya-for-testing-gguf"
|
||||
server.n_ctx = 1024
|
||||
server.n_batch = 512
|
||||
server.n_ubatch = 512
|
||||
server.n_slots = 2
|
||||
server.seed = 42
|
||||
server.server_embeddings = True
|
||||
return server
|
||||
|
||||
@staticmethod
|
||||
def tinygemma3() -> ServerProcess:
|
||||
server = ServerProcess()
|
||||
|
||||
Reference in New Issue
Block a user