mirror of
https://github.com/ggml-org/llama.cpp.git
synced 2026-10-02 19:07:25 -05:00
add docs, imorove UX a bit
This commit is contained in:
@@ -1192,6 +1192,25 @@ struct common_init_result::impl {
|
||||
std::vector<llama_sampler_seq_config> samplers_seq_config;
|
||||
};
|
||||
|
||||
common_decision_type common_get_decision_type(const struct llama_model * model) {
|
||||
char buf[64];
|
||||
if (llama_model_meta_val_str(model, "general.architecture", buf, sizeof(buf)) < 0) {
|
||||
return COMMON_DECISION_TYPE_NONE;
|
||||
}
|
||||
const std::string key = std::string(buf) + ".decision.type";
|
||||
if (llama_model_meta_val_str(model, key.c_str(), buf, sizeof(buf)) < 0) {
|
||||
return COMMON_DECISION_TYPE_NONE;
|
||||
}
|
||||
const std::string type = buf;
|
||||
if (type == "openjev") {
|
||||
return COMMON_DECISION_TYPE_OPENJEV;
|
||||
}
|
||||
if (type == "laya") {
|
||||
return COMMON_DECISION_TYPE_LAYA;
|
||||
}
|
||||
return COMMON_DECISION_TYPE_UNKNOWN;
|
||||
}
|
||||
|
||||
common_init_result::common_init_result(common_params & params, bool model_only) :
|
||||
pimpl(new impl{}) {
|
||||
auto mparams = common_model_params_to_llama(params);
|
||||
@@ -1244,6 +1263,20 @@ common_init_result::common_init_result(common_params & params, bool model_only)
|
||||
|
||||
const llama_vocab * vocab = llama_model_get_vocab(model);
|
||||
|
||||
// this decision model returns a score for each token via the embeddings output
|
||||
// TODO: maybe improve this in the future
|
||||
if (common_get_decision_type(model) == COMMON_DECISION_TYPE_LAYA) {
|
||||
params.embedding = true;
|
||||
params.pooling_type = LLAMA_POOLING_TYPE_NONE;
|
||||
|
||||
cparams.embeddings = true;
|
||||
cparams.pooling_type = LLAMA_POOLING_TYPE_NONE;
|
||||
cparams.n_outputs_max = cparams.n_batch;
|
||||
cparams.n_outputs_max_per_seq = 1;
|
||||
|
||||
LOG_INF("%s", "laya decision model detected, enabling embedding mode\n");
|
||||
}
|
||||
|
||||
// load and optionally apply lora adapters
|
||||
for (auto & la : params.lora_adapters) {
|
||||
llama_adapter_lora_ptr lora;
|
||||
|
||||
@@ -944,6 +944,16 @@ bool tty_can_use_colors();
|
||||
|
||||
struct common_sampler;
|
||||
|
||||
// typed decision models, see "<arch>.decision.type" in the model metadata
|
||||
enum common_decision_type {
|
||||
COMMON_DECISION_TYPE_NONE, // not a decision model
|
||||
COMMON_DECISION_TYPE_UNKNOWN, // a decision model of a type that is not supported
|
||||
COMMON_DECISION_TYPE_OPENJEV, // logits of one label token per option, read at the last prompt token
|
||||
COMMON_DECISION_TYPE_LAYA, // score of one marker token per option, read from the embeddings output
|
||||
};
|
||||
|
||||
common_decision_type common_get_decision_type(const struct llama_model * model);
|
||||
|
||||
// note: defines the model, context, samplers, ets. lifetimes
|
||||
struct common_init_result {
|
||||
common_init_result(common_params & params, bool model_only = false);
|
||||
|
||||
@@ -1665,6 +1665,110 @@ curl http://localhost:8080/v1/messages/count_tokens \
|
||||
{"input_tokens": 10}
|
||||
```
|
||||
|
||||
## TypeSafe-compatible API Endpoints
|
||||
|
||||
### POST `/v1/systemone`: TypeSafe-compatible System One API
|
||||
|
||||
Answers typed questions about a `state` with a decision model.
|
||||
|
||||
Follows the [TypeSafe API](https://docs.typesafe.ai/api), streaming is not supported.
|
||||
|
||||
*Options:*
|
||||
|
||||
`state`: The content to evaluate. Can be a string, an object or an array. A value that is not a string is given to the model as JSON text.
|
||||
|
||||
`questions`: An object that maps a question id to a question. Each question has these fields:
|
||||
|
||||
- `type`: One of `choice`, `score`, `noul`.
|
||||
- `instructions`: The question. Can be a string, an object or an array.
|
||||
- `criteria`: The possible answers, the shape depends on `type`:
|
||||
- `choice`: An object that maps each option to its description. The description can be `null`.
|
||||
- `score`: An array of 2 to 10 level descriptions, lowest level first.
|
||||
- `noul`: Optional. An object with the descriptions of `true` and `false`.
|
||||
|
||||
The questions of a request are answered independently, an answer does not depend on the other questions.
|
||||
|
||||
The number of options of a `choice` question is limited by the model, for example: 52 for openjev, 255 for laya. For laya, long questions and options are truncated to the token budget the model was trained with.
|
||||
|
||||
*Response:*
|
||||
|
||||
`answers`: An object that maps each question id to its answer. The fields depend on the question type:
|
||||
|
||||
- `choice`:
|
||||
- `choice`: The option with the highest probability.
|
||||
- `probabilities`: The probability of each option, they sum to 1.
|
||||
- `confidence`: A value from 0 to 1, where 0 means all options are equally likely.
|
||||
- `score`:
|
||||
- `score`: The expected level index, weighted by probability. It can be between two levels.
|
||||
- `legend`: The description of each level index.
|
||||
- `probabilities`: The probability of each level index, they sum to 1.
|
||||
- `confidence`: A value from 0 to 1.
|
||||
- `noul`:
|
||||
- `noul`: The probability that the answer is true.
|
||||
|
||||
`usage`: `input_tokens` is the number of prompt tokens of all questions. `output_tokens` is always 0.
|
||||
|
||||
The probabilities are scaled with the temperatures stored in the model file. They are not guaranteed to be calibrated for your data.
|
||||
|
||||
*Examples:*
|
||||
|
||||
```shell
|
||||
curl http://127.0.0.1:8080/v1/systemone \
|
||||
-H "Content-Type: application/json" \
|
||||
-d '{
|
||||
"state": "Customer message: I was charged twice for my order last week and nobody has replied.",
|
||||
"questions": {
|
||||
"route": {
|
||||
"type": "choice",
|
||||
"instructions": "Which team should handle this?",
|
||||
"criteria": {"billing": null, "shipping": null, "technical": null}
|
||||
},
|
||||
"angry": {
|
||||
"type": "noul",
|
||||
"instructions": "Is the customer angry?"
|
||||
},
|
||||
"urgency": {
|
||||
"type": "score",
|
||||
"instructions": "How urgent is this?",
|
||||
"criteria": ["can wait", "this week", "today", "right now"]
|
||||
}
|
||||
}
|
||||
}' | jq
|
||||
```
|
||||
|
||||
Response (values are shortened):
|
||||
|
||||
```json
|
||||
{
|
||||
"model": "openjev",
|
||||
"answers": {
|
||||
"route": {
|
||||
"type": "choice",
|
||||
"choice": "billing",
|
||||
"probabilities": {"billing": 0.9998, "shipping": 0.0001, "technical": 0.0001},
|
||||
"confidence": 0.9997
|
||||
},
|
||||
"angry": {
|
||||
"type": "noul",
|
||||
"noul": 0.6328
|
||||
},
|
||||
"urgency": {
|
||||
"type": "score",
|
||||
"score": 2.0858,
|
||||
"legend": {"0": "can wait", "1": "this week", "2": "today", "3": "right now"},
|
||||
"probabilities": {"0": 0.0023, "1": 0.116, "2": 0.6753, "3": 0.2064},
|
||||
"confidence": 0.673
|
||||
}
|
||||
},
|
||||
"usage": {
|
||||
"input_tokens": 239,
|
||||
"output_tokens": 0
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
An invalid request returns the error `400`. A model that is not a decision model returns the error `501`.
|
||||
|
||||
## Server tools
|
||||
|
||||
The server exposes a REST API under `/tools` that allows the Web UI to call server tools. This endpoint is intended to be used internally by the Web UI and subject to change or to be removed in the future.
|
||||
|
||||
@@ -5324,14 +5324,10 @@ void server_routes::init_routes() {
|
||||
this->post_systemone = [this](const server_http_req & req) {
|
||||
auto res = create_response();
|
||||
const auto & decision = ctx_server.decision;
|
||||
if (decision.type == SERVER_DECISION_TYPE_NONE) {
|
||||
if (decision.type == COMMON_DECISION_TYPE_NONE) {
|
||||
res->error(format_error_response("This model is not a decision model", ERROR_TYPE_NOT_SUPPORTED));
|
||||
return res;
|
||||
}
|
||||
if (decision.need_embd() && (!params.embedding || meta->pooling_type != LLAMA_POOLING_TYPE_NONE)) {
|
||||
res->error(format_error_response("This decision model requires `--embedding --pooling none`", ERROR_TYPE_NOT_SUPPORTED));
|
||||
return res;
|
||||
}
|
||||
|
||||
const json body = json::parse(req.body);
|
||||
const auto questions = decision.parse_questions(body);
|
||||
|
||||
@@ -26,12 +26,14 @@ static std::string decision_meta_str(const llama_model * model, const std::strin
|
||||
void server_decision_context::init(const llama_model * model) {
|
||||
*this = server_decision_context(); // the model can be reloaded
|
||||
|
||||
const std::string prefix = decision_meta_str(model, "general.architecture") + ".decision.";
|
||||
const std::string type_name = decision_meta_str(model, prefix + "type");
|
||||
if (type_name.empty()) {
|
||||
const common_decision_type model_type = common_get_decision_type(model);
|
||||
if (model_type == COMMON_DECISION_TYPE_NONE) {
|
||||
return;
|
||||
}
|
||||
|
||||
const std::string prefix = decision_meta_str(model, "general.architecture") + ".decision.";
|
||||
const std::string type_name = decision_meta_str(model, prefix + "type");
|
||||
|
||||
vocab = llama_model_get_vocab(model);
|
||||
|
||||
const char * tmpl_src = llama_model_chat_template(model, "systemone");
|
||||
@@ -57,7 +59,7 @@ void server_decision_context::init(const llama_model * model) {
|
||||
temperatures[key + prefix_temp.size()] = temp;
|
||||
}
|
||||
|
||||
if (type_name == "openjev") {
|
||||
if (model_type == COMMON_DECISION_TYPE_OPENJEV) {
|
||||
// one letter per option, each must be a single token
|
||||
const std::string letters = "ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz";
|
||||
for (const char c : letters) {
|
||||
@@ -69,8 +71,7 @@ void server_decision_context::init(const llama_model * model) {
|
||||
}
|
||||
n_options_max = labels.size();
|
||||
noul_true_first = true;
|
||||
type = SERVER_DECISION_TYPE_OPENJEV;
|
||||
} else if (type_name == "laya") {
|
||||
} else if (model_type == COMMON_DECISION_TYPE_LAYA) {
|
||||
token_marker = llama_vocab_mask(vocab);
|
||||
token_sep = llama_vocab_sep(vocab);
|
||||
if (token_marker == LLAMA_TOKEN_NULL || token_sep == LLAMA_TOKEN_NULL) {
|
||||
@@ -84,10 +85,10 @@ void server_decision_context::init(const llama_model * model) {
|
||||
throw std::runtime_error("decision model has no valid max_head_tokens");
|
||||
}
|
||||
n_options_max = 255;
|
||||
type = SERVER_DECISION_TYPE_LAYA;
|
||||
} else {
|
||||
throw std::runtime_error("unsupported decision model type: " + type_name);
|
||||
}
|
||||
type = model_type;
|
||||
|
||||
SRV_INF("decision model type: %s\n", type_name.c_str());
|
||||
}
|
||||
@@ -223,7 +224,7 @@ std::string server_decision_context::render(const json & state, const server_dec
|
||||
void server_decision_context::fill_task(const json & state, const server_decision_question & question, server_task & task) const {
|
||||
llama_tokens tokens = common_tokenize(vocab, render(state, question), false, true);
|
||||
|
||||
if (type == SERVER_DECISION_TYPE_OPENJEV) {
|
||||
if (type == COMMON_DECISION_TYPE_OPENJEV) {
|
||||
task.decision.labels.assign(labels.begin(), labels.begin() + question.options.size());
|
||||
} else {
|
||||
fill_task_laya(tokens, question, task);
|
||||
|
||||
@@ -11,12 +11,6 @@
|
||||
// typed decision models (TypeSafe /v1/systemone API)
|
||||
// the model answers each question in one forward pass, no token is generated
|
||||
|
||||
enum server_decision_type {
|
||||
SERVER_DECISION_TYPE_NONE, // not a decision model
|
||||
SERVER_DECISION_TYPE_OPENJEV, // logits of one label token per option, read at the last prompt token
|
||||
SERVER_DECISION_TYPE_LAYA, // score of one marker token per option, read from the embeddings output
|
||||
};
|
||||
|
||||
enum server_decision_question_type {
|
||||
SERVER_DECISION_QUESTION_CHOICE,
|
||||
SERVER_DECISION_QUESTION_SCORE,
|
||||
@@ -36,16 +30,13 @@ struct server_decision_question {
|
||||
};
|
||||
|
||||
struct server_decision_context {
|
||||
server_decision_type type = SERVER_DECISION_TYPE_NONE;
|
||||
common_decision_type type = COMMON_DECISION_TYPE_NONE;
|
||||
|
||||
// read the "<arch>.decision.*" metadata, type stays NONE if the model has none
|
||||
void init(const llama_model * model);
|
||||
|
||||
// true if the result is read from the embeddings of each token
|
||||
bool need_embd() const { return type == SERVER_DECISION_TYPE_LAYA; }
|
||||
|
||||
// true if the questions of a request start with the same tokens, and the model can continue from them
|
||||
bool can_share_prompt() const { return type == SERVER_DECISION_TYPE_OPENJEV; }
|
||||
bool can_share_prompt() const { return type == COMMON_DECISION_TYPE_OPENJEV; }
|
||||
|
||||
// throw std::invalid_argument on bad input
|
||||
std::vector<server_decision_question> parse_questions(const json & body) const;
|
||||
|
||||
@@ -107,17 +107,6 @@ def test_systemone_invalid_request(data: dict):
|
||||
assert "error" in res.body
|
||||
|
||||
|
||||
def test_systemone_requires_embedding():
|
||||
global server
|
||||
server.server_embeddings = False
|
||||
server.start()
|
||||
res = server.make_request("POST", "/v1/systemone", data={
|
||||
"state": TEST_STATE,
|
||||
"questions": TEST_QUESTIONS,
|
||||
})
|
||||
assert res.status_code == 501
|
||||
|
||||
|
||||
# TODO: test the shared prompt prefix, it needs a small model of a type that supports it (e.g. openjev)
|
||||
# it can be checked with GET /metrics: for one request, prompt_tokens_cached_total must grow by
|
||||
# (shared tokens * number of child tasks) and prompt_tokens_total + prompt_tokens_cached_total == usage.input_tokens
|
||||
|
||||
@@ -644,7 +644,6 @@ class ServerPreset:
|
||||
server.n_ubatch = 512
|
||||
server.n_slots = 2
|
||||
server.seed = 42
|
||||
server.server_embeddings = True
|
||||
return server
|
||||
|
||||
@staticmethod
|
||||
|
||||
Reference in New Issue
Block a user