support lev & kev

This commit is contained in:
Xuan Son Nguyen
2026-10-01 22:14:48 +02:00
parent 82f7f147a7
commit 8536d4fb08
10 changed files with 483 additions and 56 deletions
+20 -10
View File
@@ -1192,6 +1192,22 @@ struct common_init_result::impl {
std::vector<llama_sampler_seq_config> samplers_seq_config;
};
static std::map<common_decision_type, std::string> COMMON_DECISION_TYPE_NAMES = {
{ COMMON_DECISION_TYPE_OPENJEV, "openjev" },
{ COMMON_DECISION_TYPE_LEV, "lev" },
{ COMMON_DECISION_TYPE_KEV, "kev" },
{ COMMON_DECISION_TYPE_LAYA, "laya" },
};
static common_decision_type common_decision_type_from_string(const std::string & str) {
for (const auto & pair : COMMON_DECISION_TYPE_NAMES) {
if (pair.second == str) {
return pair.first;
}
}
return COMMON_DECISION_TYPE_UNKNOWN;
}
common_decision_type common_get_decision_type(const struct llama_model * model) {
char buf[64];
if (llama_model_meta_val_str(model, "general.architecture", buf, sizeof(buf)) < 0) {
@@ -1201,14 +1217,7 @@ common_decision_type common_get_decision_type(const struct llama_model * model)
if (llama_model_meta_val_str(model, key.c_str(), buf, sizeof(buf)) < 0) {
return COMMON_DECISION_TYPE_NONE;
}
const std::string type = buf;
if (type == "openjev") {
return COMMON_DECISION_TYPE_OPENJEV;
}
if (type == "laya") {
return COMMON_DECISION_TYPE_LAYA;
}
return COMMON_DECISION_TYPE_UNKNOWN;
return common_decision_type_from_string(buf);
}
common_init_result::common_init_result(common_params & params, bool model_only) :
@@ -1265,7 +1274,8 @@ common_init_result::common_init_result(common_params & params, bool model_only)
// this decision model returns a score for each token via the embeddings output
// TODO: maybe improve this in the future
if (common_get_decision_type(model) == COMMON_DECISION_TYPE_LAYA) {
const auto decision_type = common_get_decision_type(model);
if (decision_type == COMMON_DECISION_TYPE_LAYA || decision_type == COMMON_DECISION_TYPE_KEV) {
params.embedding = true;
params.pooling_type = LLAMA_POOLING_TYPE_NONE;
@@ -1274,7 +1284,7 @@ common_init_result::common_init_result(common_params & params, bool model_only)
cparams.n_outputs_max = cparams.n_batch;
cparams.n_outputs_max_per_seq = 1;
LOG_INF("%s", "laya decision model detected, enabling embedding mode\n");
LOG_INF("%s", "decision model reads the embeddings output, enabling embedding mode\n");
}
// load and optionally apply lora adapters
+3 -1
View File
@@ -947,9 +947,11 @@ struct common_sampler;
// typed decision models, see "<arch>.decision.type" in the model metadata
enum common_decision_type {
COMMON_DECISION_TYPE_NONE, // not a decision model
COMMON_DECISION_TYPE_UNKNOWN, // a decision model of a type that is not supported
COMMON_DECISION_TYPE_OPENJEV, // logits of one label token per option, read at the last prompt token
COMMON_DECISION_TYPE_LEV, // same as openjev, noul is read from a rating scale
COMMON_DECISION_TYPE_KEV, // dot product of the hidden states of the last token and of one end token per option
COMMON_DECISION_TYPE_LAYA, // score of one marker token per option, read from the embeddings output
COMMON_DECISION_TYPE_UNKNOWN, // a decision model of a type that is not supported
};
common_decision_type common_get_decision_type(const struct llama_model * model);
+2
View File
@@ -150,6 +150,8 @@ TEXT_MODEL_MAP: dict[str, str] = {
"LLaDAMoEModelLM": "llada",
"LLaDAModelLM": "llada",
"LLaMAForCausalLM": "llama",
"KevModel": "lev",
"LevModel": "lev",
"Lfm25AudioTokenizer": "lfm2",
"Lfm2BidirectionalModel": "lfm2",
"Lfm2ForCausalLM": "lfm2",
+187
View File
@@ -0,0 +1,187 @@
from __future__ import annotations
import json
from pathlib import Path
from typing import Any, Iterable, TYPE_CHECKING
import torch
if TYPE_CHECKING:
from torch import Tensor
from .base import LazyTorchTensor, ModelBase, gguf, jinja_str_or_json, logger
from .qwen import Qwen3_5TextModel
def _decision_lora_base(dir_model: Path) -> tuple[str, str | None]:
# the base model of a LoRA adapter: (repo id, revision)
with open(dir_model / "adapter_config.json", encoding="utf-8") as f:
lora_config = json.load(f)
revision = lora_config.get("revision")
if revision is None and (dir_model / "training_config.json").is_file():
with open(dir_model / "training_config.json", encoding="utf-8") as f:
revision = json.load(f).get("base_revision")
return lora_config["base_model_name_or_path"], revision
def _load_decision_lora_hparams(dir_model: Path, arch: str) -> dict[str, Any]:
from huggingface_hub import hf_hub_download
repo_id, revision = _decision_lora_base(dir_model)
with open(hf_hub_download(repo_id, "config.json", revision=revision), encoding="utf-8") as f:
hparams = json.load(f)
hparams["architectures"] = [arch]
return hparams
class _DecisionLoraMixin:
# decision model released as a LoRA adapter: the base model is downloaded and the adapter is merged into it
no_mtp = True
def __init__(self, dir_model: Path, *args, **kwargs):
from huggingface_hub import snapshot_download
from safetensors.torch import load_file
repo_id, revision = _decision_lora_base(dir_model)
logger.info(f"gguf: downloading the base model {repo_id}")
dir_base = Path(snapshot_download(repo_id, revision=revision, allow_patterns=["*.json", "*.jinja", "*.safetensors"]))
super().__init__(dir_base, *args, **kwargs)
self.dir_adapter = dir_model
self.dir_model_card = dir_model
with open(dir_model / "adapter_config.json", encoding="utf-8") as f:
lora_config = json.load(f)
self.lora_scale = lora_config["lora_alpha"] / lora_config["r"]
# "layers.0.mlp.up_proj.weight" -> {"A": tensor, "B": tensor}
self.lora: dict[str, dict[str, Tensor]] = {}
for name, tensor in load_file(dir_model / "adapter_model.safetensors").items():
base_name, _, part = name[name.index("layers."):].partition(".lora_")
self.lora.setdefault(base_name + ".weight", {})[part[0]] = tensor.float()
self.lora_merged: set[str] = set()
def modify_tensors(self, data_torch: Tensor, name: str, bid: int | None) -> Iterable[tuple[str, Tensor]]:
lora = self.lora.get(name[name.index("layers."):]) if "layers." in name else None
if lora is not None:
delta = self.lora_scale * (lora["B"] @ lora["A"])
data_torch = data_torch.float() + LazyTorchTensor.from_eager(delta)
self.lora_merged.add(name[name.index("layers."):])
yield from super().modify_tensors(data_torch, name, bid) # ty: ignore[unresolved-attribute]
def prepare_tensors(self):
super().prepare_tensors() # ty: ignore[unresolved-attribute]
if len(self.lora_merged) != len(self.lora):
raise ValueError(f"only {len(self.lora_merged)} of {len(self.lora)} LoRA tensors were merged into the base model")
@ModelBase.register_hparams_loader(lambda dir_model: (dir_model / "lev_release.json").is_file())
def _load_lev_hparams(dir_model: Path) -> dict[str, Any]:
logger.info("gguf: detected Lev checkpoint")
return _load_decision_lora_hparams(dir_model, "LevModel")
@ModelBase.register("LevModel")
@ModelBase.example("interfaze-ai/lev")
class LevModel(_DecisionLoraMixin, Qwen3_5TextModel):
model_arch = gguf.MODEL_ARCH.QWEN35
# TODO: the head for large option sets (mode B, mode_b_head.pt) is not converted, only the label readout is supported
# TODO: a description that is not text is given as JSON without the escaping of non-ASCII characters used in training
# prompt follows packages/lev/src/lev/prompt.py of https://github.com/Abhinavexists/lev (chat style, state first)
_SYSTEM_PROMPT = (
"You are a System One decision model. You read the Evidence and answer each "
"Criterion by choosing exactly one of the listed options. You never explain. "
"You answer with the single option label only."
)
def set_vocab(self):
super().set_vocab()
self.gguf_writer.add_chat_template([{"name": "systemone", "template": self._systemone_template()}])
def _systemone_template(self) -> str:
description = jinja_str_or_json("o.description")
options = (
"{{ '# Options\\n' }}{% for o in options %}{{ o.label }}. "
"{% if type == 'score' %}(level {{ o.key }} of {{ options | length - 1 }}) " + description
+ "{% else %}{{ o.key }}{% if o.description %}: " + description + "{% endif %}{% endif %}"
"{{ '\\n' }}{% endfor %}"
"{{ '\\nRespond with only the letter of ' }}"
"{% if type == 'score' %}the level that best matches.{% else %}the best option.{% endif %}"
)
# noul is answered on a rating scale, its 2 options are only used for their description
scale = "{{ '# Scale\\n0 = certainly no ... 8 = certainly yes\\n' }}"
for key, name in (("true", "yes"), ("false", "no")):
scale += (
"{% for o in options %}{% if o.key == '" + key + "' and o.description %}"
+ name + ": " + description + "{{ '\\n' }}{% endif %}{% endfor %}"
)
scale += "{{ '\\nRespond with only a digit from 0 to 8.' }}"
return (
"<|im_start|>system\n" + self._SYSTEM_PROMPT + "<|im_end|>\n"
"<|im_start|>user\n# Evidence\n" + jinja_str_or_json("state") + "\n\n# Criterion\n"
"{% if instructions %}" + jinja_str_or_json("instructions") + "{% else %}{{ id }}{% endif %}"
"{{ '\\n\\n' }}{% if type == 'noul' %}" + scale + "{% else %}" + options + "{% endif %}"
"{{ '\\n<|im_end|>\\n<|im_start|>assistant\\n<think>\\n\\n</think>\\n\\n' }}"
)
def set_gguf_parameters(self):
super().set_gguf_parameters()
self.gguf_writer.add_decision_type(gguf.DecisionType.LEV)
with open(self.dir_adapter / "calibration.json", encoding="utf-8") as f:
temperatures = json.load(f)["temperatures"]
# "choice:A:small" -> "choice.small", only the label readout (mode A) is supported
for name, value in temperatures.items():
qtype, mode, *band = name.split(":")
if mode == "A":
self.gguf_writer.add_decision_temperature(".".join([qtype] + band), value)
@ModelBase.register_hparams_loader(lambda dir_model: (dir_model / "adapter_config.json").is_file() and (dir_model / "head.pt").is_file())
def _load_kev_hparams(dir_model: Path) -> dict[str, Any]:
logger.info("gguf: detected Kev checkpoint")
return _load_decision_lora_hparams(dir_model, "KevModel")
@ModelBase.register("KevModel")
@ModelBase.example("jaredpalmer/kev-4b")
class KevModel(_DecisionLoraMixin, Qwen3_5TextModel):
model_arch = gguf.MODEL_ARCH.QWEN35
# TODO: the server needs a question and its options in one batch, the state can be in previous batches
# TODO: the optional date_facts preprocessing of the state (kev/api.py) is not supported
def __init__(self, *args, **kwargs):
super().__init__(*args, **kwargs)
self.head = torch.load(self.dir_adapter / "head.pt", map_location="cpu", weights_only=True)
def set_vocab(self):
super().set_vocab()
self.gguf_writer.add_chat_template([{"name": "systemone", "template": self._systemone_template()}])
def _systemone_template(self) -> str:
# prompt follows kev/model.py and kev/api.py of https://github.com/jaredpalmer/kev
# state, instructions and descriptions are given as text
name = "{% if type != 'noul' %}{{ o.key }}{% elif o.key == 'true' %}yes{% else %}no{% endif %}"
option = (
"{% if type == 'score' %}{% if o.description %}{{ o.description }}{% endif %}"
"{% else %}" + name + "{% if o.description %}: {{ o.description }}{% endif %}{% endif %}"
)
return (
"<|fim_prefix|>{{ state }}<|fim_middle|>{{ instructions }}"
"{% for o in options %}<|box_start|>" + option + "<|box_end|>{% endfor %}<|fim_suffix|>"
)
def set_gguf_parameters(self):
super().set_gguf_parameters()
self.gguf_writer.add_decision_type(gguf.DecisionType.KEV)
self.gguf_writer.add_embedding_length_out(2 * self.head["head_dim"])
for name in ("choice", "score", "noul"):
self.gguf_writer.add_decision_temperature(name, self.head["temperature"])
def generate_extra_tensors(self) -> Iterable[tuple[str, Tensor]]:
yield from super().generate_extra_tensors()
# pointer head: the output of a token is [q | k]
head = self.head["head"]
yield "classifier.out_proj.weight", torch.cat([head["q.weight"], head["k.weight"]], dim=0)
yield "classifier.out_proj.bias", torch.cat([head["q.bias"], head["k.bias"]], dim=0)
+3
View File
@@ -2889,6 +2889,7 @@ MODEL_TENSORS: dict[MODEL_ARCH, list[MODEL_TENSOR]] = {
MODEL_TENSOR.TOKEN_EMBD,
MODEL_TENSOR.OUTPUT_NORM,
MODEL_TENSOR.OUTPUT,
MODEL_TENSOR.CLS_OUT,
MODEL_TENSOR.ATTN_NORM,
MODEL_TENSOR.ATTN_Q,
MODEL_TENSOR.ATTN_Q_NORM,
@@ -5917,6 +5918,8 @@ class GGUFValueType(IntEnum):
class DecisionType:
LAYA = "laya" # head blocks + scorer on the hidden state of one marker token per option
OPENJEV = "openjev" # logits of one label token per option
LEV = "lev" # same as openjev, noul is read from a rating scale
KEV = "kev" # dot product of the hidden states of the last token and of one end token per option
class VisionProjectorType:
+14
View File
@@ -43,6 +43,10 @@ void llama_model_qwen35::load_arch_tensors(llama_model_loader & ml) {
output_norm = create_tensor(tn(LLM_TENSOR_OUTPUT_NORM, "weight"), { n_embd }, 0);
output = create_tensor(tn(LLM_TENSOR_OUTPUT, "weight"), { n_embd, n_vocab }, TENSOR_NOT_REQUIRED);
// optional projection of the embeddings output
cls_out = create_tensor(tn(LLM_TENSOR_CLS_OUT, "weight"), { n_embd, hparams.n_embd_out() }, TENSOR_NOT_REQUIRED);
cls_out_b = create_tensor(tn(LLM_TENSOR_CLS_OUT, "bias"), { hparams.n_embd_out() }, TENSOR_NOT_REQUIRED);
// if output is NULL, init from the input tok embed
if (output == NULL) {
output = create_tensor(tn(LLM_TENSOR_TOKEN_EMBD, "weight"), { n_embd, n_vocab }, TENSOR_DUPLICATED);
@@ -215,6 +219,16 @@ llama_model_qwen35::graph::graph(const llama_model & model, const llm_graph_para
cb(cur, "result_norm", -1);
res->t_embd = cur;
if (model.cls_out) {
ggml_tensor * embd = ggml_mul_mat(ctx0, model.cls_out, cur);
if (model.cls_out_b) {
embd = ggml_add(ctx0, embd, model.cls_out_b);
}
cb(embd, "result_embd_proj", -1);
res->t_embd = embd;
ggml_build_forward_expand(gf, embd);
}
// LM head
cur = build_lora_mm(model.output, cur, model.output_s);
+45 -17
View File
@@ -414,6 +414,10 @@ struct server_slot {
if (pooling == LLAMA_POOLING_TYPE_RANK && llama_get_causal_attn(ctx_tgt)) {
return true;
}
// a decision task reads its outputs from the last batch
if (task->type == SERVER_TASK_TYPE_DECISION) {
return true;
}
return false;
}
@@ -2232,21 +2236,39 @@ private:
res->scores.push_back(logits[label]);
}
} else {
// the prompt is evaluated in one batch, the n-th output of this slot is the n-th prompt token
// the outputs of this slot in this batch are the last tokens of the prompt
std::vector<int32_t> idx;
for (int i = 0; i < batch.size(); ++i) {
if (batch.tokens[i].output && batch.tokens[i].seq_id == slot.id) {
idx.push_back(i);
}
}
GGML_ASSERT(decision.column >= 0 && decision.column < llama_model_n_embd_out(model_tgt));
const int32_t pos_first = slot.prompt.n_tokens() - (int32_t) idx.size();
auto get_embd = [&](int32_t pos) -> const float * {
const int32_t i = pos - pos_first;
return i >= 0 && i < (int32_t) idx.size() ? llama_get_embeddings_ith(slot.ctx_tgt, idx[i]) : nullptr;
};
const int32_t n_embd_out = llama_model_n_embd_out(model_tgt);
const int32_t n_pointer = n_embd_out / 2;
const float * embd_q = decision.pointer >= 0 ? get_embd(decision.pointer) : nullptr;
GGML_ASSERT(decision.column >= 0 && decision.column < n_embd_out);
for (const int32_t marker : decision.markers) {
const float * embd = marker >= 0 && marker < (int32_t) idx.size() ? llama_get_embeddings_ith(slot.ctx_tgt, idx[marker]) : nullptr;
if (embd == nullptr) {
send_error(slot, "failed to get embeddings", ERROR_TYPE_SERVER);
const float * embd = get_embd(marker);
if (embd == nullptr || (decision.pointer >= 0 && embd_q == nullptr)) {
send_error(slot, "failed to get embeddings, the question and its options must fit in one batch", ERROR_TYPE_SERVER);
return;
}
res->scores.push_back(embd[decision.column]);
if (decision.pointer < 0) {
res->scores.push_back(embd[decision.column]);
continue;
}
float dot = 0.0f;
for (int32_t i = 0; i < n_pointer; i++) {
dot += embd_q[i] * embd[n_pointer + i];
}
res->scores.push_back(dot / sqrtf((float) n_pointer));
}
}
@@ -5339,16 +5361,17 @@ void server_routes::init_routes() {
return res;
}
// one task per question
// one task per variant of each question
auto & rd = res->rd;
{
std::vector<server_task> tasks;
tasks.reserve(questions.size());
for (const auto & question : questions) {
server_task task = server_task(SERVER_TASK_TYPE_DECISION);
task.id = rd.get_new_id();
decision.fill_task(state, question, files, ctx_server.mctx, ctx_server.init_opt, task);
tasks.push_back(std::move(task));
for (size_t variant = 0; variant < decision.n_variants(question); variant++) {
server_task task = server_task(SERVER_TASK_TYPE_DECISION);
task.id = rd.get_new_id();
decision.fill_task(state, question, variant, files, ctx_server.mctx, ctx_server.init_opt, task);
tasks.push_back(std::move(task));
}
}
if (decision.can_share_prompt()) {
tasks = server_decision_group_tasks(std::move(tasks), params.n_parallel);
@@ -5367,11 +5390,16 @@ void server_routes::init_routes() {
json answers = json::object();
int32_t n_tokens = 0;
for (size_t i = 0; i < questions.size(); i++) {
auto * result = dynamic_cast<server_task_result_decision *>(all_results.results[i].get());
GGML_ASSERT(result != nullptr);
answers[questions[i].id] = decision.format_answer(questions[i], result->scores);
n_tokens += result->n_tokens;
size_t i_result = 0;
for (const auto & question : questions) {
std::vector<std::vector<float>> scores;
for (size_t variant = 0; variant < decision.n_variants(question); variant++) {
auto * result = dynamic_cast<server_task_result_decision *>(all_results.results[i_result++].get());
GGML_ASSERT(result != nullptr);
scores.push_back(result->scores);
n_tokens += result->n_tokens;
}
answers[question.id] = decision.format_answer(question, scores);
}
res->ok(json{
+193 -22
View File
@@ -2,6 +2,7 @@
#include <algorithm>
#include <cmath>
#include <regex>
#include <stdexcept>
static const char * decision_question_type_name(server_decision_question_type type) {
@@ -13,6 +14,9 @@ static const char * decision_question_type_name(server_decision_question_type ty
return "";
}
// lev reads noul from a rating scale: 0 = certainly no, 8 = certainly yes
static const size_t DECISION_LEV_N_RATINGS = 9;
static std::string decision_meta_str(const llama_model * model, const std::string & key) {
char buf[256];
const int32_t n = llama_model_meta_val_str(model, key.c_str(), buf, sizeof(buf));
@@ -71,6 +75,36 @@ void server_decision_context::init(const llama_model * model) {
}
n_options_max = labels.size();
noul_true_first = true;
} else if (model_type == COMMON_DECISION_TYPE_LEV) {
// label codes are A..Z then AA..ZZ, only the ones that are a single token are used
std::vector<std::string> codes;
for (char a = 'A'; a <= 'Z'; a++) {
codes.push_back(std::string(1, a));
}
for (char a = 'A'; a <= 'Z'; a++) {
for (char b = 'A'; b <= 'Z'; b++) {
codes.push_back(std::string{a, b});
}
}
for (const auto & code : codes) {
const auto toks = common_tokenize(vocab, code, false, false);
if (toks.size() == 1 && labels.size() < 255) {
labels.push_back(toks[0]);
label_texts.push_back(code);
}
}
if (labels.size() < DECISION_LEV_N_RATINGS) {
throw std::runtime_error("not enough single-token labels for this decision model");
}
n_options_max = labels.size();
} else if (model_type == COMMON_DECISION_TYPE_KEV) {
// the hidden state of an option is read at the token that ends it
const auto toks = common_tokenize(vocab, "<|box_end|>", false, true);
if (toks.size() != 1) {
throw std::runtime_error("decision model has no <|box_end|> token");
}
token_marker = toks[0];
n_options_max = 255;
} else if (model_type == COMMON_DECISION_TYPE_LAYA) {
token_marker = llama_vocab_mask(vocab);
token_sep = llama_vocab_sep(vocab);
@@ -254,23 +288,124 @@ static json decision_replace_text(const json & val, const std::string & search,
return val;
}
std::string server_decision_context::render(const json & state, const server_decision_question & question, size_t n_images) const {
// sort the keys of all objects of a JSON value
static json decision_sort_keys(const json & val) {
if (val.is_array()) {
json out = json::array();
for (const auto & item : val) {
out.push_back(decision_sort_keys(item));
}
return out;
}
if (val.is_object()) {
std::map<std::string, json> sorted;
for (const auto & [key, item] : val.items()) {
sorted[key] = decision_sort_keys(item);
}
json out = json::object();
for (const auto & [key, item] : sorted) {
out[key] = item;
}
return out;
}
return val;
}
// kev flattens a JSON value into text, the keys of an object are kept as labels (kev/api.py: render)
static std::string decision_kev_render(const json & val, int indent = 0) {
const std::string pad(2 * indent, ' ');
if (val.is_null()) {
return "";
}
if (val.is_string()) {
return val.get<std::string>();
}
if (val.is_boolean()) {
return val.get<bool>() ? "True" : "False";
}
if (val.is_array()) {
std::string out;
for (const auto & item : val) {
const std::string text = decision_kev_render(item, indent + 1);
out += (out.empty() ? "" : "\n") + pad + "- " + text.substr(std::min(text.size(), text.find_first_not_of(" \t\n\r")));
}
return out;
}
if (val.is_object()) {
std::string out;
for (const auto & [key, item] : val.items()) {
const bool is_nested = item.is_object() || item.is_array();
out += (out.empty() ? "" : "\n") + pad + key + (is_nested ? ":\n" : ": ") + decision_kev_render(item, is_nested ? indent + 1 : 0);
}
return out;
}
return val.dump();
}
// kev text input: special tokens written in the text must not be parsed as such
static std::string decision_kev_text(const json & val) {
static const std::regex re_special("<\\|([A-Za-z0-9_]+)\\|>");
return std::regex_replace(decision_kev_render(val), re_special, "<\xC2\xA6$1\xC2\xA6>");
}
size_t server_decision_context::n_variants(const server_decision_question & question) const {
// lev shows the options of a choice in 2 orders, to cancel the preference for the first label
if (type == COMMON_DECISION_TYPE_LEV && question.type == SERVER_DECISION_QUESTION_CHOICE && question.options.size() > 1) {
return 2;
}
return 1;
}
size_t server_decision_context::n_outputs(const server_decision_question & question) const {
if (type == COMMON_DECISION_TYPE_LEV && question.type == SERVER_DECISION_QUESTION_NOUL) {
return DECISION_LEV_N_RATINGS;
}
return question.options.size();
}
std::string server_decision_context::render(const json & state, const server_decision_question & question, size_t variant, size_t n_images) const {
const size_t n_options = question.options.size();
// the second variant shows the options in the reverse order
json options = json::array();
for (const auto & opt : question.options) {
options.push_back(json{
for (size_t i = 0; i < n_options; i++) {
const auto & opt = question.options[variant == 0 ? i : n_options - 1 - i];
json option = json{
{"key", opt.key},
{"description", opt.description},
});
};
if (type == COMMON_DECISION_TYPE_KEV) {
option["key"] = decision_kev_text(opt.key);
if (!opt.description.is_null()) {
option["description"] = decision_kev_text(opt.description);
}
}
if (!label_texts.empty()) {
option["label"] = label_texts[i];
}
options.push_back(option);
}
// the template is given raw JSON values, it serializes the ones that are not strings
json inp = json{
{"id", question.id},
{"type", decision_question_type_name(question.type)},
{"instructions", question.instructions},
{"state", state},
{"options", options},
};
// lev was trained with sorted keys
if (type == COMMON_DECISION_TYPE_LEV) {
inp = decision_sort_keys(inp);
}
// the kev template only takes text
if (type == COMMON_DECISION_TYPE_KEV) {
inp["state"] = decision_kev_text(state);
inp["instructions"] = decision_kev_text(question.instructions);
}
// the input must not contain the marker of the options
if (!text_marker.empty()) {
inp = decision_replace_text(inp, text_marker, " ");
@@ -296,14 +431,15 @@ std::string server_decision_context::render(const json & state, const server_dec
void server_decision_context::fill_task(
const json & state,
const server_decision_question & question,
size_t variant,
const std::vector<raw_buffer> & files,
mtmd_context * mctx,
const mtmd_helper_init_opt & init_opt,
server_task & task) const {
const std::string prompt = render(state, question, files.size());
const std::string prompt = render(state, question, variant, files.size());
if (type == COMMON_DECISION_TYPE_OPENJEV) {
task.decision.labels.assign(labels.begin(), labels.begin() + question.options.size());
if (type == COMMON_DECISION_TYPE_OPENJEV || type == COMMON_DECISION_TYPE_LEV) {
task.decision.labels.assign(labels.begin(), labels.begin() + n_outputs(question));
if (!files.empty()) {
task.tokens = process_mtmd_prompt(mctx, prompt, files, init_opt);
return;
@@ -314,6 +450,18 @@ void server_decision_context::fill_task(
if (type == COMMON_DECISION_TYPE_LAYA) {
fill_task_laya(tokens, question, task);
}
if (type == COMMON_DECISION_TYPE_KEV) {
// an option is read at its end token, the question at the last token
for (size_t i = 0; i < tokens.size(); i++) {
if (tokens[i] == token_marker) {
task.decision.markers.push_back(i);
}
}
if (task.decision.markers.size() != question.options.size()) {
throw std::runtime_error("unexpected layout of the decision prompt");
}
task.decision.pointer = tokens.size() - 1;
}
task.tokens = server_tokens(tokens, false);
}
@@ -381,7 +529,14 @@ void server_decision_context::fill_task_laya(llama_tokens & tokens, const server
float server_decision_context::get_temperature(const server_decision_question & question) const {
const size_t n = question.options.size();
const std::string type_name = decision_question_type_name(question.type);
const std::string bucket = n <= 2 ? "2" : n <= 5 ? "3_5" : n <= 10 ? "6_10" : "11";
// the temperature can depend on the number of options, the buckets are the ones used to fit it
std::string bucket;
if (type == COMMON_DECISION_TYPE_LEV) {
bucket = n <= 8 ? "small" : n <= 26 ? "mid" : "large";
} else {
bucket = n <= 2 ? "2" : n <= 5 ? "3_5" : n <= 10 ? "6_10" : "11";
}
for (const auto & name : {type_name + "." + bucket, type_name}) {
const auto it = temperatures.find(name);
@@ -420,28 +575,44 @@ static double decision_confidence_score(const std::vector<double> & probs) {
return std::max(0.0, 1.0 - dist / dist_uniform);
}
json server_decision_context::format_answer(const server_decision_question & question, const std::vector<float> & scores) const {
const size_t n = question.options.size();
if (scores.size() != n) {
throw std::runtime_error("decision result does not match the number of options");
json server_decision_context::format_answer(const server_decision_question & question, const std::vector<std::vector<float>> & scores) const {
const size_t n = n_outputs(question);
if (scores.size() != n_variants(question)) {
throw std::runtime_error("decision result does not match the number of variants");
}
// softmax over the options
// softmax over the outputs of each variant, then the average of the variants
const float temperature = get_temperature(question);
const float score_max = *std::max_element(scores.begin(), scores.end());
std::vector<double> probs(n);
double sum = 0.0;
for (size_t i = 0; i < n; i++) {
probs[i] = std::exp((double) (scores[i] - score_max) / temperature);
sum += probs[i];
}
for (auto & p : probs) {
p /= sum;
std::vector<double> probs(n, 0.0);
for (size_t v = 0; v < scores.size(); v++) {
const auto & s = scores[v];
if (s.size() != n) {
throw std::runtime_error("decision result does not match the number of options");
}
const float score_max = *std::max_element(s.begin(), s.end());
std::vector<double> p(n);
double sum = 0.0;
for (size_t i = 0; i < n; i++) {
p[i] = std::exp((double) (s[i] - score_max) / temperature);
sum += p[i];
}
for (size_t i = 0; i < n; i++) {
// the second variant is in the reverse order
probs[v == 0 ? i : n - 1 - i] += p[i] / sum / scores.size();
}
}
json answer = json{{"type", decision_question_type_name(question.type)}};
if (question.type == SERVER_DECISION_QUESTION_NOUL) {
if (type == COMMON_DECISION_TYPE_LEV) {
double expected = 0.0;
for (size_t i = 0; i < n; i++) {
expected += probs[i] * i / (n - 1);
}
answer["noul"] = expected;
return answer;
}
for (size_t i = 0; i < n; i++) {
if (question.options[i].key == "true") {
answer["noul"] = probs[i];
+14 -6
View File
@@ -39,6 +39,8 @@ struct server_decision_context {
bool can_share_prompt() const {
switch (type) {
case COMMON_DECISION_TYPE_OPENJEV:
case COMMON_DECISION_TYPE_LEV:
case COMMON_DECISION_TYPE_KEV:
return true;
default:
return false;
@@ -62,18 +64,22 @@ struct server_decision_context {
// images come from "images" and from the image_url parts of a state made of chat messages
json parse_state(const json & body, std::vector<raw_buffer> & files) const;
// set the prompt of this question, and where to read its result
// number of prompts that are evaluated to answer this question, each one shows the options in a different order
size_t n_variants(const server_decision_question & question) const;
// set the prompt of one variant of this question, and where to read its result
// mctx is only used if there are files
void fill_task(
const json & state,
const server_decision_question & question,
size_t variant,
const std::vector<raw_buffer> & files,
mtmd_context * mctx,
const mtmd_helper_init_opt & init_opt,
server_task & task) const;
// scores: one raw model output per option
json format_answer(const server_decision_question & question, const std::vector<float> & scores) const;
// scores: the raw model outputs of each variant
json format_answer(const server_decision_question & question, const std::vector<std::vector<float>> & scores) const;
private:
const llama_vocab * vocab = nullptr;
@@ -83,17 +89,19 @@ private:
size_t n_options_max = 0;
bool noul_true_first = false; // noul options are [true, false] instead of [false, true]
// OPENJEV
// OPENJEV, LEV
std::vector<llama_token> labels;
std::vector<std::string> label_texts; // only if the label of an option is given to the template
// LAYA
// LAYA, KEV
llama_token token_marker = LLAMA_TOKEN_NULL;
llama_token token_sep = LLAMA_TOKEN_NULL;
std::string text_marker;
size_t max_head_tokens = 0; // question + options
size_t max_option_tokens = 48;
std::string render(const json & state, const server_decision_question & question, size_t n_images) const;
std::string render(const json & state, const server_decision_question & question, size_t variant, size_t n_images) const;
size_t n_outputs(const server_decision_question & question) const;
void fill_task_laya(llama_tokens & tokens, const server_decision_question & question, server_task & task) const;
float get_temperature(const server_decision_question & question) const;
+2
View File
@@ -181,6 +181,8 @@ struct server_task {
std::vector<llama_token> labels; // logits of these tokens, at the last prompt token
std::vector<int32_t> markers; // embeddings[column] at these prompt positions
int32_t column = 0;
// if set, embeddings is [q | k], and the output is instead the scaled dot product of q[pointer] and k[marker]
int32_t pointer = -1;
};
decision decision;