From 8536d4fb08436807a63533ade49b08ff1cc84507 Mon Sep 17 00:00:00 2001 From: Xuan Son Nguyen Date: Thu, 1 Oct 2026 22:14:48 +0200 Subject: [PATCH] support lev & kev --- common/common.cpp | 30 +++-- common/common.h | 4 +- conversion/__init__.py | 2 + conversion/lev.py | 187 +++++++++++++++++++++++++++ gguf-py/gguf/constants.py | 3 + src/models/qwen35.cpp | 14 ++ tools/server/server-context.cpp | 62 ++++++--- tools/server/server-decision.cpp | 215 +++++++++++++++++++++++++++---- tools/server/server-decision.h | 20 ++- tools/server/server-task.h | 2 + 10 files changed, 483 insertions(+), 56 deletions(-) create mode 100644 conversion/lev.py diff --git a/common/common.cpp b/common/common.cpp index 4005990ce2..1612a3a9f0 100644 --- a/common/common.cpp +++ b/common/common.cpp @@ -1192,6 +1192,22 @@ struct common_init_result::impl { std::vector samplers_seq_config; }; +static std::map COMMON_DECISION_TYPE_NAMES = { + { COMMON_DECISION_TYPE_OPENJEV, "openjev" }, + { COMMON_DECISION_TYPE_LEV, "lev" }, + { COMMON_DECISION_TYPE_KEV, "kev" }, + { COMMON_DECISION_TYPE_LAYA, "laya" }, +}; + +static common_decision_type common_decision_type_from_string(const std::string & str) { + for (const auto & pair : COMMON_DECISION_TYPE_NAMES) { + if (pair.second == str) { + return pair.first; + } + } + return COMMON_DECISION_TYPE_UNKNOWN; +} + common_decision_type common_get_decision_type(const struct llama_model * model) { char buf[64]; if (llama_model_meta_val_str(model, "general.architecture", buf, sizeof(buf)) < 0) { @@ -1201,14 +1217,7 @@ common_decision_type common_get_decision_type(const struct llama_model * model) if (llama_model_meta_val_str(model, key.c_str(), buf, sizeof(buf)) < 0) { return COMMON_DECISION_TYPE_NONE; } - const std::string type = buf; - if (type == "openjev") { - return COMMON_DECISION_TYPE_OPENJEV; - } - if (type == "laya") { - return COMMON_DECISION_TYPE_LAYA; - } - return COMMON_DECISION_TYPE_UNKNOWN; + return common_decision_type_from_string(buf); } common_init_result::common_init_result(common_params & params, bool model_only) : @@ -1265,7 +1274,8 @@ common_init_result::common_init_result(common_params & params, bool model_only) // this decision model returns a score for each token via the embeddings output // TODO: maybe improve this in the future - if (common_get_decision_type(model) == COMMON_DECISION_TYPE_LAYA) { + const auto decision_type = common_get_decision_type(model); + if (decision_type == COMMON_DECISION_TYPE_LAYA || decision_type == COMMON_DECISION_TYPE_KEV) { params.embedding = true; params.pooling_type = LLAMA_POOLING_TYPE_NONE; @@ -1274,7 +1284,7 @@ common_init_result::common_init_result(common_params & params, bool model_only) cparams.n_outputs_max = cparams.n_batch; cparams.n_outputs_max_per_seq = 1; - LOG_INF("%s", "laya decision model detected, enabling embedding mode\n"); + LOG_INF("%s", "decision model reads the embeddings output, enabling embedding mode\n"); } // load and optionally apply lora adapters diff --git a/common/common.h b/common/common.h index 580a162fe5..1d90f2211e 100644 --- a/common/common.h +++ b/common/common.h @@ -947,9 +947,11 @@ struct common_sampler; // typed decision models, see ".decision.type" in the model metadata enum common_decision_type { COMMON_DECISION_TYPE_NONE, // not a decision model - COMMON_DECISION_TYPE_UNKNOWN, // a decision model of a type that is not supported COMMON_DECISION_TYPE_OPENJEV, // logits of one label token per option, read at the last prompt token + COMMON_DECISION_TYPE_LEV, // same as openjev, noul is read from a rating scale + COMMON_DECISION_TYPE_KEV, // dot product of the hidden states of the last token and of one end token per option COMMON_DECISION_TYPE_LAYA, // score of one marker token per option, read from the embeddings output + COMMON_DECISION_TYPE_UNKNOWN, // a decision model of a type that is not supported }; common_decision_type common_get_decision_type(const struct llama_model * model); diff --git a/conversion/__init__.py b/conversion/__init__.py index fac5d95a9f..cd8c4d3bac 100644 --- a/conversion/__init__.py +++ b/conversion/__init__.py @@ -150,6 +150,8 @@ TEXT_MODEL_MAP: dict[str, str] = { "LLaDAMoEModelLM": "llada", "LLaDAModelLM": "llada", "LLaMAForCausalLM": "llama", + "KevModel": "lev", + "LevModel": "lev", "Lfm25AudioTokenizer": "lfm2", "Lfm2BidirectionalModel": "lfm2", "Lfm2ForCausalLM": "lfm2", diff --git a/conversion/lev.py b/conversion/lev.py new file mode 100644 index 0000000000..126556c4a5 --- /dev/null +++ b/conversion/lev.py @@ -0,0 +1,187 @@ +from __future__ import annotations + +import json + +from pathlib import Path +from typing import Any, Iterable, TYPE_CHECKING + +import torch + +if TYPE_CHECKING: + from torch import Tensor + +from .base import LazyTorchTensor, ModelBase, gguf, jinja_str_or_json, logger +from .qwen import Qwen3_5TextModel + + +def _decision_lora_base(dir_model: Path) -> tuple[str, str | None]: + # the base model of a LoRA adapter: (repo id, revision) + with open(dir_model / "adapter_config.json", encoding="utf-8") as f: + lora_config = json.load(f) + revision = lora_config.get("revision") + if revision is None and (dir_model / "training_config.json").is_file(): + with open(dir_model / "training_config.json", encoding="utf-8") as f: + revision = json.load(f).get("base_revision") + return lora_config["base_model_name_or_path"], revision + + +def _load_decision_lora_hparams(dir_model: Path, arch: str) -> dict[str, Any]: + from huggingface_hub import hf_hub_download + repo_id, revision = _decision_lora_base(dir_model) + with open(hf_hub_download(repo_id, "config.json", revision=revision), encoding="utf-8") as f: + hparams = json.load(f) + hparams["architectures"] = [arch] + return hparams + + +class _DecisionLoraMixin: + # decision model released as a LoRA adapter: the base model is downloaded and the adapter is merged into it + no_mtp = True + + def __init__(self, dir_model: Path, *args, **kwargs): + from huggingface_hub import snapshot_download + from safetensors.torch import load_file + + repo_id, revision = _decision_lora_base(dir_model) + logger.info(f"gguf: downloading the base model {repo_id}") + dir_base = Path(snapshot_download(repo_id, revision=revision, allow_patterns=["*.json", "*.jinja", "*.safetensors"])) + super().__init__(dir_base, *args, **kwargs) + self.dir_adapter = dir_model + self.dir_model_card = dir_model + + with open(dir_model / "adapter_config.json", encoding="utf-8") as f: + lora_config = json.load(f) + self.lora_scale = lora_config["lora_alpha"] / lora_config["r"] + + # "layers.0.mlp.up_proj.weight" -> {"A": tensor, "B": tensor} + self.lora: dict[str, dict[str, Tensor]] = {} + for name, tensor in load_file(dir_model / "adapter_model.safetensors").items(): + base_name, _, part = name[name.index("layers."):].partition(".lora_") + self.lora.setdefault(base_name + ".weight", {})[part[0]] = tensor.float() + self.lora_merged: set[str] = set() + + def modify_tensors(self, data_torch: Tensor, name: str, bid: int | None) -> Iterable[tuple[str, Tensor]]: + lora = self.lora.get(name[name.index("layers."):]) if "layers." in name else None + if lora is not None: + delta = self.lora_scale * (lora["B"] @ lora["A"]) + data_torch = data_torch.float() + LazyTorchTensor.from_eager(delta) + self.lora_merged.add(name[name.index("layers."):]) + yield from super().modify_tensors(data_torch, name, bid) # ty: ignore[unresolved-attribute] + + def prepare_tensors(self): + super().prepare_tensors() # ty: ignore[unresolved-attribute] + if len(self.lora_merged) != len(self.lora): + raise ValueError(f"only {len(self.lora_merged)} of {len(self.lora)} LoRA tensors were merged into the base model") + + +@ModelBase.register_hparams_loader(lambda dir_model: (dir_model / "lev_release.json").is_file()) +def _load_lev_hparams(dir_model: Path) -> dict[str, Any]: + logger.info("gguf: detected Lev checkpoint") + return _load_decision_lora_hparams(dir_model, "LevModel") + + +@ModelBase.register("LevModel") +@ModelBase.example("interfaze-ai/lev") +class LevModel(_DecisionLoraMixin, Qwen3_5TextModel): + model_arch = gguf.MODEL_ARCH.QWEN35 + + # TODO: the head for large option sets (mode B, mode_b_head.pt) is not converted, only the label readout is supported + # TODO: a description that is not text is given as JSON without the escaping of non-ASCII characters used in training + + # prompt follows packages/lev/src/lev/prompt.py of https://github.com/Abhinavexists/lev (chat style, state first) + _SYSTEM_PROMPT = ( + "You are a System One decision model. You read the Evidence and answer each " + "Criterion by choosing exactly one of the listed options. You never explain. " + "You answer with the single option label only." + ) + + def set_vocab(self): + super().set_vocab() + self.gguf_writer.add_chat_template([{"name": "systemone", "template": self._systemone_template()}]) + + def _systemone_template(self) -> str: + description = jinja_str_or_json("o.description") + options = ( + "{{ '# Options\\n' }}{% for o in options %}{{ o.label }}. " + "{% if type == 'score' %}(level {{ o.key }} of {{ options | length - 1 }}) " + description + + "{% else %}{{ o.key }}{% if o.description %}: " + description + "{% endif %}{% endif %}" + "{{ '\\n' }}{% endfor %}" + "{{ '\\nRespond with only the letter of ' }}" + "{% if type == 'score' %}the level that best matches.{% else %}the best option.{% endif %}" + ) + # noul is answered on a rating scale, its 2 options are only used for their description + scale = "{{ '# Scale\\n0 = certainly no ... 8 = certainly yes\\n' }}" + for key, name in (("true", "yes"), ("false", "no")): + scale += ( + "{% for o in options %}{% if o.key == '" + key + "' and o.description %}" + + name + ": " + description + "{{ '\\n' }}{% endif %}{% endfor %}" + ) + scale += "{{ '\\nRespond with only a digit from 0 to 8.' }}" + return ( + "<|im_start|>system\n" + self._SYSTEM_PROMPT + "<|im_end|>\n" + "<|im_start|>user\n# Evidence\n" + jinja_str_or_json("state") + "\n\n# Criterion\n" + "{% if instructions %}" + jinja_str_or_json("instructions") + "{% else %}{{ id }}{% endif %}" + "{{ '\\n\\n' }}{% if type == 'noul' %}" + scale + "{% else %}" + options + "{% endif %}" + "{{ '\\n<|im_end|>\\n<|im_start|>assistant\\n\\n\\n\\n\\n' }}" + ) + + def set_gguf_parameters(self): + super().set_gguf_parameters() + self.gguf_writer.add_decision_type(gguf.DecisionType.LEV) + with open(self.dir_adapter / "calibration.json", encoding="utf-8") as f: + temperatures = json.load(f)["temperatures"] + # "choice:A:small" -> "choice.small", only the label readout (mode A) is supported + for name, value in temperatures.items(): + qtype, mode, *band = name.split(":") + if mode == "A": + self.gguf_writer.add_decision_temperature(".".join([qtype] + band), value) + + +@ModelBase.register_hparams_loader(lambda dir_model: (dir_model / "adapter_config.json").is_file() and (dir_model / "head.pt").is_file()) +def _load_kev_hparams(dir_model: Path) -> dict[str, Any]: + logger.info("gguf: detected Kev checkpoint") + return _load_decision_lora_hparams(dir_model, "KevModel") + + +@ModelBase.register("KevModel") +@ModelBase.example("jaredpalmer/kev-4b") +class KevModel(_DecisionLoraMixin, Qwen3_5TextModel): + model_arch = gguf.MODEL_ARCH.QWEN35 + + # TODO: the server needs a question and its options in one batch, the state can be in previous batches + # TODO: the optional date_facts preprocessing of the state (kev/api.py) is not supported + + def __init__(self, *args, **kwargs): + super().__init__(*args, **kwargs) + self.head = torch.load(self.dir_adapter / "head.pt", map_location="cpu", weights_only=True) + + def set_vocab(self): + super().set_vocab() + self.gguf_writer.add_chat_template([{"name": "systemone", "template": self._systemone_template()}]) + + def _systemone_template(self) -> str: + # prompt follows kev/model.py and kev/api.py of https://github.com/jaredpalmer/kev + # state, instructions and descriptions are given as text + name = "{% if type != 'noul' %}{{ o.key }}{% elif o.key == 'true' %}yes{% else %}no{% endif %}" + option = ( + "{% if type == 'score' %}{% if o.description %}{{ o.description }}{% endif %}" + "{% else %}" + name + "{% if o.description %}: {{ o.description }}{% endif %}{% endif %}" + ) + return ( + "<|fim_prefix|>{{ state }}<|fim_middle|>{{ instructions }}" + "{% for o in options %}<|box_start|>" + option + "<|box_end|>{% endfor %}<|fim_suffix|>" + ) + + def set_gguf_parameters(self): + super().set_gguf_parameters() + self.gguf_writer.add_decision_type(gguf.DecisionType.KEV) + self.gguf_writer.add_embedding_length_out(2 * self.head["head_dim"]) + for name in ("choice", "score", "noul"): + self.gguf_writer.add_decision_temperature(name, self.head["temperature"]) + + def generate_extra_tensors(self) -> Iterable[tuple[str, Tensor]]: + yield from super().generate_extra_tensors() + # pointer head: the output of a token is [q | k] + head = self.head["head"] + yield "classifier.out_proj.weight", torch.cat([head["q.weight"], head["k.weight"]], dim=0) + yield "classifier.out_proj.bias", torch.cat([head["q.bias"], head["k.bias"]], dim=0) diff --git a/gguf-py/gguf/constants.py b/gguf-py/gguf/constants.py index dab8445a9a..a8fd9f0dd7 100644 --- a/gguf-py/gguf/constants.py +++ b/gguf-py/gguf/constants.py @@ -2889,6 +2889,7 @@ MODEL_TENSORS: dict[MODEL_ARCH, list[MODEL_TENSOR]] = { MODEL_TENSOR.TOKEN_EMBD, MODEL_TENSOR.OUTPUT_NORM, MODEL_TENSOR.OUTPUT, + MODEL_TENSOR.CLS_OUT, MODEL_TENSOR.ATTN_NORM, MODEL_TENSOR.ATTN_Q, MODEL_TENSOR.ATTN_Q_NORM, @@ -5917,6 +5918,8 @@ class GGUFValueType(IntEnum): class DecisionType: LAYA = "laya" # head blocks + scorer on the hidden state of one marker token per option OPENJEV = "openjev" # logits of one label token per option + LEV = "lev" # same as openjev, noul is read from a rating scale + KEV = "kev" # dot product of the hidden states of the last token and of one end token per option class VisionProjectorType: diff --git a/src/models/qwen35.cpp b/src/models/qwen35.cpp index a1e263500e..78bd913af6 100644 --- a/src/models/qwen35.cpp +++ b/src/models/qwen35.cpp @@ -43,6 +43,10 @@ void llama_model_qwen35::load_arch_tensors(llama_model_loader & ml) { output_norm = create_tensor(tn(LLM_TENSOR_OUTPUT_NORM, "weight"), { n_embd }, 0); output = create_tensor(tn(LLM_TENSOR_OUTPUT, "weight"), { n_embd, n_vocab }, TENSOR_NOT_REQUIRED); + // optional projection of the embeddings output + cls_out = create_tensor(tn(LLM_TENSOR_CLS_OUT, "weight"), { n_embd, hparams.n_embd_out() }, TENSOR_NOT_REQUIRED); + cls_out_b = create_tensor(tn(LLM_TENSOR_CLS_OUT, "bias"), { hparams.n_embd_out() }, TENSOR_NOT_REQUIRED); + // if output is NULL, init from the input tok embed if (output == NULL) { output = create_tensor(tn(LLM_TENSOR_TOKEN_EMBD, "weight"), { n_embd, n_vocab }, TENSOR_DUPLICATED); @@ -215,6 +219,16 @@ llama_model_qwen35::graph::graph(const llama_model & model, const llm_graph_para cb(cur, "result_norm", -1); res->t_embd = cur; + if (model.cls_out) { + ggml_tensor * embd = ggml_mul_mat(ctx0, model.cls_out, cur); + if (model.cls_out_b) { + embd = ggml_add(ctx0, embd, model.cls_out_b); + } + cb(embd, "result_embd_proj", -1); + res->t_embd = embd; + ggml_build_forward_expand(gf, embd); + } + // LM head cur = build_lora_mm(model.output, cur, model.output_s); diff --git a/tools/server/server-context.cpp b/tools/server/server-context.cpp index cda543f8b9..3c9dfe66c2 100644 --- a/tools/server/server-context.cpp +++ b/tools/server/server-context.cpp @@ -414,6 +414,10 @@ struct server_slot { if (pooling == LLAMA_POOLING_TYPE_RANK && llama_get_causal_attn(ctx_tgt)) { return true; } + // a decision task reads its outputs from the last batch + if (task->type == SERVER_TASK_TYPE_DECISION) { + return true; + } return false; } @@ -2232,21 +2236,39 @@ private: res->scores.push_back(logits[label]); } } else { - // the prompt is evaluated in one batch, the n-th output of this slot is the n-th prompt token + // the outputs of this slot in this batch are the last tokens of the prompt std::vector idx; for (int i = 0; i < batch.size(); ++i) { if (batch.tokens[i].output && batch.tokens[i].seq_id == slot.id) { idx.push_back(i); } } - GGML_ASSERT(decision.column >= 0 && decision.column < llama_model_n_embd_out(model_tgt)); + const int32_t pos_first = slot.prompt.n_tokens() - (int32_t) idx.size(); + auto get_embd = [&](int32_t pos) -> const float * { + const int32_t i = pos - pos_first; + return i >= 0 && i < (int32_t) idx.size() ? llama_get_embeddings_ith(slot.ctx_tgt, idx[i]) : nullptr; + }; + + const int32_t n_embd_out = llama_model_n_embd_out(model_tgt); + const int32_t n_pointer = n_embd_out / 2; + const float * embd_q = decision.pointer >= 0 ? get_embd(decision.pointer) : nullptr; + GGML_ASSERT(decision.column >= 0 && decision.column < n_embd_out); + for (const int32_t marker : decision.markers) { - const float * embd = marker >= 0 && marker < (int32_t) idx.size() ? llama_get_embeddings_ith(slot.ctx_tgt, idx[marker]) : nullptr; - if (embd == nullptr) { - send_error(slot, "failed to get embeddings", ERROR_TYPE_SERVER); + const float * embd = get_embd(marker); + if (embd == nullptr || (decision.pointer >= 0 && embd_q == nullptr)) { + send_error(slot, "failed to get embeddings, the question and its options must fit in one batch", ERROR_TYPE_SERVER); return; } - res->scores.push_back(embd[decision.column]); + if (decision.pointer < 0) { + res->scores.push_back(embd[decision.column]); + continue; + } + float dot = 0.0f; + for (int32_t i = 0; i < n_pointer; i++) { + dot += embd_q[i] * embd[n_pointer + i]; + } + res->scores.push_back(dot / sqrtf((float) n_pointer)); } } @@ -5339,16 +5361,17 @@ void server_routes::init_routes() { return res; } - // one task per question + // one task per variant of each question auto & rd = res->rd; { std::vector tasks; - tasks.reserve(questions.size()); for (const auto & question : questions) { - server_task task = server_task(SERVER_TASK_TYPE_DECISION); - task.id = rd.get_new_id(); - decision.fill_task(state, question, files, ctx_server.mctx, ctx_server.init_opt, task); - tasks.push_back(std::move(task)); + for (size_t variant = 0; variant < decision.n_variants(question); variant++) { + server_task task = server_task(SERVER_TASK_TYPE_DECISION); + task.id = rd.get_new_id(); + decision.fill_task(state, question, variant, files, ctx_server.mctx, ctx_server.init_opt, task); + tasks.push_back(std::move(task)); + } } if (decision.can_share_prompt()) { tasks = server_decision_group_tasks(std::move(tasks), params.n_parallel); @@ -5367,11 +5390,16 @@ void server_routes::init_routes() { json answers = json::object(); int32_t n_tokens = 0; - for (size_t i = 0; i < questions.size(); i++) { - auto * result = dynamic_cast(all_results.results[i].get()); - GGML_ASSERT(result != nullptr); - answers[questions[i].id] = decision.format_answer(questions[i], result->scores); - n_tokens += result->n_tokens; + size_t i_result = 0; + for (const auto & question : questions) { + std::vector> scores; + for (size_t variant = 0; variant < decision.n_variants(question); variant++) { + auto * result = dynamic_cast(all_results.results[i_result++].get()); + GGML_ASSERT(result != nullptr); + scores.push_back(result->scores); + n_tokens += result->n_tokens; + } + answers[question.id] = decision.format_answer(question, scores); } res->ok(json{ diff --git a/tools/server/server-decision.cpp b/tools/server/server-decision.cpp index 014104293c..cdec03bf4a 100644 --- a/tools/server/server-decision.cpp +++ b/tools/server/server-decision.cpp @@ -2,6 +2,7 @@ #include #include +#include #include static const char * decision_question_type_name(server_decision_question_type type) { @@ -13,6 +14,9 @@ static const char * decision_question_type_name(server_decision_question_type ty return ""; } +// lev reads noul from a rating scale: 0 = certainly no, 8 = certainly yes +static const size_t DECISION_LEV_N_RATINGS = 9; + static std::string decision_meta_str(const llama_model * model, const std::string & key) { char buf[256]; const int32_t n = llama_model_meta_val_str(model, key.c_str(), buf, sizeof(buf)); @@ -71,6 +75,36 @@ void server_decision_context::init(const llama_model * model) { } n_options_max = labels.size(); noul_true_first = true; + } else if (model_type == COMMON_DECISION_TYPE_LEV) { + // label codes are A..Z then AA..ZZ, only the ones that are a single token are used + std::vector codes; + for (char a = 'A'; a <= 'Z'; a++) { + codes.push_back(std::string(1, a)); + } + for (char a = 'A'; a <= 'Z'; a++) { + for (char b = 'A'; b <= 'Z'; b++) { + codes.push_back(std::string{a, b}); + } + } + for (const auto & code : codes) { + const auto toks = common_tokenize(vocab, code, false, false); + if (toks.size() == 1 && labels.size() < 255) { + labels.push_back(toks[0]); + label_texts.push_back(code); + } + } + if (labels.size() < DECISION_LEV_N_RATINGS) { + throw std::runtime_error("not enough single-token labels for this decision model"); + } + n_options_max = labels.size(); + } else if (model_type == COMMON_DECISION_TYPE_KEV) { + // the hidden state of an option is read at the token that ends it + const auto toks = common_tokenize(vocab, "<|box_end|>", false, true); + if (toks.size() != 1) { + throw std::runtime_error("decision model has no <|box_end|> token"); + } + token_marker = toks[0]; + n_options_max = 255; } else if (model_type == COMMON_DECISION_TYPE_LAYA) { token_marker = llama_vocab_mask(vocab); token_sep = llama_vocab_sep(vocab); @@ -254,23 +288,124 @@ static json decision_replace_text(const json & val, const std::string & search, return val; } -std::string server_decision_context::render(const json & state, const server_decision_question & question, size_t n_images) const { +// sort the keys of all objects of a JSON value +static json decision_sort_keys(const json & val) { + if (val.is_array()) { + json out = json::array(); + for (const auto & item : val) { + out.push_back(decision_sort_keys(item)); + } + return out; + } + if (val.is_object()) { + std::map sorted; + for (const auto & [key, item] : val.items()) { + sorted[key] = decision_sort_keys(item); + } + json out = json::object(); + for (const auto & [key, item] : sorted) { + out[key] = item; + } + return out; + } + return val; +} + +// kev flattens a JSON value into text, the keys of an object are kept as labels (kev/api.py: render) +static std::string decision_kev_render(const json & val, int indent = 0) { + const std::string pad(2 * indent, ' '); + if (val.is_null()) { + return ""; + } + if (val.is_string()) { + return val.get(); + } + if (val.is_boolean()) { + return val.get() ? "True" : "False"; + } + if (val.is_array()) { + std::string out; + for (const auto & item : val) { + const std::string text = decision_kev_render(item, indent + 1); + out += (out.empty() ? "" : "\n") + pad + "- " + text.substr(std::min(text.size(), text.find_first_not_of(" \t\n\r"))); + } + return out; + } + if (val.is_object()) { + std::string out; + for (const auto & [key, item] : val.items()) { + const bool is_nested = item.is_object() || item.is_array(); + out += (out.empty() ? "" : "\n") + pad + key + (is_nested ? ":\n" : ": ") + decision_kev_render(item, is_nested ? indent + 1 : 0); + } + return out; + } + return val.dump(); +} + +// kev text input: special tokens written in the text must not be parsed as such +static std::string decision_kev_text(const json & val) { + static const std::regex re_special("<\\|([A-Za-z0-9_]+)\\|>"); + return std::regex_replace(decision_kev_render(val), re_special, "<\xC2\xA6$1\xC2\xA6>"); +} + +size_t server_decision_context::n_variants(const server_decision_question & question) const { + // lev shows the options of a choice in 2 orders, to cancel the preference for the first label + if (type == COMMON_DECISION_TYPE_LEV && question.type == SERVER_DECISION_QUESTION_CHOICE && question.options.size() > 1) { + return 2; + } + return 1; +} + +size_t server_decision_context::n_outputs(const server_decision_question & question) const { + if (type == COMMON_DECISION_TYPE_LEV && question.type == SERVER_DECISION_QUESTION_NOUL) { + return DECISION_LEV_N_RATINGS; + } + return question.options.size(); +} + +std::string server_decision_context::render(const json & state, const server_decision_question & question, size_t variant, size_t n_images) const { + const size_t n_options = question.options.size(); + + // the second variant shows the options in the reverse order json options = json::array(); - for (const auto & opt : question.options) { - options.push_back(json{ + for (size_t i = 0; i < n_options; i++) { + const auto & opt = question.options[variant == 0 ? i : n_options - 1 - i]; + json option = json{ {"key", opt.key}, {"description", opt.description}, - }); + }; + if (type == COMMON_DECISION_TYPE_KEV) { + option["key"] = decision_kev_text(opt.key); + if (!opt.description.is_null()) { + option["description"] = decision_kev_text(opt.description); + } + } + if (!label_texts.empty()) { + option["label"] = label_texts[i]; + } + options.push_back(option); } // the template is given raw JSON values, it serializes the ones that are not strings json inp = json{ + {"id", question.id}, {"type", decision_question_type_name(question.type)}, {"instructions", question.instructions}, {"state", state}, {"options", options}, }; + // lev was trained with sorted keys + if (type == COMMON_DECISION_TYPE_LEV) { + inp = decision_sort_keys(inp); + } + + // the kev template only takes text + if (type == COMMON_DECISION_TYPE_KEV) { + inp["state"] = decision_kev_text(state); + inp["instructions"] = decision_kev_text(question.instructions); + } + // the input must not contain the marker of the options if (!text_marker.empty()) { inp = decision_replace_text(inp, text_marker, " "); @@ -296,14 +431,15 @@ std::string server_decision_context::render(const json & state, const server_dec void server_decision_context::fill_task( const json & state, const server_decision_question & question, + size_t variant, const std::vector & files, mtmd_context * mctx, const mtmd_helper_init_opt & init_opt, server_task & task) const { - const std::string prompt = render(state, question, files.size()); + const std::string prompt = render(state, question, variant, files.size()); - if (type == COMMON_DECISION_TYPE_OPENJEV) { - task.decision.labels.assign(labels.begin(), labels.begin() + question.options.size()); + if (type == COMMON_DECISION_TYPE_OPENJEV || type == COMMON_DECISION_TYPE_LEV) { + task.decision.labels.assign(labels.begin(), labels.begin() + n_outputs(question)); if (!files.empty()) { task.tokens = process_mtmd_prompt(mctx, prompt, files, init_opt); return; @@ -314,6 +450,18 @@ void server_decision_context::fill_task( if (type == COMMON_DECISION_TYPE_LAYA) { fill_task_laya(tokens, question, task); } + if (type == COMMON_DECISION_TYPE_KEV) { + // an option is read at its end token, the question at the last token + for (size_t i = 0; i < tokens.size(); i++) { + if (tokens[i] == token_marker) { + task.decision.markers.push_back(i); + } + } + if (task.decision.markers.size() != question.options.size()) { + throw std::runtime_error("unexpected layout of the decision prompt"); + } + task.decision.pointer = tokens.size() - 1; + } task.tokens = server_tokens(tokens, false); } @@ -381,7 +529,14 @@ void server_decision_context::fill_task_laya(llama_tokens & tokens, const server float server_decision_context::get_temperature(const server_decision_question & question) const { const size_t n = question.options.size(); const std::string type_name = decision_question_type_name(question.type); - const std::string bucket = n <= 2 ? "2" : n <= 5 ? "3_5" : n <= 10 ? "6_10" : "11"; + + // the temperature can depend on the number of options, the buckets are the ones used to fit it + std::string bucket; + if (type == COMMON_DECISION_TYPE_LEV) { + bucket = n <= 8 ? "small" : n <= 26 ? "mid" : "large"; + } else { + bucket = n <= 2 ? "2" : n <= 5 ? "3_5" : n <= 10 ? "6_10" : "11"; + } for (const auto & name : {type_name + "." + bucket, type_name}) { const auto it = temperatures.find(name); @@ -420,28 +575,44 @@ static double decision_confidence_score(const std::vector & probs) { return std::max(0.0, 1.0 - dist / dist_uniform); } -json server_decision_context::format_answer(const server_decision_question & question, const std::vector & scores) const { - const size_t n = question.options.size(); - if (scores.size() != n) { - throw std::runtime_error("decision result does not match the number of options"); +json server_decision_context::format_answer(const server_decision_question & question, const std::vector> & scores) const { + const size_t n = n_outputs(question); + if (scores.size() != n_variants(question)) { + throw std::runtime_error("decision result does not match the number of variants"); } - // softmax over the options + // softmax over the outputs of each variant, then the average of the variants const float temperature = get_temperature(question); - const float score_max = *std::max_element(scores.begin(), scores.end()); - std::vector probs(n); - double sum = 0.0; - for (size_t i = 0; i < n; i++) { - probs[i] = std::exp((double) (scores[i] - score_max) / temperature); - sum += probs[i]; - } - for (auto & p : probs) { - p /= sum; + std::vector probs(n, 0.0); + for (size_t v = 0; v < scores.size(); v++) { + const auto & s = scores[v]; + if (s.size() != n) { + throw std::runtime_error("decision result does not match the number of options"); + } + const float score_max = *std::max_element(s.begin(), s.end()); + std::vector p(n); + double sum = 0.0; + for (size_t i = 0; i < n; i++) { + p[i] = std::exp((double) (s[i] - score_max) / temperature); + sum += p[i]; + } + for (size_t i = 0; i < n; i++) { + // the second variant is in the reverse order + probs[v == 0 ? i : n - 1 - i] += p[i] / sum / scores.size(); + } } json answer = json{{"type", decision_question_type_name(question.type)}}; if (question.type == SERVER_DECISION_QUESTION_NOUL) { + if (type == COMMON_DECISION_TYPE_LEV) { + double expected = 0.0; + for (size_t i = 0; i < n; i++) { + expected += probs[i] * i / (n - 1); + } + answer["noul"] = expected; + return answer; + } for (size_t i = 0; i < n; i++) { if (question.options[i].key == "true") { answer["noul"] = probs[i]; diff --git a/tools/server/server-decision.h b/tools/server/server-decision.h index 937a2620ed..5e38feb441 100644 --- a/tools/server/server-decision.h +++ b/tools/server/server-decision.h @@ -39,6 +39,8 @@ struct server_decision_context { bool can_share_prompt() const { switch (type) { case COMMON_DECISION_TYPE_OPENJEV: + case COMMON_DECISION_TYPE_LEV: + case COMMON_DECISION_TYPE_KEV: return true; default: return false; @@ -62,18 +64,22 @@ struct server_decision_context { // images come from "images" and from the image_url parts of a state made of chat messages json parse_state(const json & body, std::vector & files) const; - // set the prompt of this question, and where to read its result + // number of prompts that are evaluated to answer this question, each one shows the options in a different order + size_t n_variants(const server_decision_question & question) const; + + // set the prompt of one variant of this question, and where to read its result // mctx is only used if there are files void fill_task( const json & state, const server_decision_question & question, + size_t variant, const std::vector & files, mtmd_context * mctx, const mtmd_helper_init_opt & init_opt, server_task & task) const; - // scores: one raw model output per option - json format_answer(const server_decision_question & question, const std::vector & scores) const; + // scores: the raw model outputs of each variant + json format_answer(const server_decision_question & question, const std::vector> & scores) const; private: const llama_vocab * vocab = nullptr; @@ -83,17 +89,19 @@ private: size_t n_options_max = 0; bool noul_true_first = false; // noul options are [true, false] instead of [false, true] - // OPENJEV + // OPENJEV, LEV std::vector labels; + std::vector label_texts; // only if the label of an option is given to the template - // LAYA + // LAYA, KEV llama_token token_marker = LLAMA_TOKEN_NULL; llama_token token_sep = LLAMA_TOKEN_NULL; std::string text_marker; size_t max_head_tokens = 0; // question + options size_t max_option_tokens = 48; - std::string render(const json & state, const server_decision_question & question, size_t n_images) const; + std::string render(const json & state, const server_decision_question & question, size_t variant, size_t n_images) const; + size_t n_outputs(const server_decision_question & question) const; void fill_task_laya(llama_tokens & tokens, const server_decision_question & question, server_task & task) const; float get_temperature(const server_decision_question & question) const; diff --git a/tools/server/server-task.h b/tools/server/server-task.h index 05994ab5e5..848dfa0a44 100644 --- a/tools/server/server-task.h +++ b/tools/server/server-task.h @@ -181,6 +181,8 @@ struct server_task { std::vector labels; // logits of these tokens, at the last prompt token std::vector markers; // embeddings[column] at these prompt positions int32_t column = 0; + // if set, embeddings is [q | k], and the output is instead the scaled dot product of q[pointer] and k[marker] + int32_t pointer = -1; }; decision decision;