mirror of
https://github.com/ggml-org/llama.cpp.git
synced 2026-10-02 02:47:26 -05:00
init support for clef (text only)
This commit is contained in:
+13
-3
@@ -3,6 +3,9 @@
|
||||
|
||||
#include "build-info.h"
|
||||
#include "common.h"
|
||||
|
||||
#include "../src/llama-ext.h"
|
||||
|
||||
#include "fit.h"
|
||||
#include "log.h"
|
||||
#include "llama.h"
|
||||
@@ -1208,6 +1211,9 @@ common_decision_type common_get_decision_type(const struct llama_model * model)
|
||||
if (type == "laya") {
|
||||
return COMMON_DECISION_TYPE_LAYA;
|
||||
}
|
||||
if (type == "clef") {
|
||||
return COMMON_DECISION_TYPE_CLEF;
|
||||
}
|
||||
return COMMON_DECISION_TYPE_UNKNOWN;
|
||||
}
|
||||
|
||||
@@ -1263,9 +1269,10 @@ common_init_result::common_init_result(common_params & params, bool model_only)
|
||||
|
||||
const llama_vocab * vocab = llama_model_get_vocab(model);
|
||||
|
||||
// this decision model returns a score for each token via the embeddings output
|
||||
// these decision models return a score for each token via the embeddings output
|
||||
// TODO: maybe improve this in the future
|
||||
if (common_get_decision_type(model) == COMMON_DECISION_TYPE_LAYA) {
|
||||
const auto decision_type = common_get_decision_type(model);
|
||||
if (decision_type == COMMON_DECISION_TYPE_LAYA || decision_type == COMMON_DECISION_TYPE_CLEF) {
|
||||
params.embedding = true;
|
||||
params.pooling_type = LLAMA_POOLING_TYPE_NONE;
|
||||
|
||||
@@ -1274,7 +1281,7 @@ common_init_result::common_init_result(common_params & params, bool model_only)
|
||||
cparams.n_outputs_max = cparams.n_batch;
|
||||
cparams.n_outputs_max_per_seq = 1;
|
||||
|
||||
LOG_INF("%s", "laya decision model detected, enabling embedding mode\n");
|
||||
LOG_INF("%s", "decision model detected, enabling embedding mode\n");
|
||||
}
|
||||
|
||||
// load and optionally apply lora adapters
|
||||
@@ -2211,6 +2218,9 @@ llama_batch_ext * common_batch::get_sub_batch(int32_t off, int32_t n) {
|
||||
if (t.output) {
|
||||
llama_batch_ext_set_output_logits(res, idx, true);
|
||||
}
|
||||
if (t.decision_order != 0) {
|
||||
llama_batch_ext_set_decision_order(res, idx, t.decision_order);
|
||||
}
|
||||
}
|
||||
|
||||
return res;
|
||||
|
||||
@@ -950,6 +950,7 @@ enum common_decision_type {
|
||||
COMMON_DECISION_TYPE_UNKNOWN, // a decision model of a type that is not supported
|
||||
COMMON_DECISION_TYPE_OPENJEV, // logits of one label token per option, read at the last prompt token
|
||||
COMMON_DECISION_TYPE_LAYA, // score of one marker token per option, read from the embeddings output
|
||||
COMMON_DECISION_TYPE_CLEF, // all questions in one prompt, score of option i read from the embeddings output at row i
|
||||
};
|
||||
|
||||
common_decision_type common_get_decision_type(const struct llama_model * model);
|
||||
@@ -1051,6 +1052,7 @@ struct common_batch {
|
||||
bool output;
|
||||
llama_embd embd; // non-owning view of the data passed to add_embd()/set_embd(), data == NULL if none
|
||||
std::vector<llama_seq_id> seq_ids_extra; // see add_seq()
|
||||
int32_t decision_order = 0; // see llama_batch_ext_set_decision_order()
|
||||
};
|
||||
|
||||
std::vector<token> tokens; // mirror of the entries, tokens[i] describes batch index i
|
||||
|
||||
@@ -42,6 +42,7 @@ TEXT_MODEL_MAP: dict[str, str] = {
|
||||
"ChameleonForConditionalGeneration": "chameleon",
|
||||
"ChatGLMForConditionalGeneration": "chatglm",
|
||||
"ChatGLMModel": "chatglm",
|
||||
"ClefModel": "clef",
|
||||
"CodeShellForCausalLM": "codeshell",
|
||||
"CogVLMForCausalLM": "cogvlm",
|
||||
"Cohere2MoeForCausalLM": "command_r",
|
||||
|
||||
@@ -0,0 +1,132 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import math
|
||||
|
||||
from pathlib import Path
|
||||
from typing import Any, Iterable, TYPE_CHECKING
|
||||
|
||||
import torch
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from torch import Tensor
|
||||
|
||||
from .base import ModelBase, gguf, logger
|
||||
from .qwen import Qwen3_5TextModel
|
||||
|
||||
|
||||
def _is_clef_checkpoint(dir_model: Path) -> bool:
|
||||
return (dir_model / "joint_head_config.json").is_file() and (dir_model / "config.json").is_file()
|
||||
|
||||
|
||||
@ModelBase.register_hparams_loader(_is_clef_checkpoint)
|
||||
def _load_clef_hparams(dir_model: Path) -> dict[str, Any]:
|
||||
logger.info("gguf: detected Clef checkpoint")
|
||||
hparams = ModelBase.load_hparams(dir_model, False, guess=False)
|
||||
hparams["architectures"] = ["ClefModel"]
|
||||
with open(dir_model / "joint_head_config.json", encoding="utf-8") as f:
|
||||
hparams["decision"] = json.load(f)
|
||||
return hparams
|
||||
|
||||
|
||||
# TODO: image input needs token and embedding entries in the same batch, see https://github.com/ggml-org/llama.cpp/pull/29622
|
||||
@ModelBase.register("ClefModel")
|
||||
class ClefModel(Qwen3_5TextModel):
|
||||
model_arch = gguf.MODEL_ARCH.CLEF
|
||||
no_mtp = True # the checkpoint has no MTP head
|
||||
|
||||
# prompt follows joint_schema_model.py of the model repo
|
||||
_SYSTEM_PROMPT = (
|
||||
"Read the complete state and schema. Decide every field jointly. Each answer "
|
||||
"must be exactly one of that field's allowed options."
|
||||
)
|
||||
# the pieces of the prompt are tokenized one by one, this separates them
|
||||
_PIECE_SEP = "<<clef:sep>>"
|
||||
# start of a piece: state, question span, option span
|
||||
_PIECE_STATE, _PIECE_QUESTION, _PIECE_OPTION = "<<clef:state>>", "<<clef:question>>", "<<clef:option>>"
|
||||
|
||||
def __init__(self, *args, **kwargs):
|
||||
super().__init__(*args, **kwargs)
|
||||
head = self.hparams["decision"]
|
||||
self._n_routing = head["routing_layers"]
|
||||
# the head blocks are named dec.blk.N, routing blocks first
|
||||
self.tensor_map = gguf.get_tensor_name_map(self.model_arch, max(self.block_count, self._n_routing + head["layers"]))
|
||||
self._scales: dict[str, float] = {}
|
||||
|
||||
def set_vocab(self):
|
||||
super().set_vocab()
|
||||
self.gguf_writer.add_chat_template([{"name": "systemone", "template": self._systemone_template()}])
|
||||
|
||||
@classmethod
|
||||
def _systemone_template(cls) -> str:
|
||||
def text(value: str) -> str:
|
||||
return "{{ " + json.dumps(value) + " }}"
|
||||
|
||||
sep = text(cls._PIECE_SEP)
|
||||
# state, instructions and option text are given as strings
|
||||
return (
|
||||
text(f"<|im_start|>system\n{cls._SYSTEM_PROMPT}<|im_end|>\n<|im_start|>user\nSTATE:\n")
|
||||
+ sep + text(cls._PIECE_STATE) + "{{ state }}"
|
||||
+ sep + text("\n\nSCHEMA FIELDS:\n")
|
||||
+ "{% for q in questions %}"
|
||||
+ sep + text("\nFIELD ") + "{{ loop.index }}" + text("\nID: ") + "{{ q.id }}"
|
||||
+ text("\nTYPE: ") + "{{ q.type }}" + text("\nINSTRUCTION: ")
|
||||
+ sep + text(cls._PIECE_QUESTION) + "{{ q.instructions }}"
|
||||
+ sep + text("\nALLOWED OPTIONS:\n")
|
||||
+ "{% for o in q.options %}"
|
||||
+ sep + text("OPTION ") + "{{ loop.index }}" + text(": ")
|
||||
+ sep + text(cls._PIECE_OPTION) + "{{ o.text }}"
|
||||
+ sep + text("\n")
|
||||
+ "{% endfor %}"
|
||||
+ sep + text("END FIELD\n")
|
||||
+ "{% endfor %}"
|
||||
+ sep + text("\n<|im_end|>\n<|im_start|>assistant\n<think>\n\n</think>\n\nJOINT SCHEMA DECISIONS:")
|
||||
)
|
||||
|
||||
def set_gguf_parameters(self):
|
||||
super().set_gguf_parameters()
|
||||
head = self.hparams["decision"]
|
||||
self.gguf_writer.add_decision_type(gguf.DecisionType.CLEF)
|
||||
self.gguf_writer.add_decision_routing_block_count(head["routing_layers"])
|
||||
self.gguf_writer.add_decision_block_count(head["layers"])
|
||||
self.gguf_writer.add_decision_head_count(head["heads"])
|
||||
|
||||
def generate_extra_tensors(self) -> Iterable[tuple[str, Tensor]]:
|
||||
yield from super().generate_extra_tensors()
|
||||
from safetensors.torch import load_file
|
||||
for name, data in load_file(self.dir_model / "joint_head.safetensors").items():
|
||||
yield "joint_head." + name, data
|
||||
|
||||
def modify_tensors(self, data_torch: Tensor, name: str, bid: int | None) -> Iterable[tuple[str, Tensor]]:
|
||||
if not name.startswith("joint_head."):
|
||||
yield from super().modify_tensors(data_torch, name, bid)
|
||||
return
|
||||
|
||||
parts = name.split(".")
|
||||
|
||||
# learned scalars, stored as the values used at inference
|
||||
if len(parts) == 2 and data_torch.ndim == 0:
|
||||
value = float(data_torch)
|
||||
if parts[1] == "residual_gate":
|
||||
self._scales[parts[1]] = 1.0 / (1.0 + math.exp(-value))
|
||||
else:
|
||||
self._scales[parts[1]] = math.exp(min(value, math.log(100.0)))
|
||||
if len(self._scales) == 3:
|
||||
scales = [self._scales[k] for k in ("prior_logit_scale", "joint_logit_scale", "residual_gate")]
|
||||
yield self.format_tensor_name(gguf.MODEL_TENSOR.DECISION_SCALES, suffix=""), torch.tensor(scales, dtype=torch.float32)
|
||||
return
|
||||
|
||||
# routing blocks come first
|
||||
if parts[1] == "layers":
|
||||
parts[2] = str(int(parts[2]) + self._n_routing)
|
||||
name = ".".join(parts)
|
||||
|
||||
# nn.MultiheadAttention keeps q, k, v in one tensor
|
||||
for suffix in ("weight", "bias"):
|
||||
if name.endswith(".in_proj_" + suffix):
|
||||
prefix = name[:-len("in_proj_" + suffix)]
|
||||
for x, data in zip("qkv", data_torch.chunk(3, dim=0)):
|
||||
yield self.map_tensor_name(prefix + x + "." + suffix), data
|
||||
return
|
||||
|
||||
yield self.map_tensor_name(name), data_torch
|
||||
@@ -324,6 +324,8 @@ class Keys:
|
||||
TYPE = "{arch}.decision.type"
|
||||
# note: single-use-case keys can be hard-coded in cpp code
|
||||
BLOCK_COUNT = "{arch}.decision.block_count"
|
||||
ROUTING_BLOCK_COUNT = "{arch}.decision.routing_block_count"
|
||||
HEAD_COUNT = "{arch}.decision.head_count"
|
||||
MAX_HEAD_TOKENS = "{arch}.decision.max_head_tokens"
|
||||
TEMPERATURE = "{arch}.decision.temperature.{name}" # name: "<type>" or "<type>.<n_opt bucket>"
|
||||
|
||||
@@ -537,6 +539,7 @@ class MODEL_ARCH(IntEnum):
|
||||
QWEN3VLMOE = auto()
|
||||
QWEN35 = auto()
|
||||
QWEN35MOE = auto()
|
||||
CLEF = auto()
|
||||
QWEN4EXP = auto()
|
||||
PHI2 = auto()
|
||||
PHI3 = auto()
|
||||
@@ -871,6 +874,18 @@ class MODEL_TENSOR(IntEnum):
|
||||
DEC_FFN_DOWN = auto()
|
||||
DEC_FFN_UP = auto()
|
||||
DEC_OUTPUT_NORM = auto()
|
||||
DEC_CROSS_ATTN_NORM_KV = auto()
|
||||
DECISION_HIDDEN_NORM = auto()
|
||||
DECISION_PROJ_MEMORY = auto()
|
||||
DECISION_PROJ_QUESTION = auto()
|
||||
DECISION_PROJ_OPTION_QUESTION = auto()
|
||||
DECISION_PROJ_GLOBAL = auto()
|
||||
DECISION_PROJ_OPTION_CONTEXT = auto()
|
||||
DECISION_PROJ_OPTION_LEXICAL = auto()
|
||||
DECISION_OPTION_SUMMARY_NORM = auto()
|
||||
DECISION_FIELD_NORM = auto()
|
||||
DECISION_OPTION_NORM = auto()
|
||||
DECISION_SCALES = auto()
|
||||
ENC_ATTN_NORM = auto()
|
||||
ENC_ATTN_Q = auto()
|
||||
ENC_ATTN_K = auto()
|
||||
@@ -1301,6 +1316,7 @@ MODEL_ARCH_NAMES: dict[MODEL_ARCH, str] = {
|
||||
MODEL_ARCH.QWEN3VLMOE: "qwen3vlmoe",
|
||||
MODEL_ARCH.QWEN35: "qwen35",
|
||||
MODEL_ARCH.QWEN35MOE: "qwen35moe",
|
||||
MODEL_ARCH.CLEF: "clef",
|
||||
MODEL_ARCH.QWEN4EXP: "qwen4exp",
|
||||
MODEL_ARCH.PHI2: "phi2",
|
||||
MODEL_ARCH.PHI3: "phi3",
|
||||
@@ -1634,6 +1650,18 @@ TENSOR_NAMES: dict[MODEL_TENSOR, str] = {
|
||||
MODEL_TENSOR.DEC_FFN_DOWN: "dec.blk.{bid}.ffn_down",
|
||||
MODEL_TENSOR.DEC_FFN_UP: "dec.blk.{bid}.ffn_up",
|
||||
MODEL_TENSOR.DEC_OUTPUT_NORM: "dec.output_norm",
|
||||
MODEL_TENSOR.DEC_CROSS_ATTN_NORM_KV: "dec.blk.{bid}.cross_attn_norm_kv",
|
||||
MODEL_TENSOR.DECISION_HIDDEN_NORM: "decision.hidden_norm",
|
||||
MODEL_TENSOR.DECISION_PROJ_MEMORY: "decision.proj_memory",
|
||||
MODEL_TENSOR.DECISION_PROJ_QUESTION: "decision.proj_question",
|
||||
MODEL_TENSOR.DECISION_PROJ_OPTION_QUESTION: "decision.proj_option_question",
|
||||
MODEL_TENSOR.DECISION_PROJ_GLOBAL: "decision.proj_global",
|
||||
MODEL_TENSOR.DECISION_PROJ_OPTION_CONTEXT: "decision.proj_option_context",
|
||||
MODEL_TENSOR.DECISION_PROJ_OPTION_LEXICAL: "decision.proj_option_lexical",
|
||||
MODEL_TENSOR.DECISION_OPTION_SUMMARY_NORM: "decision.option_summary_norm",
|
||||
MODEL_TENSOR.DECISION_FIELD_NORM: "decision.field_norm",
|
||||
MODEL_TENSOR.DECISION_OPTION_NORM: "decision.option_norm",
|
||||
MODEL_TENSOR.DECISION_SCALES: "decision.scales",
|
||||
MODEL_TENSOR.ENC_ATTN_NORM: "enc.blk.{bid}.attn_norm",
|
||||
MODEL_TENSOR.ENC_ATTN_Q: "enc.blk.{bid}.attn_q",
|
||||
MODEL_TENSOR.ENC_ATTN_K: "enc.blk.{bid}.attn_k",
|
||||
@@ -2917,6 +2945,60 @@ MODEL_TENSORS: dict[MODEL_ARCH, list[MODEL_TENSOR]] = {
|
||||
MODEL_TENSOR.NEXTN_SHARED_HEAD_HEAD,
|
||||
MODEL_TENSOR.NEXTN_SHARED_HEAD_NORM,
|
||||
],
|
||||
MODEL_ARCH.CLEF: [
|
||||
MODEL_TENSOR.TOKEN_EMBD,
|
||||
MODEL_TENSOR.OUTPUT_NORM,
|
||||
MODEL_TENSOR.OUTPUT,
|
||||
MODEL_TENSOR.ATTN_NORM,
|
||||
MODEL_TENSOR.ATTN_Q,
|
||||
MODEL_TENSOR.ATTN_Q_NORM,
|
||||
MODEL_TENSOR.ATTN_K,
|
||||
MODEL_TENSOR.ATTN_K_NORM,
|
||||
MODEL_TENSOR.ATTN_V,
|
||||
MODEL_TENSOR.ATTN_OUT,
|
||||
MODEL_TENSOR.ATTN_POST_NORM,
|
||||
MODEL_TENSOR.ATTN_GATE,
|
||||
MODEL_TENSOR.ATTN_QKV,
|
||||
MODEL_TENSOR.FFN_GATE,
|
||||
MODEL_TENSOR.FFN_DOWN,
|
||||
MODEL_TENSOR.FFN_UP,
|
||||
MODEL_TENSOR.SSM_A,
|
||||
MODEL_TENSOR.SSM_CONV1D,
|
||||
MODEL_TENSOR.SSM_DT,
|
||||
MODEL_TENSOR.SSM_NORM,
|
||||
MODEL_TENSOR.SSM_BETA,
|
||||
MODEL_TENSOR.SSM_ALPHA,
|
||||
MODEL_TENSOR.SSM_OUT,
|
||||
# decision head
|
||||
MODEL_TENSOR.TOKEN_TYPES,
|
||||
MODEL_TENSOR.CLS,
|
||||
MODEL_TENSOR.CLS_OUT,
|
||||
MODEL_TENSOR.DEC_ATTN_NORM,
|
||||
MODEL_TENSOR.DEC_ATTN_Q,
|
||||
MODEL_TENSOR.DEC_ATTN_K,
|
||||
MODEL_TENSOR.DEC_ATTN_V,
|
||||
MODEL_TENSOR.DEC_ATTN_OUT,
|
||||
MODEL_TENSOR.DEC_CROSS_ATTN_NORM,
|
||||
MODEL_TENSOR.DEC_CROSS_ATTN_NORM_KV,
|
||||
MODEL_TENSOR.DEC_CROSS_ATTN_Q,
|
||||
MODEL_TENSOR.DEC_CROSS_ATTN_K,
|
||||
MODEL_TENSOR.DEC_CROSS_ATTN_V,
|
||||
MODEL_TENSOR.DEC_CROSS_ATTN_OUT,
|
||||
MODEL_TENSOR.DEC_FFN_NORM,
|
||||
MODEL_TENSOR.DEC_FFN_DOWN,
|
||||
MODEL_TENSOR.DEC_FFN_UP,
|
||||
MODEL_TENSOR.DECISION_HIDDEN_NORM,
|
||||
MODEL_TENSOR.DECISION_PROJ_MEMORY,
|
||||
MODEL_TENSOR.DECISION_PROJ_QUESTION,
|
||||
MODEL_TENSOR.DECISION_PROJ_OPTION_QUESTION,
|
||||
MODEL_TENSOR.DECISION_PROJ_GLOBAL,
|
||||
MODEL_TENSOR.DECISION_PROJ_OPTION_CONTEXT,
|
||||
MODEL_TENSOR.DECISION_PROJ_OPTION_LEXICAL,
|
||||
MODEL_TENSOR.DECISION_OPTION_SUMMARY_NORM,
|
||||
MODEL_TENSOR.DECISION_FIELD_NORM,
|
||||
MODEL_TENSOR.DECISION_OPTION_NORM,
|
||||
MODEL_TENSOR.DECISION_SCALES,
|
||||
],
|
||||
MODEL_ARCH.QWEN35MOE: [
|
||||
MODEL_TENSOR.TOKEN_EMBD,
|
||||
MODEL_TENSOR.OUTPUT_NORM,
|
||||
@@ -5917,6 +5999,7 @@ class GGUFValueType(IntEnum):
|
||||
class DecisionType:
|
||||
LAYA = "laya" # head blocks + scorer on the hidden state of one marker token per option
|
||||
OPENJEV = "openjev" # logits of one label token per option
|
||||
CLEF = "clef" # joint head over all questions, one score per option
|
||||
|
||||
|
||||
class VisionProjectorType:
|
||||
|
||||
@@ -1346,6 +1346,12 @@ class GGUFWriter:
|
||||
def add_decision_block_count(self, value: int) -> None:
|
||||
self.add_uint32(Keys.Decision.BLOCK_COUNT.format(arch=self.arch), value)
|
||||
|
||||
def add_decision_routing_block_count(self, value: int) -> None:
|
||||
self.add_uint32(Keys.Decision.ROUTING_BLOCK_COUNT.format(arch=self.arch), value)
|
||||
|
||||
def add_decision_head_count(self, value: int) -> None:
|
||||
self.add_uint32(Keys.Decision.HEAD_COUNT.format(arch=self.arch), value)
|
||||
|
||||
def add_decision_max_head_tokens(self, value: int) -> None:
|
||||
self.add_uint32(Keys.Decision.MAX_HEAD_TOKENS.format(arch=self.arch), value)
|
||||
|
||||
|
||||
@@ -49,6 +49,7 @@ class TensorNameMap:
|
||||
MODEL_TENSOR.TOKEN_TYPES: (
|
||||
"embeddings.token_type_embeddings", # bert nomic-bert
|
||||
"type_emb", # laya
|
||||
"joint_head.type_embedding", # clef
|
||||
),
|
||||
|
||||
# Normalization of token embeddings
|
||||
@@ -1181,22 +1182,27 @@ class TensorNameMap:
|
||||
|
||||
MODEL_TENSOR.DEC_ATTN_NORM: (
|
||||
"decoder.block.{bid}.layer.0.layer_norm", # t5
|
||||
"joint_head.layers.{bid}.norm1", # clef
|
||||
),
|
||||
|
||||
MODEL_TENSOR.DEC_ATTN_Q: (
|
||||
"decoder.block.{bid}.layer.0.SelfAttention.q", # t5
|
||||
"joint_head.layers.{bid}.self_attn.q", # clef
|
||||
),
|
||||
|
||||
MODEL_TENSOR.DEC_ATTN_K: (
|
||||
"decoder.block.{bid}.layer.0.SelfAttention.k", # t5
|
||||
"joint_head.layers.{bid}.self_attn.k", # clef
|
||||
),
|
||||
|
||||
MODEL_TENSOR.DEC_ATTN_V: (
|
||||
"decoder.block.{bid}.layer.0.SelfAttention.v", # t5
|
||||
"joint_head.layers.{bid}.self_attn.v", # clef
|
||||
),
|
||||
|
||||
MODEL_TENSOR.DEC_ATTN_OUT: (
|
||||
"decoder.block.{bid}.layer.0.SelfAttention.o", # t5
|
||||
"joint_head.layers.{bid}.self_attn.out_proj", # clef
|
||||
),
|
||||
|
||||
MODEL_TENSOR.DEC_ATTN_REL_B: (
|
||||
@@ -1205,22 +1211,32 @@ class TensorNameMap:
|
||||
|
||||
MODEL_TENSOR.DEC_CROSS_ATTN_NORM: (
|
||||
"decoder.block.{bid}.layer.1.layer_norm", # t5
|
||||
"joint_head.layers.{bid}.norm2", # clef
|
||||
"joint_head.evidence_layers.{bid}.query_norm", # clef
|
||||
),
|
||||
|
||||
MODEL_TENSOR.DEC_CROSS_ATTN_Q: (
|
||||
"decoder.block.{bid}.layer.1.EncDecAttention.q", # t5
|
||||
"joint_head.layers.{bid}.multihead_attn.q", # clef
|
||||
"joint_head.evidence_layers.{bid}.attention.q", # clef
|
||||
),
|
||||
|
||||
MODEL_TENSOR.DEC_CROSS_ATTN_K: (
|
||||
"decoder.block.{bid}.layer.1.EncDecAttention.k", # t5
|
||||
"joint_head.layers.{bid}.multihead_attn.k", # clef
|
||||
"joint_head.evidence_layers.{bid}.attention.k", # clef
|
||||
),
|
||||
|
||||
MODEL_TENSOR.DEC_CROSS_ATTN_V: (
|
||||
"decoder.block.{bid}.layer.1.EncDecAttention.v", # t5
|
||||
"joint_head.layers.{bid}.multihead_attn.v", # clef
|
||||
"joint_head.evidence_layers.{bid}.attention.v", # clef
|
||||
),
|
||||
|
||||
MODEL_TENSOR.DEC_CROSS_ATTN_OUT: (
|
||||
"decoder.block.{bid}.layer.1.EncDecAttention.o", # t5
|
||||
"joint_head.layers.{bid}.multihead_attn.out_proj", # clef
|
||||
"joint_head.evidence_layers.{bid}.attention.out_proj", # clef
|
||||
),
|
||||
|
||||
MODEL_TENSOR.DEC_CROSS_ATTN_REL_B: (
|
||||
@@ -1229,6 +1245,8 @@ class TensorNameMap:
|
||||
|
||||
MODEL_TENSOR.DEC_FFN_NORM: (
|
||||
"decoder.block.{bid}.layer.2.layer_norm", # t5
|
||||
"joint_head.layers.{bid}.norm3", # clef
|
||||
"joint_head.evidence_layers.{bid}.feedforward_norm", # clef
|
||||
),
|
||||
|
||||
MODEL_TENSOR.DEC_FFN_GATE: (
|
||||
@@ -1238,16 +1256,64 @@ class TensorNameMap:
|
||||
MODEL_TENSOR.DEC_FFN_UP: (
|
||||
"decoder.block.{bid}.layer.2.DenseReluDense.wi", # t5
|
||||
"decoder.block.{bid}.layer.2.DenseReluDense.wi_1", # flan-t5
|
||||
"joint_head.layers.{bid}.linear1", # clef
|
||||
"joint_head.evidence_layers.{bid}.feedforward.0", # clef
|
||||
),
|
||||
|
||||
MODEL_TENSOR.DEC_FFN_DOWN: (
|
||||
"decoder.block.{bid}.layer.2.DenseReluDense.wo", # t5
|
||||
"joint_head.layers.{bid}.linear2", # clef
|
||||
"joint_head.evidence_layers.{bid}.feedforward.3", # clef
|
||||
),
|
||||
|
||||
MODEL_TENSOR.DEC_OUTPUT_NORM: (
|
||||
"decoder.final_layer_norm", # t5
|
||||
),
|
||||
|
||||
MODEL_TENSOR.DEC_CROSS_ATTN_NORM_KV: (
|
||||
"joint_head.evidence_layers.{bid}.memory_norm", # clef
|
||||
),
|
||||
|
||||
MODEL_TENSOR.DECISION_HIDDEN_NORM: (
|
||||
"joint_head.hidden_norm", # clef
|
||||
),
|
||||
|
||||
MODEL_TENSOR.DECISION_PROJ_MEMORY: (
|
||||
"joint_head.memory_projection", # clef
|
||||
),
|
||||
|
||||
MODEL_TENSOR.DECISION_PROJ_QUESTION: (
|
||||
"joint_head.question_projection", # clef
|
||||
),
|
||||
|
||||
MODEL_TENSOR.DECISION_PROJ_OPTION_QUESTION: (
|
||||
"joint_head.option_question_projection", # clef
|
||||
),
|
||||
|
||||
MODEL_TENSOR.DECISION_PROJ_GLOBAL: (
|
||||
"joint_head.global_projection", # clef
|
||||
),
|
||||
|
||||
MODEL_TENSOR.DECISION_PROJ_OPTION_CONTEXT: (
|
||||
"joint_head.option_context_projection", # clef
|
||||
),
|
||||
|
||||
MODEL_TENSOR.DECISION_PROJ_OPTION_LEXICAL: (
|
||||
"joint_head.option_lexical_projection", # clef
|
||||
),
|
||||
|
||||
MODEL_TENSOR.DECISION_OPTION_SUMMARY_NORM: (
|
||||
"joint_head.option_summary_norm", # clef
|
||||
),
|
||||
|
||||
MODEL_TENSOR.DECISION_FIELD_NORM: (
|
||||
"joint_head.field_norm", # clef
|
||||
),
|
||||
|
||||
MODEL_TENSOR.DECISION_OPTION_NORM: (
|
||||
"joint_head.option_norm", # clef
|
||||
),
|
||||
|
||||
MODEL_TENSOR.ENC_ATTN_NORM: (
|
||||
"encoder.block.{bid}.layer.0.layer_norm", # t5
|
||||
),
|
||||
@@ -1449,11 +1515,13 @@ class TensorNameMap:
|
||||
"dense", # neobert
|
||||
"head.dense", # modern-bert
|
||||
"scorer.1", # laya
|
||||
"joint_head.residual_scorer.0", # clef
|
||||
),
|
||||
|
||||
MODEL_TENSOR.CLS_OUT: (
|
||||
"classifier.out_proj", # roberta
|
||||
"scorer.3", # laya
|
||||
"joint_head.residual_scorer.3", # clef
|
||||
),
|
||||
|
||||
MODEL_TENSOR.CLS_NORM: (
|
||||
|
||||
+28
-1
@@ -40,6 +40,7 @@ static const std::map<llm_arch, const char *> LLM_ARCH_NAMES = {
|
||||
{ LLM_ARCH_QWEN3VLMOE, "qwen3vlmoe" },
|
||||
{ LLM_ARCH_QWEN35, "qwen35" },
|
||||
{ LLM_ARCH_QWEN35MOE, "qwen35moe" },
|
||||
{ LLM_ARCH_CLEF, "clef" },
|
||||
{ LLM_ARCH_QWEN4EXP, "qwen4exp" },
|
||||
{ LLM_ARCH_PHI2, "phi2" },
|
||||
{ LLM_ARCH_PHI3, "phi3" },
|
||||
@@ -366,7 +367,9 @@ static const std::map<llm_kv, const char *> LLM_KV_NAMES = {
|
||||
{ LLM_KV_CLASSIFIER_OUTPUT_LABELS, "%s.classifier.output_labels" },
|
||||
{ LLM_KV_CLASSIFIER_POOLING_TYPE, "%s.classifier.pooling_type" },
|
||||
|
||||
{ LLM_KV_DECISION_BLOCK_COUNT, "%s.decision.block_count" },
|
||||
{ LLM_KV_DECISION_BLOCK_COUNT, "%s.decision.block_count" },
|
||||
{ LLM_KV_DECISION_ROUTING_BLOCK_COUNT, "%s.decision.routing_block_count" },
|
||||
{ LLM_KV_DECISION_HEAD_COUNT, "%s.decision.head_count" },
|
||||
|
||||
{ LLM_KV_TARGET_LAYERS, "%s.target_layers" },
|
||||
{ LLM_KV_TARGET_HIDDEN_SIZE, "%s.target_hidden_size" },
|
||||
@@ -599,6 +602,18 @@ static const std::map<llm_tensor, const char *> LLM_TENSOR_NAMES = {
|
||||
{ LLM_TENSOR_ATTN_SUB_NORM, "blk.%d.attn_sub_norm" },
|
||||
{ LLM_TENSOR_FFN_SUB_NORM, "blk.%d.ffn_sub_norm" },
|
||||
{ LLM_TENSOR_DEC_OUTPUT_NORM, "dec.output_norm" },
|
||||
{ LLM_TENSOR_DEC_CROSS_ATTN_NORM_KV, "dec.blk.%d.cross_attn_norm_kv" },
|
||||
{ LLM_TENSOR_DECISION_HIDDEN_NORM, "decision.hidden_norm" },
|
||||
{ LLM_TENSOR_DECISION_PROJ_MEMORY, "decision.proj_memory" },
|
||||
{ LLM_TENSOR_DECISION_PROJ_QUESTION, "decision.proj_question" },
|
||||
{ LLM_TENSOR_DECISION_PROJ_OPTION_QUESTION, "decision.proj_option_question" },
|
||||
{ LLM_TENSOR_DECISION_PROJ_GLOBAL, "decision.proj_global" },
|
||||
{ LLM_TENSOR_DECISION_PROJ_OPTION_CONTEXT, "decision.proj_option_context" },
|
||||
{ LLM_TENSOR_DECISION_PROJ_OPTION_LEXICAL, "decision.proj_option_lexical" },
|
||||
{ LLM_TENSOR_DECISION_OPTION_SUMMARY_NORM, "decision.option_summary_norm" },
|
||||
{ LLM_TENSOR_DECISION_FIELD_NORM, "decision.field_norm" },
|
||||
{ LLM_TENSOR_DECISION_OPTION_NORM, "decision.option_norm" },
|
||||
{ LLM_TENSOR_DECISION_SCALES, "decision.scales" },
|
||||
{ LLM_TENSOR_DEC_ATTN_NORM, "dec.blk.%d.attn_norm" },
|
||||
{ LLM_TENSOR_DEC_ATTN_Q, "dec.blk.%d.attn_q" },
|
||||
{ LLM_TENSOR_DEC_ATTN_K, "dec.blk.%d.attn_k" },
|
||||
@@ -744,6 +759,18 @@ static const std::map<llm_tensor, llm_tensor_info> LLM_TENSOR_INFOS = {
|
||||
{LLM_TENSOR_OUTPUT_NORM, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_MUL}},
|
||||
{LLM_TENSOR_OUTPUT_NORM_LFM2, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_MUL}},
|
||||
{LLM_TENSOR_DEC_OUTPUT_NORM, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_MUL}},
|
||||
{LLM_TENSOR_DEC_CROSS_ATTN_NORM_KV, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL}},
|
||||
{LLM_TENSOR_DECISION_HIDDEN_NORM, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_MUL}},
|
||||
{LLM_TENSOR_DECISION_PROJ_MEMORY, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_MUL_MAT}},
|
||||
{LLM_TENSOR_DECISION_PROJ_QUESTION, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_MUL_MAT}},
|
||||
{LLM_TENSOR_DECISION_PROJ_OPTION_QUESTION, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_MUL_MAT}},
|
||||
{LLM_TENSOR_DECISION_PROJ_GLOBAL, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_MUL_MAT}},
|
||||
{LLM_TENSOR_DECISION_PROJ_OPTION_CONTEXT, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_MUL_MAT}},
|
||||
{LLM_TENSOR_DECISION_PROJ_OPTION_LEXICAL, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_MUL_MAT}},
|
||||
{LLM_TENSOR_DECISION_OPTION_SUMMARY_NORM, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_MUL}},
|
||||
{LLM_TENSOR_DECISION_FIELD_NORM, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_MUL}},
|
||||
{LLM_TENSOR_DECISION_OPTION_NORM, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_MUL}},
|
||||
{LLM_TENSOR_DECISION_SCALES, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_MUL}},
|
||||
{LLM_TENSOR_ENC_OUTPUT_NORM, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_MUL}},
|
||||
{LLM_TENSOR_ROPE_FREQS, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_ROPE}},
|
||||
{LLM_TENSOR_ROPE_FACTORS_LONG, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_ROPE}},
|
||||
|
||||
@@ -45,6 +45,7 @@ enum llm_arch {
|
||||
LLM_ARCH_QWEN3VLMOE,
|
||||
LLM_ARCH_QWEN35,
|
||||
LLM_ARCH_QWEN35MOE,
|
||||
LLM_ARCH_CLEF,
|
||||
LLM_ARCH_QWEN4EXP,
|
||||
LLM_ARCH_PHI2,
|
||||
LLM_ARCH_PHI3,
|
||||
@@ -413,6 +414,8 @@ enum llm_kv {
|
||||
LLM_KV_CLASSIFIER_POOLING_TYPE,
|
||||
|
||||
LLM_KV_DECISION_BLOCK_COUNT,
|
||||
LLM_KV_DECISION_ROUTING_BLOCK_COUNT,
|
||||
LLM_KV_DECISION_HEAD_COUNT,
|
||||
|
||||
LLM_KV_TARGET_LAYERS,
|
||||
LLM_KV_TARGET_HIDDEN_SIZE,
|
||||
@@ -646,6 +649,18 @@ enum llm_tensor {
|
||||
LLM_TENSOR_DEC_FFN_DOWN,
|
||||
LLM_TENSOR_DEC_FFN_UP,
|
||||
LLM_TENSOR_DEC_OUTPUT_NORM,
|
||||
LLM_TENSOR_DEC_CROSS_ATTN_NORM_KV,
|
||||
LLM_TENSOR_DECISION_HIDDEN_NORM,
|
||||
LLM_TENSOR_DECISION_PROJ_MEMORY,
|
||||
LLM_TENSOR_DECISION_PROJ_QUESTION,
|
||||
LLM_TENSOR_DECISION_PROJ_OPTION_QUESTION,
|
||||
LLM_TENSOR_DECISION_PROJ_GLOBAL,
|
||||
LLM_TENSOR_DECISION_PROJ_OPTION_CONTEXT,
|
||||
LLM_TENSOR_DECISION_PROJ_OPTION_LEXICAL,
|
||||
LLM_TENSOR_DECISION_OPTION_SUMMARY_NORM,
|
||||
LLM_TENSOR_DECISION_FIELD_NORM,
|
||||
LLM_TENSOR_DECISION_OPTION_NORM,
|
||||
LLM_TENSOR_DECISION_SCALES,
|
||||
LLM_TENSOR_ENC_ATTN_NORM,
|
||||
LLM_TENSOR_ENC_ATTN_Q,
|
||||
LLM_TENSOR_ENC_ATTN_K,
|
||||
|
||||
@@ -168,6 +168,16 @@ bool llama_batch_allocr::init(
|
||||
}
|
||||
}
|
||||
|
||||
for (int32_t i = 0; i < n_tok; ++i) {
|
||||
if (batch_inp.tokens[i].decision_order != 0) {
|
||||
decision_order.resize(n_tok, 0);
|
||||
break;
|
||||
}
|
||||
}
|
||||
for (size_t i = 0; i < decision_order.size(); ++i) {
|
||||
decision_order[i] = batch_inp.tokens[i].decision_order;
|
||||
}
|
||||
|
||||
//
|
||||
// set up the internal llama_batch to point to our owned arrays
|
||||
//
|
||||
@@ -765,6 +775,7 @@ void llama_batch_allocr::clear() {
|
||||
seq_id .clear();
|
||||
seq_id_unq .clear();
|
||||
output .clear();
|
||||
decision_order.clear();
|
||||
|
||||
for (auto & cur : seq_pos) {
|
||||
cur.clear();
|
||||
@@ -826,6 +837,10 @@ llama_ubatch llama_batch_allocr::ubatch_add(const std::vector<int32_t> & idxs, u
|
||||
udata->n_seq_id[i] = batch.n_seq_id[idxs[i]];
|
||||
udata->output[i] = batch.logits[idxs[i]];
|
||||
|
||||
if (!decision_order.empty()) {
|
||||
udata->decision_order.push_back(decision_order[idxs[i]]);
|
||||
}
|
||||
|
||||
for (int s = 0; s < udata->n_seq_id[i]; ++s) {
|
||||
const llama_seq_id seq_id = batch.seq_id[idxs[i]][s];
|
||||
|
||||
@@ -870,6 +885,10 @@ llama_ubatch llama_batch_allocr::ubatch_add(const std::vector<int32_t> & idxs, u
|
||||
/*.data =*/ std::move(udata),
|
||||
};
|
||||
|
||||
if (!res.data->decision_order.empty()) {
|
||||
res.decision_order = res.data->decision_order.data();
|
||||
}
|
||||
|
||||
if (debug > 0) {
|
||||
LLAMA_LOG_DEBUG("%s: added ubatch to split:\n", __func__);
|
||||
|
||||
@@ -1175,6 +1194,14 @@ bool llama_batch_ext::set_output(int32_t idx, bool output_last) {
|
||||
return true;
|
||||
}
|
||||
|
||||
bool llama_batch_ext::set_decision_order(int32_t idx, int32_t order) {
|
||||
if (idx < 0 || idx >= (int32_t) tokens.size()) {
|
||||
return false;
|
||||
}
|
||||
tokens[idx].decision_order = order;
|
||||
return true;
|
||||
}
|
||||
|
||||
// llama_batch_ext C API
|
||||
|
||||
llama_batch_ext * llama_batch_ext_init(llama_context * ctx) {
|
||||
@@ -1243,6 +1270,10 @@ bool llama_batch_ext_set_output_logits(llama_batch_ext * batch, int32_t idx, boo
|
||||
return batch->set_output(idx, value);
|
||||
}
|
||||
|
||||
bool llama_batch_ext_set_decision_order(llama_batch_ext * batch, int32_t idx, int32_t order) {
|
||||
return batch->set_decision_order(idx, order);
|
||||
}
|
||||
|
||||
// llama_batch_compat
|
||||
|
||||
void llama_batch_compat::init(llama_batch_ext & dst, const llama_batch & batch_inp, size_t n_embd_row) {
|
||||
|
||||
@@ -63,12 +63,16 @@ struct llama_ubatch {
|
||||
std::vector<int32_t> seq_idx;
|
||||
std::vector<int8_t> output;
|
||||
std::vector<int32_t> batch_idxs; // original batch index for each token
|
||||
std::vector<int32_t> decision_order;
|
||||
|
||||
std::vector<llama_seq_id> seq_id_data;
|
||||
};
|
||||
|
||||
// the llama_ubatch pointers above point to this data if set. otherwise - point to external non-owning data
|
||||
std::shared_ptr<data_t> data;
|
||||
|
||||
// [n_tokens], see llama_batch_ext_set_decision_order(), NULL if no entry has one
|
||||
int32_t * decision_order = nullptr;
|
||||
};
|
||||
|
||||
struct llama_hparams;
|
||||
@@ -96,6 +100,7 @@ struct llama_batch_ext {
|
||||
bool has_embd = false; // whether embd_off is set
|
||||
size_t embd_off = 0; // index offset in the embd array
|
||||
bool output = false; // TODO: have dedicated output flags
|
||||
int32_t decision_order = 0; // see llama_batch_ext_set_decision_order()
|
||||
std::unordered_set<llama_seq_id> seq_ids;
|
||||
std::array<llama_pos, GGML_MROPE_SECTIONS> pos = {0, 0, 0, 0};
|
||||
};
|
||||
@@ -125,6 +130,7 @@ struct llama_batch_ext {
|
||||
bool set_token_embd(int32_t idx, llama_embd embd_in);
|
||||
bool set_token_pos(int32_t idx, const llama_pos * pos_in);
|
||||
bool set_output(int32_t idx, bool output_last);
|
||||
bool set_decision_order(int32_t idx, int32_t order);
|
||||
};
|
||||
|
||||
// a helper for sanitizing, fulfilling and splitting a batch
|
||||
@@ -202,6 +208,7 @@ private:
|
||||
std::vector<llama_seq_id> seq_id_unq;
|
||||
std::vector<int32_t> seq_idx;
|
||||
std::vector<int8_t> output;
|
||||
std::vector<int32_t> decision_order; // empty if no entry has one
|
||||
|
||||
using pos_set_t = std::set<llama_pos>;
|
||||
using seq_cpl_t = std::vector<bool>;
|
||||
|
||||
@@ -2406,6 +2406,7 @@ uint32_t llama_context::graph_max_nodes(uint32_t n_tokens) const {
|
||||
model.arch == LLM_ARCH_BAILINGMOE3 ||
|
||||
model.arch == LLM_ARCH_QWEN35 ||
|
||||
model.arch == LLM_ARCH_QWEN35MOE ||
|
||||
model.arch == LLM_ARCH_CLEF ||
|
||||
model.arch == LLM_ARCH_QWEN4EXP ||
|
||||
model.arch == LLM_ARCH_DEEPSEEK4 ||
|
||||
(model.arch == LLM_ARCH_DFLASH && model.hparams.dsv4_hc_mult > 0) ||
|
||||
|
||||
@@ -100,6 +100,19 @@ LLAMA_API void llama_set_embeddings_nextn(struct llama_context * ctx, bool value
|
||||
// chain multiple trained NextN heads. Default 0 (first head).
|
||||
LLAMA_API void llama_set_nextn_layer_offset(struct llama_context * ctx, int32_t offset);
|
||||
|
||||
// Marks the entries that a joint decision head (clef) reads, the default is 0
|
||||
// A run of entries with the same value is one span, spans must be separated by entries with value 0
|
||||
// An option belongs to the last question before it
|
||||
enum llama_decision_order {
|
||||
LLAMA_DECISION_ORDER_NONE = 0, // not read by the head
|
||||
LLAMA_DECISION_ORDER_QUESTION_NOUL = 1, // text of a question
|
||||
LLAMA_DECISION_ORDER_QUESTION_CHOICE = 2,
|
||||
LLAMA_DECISION_ORDER_QUESTION_SCORE = 3,
|
||||
LLAMA_DECISION_ORDER_OPTION = 4, // text of an option
|
||||
};
|
||||
// The embeddings output has one value per entry: row i is the score of option i
|
||||
LLAMA_API bool llama_batch_ext_set_decision_order(struct llama_batch_ext * batch, int32_t idx, int32_t order);
|
||||
|
||||
// mirrors:
|
||||
// LLAMA_API float * llama_get_embeddings(struct llama_context * ctx);
|
||||
LLAMA_API float * llama_get_embeddings_nextn(struct llama_context * ctx);
|
||||
|
||||
@@ -20,6 +20,7 @@ bool llama_model_saver_supports_arch(llm_arch arch) {
|
||||
case LLM_ARCH_T5:
|
||||
case LLM_ARCH_APERTUS:
|
||||
case LLM_ARCH_STEP35:
|
||||
case LLM_ARCH_CLEF: // the head tensors are not saved
|
||||
return false;
|
||||
default:
|
||||
return true;
|
||||
|
||||
+12
-4
@@ -326,6 +326,8 @@ static llama_model * llama_model_mapping(llm_arch arch, const llama_model_params
|
||||
return new llama_model_qwen35(params);
|
||||
case LLM_ARCH_QWEN35MOE:
|
||||
return new llama_model_qwen35moe(params);
|
||||
case LLM_ARCH_CLEF:
|
||||
return new llama_model_clef(params);
|
||||
case LLM_ARCH_QWEN4EXP:
|
||||
return new llama_model_qwen4exp(params);
|
||||
case LLM_ARCH_MISTRAL3:
|
||||
@@ -607,7 +609,7 @@ struct ggml_backend_meta_split_state llama_meta_device_get_split_state(const str
|
||||
auto get_split_segments = [&](int axis, uint32_t il) -> std::vector<std::pair<int64_t, uint32_t>> {
|
||||
// TODO: clarify why this is necessary specifically for these models
|
||||
// TODO: deduplicate condition [TAG_SPLIT_QGATE_QWEN]
|
||||
if (ud->model->arch == LLM_ARCH_QWEN3NEXT || ud->model->arch == LLM_ARCH_QWEN35 || ud->model->arch == LLM_ARCH_QWEN35MOE ||
|
||||
if (ud->model->arch == LLM_ARCH_QWEN3NEXT || ud->model->arch == LLM_ARCH_QWEN35 || ud->model->arch == LLM_ARCH_QWEN35MOE || ud->model->arch == LLM_ARCH_CLEF ||
|
||||
ud->model->arch == LLM_ARCH_QWEN4EXP) {
|
||||
|
||||
// fused full attention layers with Q gate tensors that need n_embd doubled:
|
||||
@@ -762,7 +764,7 @@ struct ggml_backend_meta_split_state llama_meta_device_get_split_state(const str
|
||||
GGML_ASSERT(segments.size() == 1);
|
||||
// some models have Q gate tensors, for those cases the granularity needs to be doubled:
|
||||
// TODO: deduplicate condition [TAG_SPLIT_QGATE_QWEN]
|
||||
if (ud->model->arch == LLM_ARCH_QWEN3NEXT || ud->model->arch == LLM_ARCH_QWEN35 || ud->model->arch == LLM_ARCH_QWEN35MOE ||
|
||||
if (ud->model->arch == LLM_ARCH_QWEN3NEXT || ud->model->arch == LLM_ARCH_QWEN35 || ud->model->arch == LLM_ARCH_QWEN35MOE || ud->model->arch == LLM_ARCH_CLEF ||
|
||||
ud->model->arch == LLM_ARCH_QWEN4EXP) {
|
||||
return {std::lcm(2*n_embd_q, blck_size_perf)};
|
||||
}
|
||||
@@ -795,7 +797,7 @@ struct ggml_backend_meta_split_state llama_meta_device_get_split_state(const str
|
||||
if (std::regex_match(tensor_name, pattern_qkv_weight) || std::regex_match(tensor_name, pattern_qkv_bias)) {
|
||||
// fused full attention layers need Q gate tensors handled like above:
|
||||
// TODO: deduplicate condition [TAG_SPLIT_QGATE_QWEN]
|
||||
if (ud->model->arch == LLM_ARCH_QWEN3NEXT || ud->model->arch == LLM_ARCH_QWEN35 || ud->model->arch == LLM_ARCH_QWEN35MOE ||
|
||||
if (ud->model->arch == LLM_ARCH_QWEN3NEXT || ud->model->arch == LLM_ARCH_QWEN35 || ud->model->arch == LLM_ARCH_QWEN35MOE || ud->model->arch == LLM_ARCH_CLEF ||
|
||||
ud->model->arch == LLM_ARCH_QWEN4EXP) {
|
||||
return {std::lcm(2*n_embd_q, blck_size_perf), granularity_kv};
|
||||
}
|
||||
@@ -1961,7 +1963,11 @@ bool llama_model_base::load_tensors(llama_model_loader & ml) {
|
||||
}
|
||||
|
||||
ggml_tensor * llama_model_base::create_tensor(llama_model_loader & ml, const LLM_TN_IMPL & tn, const std::initializer_list<int64_t> & ne, int flags) {
|
||||
const buft_list_t * buft_list_layer = tn.bid == -1 ? nullptr : pimpl->dev_layer.at(tn.bid).buft_list;
|
||||
const buft_list_t * buft_list_layer = nullptr;
|
||||
if (tn.bid != -1) {
|
||||
// blocks that are not model layers (e.g. the blocks of a head) go with the output
|
||||
buft_list_layer = (size_t) tn.bid < pimpl->dev_layer.size() ? pimpl->dev_layer.at(tn.bid).buft_list : pimpl->dev_output.buft_list;
|
||||
}
|
||||
return ml.create_tensor(
|
||||
hparams, &pimpl->cpu_buft_list, pimpl->dev_input.buft_list, pimpl->dev_output.buft_list, buft_list_layer,
|
||||
tn, ne, flags);
|
||||
@@ -2369,6 +2375,7 @@ llama_memory_i * llama_model::create_memory(const llama_memory_params & params,
|
||||
case LLM_ARCH_LLADA:
|
||||
case LLM_ARCH_LLADA_MOE:
|
||||
case LLM_ARCH_RND1:
|
||||
case LLM_ARCH_CLEF:
|
||||
{
|
||||
res = nullptr;
|
||||
} break;
|
||||
@@ -3200,6 +3207,7 @@ llama_rope_type llama_model_rope_type(const llama_model * model) {
|
||||
case LLM_ARCH_QWEN3VLMOE:
|
||||
case LLM_ARCH_QWEN35:
|
||||
case LLM_ARCH_QWEN35MOE:
|
||||
case LLM_ARCH_CLEF:
|
||||
case LLM_ARCH_QWEN4EXP:
|
||||
case LLM_ARCH_QWEN3TTS:
|
||||
return LLAMA_ROPE_TYPE_IMROPE;
|
||||
|
||||
@@ -0,0 +1,612 @@
|
||||
#include "models.h"
|
||||
|
||||
#include <cmath>
|
||||
|
||||
// torch.nn.LayerNorm default
|
||||
static const float CLEF_HEAD_NORM_EPS = 1e-5f;
|
||||
|
||||
void llama_model_clef::load_arch_hparams(llama_model_loader & ml) {
|
||||
llama_model_qwen35::load_arch_hparams(ml);
|
||||
|
||||
ml.get_key(LLM_KV_DECISION_ROUTING_BLOCK_COUNT, n_layer_routing);
|
||||
ml.get_key(LLM_KV_DECISION_BLOCK_COUNT, n_layer_joint);
|
||||
ml.get_key(LLM_KV_DECISION_HEAD_COUNT, n_head_decision);
|
||||
|
||||
if (n_head_decision == 0) {
|
||||
throw std::runtime_error("invalid number of heads in the decision head");
|
||||
}
|
||||
|
||||
hparams.f_norm_eps = CLEF_HEAD_NORM_EPS;
|
||||
|
||||
// the output is one score per token, see llama_batch_ext_set_decision_order()
|
||||
hparams.n_embd_out_impl = 1;
|
||||
}
|
||||
|
||||
void llama_model_clef::load_arch_tensors(llama_model_loader & ml) {
|
||||
llama_model_qwen35::load_arch_tensors(ml);
|
||||
|
||||
LLAMA_LOAD_LOCALS;
|
||||
|
||||
const auto * w_memory = ml.get_weight(tn(LLM_TENSOR_DECISION_PROJ_MEMORY, "weight").str().c_str());
|
||||
const auto * w_ffn = ml.get_weight(tn(LLM_TENSOR_DEC_FFN_UP, "weight", 0).str().c_str());
|
||||
if (w_memory == nullptr || w_ffn == nullptr) {
|
||||
throw std::runtime_error("the decision head is missing");
|
||||
}
|
||||
const int64_t n_embd_h = w_memory->tensor->ne[1];
|
||||
const int64_t n_ff_h = w_ffn->tensor->ne[1];
|
||||
|
||||
if (n_embd_h % n_head_decision != 0) {
|
||||
throw std::runtime_error("invalid width of the decision head");
|
||||
}
|
||||
|
||||
auto load_norm = [&](norm & n, llm_tensor type, int64_t size, int il = -1) {
|
||||
n.w = il < 0 ? create_tensor(tn(type, "weight"), {size}, 0) : create_tensor(tn(type, "weight", il), {size}, 0);
|
||||
n.b = il < 0 ? create_tensor(tn(type, "bias"), {size}, 0) : create_tensor(tn(type, "bias", il), {size}, 0);
|
||||
};
|
||||
|
||||
auto load_attn = [&](attn & a, llm_tensor q, llm_tensor k, llm_tensor v, llm_tensor o, int il) {
|
||||
a.wq = create_tensor(tn(q, "weight", il), {n_embd_h, n_embd_h}, 0);
|
||||
a.bq = create_tensor(tn(q, "bias", il), {n_embd_h}, 0);
|
||||
a.wk = create_tensor(tn(k, "weight", il), {n_embd_h, n_embd_h}, 0);
|
||||
a.bk = create_tensor(tn(k, "bias", il), {n_embd_h}, 0);
|
||||
a.wv = create_tensor(tn(v, "weight", il), {n_embd_h, n_embd_h}, 0);
|
||||
a.bv = create_tensor(tn(v, "bias", il), {n_embd_h}, 0);
|
||||
a.wo = create_tensor(tn(o, "weight", il), {n_embd_h, n_embd_h}, 0);
|
||||
a.bo = create_tensor(tn(o, "bias", il), {n_embd_h}, 0);
|
||||
};
|
||||
|
||||
head_layers.resize(n_layer_routing + n_layer_joint);
|
||||
for (int il = 0; il < (int) head_layers.size(); ++il) {
|
||||
auto & layer = head_layers[il];
|
||||
|
||||
if (il < (int) n_layer_routing) {
|
||||
load_norm(layer.cross_norm_kv, LLM_TENSOR_DEC_CROSS_ATTN_NORM_KV, n_embd_h, il);
|
||||
} else {
|
||||
load_norm(layer.self_norm, LLM_TENSOR_DEC_ATTN_NORM, n_embd_h, il);
|
||||
load_attn(layer.self_attn, LLM_TENSOR_DEC_ATTN_Q, LLM_TENSOR_DEC_ATTN_K, LLM_TENSOR_DEC_ATTN_V, LLM_TENSOR_DEC_ATTN_OUT, il);
|
||||
}
|
||||
|
||||
load_norm(layer.cross_norm, LLM_TENSOR_DEC_CROSS_ATTN_NORM, n_embd_h, il);
|
||||
load_attn(layer.cross_attn, LLM_TENSOR_DEC_CROSS_ATTN_Q, LLM_TENSOR_DEC_CROSS_ATTN_K, LLM_TENSOR_DEC_CROSS_ATTN_V, LLM_TENSOR_DEC_CROSS_ATTN_OUT, il);
|
||||
|
||||
load_norm(layer.ffn_norm, LLM_TENSOR_DEC_FFN_NORM, n_embd_h, il);
|
||||
layer.ffn_up = create_tensor(tn(LLM_TENSOR_DEC_FFN_UP, "weight", il), {n_embd_h, n_ff_h}, 0);
|
||||
layer.ffn_up_b = create_tensor(tn(LLM_TENSOR_DEC_FFN_UP, "bias", il), {n_ff_h}, 0);
|
||||
layer.ffn_down = create_tensor(tn(LLM_TENSOR_DEC_FFN_DOWN, "weight", il), {n_ff_h, n_embd_h}, 0);
|
||||
layer.ffn_down_b = create_tensor(tn(LLM_TENSOR_DEC_FFN_DOWN, "bias", il), {n_embd_h}, 0);
|
||||
}
|
||||
|
||||
load_norm(hidden_norm, LLM_TENSOR_DECISION_HIDDEN_NORM, n_embd);
|
||||
load_norm(option_summary_norm, LLM_TENSOR_DECISION_OPTION_SUMMARY_NORM, n_embd_h);
|
||||
load_norm(field_norm, LLM_TENSOR_DECISION_FIELD_NORM, n_embd_h);
|
||||
load_norm(option_norm, LLM_TENSOR_DECISION_OPTION_NORM, n_embd_h);
|
||||
|
||||
proj_memory = create_tensor(tn(LLM_TENSOR_DECISION_PROJ_MEMORY, "weight"), {n_embd, n_embd_h}, 0);
|
||||
proj_question = create_tensor(tn(LLM_TENSOR_DECISION_PROJ_QUESTION, "weight"), {n_embd, n_embd_h}, 0);
|
||||
proj_option_question = create_tensor(tn(LLM_TENSOR_DECISION_PROJ_OPTION_QUESTION, "weight"), {n_embd, n_embd_h}, 0);
|
||||
proj_global = create_tensor(tn(LLM_TENSOR_DECISION_PROJ_GLOBAL, "weight"), {n_embd, n_embd_h}, 0);
|
||||
proj_option_context = create_tensor(tn(LLM_TENSOR_DECISION_PROJ_OPTION_CONTEXT, "weight"), {n_embd, n_embd_h}, 0);
|
||||
proj_option_lexical = create_tensor(tn(LLM_TENSOR_DECISION_PROJ_OPTION_LEXICAL, "weight"), {n_embd, n_embd_h}, 0);
|
||||
|
||||
scales = create_tensor(tn(LLM_TENSOR_DECISION_SCALES), {3}, 0);
|
||||
type_embd = create_tensor(tn(LLM_TENSOR_TOKEN_TYPES, "weight"), {n_embd_h, 3}, 0);
|
||||
cls = create_tensor(tn(LLM_TENSOR_CLS, "weight"), {4 * n_embd_h, n_embd_h}, 0);
|
||||
cls_b = create_tensor(tn(LLM_TENSOR_CLS, "bias"), {n_embd_h}, 0);
|
||||
cls_out = create_tensor(tn(LLM_TENSOR_CLS_OUT, "weight"), {n_embd_h, 1}, 0);
|
||||
cls_out_b = create_tensor(tn(LLM_TENSOR_CLS_OUT, "bias"), {1}, 0);
|
||||
}
|
||||
|
||||
std::unique_ptr<llm_graph_context> llama_model_clef::build_arch_graph(const llm_graph_params & params) const {
|
||||
return std::make_unique<graph>(*this, params);
|
||||
}
|
||||
|
||||
// spans read by the head, [start, end) in ubatch token indices
|
||||
struct clef_spans {
|
||||
struct question {
|
||||
int32_t type; // noul, choice, score
|
||||
int32_t start;
|
||||
int32_t end;
|
||||
};
|
||||
struct option {
|
||||
int32_t question;
|
||||
int32_t start;
|
||||
int32_t end;
|
||||
};
|
||||
std::vector<question> questions;
|
||||
std::vector<option> options;
|
||||
bool valid = false; // at least one question, and each question has an option
|
||||
};
|
||||
|
||||
// see llama_batch_ext_set_decision_order()
|
||||
static clef_spans clef_get_spans(const llama_ubatch & ubatch) {
|
||||
const int32_t ORDER_QUESTION_FIRST = 1;
|
||||
const int32_t ORDER_QUESTION_LAST = 3;
|
||||
const int32_t ORDER_OPTION = 4;
|
||||
|
||||
clef_spans res;
|
||||
if (ubatch.decision_order == nullptr) {
|
||||
return res;
|
||||
}
|
||||
|
||||
const int32_t n_tokens = ubatch.n_tokens;
|
||||
std::vector<bool> has_option;
|
||||
for (int32_t i = 0; i < n_tokens;) {
|
||||
const int32_t order = ubatch.decision_order[i];
|
||||
int32_t end = i + 1;
|
||||
while (end < n_tokens && ubatch.decision_order[end] == order) {
|
||||
end++;
|
||||
}
|
||||
if (order >= ORDER_QUESTION_FIRST && order <= ORDER_QUESTION_LAST) {
|
||||
res.questions.push_back({ order - ORDER_QUESTION_FIRST, i, end });
|
||||
has_option.push_back(false);
|
||||
} else if (order == ORDER_OPTION) {
|
||||
if (res.questions.empty()) {
|
||||
return res;
|
||||
}
|
||||
res.options.push_back({ (int32_t) res.questions.size() - 1, i, end });
|
||||
has_option.back() = true;
|
||||
} else if (order != 0) {
|
||||
return res;
|
||||
}
|
||||
i = end;
|
||||
}
|
||||
|
||||
res.valid = !res.questions.empty() && std::find(has_option.begin(), has_option.end(), false) == has_option.end();
|
||||
return res;
|
||||
}
|
||||
|
||||
// pooling matrices and indices computed from the spans
|
||||
class llama_model_clef::input_decision : public llm_graph_input_i {
|
||||
public:
|
||||
input_decision(const llama_ubatch & ubatch) : n_tokens(ubatch.n_tokens) {
|
||||
const auto spans = clef_get_spans(ubatch);
|
||||
if (spans.valid) {
|
||||
n_questions = spans.questions.size();
|
||||
n_options = spans.options.size();
|
||||
}
|
||||
}
|
||||
|
||||
void set_input(const llama_ubatch * ubatch) override {
|
||||
if (tokens == nullptr) {
|
||||
return;
|
||||
}
|
||||
GGML_ASSERT(ubatch->token);
|
||||
ggml_backend_tensor_set(tokens, ubatch->token, 0, n_tokens * sizeof(llama_token));
|
||||
|
||||
const auto spans = clef_get_spans(*ubatch);
|
||||
GGML_ASSERT(spans.valid && spans.questions.size() == n_questions && spans.options.size() == n_options);
|
||||
|
||||
std::vector<float> pool_q_data(n_tokens * n_questions, 0.0f);
|
||||
std::vector<int32_t> types(n_questions);
|
||||
for (size_t i = 0; i < n_questions; i++) {
|
||||
const auto & q = spans.questions[i];
|
||||
for (int32_t j = q.start; j < q.end; j++) {
|
||||
pool_q_data[i * n_tokens + j] = 1.0f / (q.end - q.start);
|
||||
}
|
||||
types[i] = q.type;
|
||||
}
|
||||
|
||||
std::vector<float> pool_o_data(n_tokens * n_options, 0.0f);
|
||||
std::vector<int32_t> owner(n_options);
|
||||
std::vector<float> mask(n_options * n_questions, -INFINITY);
|
||||
for (size_t i = 0; i < n_options; i++) {
|
||||
const auto & o = spans.options[i];
|
||||
for (int32_t j = o.start; j < o.end; j++) {
|
||||
pool_o_data[i * n_tokens + j] = 1.0f / (o.end - o.start);
|
||||
}
|
||||
owner[i] = o.question;
|
||||
mask[o.question * n_options + i] = 0.0f;
|
||||
}
|
||||
|
||||
ggml_backend_tensor_set(pool_q, pool_q_data.data(), 0, ggml_nbytes(pool_q));
|
||||
ggml_backend_tensor_set(pool_o, pool_o_data.data(), 0, ggml_nbytes(pool_o));
|
||||
ggml_backend_tensor_set(question_type, types.data(), 0, ggml_nbytes(question_type));
|
||||
ggml_backend_tensor_set(option_question, owner.data(), 0, ggml_nbytes(option_question));
|
||||
ggml_backend_tensor_set(option_mask, mask.data(), 0, ggml_nbytes(option_mask));
|
||||
}
|
||||
|
||||
bool can_reuse(const llm_graph_params & params) override {
|
||||
// the values are computed again in set_input(), only the shapes must match
|
||||
const auto spans = clef_get_spans(params.ubatch);
|
||||
const size_t n_q = spans.valid ? spans.questions.size() : 0;
|
||||
const size_t n_o = spans.valid ? spans.options.size() : 0;
|
||||
return n_q == n_questions && n_o == n_options;
|
||||
}
|
||||
|
||||
ggml_tensor * tokens = nullptr; // I32 [n_tokens]
|
||||
ggml_tensor * pool_q = nullptr; // F32 [n_tokens, n_questions], mean over the span of the question
|
||||
ggml_tensor * pool_o = nullptr; // F32 [n_tokens, n_options], mean over the span of the option
|
||||
ggml_tensor * question_type = nullptr; // I32 [n_questions]
|
||||
ggml_tensor * option_question = nullptr; // I32 [n_options]
|
||||
ggml_tensor * option_mask = nullptr; // F32 [n_options, n_questions], 0 if the option belongs to the question, else -inf
|
||||
|
||||
const int64_t n_tokens;
|
||||
size_t n_questions = 0; // 0 if the batch has no valid spans
|
||||
size_t n_options = 0;
|
||||
};
|
||||
|
||||
// the backbone is copied from llama_model_qwen35::graph, without the memory module
|
||||
llama_model_clef::graph::graph(const llama_model & model_base, const llm_graph_params & params) :
|
||||
llm_build_delta_net_base(params), model(static_cast<const llama_model_clef &>(model_base)) {
|
||||
const int64_t n_embd_head = hparams.n_embd_head_v();
|
||||
|
||||
GGML_ASSERT(n_embd_head == hparams.n_embd_head_k());
|
||||
|
||||
int sections[4];
|
||||
std::copy(std::begin(hparams.rope_sections), std::begin(hparams.rope_sections) + 4, sections);
|
||||
|
||||
ggml_tensor * cur;
|
||||
ggml_tensor * inpL;
|
||||
|
||||
inpL = build_inp_embd(model.tok_embd);
|
||||
|
||||
ggml_tensor * inp_pos = build_inp_pos();
|
||||
|
||||
auto * inp_attn = build_attn_inp_causal();
|
||||
|
||||
auto inp_decision_ptr = std::make_unique<input_decision>(ubatch);
|
||||
auto * inp_decision = inp_decision_ptr.get();
|
||||
res->add_input(std::move(inp_decision_ptr));
|
||||
|
||||
for (int il = 0; il < n_layer; ++il) {
|
||||
ggml_tensor * inpSA = inpL;
|
||||
|
||||
cur = build_norm(inpL, model.layers[il].attn_norm, nullptr, LLM_NORM_RMS, il);
|
||||
cb(cur, "attn_norm", il);
|
||||
|
||||
if (hparams.is_recr(il)) {
|
||||
cur = build_layer_attn_linear(cur, il);
|
||||
} else {
|
||||
cur = build_layer_attn(inp_attn, cur, inp_pos, sections, il);
|
||||
}
|
||||
|
||||
cur = ggml_add(ctx0, cur, inpSA);
|
||||
cb(cur, "attn_residual", il);
|
||||
|
||||
ggml_tensor * ffn_residual = cur;
|
||||
|
||||
cur = build_norm(cur, model.layers[il].attn_post_norm, nullptr, LLM_NORM_RMS, il);
|
||||
cb(cur, "attn_post_norm", il);
|
||||
|
||||
cur = build_ffn(cur,
|
||||
model.layers[il].ffn_up, NULL, model.layers[il].ffn_up_s,
|
||||
model.layers[il].ffn_gate, NULL, model.layers[il].ffn_gate_s,
|
||||
model.layers[il].ffn_down, NULL, model.layers[il].ffn_down_s,
|
||||
NULL,
|
||||
LLM_FFN_SILU, LLM_FFN_PAR, il);
|
||||
cb(cur, "ffn_out", il);
|
||||
|
||||
cur = ggml_add(ctx0, cur, ffn_residual);
|
||||
cb(cur, "l_out", il);
|
||||
|
||||
inpL = cur;
|
||||
}
|
||||
|
||||
cur = build_norm(inpL, model.output_norm, nullptr, LLM_NORM_RMS, -1);
|
||||
cb(cur, "result_norm", -1);
|
||||
|
||||
// TODO: support multiple sequences
|
||||
const bool is_single_seq = ubatch.n_seqs_unq == 1;
|
||||
|
||||
if (inp_decision->n_questions > 0 && is_single_seq) {
|
||||
cur = build_head(cur, inp_decision);
|
||||
// row i of the output is the score of option i
|
||||
cur = ggml_pad(ctx0, cur, 0, n_tokens - cur->ne[1], 0, 0);
|
||||
} else {
|
||||
if (ubatch.decision_order != nullptr) {
|
||||
LLAMA_LOG_ERROR("%s: %s, the head is not evaluated\n", __func__, is_single_seq ? "invalid decision order" : "more than one sequence in the batch");
|
||||
}
|
||||
cur = ggml_scale(ctx0, ggml_cont(ctx0, ggml_view_2d(ctx0, cur, 1, n_tokens, cur->nb[1], 0)), 0.0f);
|
||||
}
|
||||
cb(cur, "result_decision", -1);
|
||||
|
||||
res->t_embd = cur;
|
||||
|
||||
ggml_build_forward_expand(gf, cur);
|
||||
}
|
||||
|
||||
// same as build_attn_inp_no_cache(), but the mask is causal even if the batch is processed by the encoder path
|
||||
llm_graph_input_attn_no_cache * llama_model_clef::graph::build_attn_inp_causal() {
|
||||
llama_cparams cparams_causal = cparams;
|
||||
cparams_causal.causal_attn = true;
|
||||
|
||||
auto inp = std::make_unique<llm_graph_input_attn_no_cache>(hparams, cparams_causal);
|
||||
|
||||
const auto type_mask = cparams.flash_attn ? GGML_TYPE_F16 : GGML_TYPE_F32;
|
||||
|
||||
inp->self_kq_mask = ggml_new_tensor_4d(ctx0, type_mask, n_tokens, n_tokens, 1, 1);
|
||||
ggml_set_input(inp->self_kq_mask);
|
||||
cb(inp->self_kq_mask, "self_kq_mask", -1);
|
||||
|
||||
inp->self_kq_mask_cnv = inp->self_kq_mask;
|
||||
|
||||
return (llm_graph_input_attn_no_cache *) res->add_input(std::move(inp));
|
||||
}
|
||||
|
||||
ggml_tensor * llama_model_clef::graph::build_layer_attn(
|
||||
llm_graph_input_attn_no_cache * inp,
|
||||
ggml_tensor * cur,
|
||||
ggml_tensor * inp_pos,
|
||||
int * sections,
|
||||
int il) {
|
||||
const int64_t n_embd_head = hparams.n_embd_head_v();
|
||||
|
||||
// the Q projection outputs query + gate
|
||||
auto [Qcur_full, Kcur, Vcur] = build_qkv(model.layers[il], cur,
|
||||
n_embd_head * 2, n_head,
|
||||
n_embd_head, n_head_kv,
|
||||
n_embd_head, n_head_kv,
|
||||
il, false);
|
||||
|
||||
ggml_tensor * Qcur = ggml_view_3d(ctx0, Qcur_full, n_embd_head, n_head, n_tokens,
|
||||
ggml_element_size(Qcur_full) * n_embd_head * 2,
|
||||
ggml_element_size(Qcur_full) * n_embd_head * 2 * n_head, 0);
|
||||
|
||||
Qcur = build_norm(Qcur, model.layers[il].attn_q_norm, nullptr, LLM_NORM_RMS, il);
|
||||
cb(Qcur, "Qcur_normed", il);
|
||||
|
||||
Kcur = ggml_reshape_3d(ctx0, Kcur, n_embd_head, n_head_kv, n_tokens);
|
||||
Kcur = build_norm(Kcur, model.layers[il].attn_k_norm, nullptr, LLM_NORM_RMS, il);
|
||||
cb(Kcur, "Kcur_normed", il);
|
||||
|
||||
ggml_tensor * gate = ggml_view_3d(ctx0, Qcur_full, n_embd_head, n_head, n_tokens,
|
||||
ggml_element_size(Qcur_full) * n_embd_head * 2,
|
||||
ggml_element_size(Qcur_full) * n_embd_head * 2 * n_head,
|
||||
ggml_element_size(Qcur_full) * n_embd_head);
|
||||
gate = ggml_cont_2d(ctx0, gate, n_embd_head * n_head, n_tokens);
|
||||
|
||||
Vcur = ggml_reshape_3d(ctx0, Vcur, n_embd_head, n_head_kv, n_tokens);
|
||||
|
||||
Qcur = ggml_rope_multi(
|
||||
ctx0, Qcur, inp_pos, nullptr,
|
||||
n_rot, sections, rope_type, n_ctx_orig, freq_base, freq_scale,
|
||||
ext_factor, attn_factor, beta_fast, beta_slow
|
||||
);
|
||||
|
||||
Kcur = ggml_rope_multi(
|
||||
ctx0, Kcur, inp_pos, nullptr,
|
||||
n_rot, sections, rope_type, n_ctx_orig, freq_base, freq_scale,
|
||||
ext_factor, attn_factor, beta_fast, beta_slow
|
||||
);
|
||||
|
||||
const float kq_scale = hparams.f_attention_scale == 0.0f ? 1.0f / sqrtf(float(n_embd_head)) : hparams.f_attention_scale;
|
||||
|
||||
cur = build_attn(inp,
|
||||
nullptr, nullptr, nullptr,
|
||||
Qcur, Kcur, Vcur, nullptr, nullptr, nullptr, kq_scale, il);
|
||||
cb(cur, "attn_pregate", il);
|
||||
|
||||
cur = ggml_mul(ctx0, cur, ggml_sigmoid(ctx0, gate));
|
||||
|
||||
cur = build_lora_mm(model.layers[il].wo, cur, model.layers[il].wo_s);
|
||||
cb(cur, "attn_output", il);
|
||||
|
||||
return cur;
|
||||
}
|
||||
|
||||
// gated delta net over the whole batch, the conv and recurrent states start from zero and are not kept
|
||||
ggml_tensor * llama_model_clef::graph::build_layer_attn_linear(
|
||||
ggml_tensor * cur,
|
||||
int il) {
|
||||
const auto & layer = model.layers[il];
|
||||
|
||||
const int64_t d_inner = hparams.ssm_d_inner;
|
||||
const int64_t head_k_dim = hparams.ssm_d_state;
|
||||
const int64_t num_k_heads = hparams.ssm_n_group;
|
||||
const int64_t num_v_heads = hparams.ssm_dt_rank;
|
||||
const int64_t head_v_dim = d_inner / num_v_heads;
|
||||
const int64_t n_seqs = 1;
|
||||
|
||||
ggml_tensor * qkv_mixed = build_lora_mm(layer.wqkv, cur, layer.wqkv_s);
|
||||
qkv_mixed = ggml_reshape_3d(ctx0, qkv_mixed, qkv_mixed->ne[0], n_tokens, n_seqs);
|
||||
|
||||
ggml_tensor * z = build_lora_mm(layer.wqkv_gate, cur, layer.wqkv_gate_s);
|
||||
|
||||
ggml_tensor * beta = build_lora_mm(layer.ssm_beta, cur, layer.ssm_beta_s);
|
||||
beta = ggml_reshape_4d(ctx0, beta, 1, num_v_heads, n_tokens, n_seqs);
|
||||
beta = ggml_sigmoid(ctx0, beta);
|
||||
|
||||
ggml_tensor * alpha = build_lora_mm(layer.ssm_alpha, cur, layer.ssm_alpha_s);
|
||||
alpha = ggml_reshape_3d(ctx0, alpha, num_v_heads, n_tokens, n_seqs);
|
||||
|
||||
ggml_tensor * gate = ggml_softplus(ctx0, ggml_add(ctx0, alpha, layer.ssm_dt));
|
||||
gate = ggml_mul(ctx0, gate, layer.ssm_a);
|
||||
gate = ggml_reshape_4d(ctx0, gate, 1, num_v_heads, n_tokens, n_seqs);
|
||||
|
||||
const int64_t conv_kernel_size = layer.ssm_conv1d->ne[0];
|
||||
const int64_t conv_channels = d_inner + 2 * num_k_heads * head_k_dim;
|
||||
|
||||
ggml_tensor * conv_states = ggml_fill(ctx0, ggml_new_tensor_3d(ctx0, GGML_TYPE_F32, conv_kernel_size - 1, conv_channels, n_seqs), 0.0f);
|
||||
ggml_tensor * conv_input = ggml_concat(ctx0, conv_states, ggml_transpose(ctx0, qkv_mixed), 0);
|
||||
|
||||
ggml_tensor * conv_out = ggml_silu(ctx0, ggml_ssm_conv(ctx0, conv_input, layer.ssm_conv1d));
|
||||
cb(conv_out, "conv_output_silu", il);
|
||||
|
||||
const int64_t qkv_dim = head_k_dim * num_k_heads * 2 + head_v_dim * num_v_heads;
|
||||
const int64_t nb1_qkv = ggml_row_size(conv_out->type, qkv_dim);
|
||||
|
||||
ggml_tensor * q_conv = ggml_view_4d(ctx0, conv_out, head_k_dim, num_k_heads, n_tokens, n_seqs,
|
||||
ggml_row_size(conv_out->type, head_k_dim), nb1_qkv, nb1_qkv * n_tokens,
|
||||
0);
|
||||
ggml_tensor * k_conv = ggml_view_4d(ctx0, conv_out, head_k_dim, num_k_heads, n_tokens, n_seqs,
|
||||
ggml_row_size(conv_out->type, head_k_dim), nb1_qkv, nb1_qkv * n_tokens,
|
||||
head_k_dim * num_k_heads * ggml_element_size(conv_out));
|
||||
ggml_tensor * v_conv = ggml_view_4d(ctx0, conv_out, head_v_dim, num_v_heads, n_tokens, n_seqs,
|
||||
ggml_row_size(conv_out->type, head_v_dim), nb1_qkv, nb1_qkv * n_tokens,
|
||||
ggml_row_size(conv_out->type, 2 * head_k_dim * num_k_heads));
|
||||
|
||||
q_conv = build_gdn_l2_norm(ctx0, q_conv, hparams.f_norm_rms_eps);
|
||||
k_conv = build_gdn_l2_norm(ctx0, k_conv, hparams.f_norm_rms_eps);
|
||||
|
||||
// note: need explicit repeat only if we are not using the fused GDN
|
||||
if (num_k_heads != num_v_heads && (!cparams.fused_gdn_ar || !cparams.fused_gdn_ch)) {
|
||||
GGML_ASSERT(num_v_heads % num_k_heads == 0);
|
||||
q_conv = ggml_repeat_4d(ctx0, q_conv, head_k_dim, num_v_heads, n_tokens, n_seqs);
|
||||
k_conv = ggml_repeat_4d(ctx0, k_conv, head_k_dim, num_v_heads, n_tokens, n_seqs);
|
||||
}
|
||||
|
||||
ggml_tensor * state = ggml_fill(ctx0, ggml_new_tensor_4d(ctx0, GGML_TYPE_F32, head_v_dim, head_v_dim, num_v_heads, n_seqs), 0.0f);
|
||||
|
||||
ggml_tensor * output = build_delta_net(q_conv, k_conv, v_conv, gate, beta, state, il).first;
|
||||
cb(output, "attn_output", il);
|
||||
|
||||
// gated normalization
|
||||
ggml_tensor * z_4d = ggml_reshape_4d(ctx0, z, head_v_dim, num_v_heads, n_tokens, n_seqs);
|
||||
output = build_norm(output, layer.ssm_norm, nullptr, LLM_NORM_RMS, il);
|
||||
output = ggml_mul(ctx0, output, ggml_silu(ctx0, z_4d));
|
||||
|
||||
output = ggml_reshape_3d(ctx0, output, head_v_dim * num_v_heads, n_tokens, n_seqs);
|
||||
|
||||
cur = build_lora_mm(layer.ssm_out, output, layer.ssm_out_s);
|
||||
cb(cur, "linear_attn_out", il);
|
||||
|
||||
return ggml_reshape_2d(ctx0, cur, n_embd, n_tokens);
|
||||
}
|
||||
|
||||
ggml_tensor * llama_model_clef::graph::build_head_norm(ggml_tensor * cur, const norm & n) {
|
||||
return build_norm(cur, n.w, n.b, LLM_NORM, -1);
|
||||
}
|
||||
|
||||
// q: [n_embd_h, n_q], kv: [n_embd_h, n_kv], no mask
|
||||
ggml_tensor * llama_model_clef::graph::build_head_attn(ggml_tensor * q, ggml_tensor * kv, const attn & a) {
|
||||
const int64_t n_embd_h = q->ne[0];
|
||||
const int64_t n_head_h = model.n_head_decision;
|
||||
const int64_t d_head = n_embd_h / n_head_h;
|
||||
const int64_t n_q = q->ne[1];
|
||||
const int64_t n_kv = kv->ne[1];
|
||||
|
||||
ggml_tensor * Qcur = ggml_add(ctx0, ggml_mul_mat(ctx0, a.wq, q), a.bq);
|
||||
ggml_tensor * Kcur = ggml_add(ctx0, ggml_mul_mat(ctx0, a.wk, kv), a.bk);
|
||||
ggml_tensor * Vcur = ggml_add(ctx0, ggml_mul_mat(ctx0, a.wv, kv), a.bv);
|
||||
|
||||
Qcur = ggml_permute(ctx0, ggml_reshape_3d(ctx0, Qcur, d_head, n_head_h, n_q), 0, 2, 1, 3); // [d_head, n_q, n_head]
|
||||
Kcur = ggml_permute(ctx0, ggml_reshape_3d(ctx0, Kcur, d_head, n_head_h, n_kv), 0, 2, 1, 3); // [d_head, n_kv, n_head]
|
||||
Vcur = ggml_permute(ctx0, ggml_reshape_3d(ctx0, Vcur, d_head, n_head_h, n_kv), 1, 2, 0, 3); // [n_kv, d_head, n_head]
|
||||
Vcur = ggml_cont(ctx0, Vcur);
|
||||
|
||||
ggml_tensor * kq = ggml_mul_mat(ctx0, Kcur, Qcur); // [n_kv, n_q, n_head]
|
||||
kq = ggml_soft_max_ext(ctx0, kq, nullptr, 1.0f / sqrtf(float(d_head)), 0.0f);
|
||||
|
||||
ggml_tensor * cur = ggml_mul_mat(ctx0, Vcur, kq); // [d_head, n_q, n_head]
|
||||
cur = ggml_cont_2d(ctx0, ggml_permute(ctx0, cur, 0, 2, 1, 3), n_embd_h, n_q);
|
||||
|
||||
return ggml_add(ctx0, ggml_mul_mat(ctx0, a.wo, cur), a.bo);
|
||||
}
|
||||
|
||||
ggml_tensor * llama_model_clef::graph::build_head_ffn(ggml_tensor * cur, const head_layer & layer) {
|
||||
cur = build_head_norm(cur, layer.ffn_norm);
|
||||
cur = ggml_add(ctx0, ggml_mul_mat(ctx0, layer.ffn_up, cur), layer.ffn_up_b);
|
||||
cur = ggml_gelu_erf(ctx0, cur);
|
||||
return ggml_add(ctx0, ggml_mul_mat(ctx0, layer.ffn_down, cur), layer.ffn_down_b);
|
||||
}
|
||||
|
||||
// ref: JointSchemaHead in joint_schema_model.py of the model repo
|
||||
ggml_tensor * llama_model_clef::graph::build_head(ggml_tensor * hidden, input_decision * inp) {
|
||||
const int64_t n_questions = inp->n_questions;
|
||||
const int64_t n_options = inp->n_options;
|
||||
|
||||
inp->tokens = ggml_new_tensor_1d(ctx0, GGML_TYPE_I32, n_tokens);
|
||||
inp->pool_q = ggml_new_tensor_2d(ctx0, GGML_TYPE_F32, n_tokens, n_questions);
|
||||
inp->pool_o = ggml_new_tensor_2d(ctx0, GGML_TYPE_F32, n_tokens, n_options);
|
||||
inp->question_type = ggml_new_tensor_1d(ctx0, GGML_TYPE_I32, n_questions);
|
||||
inp->option_question = ggml_new_tensor_1d(ctx0, GGML_TYPE_I32, n_options);
|
||||
inp->option_mask = ggml_new_tensor_2d(ctx0, GGML_TYPE_F32, n_options, n_questions);
|
||||
for (ggml_tensor * t : { inp->tokens, inp->pool_q, inp->pool_o, inp->question_type, inp->option_question, inp->option_mask }) {
|
||||
ggml_set_input(t);
|
||||
}
|
||||
|
||||
hidden = build_head_norm(hidden, model.hidden_norm);
|
||||
|
||||
ggml_tensor * memory = ggml_mul_mat(ctx0, model.proj_memory, hidden); // [n_embd_h, n_tokens]
|
||||
|
||||
const int64_t n_embd_h = memory->ne[0];
|
||||
|
||||
// hidden state of the last token
|
||||
ggml_tensor * global = ggml_view_2d(ctx0, hidden, n_embd, 1, hidden->nb[1], (n_tokens - 1) * hidden->nb[1]);
|
||||
|
||||
// mean of the hidden states over each span
|
||||
ggml_tensor * hidden_t = ggml_cont(ctx0, ggml_transpose(ctx0, hidden));
|
||||
ggml_tensor * question = ggml_mul_mat(ctx0, hidden_t, inp->pool_q); // [n_embd, n_questions]
|
||||
ggml_tensor * opt_ctx = ggml_mul_mat(ctx0, hidden_t, inp->pool_o); // [n_embd, n_options]
|
||||
|
||||
// mean of the output embeddings of the tokens of each option
|
||||
ggml_tensor * lexical = ggml_get_rows(ctx0, model.output, inp->tokens);
|
||||
lexical = ggml_mul_mat(ctx0, ggml_cont(ctx0, ggml_transpose(ctx0, lexical)), inp->pool_o); // [n_embd, n_options]
|
||||
|
||||
// one query per option
|
||||
ggml_tensor * options = ggml_mul_mat(ctx0, model.proj_option_context, opt_ctx);
|
||||
options = ggml_add(ctx0, options, ggml_mul_mat(ctx0, model.proj_option_lexical, lexical));
|
||||
options = ggml_add(ctx0, options, ggml_get_rows(ctx0, ggml_mul_mat(ctx0, model.proj_option_question, question), inp->option_question));
|
||||
|
||||
// the options read the prompt
|
||||
for (uint32_t il = 0; il < model.n_layer_routing; ++il) {
|
||||
const auto & layer = model.head_layers[il];
|
||||
|
||||
ggml_tensor * cur = build_head_attn(
|
||||
build_head_norm(options, layer.cross_norm),
|
||||
build_head_norm(memory, layer.cross_norm_kv),
|
||||
layer.cross_attn);
|
||||
options = ggml_add(ctx0, options, cur);
|
||||
options = ggml_add(ctx0, options, build_head_ffn(options, layer));
|
||||
}
|
||||
cb(options, "decision_options", -1);
|
||||
|
||||
// one vector per question: its text, a summary of its options, the end of the prompt and its type
|
||||
ggml_tensor * fields = ggml_mul_mat(ctx0, model.proj_question, question); // [n_embd_h, n_questions]
|
||||
{
|
||||
// each question weights its own options
|
||||
ggml_tensor * weights = ggml_mul_mat(ctx0, options, fields); // [n_options, n_questions]
|
||||
weights = ggml_scale(ctx0, weights, 1.0f / sqrtf(float(n_embd_h)));
|
||||
weights = ggml_soft_max(ctx0, ggml_add(ctx0, weights, inp->option_mask));
|
||||
|
||||
ggml_tensor * summary = ggml_mul_mat(ctx0, ggml_cont(ctx0, ggml_transpose(ctx0, options)), weights);
|
||||
|
||||
fields = ggml_add(ctx0, fields, build_head_norm(summary, model.option_summary_norm));
|
||||
fields = ggml_add(ctx0, fields, ggml_mul_mat(ctx0, model.proj_global, global));
|
||||
fields = ggml_add(ctx0, fields, ggml_get_rows(ctx0, model.type_embd, inp->question_type));
|
||||
}
|
||||
|
||||
// the questions read each other and the prompt
|
||||
for (uint32_t il = model.n_layer_routing; il < model.head_layers.size(); ++il) {
|
||||
const auto & layer = model.head_layers[il];
|
||||
|
||||
ggml_tensor * cur = build_head_norm(fields, layer.self_norm);
|
||||
fields = ggml_add(ctx0, fields, build_head_attn(cur, cur, layer.self_attn));
|
||||
|
||||
cur = build_head_norm(fields, layer.cross_norm);
|
||||
fields = ggml_add(ctx0, fields, build_head_attn(cur, memory, layer.cross_attn));
|
||||
|
||||
fields = ggml_add(ctx0, fields, build_head_ffn(fields, layer));
|
||||
}
|
||||
fields = build_head_norm(fields, model.field_norm);
|
||||
cb(fields, "decision_fields", -1);
|
||||
|
||||
const float eps = 1e-12f;
|
||||
|
||||
// prior: the output embeddings of the option against the question and the end of the prompt
|
||||
ggml_tensor * anchor = ggml_l2_norm(ctx0, ggml_add(ctx0, question, global), eps);
|
||||
anchor = ggml_get_rows(ctx0, anchor, inp->option_question);
|
||||
ggml_tensor * prior = ggml_sum_rows(ctx0, ggml_mul(ctx0, ggml_l2_norm(ctx0, lexical, eps), anchor)); // [1, n_options]
|
||||
|
||||
// joint: each option against the vector of its question
|
||||
options = build_head_norm(options, model.option_norm);
|
||||
ggml_tensor * field = ggml_get_rows(ctx0, fields, inp->option_question); // [n_embd_h, n_options]
|
||||
|
||||
ggml_tensor * cosine = ggml_sum_rows(ctx0, ggml_mul(ctx0, ggml_l2_norm(ctx0, field, eps), ggml_l2_norm(ctx0, options, eps)));
|
||||
|
||||
ggml_tensor * features = ggml_concat(ctx0, field, options, 0);
|
||||
features = ggml_concat(ctx0, features, ggml_mul(ctx0, field, options), 0);
|
||||
features = ggml_concat(ctx0, features, ggml_abs(ctx0, ggml_sub(ctx0, field, options)), 0);
|
||||
|
||||
ggml_tensor * residual = ggml_add(ctx0, ggml_mul_mat(ctx0, model.cls, features), model.cls_b);
|
||||
residual = ggml_gelu_erf(ctx0, residual);
|
||||
residual = ggml_add(ctx0, ggml_mul_mat(ctx0, model.cls_out, residual), model.cls_out_b); // [1, n_options]
|
||||
|
||||
auto scale = [&](int i) {
|
||||
return ggml_view_1d(ctx0, model.scales, 1, i * ggml_element_size(model.scales));
|
||||
};
|
||||
|
||||
ggml_tensor * joint = ggml_add(ctx0, ggml_mul(ctx0, cosine, scale(1)), residual);
|
||||
|
||||
return ggml_add(ctx0, ggml_mul(ctx0, prior, scale(0)), ggml_mul(ctx0, joint, scale(2)));
|
||||
}
|
||||
@@ -2391,6 +2391,95 @@ struct llama_model_qwen35 : public llama_model_base {
|
||||
};
|
||||
|
||||
|
||||
// Qwen3.5 backbone evaluated without memory, with a joint decision head on top of its hidden states
|
||||
struct llama_model_clef : public llama_model_qwen35 {
|
||||
llama_model_clef(const struct llama_model_params & params) : llama_model_qwen35(params) {}
|
||||
void load_arch_hparams(llama_model_loader & ml) override;
|
||||
void load_arch_tensors(llama_model_loader & ml) override;
|
||||
|
||||
struct attn {
|
||||
ggml_tensor * wq = nullptr;
|
||||
ggml_tensor * bq = nullptr;
|
||||
ggml_tensor * wk = nullptr;
|
||||
ggml_tensor * bk = nullptr;
|
||||
ggml_tensor * wv = nullptr;
|
||||
ggml_tensor * bv = nullptr;
|
||||
ggml_tensor * wo = nullptr;
|
||||
ggml_tensor * bo = nullptr;
|
||||
};
|
||||
|
||||
struct norm {
|
||||
ggml_tensor * w = nullptr;
|
||||
ggml_tensor * b = nullptr;
|
||||
};
|
||||
|
||||
// the first n_layer_routing blocks have no self attention
|
||||
struct head_layer {
|
||||
norm self_norm;
|
||||
attn self_attn;
|
||||
norm cross_norm;
|
||||
norm cross_norm_kv; // routing blocks only
|
||||
attn cross_attn;
|
||||
norm ffn_norm;
|
||||
ggml_tensor * ffn_up = nullptr;
|
||||
ggml_tensor * ffn_up_b = nullptr;
|
||||
ggml_tensor * ffn_down = nullptr;
|
||||
ggml_tensor * ffn_down_b = nullptr;
|
||||
};
|
||||
|
||||
uint32_t n_layer_routing = 0;
|
||||
uint32_t n_layer_joint = 0;
|
||||
uint32_t n_head_decision = 0;
|
||||
|
||||
std::vector<head_layer> head_layers; // routing blocks, then joint blocks
|
||||
|
||||
norm hidden_norm;
|
||||
norm option_summary_norm;
|
||||
norm field_norm;
|
||||
norm option_norm;
|
||||
|
||||
ggml_tensor * proj_memory = nullptr;
|
||||
ggml_tensor * proj_question = nullptr;
|
||||
ggml_tensor * proj_option_question = nullptr;
|
||||
ggml_tensor * proj_global = nullptr;
|
||||
ggml_tensor * proj_option_context = nullptr;
|
||||
ggml_tensor * proj_option_lexical = nullptr;
|
||||
ggml_tensor * scales = nullptr; // prior scale, joint scale, residual gate
|
||||
|
||||
class input_decision;
|
||||
|
||||
struct graph : public llm_build_delta_net_base {
|
||||
graph(const llama_model & model, const llm_graph_params & params);
|
||||
private:
|
||||
llm_graph_input_attn_no_cache * build_attn_inp_causal();
|
||||
|
||||
ggml_tensor * build_layer_attn(
|
||||
llm_graph_input_attn_no_cache * inp_attn,
|
||||
ggml_tensor * cur,
|
||||
ggml_tensor * inp_pos,
|
||||
int * sections,
|
||||
int il);
|
||||
|
||||
ggml_tensor * build_layer_attn_linear(
|
||||
ggml_tensor * cur,
|
||||
int il);
|
||||
|
||||
// hidden: [n_embd, n_tokens], returns one score per option: [1, n_options]
|
||||
ggml_tensor * build_head(
|
||||
ggml_tensor * hidden,
|
||||
input_decision * inp);
|
||||
|
||||
ggml_tensor * build_head_norm(ggml_tensor * cur, const norm & n);
|
||||
ggml_tensor * build_head_attn(ggml_tensor * q, ggml_tensor * kv, const attn & a);
|
||||
ggml_tensor * build_head_ffn (ggml_tensor * cur, const head_layer & layer);
|
||||
|
||||
const llama_model_clef & model;
|
||||
};
|
||||
|
||||
std::unique_ptr<llm_graph_context> build_arch_graph(const llm_graph_params & params) const override;
|
||||
};
|
||||
|
||||
|
||||
struct llama_model_qwen4exp : public llama_model_base {
|
||||
llama_model_qwen4exp(const struct llama_model_params & params) : llama_model_base(params) {}
|
||||
|
||||
|
||||
@@ -685,6 +685,9 @@ static bool arch_supported(const llm_arch arch) {
|
||||
if (arch == LLM_ARCH_PLM) {
|
||||
return false; // TODO tensor shapes
|
||||
}
|
||||
if (arch == LLM_ARCH_CLEF) {
|
||||
return false; // TODO decision head tensors
|
||||
}
|
||||
if (arch == LLM_ARCH_DEEPSEEK2OCR) {
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -1688,13 +1688,15 @@ Follows the [TypeSafe API](https://docs.typesafe.ai/api), streaming is not suppo
|
||||
- `score`: An array of 2 to 10 level descriptions, lowest level first.
|
||||
- `noul`: Optional. An object with the descriptions of `true` and `false`.
|
||||
|
||||
The questions of a request are answered independently, an answer does not depend on the other questions.
|
||||
The questions of a request are answered independently, an answer does not depend on the other questions. The exception is clef: it reads all the questions in one prompt and decides them jointly.
|
||||
|
||||
The number of options of a `choice` question is limited by the model, for example: 52 for openjev, 255 for laya. For laya, long questions and options are truncated to the token budget the model was trained with.
|
||||
The number of options of a `choice` question is limited by the model, for example: 52 for openjev, 255 for laya and clef. For laya, long questions and options are truncated to the token budget the model was trained with.
|
||||
|
||||
For laya and clef, the whole prompt is evaluated in one batch: it must fit in `--ubatch-size`. A server that runs clef only serves this endpoint, text generation is not available.
|
||||
|
||||
*Image input:*
|
||||
|
||||
Image input needs a model that supports it (for example: openjev) and its multimodal projector, see `--mmproj`.
|
||||
Image input needs a model that supports it (for example: openjev) and its multimodal projector, see `--mmproj`. Image input is not supported yet for clef.
|
||||
|
||||
Images can be given in two ways, and both can be used in the same request:
|
||||
|
||||
|
||||
@@ -118,6 +118,7 @@ struct server_batch {
|
||||
llama_pos pos;
|
||||
bool output;
|
||||
bool is_prompt; // for stats tracking
|
||||
int32_t decision_order = 0;
|
||||
};
|
||||
std::vector<token> tokens;
|
||||
int32_t n_tokens_alloc = 0;
|
||||
@@ -177,6 +178,11 @@ struct server_batch {
|
||||
tokens[idx].output = output;
|
||||
}
|
||||
|
||||
void set_decision_order(int32_t idx, int32_t order) {
|
||||
GGML_ASSERT(idx >= 0 && idx < (int32_t)tokens.size());
|
||||
tokens[idx].decision_order = order;
|
||||
}
|
||||
|
||||
// render the sub-batch [off, off + n_tokens) into view, index i in view is index off + i here
|
||||
void render(int32_t off, int32_t n_tokens) {
|
||||
GGML_ASSERT(off >= 0 && off < size());
|
||||
@@ -192,6 +198,7 @@ struct server_batch {
|
||||
} else {
|
||||
view.add(t.token, t.pos, t.id_slot, t.output);
|
||||
}
|
||||
view.tokens.back().decision_order = t.decision_order;
|
||||
}
|
||||
}
|
||||
};
|
||||
@@ -420,6 +427,11 @@ struct server_slot {
|
||||
bool can_batch_with(server_slot & other_slot) const {
|
||||
GGML_ASSERT(task);
|
||||
|
||||
// a joint decision head reads the whole batch
|
||||
if (!task->decision.order.empty() || !other_slot.task->decision.order.empty()) {
|
||||
return false;
|
||||
}
|
||||
|
||||
return task->type == other_slot.task->type
|
||||
&& inp_embd.size() == other_slot.inp_embd.size()
|
||||
&& are_lora_equal(lora, other_slot.lora);
|
||||
@@ -3634,6 +3646,9 @@ private:
|
||||
/* pos = */ slot.prompt.tokens.pos_next(),
|
||||
/* output = */ slot.need_embd(),
|
||||
/* is_prompt = */ true);
|
||||
if (!slot.task->decision.order.empty()) {
|
||||
batch.set_decision_order(batch.size() - 1, slot.task->decision.order[slot.prompt.n_tokens()]);
|
||||
}
|
||||
slot.prompt.tokens.push_back(cur_tok);
|
||||
|
||||
// break at the last user message, or at user messages at least min step past the last checkpoint
|
||||
@@ -5339,12 +5354,21 @@ void server_routes::init_routes() {
|
||||
return res;
|
||||
}
|
||||
|
||||
// one task per question
|
||||
// one task per question, or one task for all of them
|
||||
auto & rd = res->rd;
|
||||
{
|
||||
std::vector<server_task> tasks;
|
||||
tasks.reserve(questions.size());
|
||||
if (decision.is_joint()) {
|
||||
server_task task = server_task(SERVER_TASK_TYPE_DECISION);
|
||||
task.id = rd.get_new_id();
|
||||
decision.fill_task_joint(state, questions, task);
|
||||
tasks.push_back(std::move(task));
|
||||
}
|
||||
for (const auto & question : questions) {
|
||||
if (decision.is_joint()) {
|
||||
break;
|
||||
}
|
||||
server_task task = server_task(SERVER_TASK_TYPE_DECISION);
|
||||
task.id = rd.get_new_id();
|
||||
decision.fill_task(state, question, files, ctx_server.mctx, ctx_server.init_opt, task);
|
||||
@@ -5367,7 +5391,16 @@ void server_routes::init_routes() {
|
||||
|
||||
json answers = json::object();
|
||||
int32_t n_tokens = 0;
|
||||
for (size_t i = 0; i < questions.size(); i++) {
|
||||
if (decision.is_joint()) {
|
||||
auto * result = dynamic_cast<server_task_result_decision *>(all_results.results[0].get());
|
||||
GGML_ASSERT(result != nullptr);
|
||||
const auto scores = decision.split_scores(questions, result->scores);
|
||||
for (size_t i = 0; i < questions.size(); i++) {
|
||||
answers[questions[i].id] = decision.format_answer(questions[i], scores[i]);
|
||||
}
|
||||
n_tokens = result->n_tokens;
|
||||
}
|
||||
for (size_t i = 0; i < questions.size() && !decision.is_joint(); i++) {
|
||||
auto * result = dynamic_cast<server_task_result_decision *>(all_results.results[i].get());
|
||||
GGML_ASSERT(result != nullptr);
|
||||
answers[questions[i].id] = decision.format_answer(questions[i], result->scores);
|
||||
|
||||
@@ -85,6 +85,9 @@ void server_decision_context::init(const llama_model * model) {
|
||||
throw std::runtime_error("decision model has no valid max_head_tokens");
|
||||
}
|
||||
n_options_max = 255;
|
||||
} else if (model_type == COMMON_DECISION_TYPE_CLEF) {
|
||||
n_options_max = 255;
|
||||
noul_true_first = true;
|
||||
} else {
|
||||
throw std::runtime_error("unsupported decision model type: " + type_name);
|
||||
}
|
||||
@@ -374,6 +377,185 @@ void server_decision_context::fill_task_laya(llama_tokens & tokens, const server
|
||||
task.decision.column = question.type;
|
||||
}
|
||||
|
||||
//
|
||||
// joint prompt (clef)
|
||||
//
|
||||
|
||||
// the template separates the pieces of the prompt with these, they are tokenized one by one
|
||||
static const std::string CLEF_MARKER = "<<clef:";
|
||||
static const std::string CLEF_PIECE_SEP = "<<clef:sep>>";
|
||||
static const std::string CLEF_PIECE_STATE = "<<clef:state>>";
|
||||
static const std::string CLEF_PIECE_QUESTION = "<<clef:question>>";
|
||||
static const std::string CLEF_PIECE_OPTION = "<<clef:option>>";
|
||||
|
||||
// same value with the keys of all objects sorted
|
||||
static json decision_sort_keys(const json & val) {
|
||||
if (val.is_array()) {
|
||||
json out = json::array();
|
||||
for (const auto & item : val) {
|
||||
out.push_back(decision_sort_keys(item));
|
||||
}
|
||||
return out;
|
||||
}
|
||||
if (val.is_object()) {
|
||||
std::map<std::string, json> sorted;
|
||||
for (const auto & [key, item] : val.items()) {
|
||||
sorted[key] = decision_sort_keys(item);
|
||||
}
|
||||
json out = json::object();
|
||||
for (const auto & [key, item] : sorted) {
|
||||
out[key] = item;
|
||||
}
|
||||
return out;
|
||||
}
|
||||
return val;
|
||||
}
|
||||
|
||||
// strings are used as is, other values are compact JSON with sorted keys
|
||||
static std::string clef_render(const json & val) {
|
||||
return val.is_string() ? val.get<std::string>() : decision_sort_keys(val).dump();
|
||||
}
|
||||
|
||||
std::vector<size_t> server_decision_context::prompt_order(const server_decision_question & question) const {
|
||||
std::vector<size_t> order(question.options.size());
|
||||
for (size_t i = 0; i < order.size(); i++) {
|
||||
order[i] = i;
|
||||
}
|
||||
if (type == COMMON_DECISION_TYPE_CLEF && question.type == SERVER_DECISION_QUESTION_CHOICE) {
|
||||
std::sort(order.begin(), order.end(), [&](size_t a, size_t b) {
|
||||
return question.options[a].key < question.options[b].key;
|
||||
});
|
||||
}
|
||||
return order;
|
||||
}
|
||||
|
||||
void server_decision_context::fill_task_joint(const json & state, const std::vector<server_decision_question> & questions, server_task & task) const {
|
||||
auto clean = [](const json & val) {
|
||||
return decision_replace_text(val, CLEF_MARKER, "<<clef ");
|
||||
};
|
||||
|
||||
json inp_questions = json::array();
|
||||
for (const auto & question : questions) {
|
||||
json options = json::array();
|
||||
for (const size_t i : prompt_order(question)) {
|
||||
const auto & opt = question.options[i];
|
||||
|
||||
json description = opt.description;
|
||||
if (description.is_null() && question.type == SERVER_DECISION_QUESTION_NOUL) {
|
||||
description = opt.key == "true"
|
||||
? "The proposition is true or the answer is yes."
|
||||
: "The proposition is false or the answer is no.";
|
||||
}
|
||||
|
||||
// keys in sorted order
|
||||
json semantics = json::object();
|
||||
if (!description.is_null()) {
|
||||
semantics["description"] = decision_sort_keys(description);
|
||||
}
|
||||
semantics["option_id"] = opt.key;
|
||||
|
||||
options.push_back(json{{"text", semantics.dump()}});
|
||||
}
|
||||
|
||||
inp_questions.push_back(json{
|
||||
{"id", question.id},
|
||||
{"type", decision_question_type_name(question.type)},
|
||||
{"instructions", clef_render(question.instructions)},
|
||||
{"options", options},
|
||||
});
|
||||
}
|
||||
|
||||
const json inp = clean(json{
|
||||
{"state", clef_render(state)},
|
||||
{"questions", inp_questions},
|
||||
});
|
||||
|
||||
jinja::context ctx(tmpl->source());
|
||||
jinja::global_from_json(ctx, inp, false);
|
||||
jinja::runtime runtime(ctx);
|
||||
const jinja::value results = runtime.execute(tmpl->prog);
|
||||
const std::string prompt = jinja::runtime::gather_string_parts(results)->as_string().str();
|
||||
|
||||
// the model was trained with the pieces tokenized one by one
|
||||
llama_tokens tokens;
|
||||
int32_t i_question = -1;
|
||||
size_t n_questions = 0;
|
||||
size_t n_options = 0;
|
||||
for (std::string piece : string_split(prompt, CLEF_PIECE_SEP)) {
|
||||
enum { PIECE_TEXT, PIECE_QUESTION, PIECE_OPTION } kind = PIECE_TEXT;
|
||||
if (string_starts_with(piece, CLEF_PIECE_QUESTION)) {
|
||||
piece = piece.substr(CLEF_PIECE_QUESTION.size());
|
||||
kind = PIECE_QUESTION;
|
||||
} else if (string_starts_with(piece, CLEF_PIECE_OPTION)) {
|
||||
piece = piece.substr(CLEF_PIECE_OPTION.size());
|
||||
kind = PIECE_OPTION;
|
||||
} else if (string_starts_with(piece, CLEF_PIECE_STATE)) {
|
||||
piece = piece.substr(CLEF_PIECE_STATE.size());
|
||||
}
|
||||
|
||||
const int32_t start = tokens.size();
|
||||
const llama_tokens piece_tokens = common_tokenize(vocab, piece, false, true);
|
||||
tokens.insert(tokens.end(), piece_tokens.begin(), piece_tokens.end());
|
||||
const int32_t end = tokens.size();
|
||||
|
||||
if (kind == PIECE_TEXT) {
|
||||
continue;
|
||||
}
|
||||
if (start == end) {
|
||||
throw std::invalid_argument("the instructions and the options of a question must not be empty");
|
||||
}
|
||||
// see llama_batch_ext_set_decision_order(): question of type noul, choice, score, or option
|
||||
int32_t order = 4;
|
||||
if (kind == PIECE_QUESTION) {
|
||||
i_question++;
|
||||
switch (questions.at(i_question).type) {
|
||||
case SERVER_DECISION_QUESTION_NOUL: order = 1; break;
|
||||
case SERVER_DECISION_QUESTION_CHOICE: order = 2; break;
|
||||
case SERVER_DECISION_QUESTION_SCORE: order = 3; break;
|
||||
}
|
||||
n_questions++;
|
||||
} else {
|
||||
// the score of option i is returned at row i
|
||||
task.decision.markers.push_back(n_options);
|
||||
n_options++;
|
||||
}
|
||||
task.decision.order.resize(tokens.size(), 0);
|
||||
std::fill(task.decision.order.begin() + start, task.decision.order.end(), order);
|
||||
}
|
||||
task.decision.order.resize(tokens.size(), 0);
|
||||
|
||||
size_t n_options_exp = 0;
|
||||
for (const auto & question : questions) {
|
||||
n_options_exp += question.options.size();
|
||||
}
|
||||
if (n_questions != questions.size() || n_options != n_options_exp) {
|
||||
throw std::runtime_error("unexpected layout of the decision prompt");
|
||||
}
|
||||
|
||||
task.decision.column = 0;
|
||||
task.tokens = server_tokens(tokens, false);
|
||||
}
|
||||
|
||||
std::vector<std::vector<float>> server_decision_context::split_scores(
|
||||
const std::vector<server_decision_question> & questions,
|
||||
const std::vector<float> & scores) const {
|
||||
std::vector<std::vector<float>> result;
|
||||
size_t offset = 0;
|
||||
for (const auto & question : questions) {
|
||||
const auto order = prompt_order(question);
|
||||
if (offset + order.size() > scores.size()) {
|
||||
throw std::runtime_error("decision result does not match the number of options");
|
||||
}
|
||||
std::vector<float> cur(order.size());
|
||||
for (size_t i = 0; i < order.size(); i++) {
|
||||
cur[order[i]] = scores[offset + i];
|
||||
}
|
||||
offset += order.size();
|
||||
result.push_back(std::move(cur));
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
//
|
||||
// answer
|
||||
//
|
||||
|
||||
@@ -45,7 +45,13 @@ struct server_decision_context {
|
||||
}
|
||||
}
|
||||
|
||||
// true if all the questions of a request go in one prompt, see fill_task_joint()
|
||||
bool is_joint() const {
|
||||
return type == COMMON_DECISION_TYPE_CLEF;
|
||||
}
|
||||
|
||||
// true if the prompt of the model has a place for images
|
||||
// TODO: clef needs token and embedding entries in the same batch, see https://github.com/ggml-org/llama.cpp/pull/29622
|
||||
bool can_use_images() const {
|
||||
switch (type) {
|
||||
case COMMON_DECISION_TYPE_OPENJEV:
|
||||
@@ -72,6 +78,12 @@ struct server_decision_context {
|
||||
const mtmd_helper_init_opt & init_opt,
|
||||
server_task & task) const;
|
||||
|
||||
// set the prompt of all the questions, and where to read their results
|
||||
void fill_task_joint(const json & state, const std::vector<server_decision_question> & questions, server_task & task) const;
|
||||
|
||||
// result of fill_task_joint() -> scores of each question, in the order of its options
|
||||
std::vector<std::vector<float>> split_scores(const std::vector<server_decision_question> & questions, const std::vector<float> & scores) const;
|
||||
|
||||
// scores: one raw model output per option
|
||||
json format_answer(const server_decision_question & question, const std::vector<float> & scores) const;
|
||||
|
||||
@@ -96,6 +108,9 @@ private:
|
||||
std::string render(const json & state, const server_decision_question & question, size_t n_images) const;
|
||||
void fill_task_laya(llama_tokens & tokens, const server_decision_question & question, server_task & task) const;
|
||||
|
||||
// indices of the options of a question, in the order they have in the prompt
|
||||
std::vector<size_t> prompt_order(const server_decision_question & question) const;
|
||||
|
||||
float get_temperature(const server_decision_question & question) const;
|
||||
};
|
||||
|
||||
|
||||
@@ -181,6 +181,10 @@ struct server_task {
|
||||
std::vector<llama_token> labels; // logits of these tokens, at the last prompt token
|
||||
std::vector<int32_t> markers; // embeddings[column] at these prompt positions
|
||||
int32_t column = 0;
|
||||
|
||||
|
||||
// for a joint head: one value per prompt token, see llama_batch_ext_set_decision_order()
|
||||
std::vector<int32_t> order;
|
||||
};
|
||||
decision decision;
|
||||
|
||||
|
||||
Reference in New Issue
Block a user