init support for clef (text only)

This commit is contained in:
Xuan Son Nguyen
2026-10-02 01:16:03 +02:00
parent 7c760c1e4c
commit aace2e501b
23 changed files with 1358 additions and 13 deletions
+13 -3
View File
@@ -3,6 +3,9 @@
#include "build-info.h"
#include "common.h"
#include "../src/llama-ext.h"
#include "fit.h"
#include "log.h"
#include "llama.h"
@@ -1208,6 +1211,9 @@ common_decision_type common_get_decision_type(const struct llama_model * model)
if (type == "laya") {
return COMMON_DECISION_TYPE_LAYA;
}
if (type == "clef") {
return COMMON_DECISION_TYPE_CLEF;
}
return COMMON_DECISION_TYPE_UNKNOWN;
}
@@ -1263,9 +1269,10 @@ common_init_result::common_init_result(common_params & params, bool model_only)
const llama_vocab * vocab = llama_model_get_vocab(model);
// this decision model returns a score for each token via the embeddings output
// these decision models return a score for each token via the embeddings output
// TODO: maybe improve this in the future
if (common_get_decision_type(model) == COMMON_DECISION_TYPE_LAYA) {
const auto decision_type = common_get_decision_type(model);
if (decision_type == COMMON_DECISION_TYPE_LAYA || decision_type == COMMON_DECISION_TYPE_CLEF) {
params.embedding = true;
params.pooling_type = LLAMA_POOLING_TYPE_NONE;
@@ -1274,7 +1281,7 @@ common_init_result::common_init_result(common_params & params, bool model_only)
cparams.n_outputs_max = cparams.n_batch;
cparams.n_outputs_max_per_seq = 1;
LOG_INF("%s", "laya decision model detected, enabling embedding mode\n");
LOG_INF("%s", "decision model detected, enabling embedding mode\n");
}
// load and optionally apply lora adapters
@@ -2211,6 +2218,9 @@ llama_batch_ext * common_batch::get_sub_batch(int32_t off, int32_t n) {
if (t.output) {
llama_batch_ext_set_output_logits(res, idx, true);
}
if (t.decision_order != 0) {
llama_batch_ext_set_decision_order(res, idx, t.decision_order);
}
}
return res;
+2
View File
@@ -950,6 +950,7 @@ enum common_decision_type {
COMMON_DECISION_TYPE_UNKNOWN, // a decision model of a type that is not supported
COMMON_DECISION_TYPE_OPENJEV, // logits of one label token per option, read at the last prompt token
COMMON_DECISION_TYPE_LAYA, // score of one marker token per option, read from the embeddings output
COMMON_DECISION_TYPE_CLEF, // all questions in one prompt, score of option i read from the embeddings output at row i
};
common_decision_type common_get_decision_type(const struct llama_model * model);
@@ -1051,6 +1052,7 @@ struct common_batch {
bool output;
llama_embd embd; // non-owning view of the data passed to add_embd()/set_embd(), data == NULL if none
std::vector<llama_seq_id> seq_ids_extra; // see add_seq()
int32_t decision_order = 0; // see llama_batch_ext_set_decision_order()
};
std::vector<token> tokens; // mirror of the entries, tokens[i] describes batch index i
+1
View File
@@ -42,6 +42,7 @@ TEXT_MODEL_MAP: dict[str, str] = {
"ChameleonForConditionalGeneration": "chameleon",
"ChatGLMForConditionalGeneration": "chatglm",
"ChatGLMModel": "chatglm",
"ClefModel": "clef",
"CodeShellForCausalLM": "codeshell",
"CogVLMForCausalLM": "cogvlm",
"Cohere2MoeForCausalLM": "command_r",
+132
View File
@@ -0,0 +1,132 @@
from __future__ import annotations
import json
import math
from pathlib import Path
from typing import Any, Iterable, TYPE_CHECKING
import torch
if TYPE_CHECKING:
from torch import Tensor
from .base import ModelBase, gguf, logger
from .qwen import Qwen3_5TextModel
def _is_clef_checkpoint(dir_model: Path) -> bool:
return (dir_model / "joint_head_config.json").is_file() and (dir_model / "config.json").is_file()
@ModelBase.register_hparams_loader(_is_clef_checkpoint)
def _load_clef_hparams(dir_model: Path) -> dict[str, Any]:
logger.info("gguf: detected Clef checkpoint")
hparams = ModelBase.load_hparams(dir_model, False, guess=False)
hparams["architectures"] = ["ClefModel"]
with open(dir_model / "joint_head_config.json", encoding="utf-8") as f:
hparams["decision"] = json.load(f)
return hparams
# TODO: image input needs token and embedding entries in the same batch, see https://github.com/ggml-org/llama.cpp/pull/29622
@ModelBase.register("ClefModel")
class ClefModel(Qwen3_5TextModel):
model_arch = gguf.MODEL_ARCH.CLEF
no_mtp = True # the checkpoint has no MTP head
# prompt follows joint_schema_model.py of the model repo
_SYSTEM_PROMPT = (
"Read the complete state and schema. Decide every field jointly. Each answer "
"must be exactly one of that field's allowed options."
)
# the pieces of the prompt are tokenized one by one, this separates them
_PIECE_SEP = "<<clef:sep>>"
# start of a piece: state, question span, option span
_PIECE_STATE, _PIECE_QUESTION, _PIECE_OPTION = "<<clef:state>>", "<<clef:question>>", "<<clef:option>>"
def __init__(self, *args, **kwargs):
super().__init__(*args, **kwargs)
head = self.hparams["decision"]
self._n_routing = head["routing_layers"]
# the head blocks are named dec.blk.N, routing blocks first
self.tensor_map = gguf.get_tensor_name_map(self.model_arch, max(self.block_count, self._n_routing + head["layers"]))
self._scales: dict[str, float] = {}
def set_vocab(self):
super().set_vocab()
self.gguf_writer.add_chat_template([{"name": "systemone", "template": self._systemone_template()}])
@classmethod
def _systemone_template(cls) -> str:
def text(value: str) -> str:
return "{{ " + json.dumps(value) + " }}"
sep = text(cls._PIECE_SEP)
# state, instructions and option text are given as strings
return (
text(f"<|im_start|>system\n{cls._SYSTEM_PROMPT}<|im_end|>\n<|im_start|>user\nSTATE:\n")
+ sep + text(cls._PIECE_STATE) + "{{ state }}"
+ sep + text("\n\nSCHEMA FIELDS:\n")
+ "{% for q in questions %}"
+ sep + text("\nFIELD ") + "{{ loop.index }}" + text("\nID: ") + "{{ q.id }}"
+ text("\nTYPE: ") + "{{ q.type }}" + text("\nINSTRUCTION: ")
+ sep + text(cls._PIECE_QUESTION) + "{{ q.instructions }}"
+ sep + text("\nALLOWED OPTIONS:\n")
+ "{% for o in q.options %}"
+ sep + text("OPTION ") + "{{ loop.index }}" + text(": ")
+ sep + text(cls._PIECE_OPTION) + "{{ o.text }}"
+ sep + text("\n")
+ "{% endfor %}"
+ sep + text("END FIELD\n")
+ "{% endfor %}"
+ sep + text("\n<|im_end|>\n<|im_start|>assistant\n<think>\n\n</think>\n\nJOINT SCHEMA DECISIONS:")
)
def set_gguf_parameters(self):
super().set_gguf_parameters()
head = self.hparams["decision"]
self.gguf_writer.add_decision_type(gguf.DecisionType.CLEF)
self.gguf_writer.add_decision_routing_block_count(head["routing_layers"])
self.gguf_writer.add_decision_block_count(head["layers"])
self.gguf_writer.add_decision_head_count(head["heads"])
def generate_extra_tensors(self) -> Iterable[tuple[str, Tensor]]:
yield from super().generate_extra_tensors()
from safetensors.torch import load_file
for name, data in load_file(self.dir_model / "joint_head.safetensors").items():
yield "joint_head." + name, data
def modify_tensors(self, data_torch: Tensor, name: str, bid: int | None) -> Iterable[tuple[str, Tensor]]:
if not name.startswith("joint_head."):
yield from super().modify_tensors(data_torch, name, bid)
return
parts = name.split(".")
# learned scalars, stored as the values used at inference
if len(parts) == 2 and data_torch.ndim == 0:
value = float(data_torch)
if parts[1] == "residual_gate":
self._scales[parts[1]] = 1.0 / (1.0 + math.exp(-value))
else:
self._scales[parts[1]] = math.exp(min(value, math.log(100.0)))
if len(self._scales) == 3:
scales = [self._scales[k] for k in ("prior_logit_scale", "joint_logit_scale", "residual_gate")]
yield self.format_tensor_name(gguf.MODEL_TENSOR.DECISION_SCALES, suffix=""), torch.tensor(scales, dtype=torch.float32)
return
# routing blocks come first
if parts[1] == "layers":
parts[2] = str(int(parts[2]) + self._n_routing)
name = ".".join(parts)
# nn.MultiheadAttention keeps q, k, v in one tensor
for suffix in ("weight", "bias"):
if name.endswith(".in_proj_" + suffix):
prefix = name[:-len("in_proj_" + suffix)]
for x, data in zip("qkv", data_torch.chunk(3, dim=0)):
yield self.map_tensor_name(prefix + x + "." + suffix), data
return
yield self.map_tensor_name(name), data_torch
+83
View File
@@ -324,6 +324,8 @@ class Keys:
TYPE = "{arch}.decision.type"
# note: single-use-case keys can be hard-coded in cpp code
BLOCK_COUNT = "{arch}.decision.block_count"
ROUTING_BLOCK_COUNT = "{arch}.decision.routing_block_count"
HEAD_COUNT = "{arch}.decision.head_count"
MAX_HEAD_TOKENS = "{arch}.decision.max_head_tokens"
TEMPERATURE = "{arch}.decision.temperature.{name}" # name: "<type>" or "<type>.<n_opt bucket>"
@@ -537,6 +539,7 @@ class MODEL_ARCH(IntEnum):
QWEN3VLMOE = auto()
QWEN35 = auto()
QWEN35MOE = auto()
CLEF = auto()
QWEN4EXP = auto()
PHI2 = auto()
PHI3 = auto()
@@ -871,6 +874,18 @@ class MODEL_TENSOR(IntEnum):
DEC_FFN_DOWN = auto()
DEC_FFN_UP = auto()
DEC_OUTPUT_NORM = auto()
DEC_CROSS_ATTN_NORM_KV = auto()
DECISION_HIDDEN_NORM = auto()
DECISION_PROJ_MEMORY = auto()
DECISION_PROJ_QUESTION = auto()
DECISION_PROJ_OPTION_QUESTION = auto()
DECISION_PROJ_GLOBAL = auto()
DECISION_PROJ_OPTION_CONTEXT = auto()
DECISION_PROJ_OPTION_LEXICAL = auto()
DECISION_OPTION_SUMMARY_NORM = auto()
DECISION_FIELD_NORM = auto()
DECISION_OPTION_NORM = auto()
DECISION_SCALES = auto()
ENC_ATTN_NORM = auto()
ENC_ATTN_Q = auto()
ENC_ATTN_K = auto()
@@ -1301,6 +1316,7 @@ MODEL_ARCH_NAMES: dict[MODEL_ARCH, str] = {
MODEL_ARCH.QWEN3VLMOE: "qwen3vlmoe",
MODEL_ARCH.QWEN35: "qwen35",
MODEL_ARCH.QWEN35MOE: "qwen35moe",
MODEL_ARCH.CLEF: "clef",
MODEL_ARCH.QWEN4EXP: "qwen4exp",
MODEL_ARCH.PHI2: "phi2",
MODEL_ARCH.PHI3: "phi3",
@@ -1634,6 +1650,18 @@ TENSOR_NAMES: dict[MODEL_TENSOR, str] = {
MODEL_TENSOR.DEC_FFN_DOWN: "dec.blk.{bid}.ffn_down",
MODEL_TENSOR.DEC_FFN_UP: "dec.blk.{bid}.ffn_up",
MODEL_TENSOR.DEC_OUTPUT_NORM: "dec.output_norm",
MODEL_TENSOR.DEC_CROSS_ATTN_NORM_KV: "dec.blk.{bid}.cross_attn_norm_kv",
MODEL_TENSOR.DECISION_HIDDEN_NORM: "decision.hidden_norm",
MODEL_TENSOR.DECISION_PROJ_MEMORY: "decision.proj_memory",
MODEL_TENSOR.DECISION_PROJ_QUESTION: "decision.proj_question",
MODEL_TENSOR.DECISION_PROJ_OPTION_QUESTION: "decision.proj_option_question",
MODEL_TENSOR.DECISION_PROJ_GLOBAL: "decision.proj_global",
MODEL_TENSOR.DECISION_PROJ_OPTION_CONTEXT: "decision.proj_option_context",
MODEL_TENSOR.DECISION_PROJ_OPTION_LEXICAL: "decision.proj_option_lexical",
MODEL_TENSOR.DECISION_OPTION_SUMMARY_NORM: "decision.option_summary_norm",
MODEL_TENSOR.DECISION_FIELD_NORM: "decision.field_norm",
MODEL_TENSOR.DECISION_OPTION_NORM: "decision.option_norm",
MODEL_TENSOR.DECISION_SCALES: "decision.scales",
MODEL_TENSOR.ENC_ATTN_NORM: "enc.blk.{bid}.attn_norm",
MODEL_TENSOR.ENC_ATTN_Q: "enc.blk.{bid}.attn_q",
MODEL_TENSOR.ENC_ATTN_K: "enc.blk.{bid}.attn_k",
@@ -2917,6 +2945,60 @@ MODEL_TENSORS: dict[MODEL_ARCH, list[MODEL_TENSOR]] = {
MODEL_TENSOR.NEXTN_SHARED_HEAD_HEAD,
MODEL_TENSOR.NEXTN_SHARED_HEAD_NORM,
],
MODEL_ARCH.CLEF: [
MODEL_TENSOR.TOKEN_EMBD,
MODEL_TENSOR.OUTPUT_NORM,
MODEL_TENSOR.OUTPUT,
MODEL_TENSOR.ATTN_NORM,
MODEL_TENSOR.ATTN_Q,
MODEL_TENSOR.ATTN_Q_NORM,
MODEL_TENSOR.ATTN_K,
MODEL_TENSOR.ATTN_K_NORM,
MODEL_TENSOR.ATTN_V,
MODEL_TENSOR.ATTN_OUT,
MODEL_TENSOR.ATTN_POST_NORM,
MODEL_TENSOR.ATTN_GATE,
MODEL_TENSOR.ATTN_QKV,
MODEL_TENSOR.FFN_GATE,
MODEL_TENSOR.FFN_DOWN,
MODEL_TENSOR.FFN_UP,
MODEL_TENSOR.SSM_A,
MODEL_TENSOR.SSM_CONV1D,
MODEL_TENSOR.SSM_DT,
MODEL_TENSOR.SSM_NORM,
MODEL_TENSOR.SSM_BETA,
MODEL_TENSOR.SSM_ALPHA,
MODEL_TENSOR.SSM_OUT,
# decision head
MODEL_TENSOR.TOKEN_TYPES,
MODEL_TENSOR.CLS,
MODEL_TENSOR.CLS_OUT,
MODEL_TENSOR.DEC_ATTN_NORM,
MODEL_TENSOR.DEC_ATTN_Q,
MODEL_TENSOR.DEC_ATTN_K,
MODEL_TENSOR.DEC_ATTN_V,
MODEL_TENSOR.DEC_ATTN_OUT,
MODEL_TENSOR.DEC_CROSS_ATTN_NORM,
MODEL_TENSOR.DEC_CROSS_ATTN_NORM_KV,
MODEL_TENSOR.DEC_CROSS_ATTN_Q,
MODEL_TENSOR.DEC_CROSS_ATTN_K,
MODEL_TENSOR.DEC_CROSS_ATTN_V,
MODEL_TENSOR.DEC_CROSS_ATTN_OUT,
MODEL_TENSOR.DEC_FFN_NORM,
MODEL_TENSOR.DEC_FFN_DOWN,
MODEL_TENSOR.DEC_FFN_UP,
MODEL_TENSOR.DECISION_HIDDEN_NORM,
MODEL_TENSOR.DECISION_PROJ_MEMORY,
MODEL_TENSOR.DECISION_PROJ_QUESTION,
MODEL_TENSOR.DECISION_PROJ_OPTION_QUESTION,
MODEL_TENSOR.DECISION_PROJ_GLOBAL,
MODEL_TENSOR.DECISION_PROJ_OPTION_CONTEXT,
MODEL_TENSOR.DECISION_PROJ_OPTION_LEXICAL,
MODEL_TENSOR.DECISION_OPTION_SUMMARY_NORM,
MODEL_TENSOR.DECISION_FIELD_NORM,
MODEL_TENSOR.DECISION_OPTION_NORM,
MODEL_TENSOR.DECISION_SCALES,
],
MODEL_ARCH.QWEN35MOE: [
MODEL_TENSOR.TOKEN_EMBD,
MODEL_TENSOR.OUTPUT_NORM,
@@ -5917,6 +5999,7 @@ class GGUFValueType(IntEnum):
class DecisionType:
LAYA = "laya" # head blocks + scorer on the hidden state of one marker token per option
OPENJEV = "openjev" # logits of one label token per option
CLEF = "clef" # joint head over all questions, one score per option
class VisionProjectorType:
+6
View File
@@ -1346,6 +1346,12 @@ class GGUFWriter:
def add_decision_block_count(self, value: int) -> None:
self.add_uint32(Keys.Decision.BLOCK_COUNT.format(arch=self.arch), value)
def add_decision_routing_block_count(self, value: int) -> None:
self.add_uint32(Keys.Decision.ROUTING_BLOCK_COUNT.format(arch=self.arch), value)
def add_decision_head_count(self, value: int) -> None:
self.add_uint32(Keys.Decision.HEAD_COUNT.format(arch=self.arch), value)
def add_decision_max_head_tokens(self, value: int) -> None:
self.add_uint32(Keys.Decision.MAX_HEAD_TOKENS.format(arch=self.arch), value)
+68
View File
@@ -49,6 +49,7 @@ class TensorNameMap:
MODEL_TENSOR.TOKEN_TYPES: (
"embeddings.token_type_embeddings", # bert nomic-bert
"type_emb", # laya
"joint_head.type_embedding", # clef
),
# Normalization of token embeddings
@@ -1181,22 +1182,27 @@ class TensorNameMap:
MODEL_TENSOR.DEC_ATTN_NORM: (
"decoder.block.{bid}.layer.0.layer_norm", # t5
"joint_head.layers.{bid}.norm1", # clef
),
MODEL_TENSOR.DEC_ATTN_Q: (
"decoder.block.{bid}.layer.0.SelfAttention.q", # t5
"joint_head.layers.{bid}.self_attn.q", # clef
),
MODEL_TENSOR.DEC_ATTN_K: (
"decoder.block.{bid}.layer.0.SelfAttention.k", # t5
"joint_head.layers.{bid}.self_attn.k", # clef
),
MODEL_TENSOR.DEC_ATTN_V: (
"decoder.block.{bid}.layer.0.SelfAttention.v", # t5
"joint_head.layers.{bid}.self_attn.v", # clef
),
MODEL_TENSOR.DEC_ATTN_OUT: (
"decoder.block.{bid}.layer.0.SelfAttention.o", # t5
"joint_head.layers.{bid}.self_attn.out_proj", # clef
),
MODEL_TENSOR.DEC_ATTN_REL_B: (
@@ -1205,22 +1211,32 @@ class TensorNameMap:
MODEL_TENSOR.DEC_CROSS_ATTN_NORM: (
"decoder.block.{bid}.layer.1.layer_norm", # t5
"joint_head.layers.{bid}.norm2", # clef
"joint_head.evidence_layers.{bid}.query_norm", # clef
),
MODEL_TENSOR.DEC_CROSS_ATTN_Q: (
"decoder.block.{bid}.layer.1.EncDecAttention.q", # t5
"joint_head.layers.{bid}.multihead_attn.q", # clef
"joint_head.evidence_layers.{bid}.attention.q", # clef
),
MODEL_TENSOR.DEC_CROSS_ATTN_K: (
"decoder.block.{bid}.layer.1.EncDecAttention.k", # t5
"joint_head.layers.{bid}.multihead_attn.k", # clef
"joint_head.evidence_layers.{bid}.attention.k", # clef
),
MODEL_TENSOR.DEC_CROSS_ATTN_V: (
"decoder.block.{bid}.layer.1.EncDecAttention.v", # t5
"joint_head.layers.{bid}.multihead_attn.v", # clef
"joint_head.evidence_layers.{bid}.attention.v", # clef
),
MODEL_TENSOR.DEC_CROSS_ATTN_OUT: (
"decoder.block.{bid}.layer.1.EncDecAttention.o", # t5
"joint_head.layers.{bid}.multihead_attn.out_proj", # clef
"joint_head.evidence_layers.{bid}.attention.out_proj", # clef
),
MODEL_TENSOR.DEC_CROSS_ATTN_REL_B: (
@@ -1229,6 +1245,8 @@ class TensorNameMap:
MODEL_TENSOR.DEC_FFN_NORM: (
"decoder.block.{bid}.layer.2.layer_norm", # t5
"joint_head.layers.{bid}.norm3", # clef
"joint_head.evidence_layers.{bid}.feedforward_norm", # clef
),
MODEL_TENSOR.DEC_FFN_GATE: (
@@ -1238,16 +1256,64 @@ class TensorNameMap:
MODEL_TENSOR.DEC_FFN_UP: (
"decoder.block.{bid}.layer.2.DenseReluDense.wi", # t5
"decoder.block.{bid}.layer.2.DenseReluDense.wi_1", # flan-t5
"joint_head.layers.{bid}.linear1", # clef
"joint_head.evidence_layers.{bid}.feedforward.0", # clef
),
MODEL_TENSOR.DEC_FFN_DOWN: (
"decoder.block.{bid}.layer.2.DenseReluDense.wo", # t5
"joint_head.layers.{bid}.linear2", # clef
"joint_head.evidence_layers.{bid}.feedforward.3", # clef
),
MODEL_TENSOR.DEC_OUTPUT_NORM: (
"decoder.final_layer_norm", # t5
),
MODEL_TENSOR.DEC_CROSS_ATTN_NORM_KV: (
"joint_head.evidence_layers.{bid}.memory_norm", # clef
),
MODEL_TENSOR.DECISION_HIDDEN_NORM: (
"joint_head.hidden_norm", # clef
),
MODEL_TENSOR.DECISION_PROJ_MEMORY: (
"joint_head.memory_projection", # clef
),
MODEL_TENSOR.DECISION_PROJ_QUESTION: (
"joint_head.question_projection", # clef
),
MODEL_TENSOR.DECISION_PROJ_OPTION_QUESTION: (
"joint_head.option_question_projection", # clef
),
MODEL_TENSOR.DECISION_PROJ_GLOBAL: (
"joint_head.global_projection", # clef
),
MODEL_TENSOR.DECISION_PROJ_OPTION_CONTEXT: (
"joint_head.option_context_projection", # clef
),
MODEL_TENSOR.DECISION_PROJ_OPTION_LEXICAL: (
"joint_head.option_lexical_projection", # clef
),
MODEL_TENSOR.DECISION_OPTION_SUMMARY_NORM: (
"joint_head.option_summary_norm", # clef
),
MODEL_TENSOR.DECISION_FIELD_NORM: (
"joint_head.field_norm", # clef
),
MODEL_TENSOR.DECISION_OPTION_NORM: (
"joint_head.option_norm", # clef
),
MODEL_TENSOR.ENC_ATTN_NORM: (
"encoder.block.{bid}.layer.0.layer_norm", # t5
),
@@ -1449,11 +1515,13 @@ class TensorNameMap:
"dense", # neobert
"head.dense", # modern-bert
"scorer.1", # laya
"joint_head.residual_scorer.0", # clef
),
MODEL_TENSOR.CLS_OUT: (
"classifier.out_proj", # roberta
"scorer.3", # laya
"joint_head.residual_scorer.3", # clef
),
MODEL_TENSOR.CLS_NORM: (
+28 -1
View File
@@ -40,6 +40,7 @@ static const std::map<llm_arch, const char *> LLM_ARCH_NAMES = {
{ LLM_ARCH_QWEN3VLMOE, "qwen3vlmoe" },
{ LLM_ARCH_QWEN35, "qwen35" },
{ LLM_ARCH_QWEN35MOE, "qwen35moe" },
{ LLM_ARCH_CLEF, "clef" },
{ LLM_ARCH_QWEN4EXP, "qwen4exp" },
{ LLM_ARCH_PHI2, "phi2" },
{ LLM_ARCH_PHI3, "phi3" },
@@ -366,7 +367,9 @@ static const std::map<llm_kv, const char *> LLM_KV_NAMES = {
{ LLM_KV_CLASSIFIER_OUTPUT_LABELS, "%s.classifier.output_labels" },
{ LLM_KV_CLASSIFIER_POOLING_TYPE, "%s.classifier.pooling_type" },
{ LLM_KV_DECISION_BLOCK_COUNT, "%s.decision.block_count" },
{ LLM_KV_DECISION_BLOCK_COUNT, "%s.decision.block_count" },
{ LLM_KV_DECISION_ROUTING_BLOCK_COUNT, "%s.decision.routing_block_count" },
{ LLM_KV_DECISION_HEAD_COUNT, "%s.decision.head_count" },
{ LLM_KV_TARGET_LAYERS, "%s.target_layers" },
{ LLM_KV_TARGET_HIDDEN_SIZE, "%s.target_hidden_size" },
@@ -599,6 +602,18 @@ static const std::map<llm_tensor, const char *> LLM_TENSOR_NAMES = {
{ LLM_TENSOR_ATTN_SUB_NORM, "blk.%d.attn_sub_norm" },
{ LLM_TENSOR_FFN_SUB_NORM, "blk.%d.ffn_sub_norm" },
{ LLM_TENSOR_DEC_OUTPUT_NORM, "dec.output_norm" },
{ LLM_TENSOR_DEC_CROSS_ATTN_NORM_KV, "dec.blk.%d.cross_attn_norm_kv" },
{ LLM_TENSOR_DECISION_HIDDEN_NORM, "decision.hidden_norm" },
{ LLM_TENSOR_DECISION_PROJ_MEMORY, "decision.proj_memory" },
{ LLM_TENSOR_DECISION_PROJ_QUESTION, "decision.proj_question" },
{ LLM_TENSOR_DECISION_PROJ_OPTION_QUESTION, "decision.proj_option_question" },
{ LLM_TENSOR_DECISION_PROJ_GLOBAL, "decision.proj_global" },
{ LLM_TENSOR_DECISION_PROJ_OPTION_CONTEXT, "decision.proj_option_context" },
{ LLM_TENSOR_DECISION_PROJ_OPTION_LEXICAL, "decision.proj_option_lexical" },
{ LLM_TENSOR_DECISION_OPTION_SUMMARY_NORM, "decision.option_summary_norm" },
{ LLM_TENSOR_DECISION_FIELD_NORM, "decision.field_norm" },
{ LLM_TENSOR_DECISION_OPTION_NORM, "decision.option_norm" },
{ LLM_TENSOR_DECISION_SCALES, "decision.scales" },
{ LLM_TENSOR_DEC_ATTN_NORM, "dec.blk.%d.attn_norm" },
{ LLM_TENSOR_DEC_ATTN_Q, "dec.blk.%d.attn_q" },
{ LLM_TENSOR_DEC_ATTN_K, "dec.blk.%d.attn_k" },
@@ -744,6 +759,18 @@ static const std::map<llm_tensor, llm_tensor_info> LLM_TENSOR_INFOS = {
{LLM_TENSOR_OUTPUT_NORM, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_MUL}},
{LLM_TENSOR_OUTPUT_NORM_LFM2, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_MUL}},
{LLM_TENSOR_DEC_OUTPUT_NORM, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_MUL}},
{LLM_TENSOR_DEC_CROSS_ATTN_NORM_KV, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL}},
{LLM_TENSOR_DECISION_HIDDEN_NORM, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_MUL}},
{LLM_TENSOR_DECISION_PROJ_MEMORY, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_MUL_MAT}},
{LLM_TENSOR_DECISION_PROJ_QUESTION, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_MUL_MAT}},
{LLM_TENSOR_DECISION_PROJ_OPTION_QUESTION, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_MUL_MAT}},
{LLM_TENSOR_DECISION_PROJ_GLOBAL, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_MUL_MAT}},
{LLM_TENSOR_DECISION_PROJ_OPTION_CONTEXT, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_MUL_MAT}},
{LLM_TENSOR_DECISION_PROJ_OPTION_LEXICAL, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_MUL_MAT}},
{LLM_TENSOR_DECISION_OPTION_SUMMARY_NORM, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_MUL}},
{LLM_TENSOR_DECISION_FIELD_NORM, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_MUL}},
{LLM_TENSOR_DECISION_OPTION_NORM, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_MUL}},
{LLM_TENSOR_DECISION_SCALES, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_MUL}},
{LLM_TENSOR_ENC_OUTPUT_NORM, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_MUL}},
{LLM_TENSOR_ROPE_FREQS, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_ROPE}},
{LLM_TENSOR_ROPE_FACTORS_LONG, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_ROPE}},
+15
View File
@@ -45,6 +45,7 @@ enum llm_arch {
LLM_ARCH_QWEN3VLMOE,
LLM_ARCH_QWEN35,
LLM_ARCH_QWEN35MOE,
LLM_ARCH_CLEF,
LLM_ARCH_QWEN4EXP,
LLM_ARCH_PHI2,
LLM_ARCH_PHI3,
@@ -413,6 +414,8 @@ enum llm_kv {
LLM_KV_CLASSIFIER_POOLING_TYPE,
LLM_KV_DECISION_BLOCK_COUNT,
LLM_KV_DECISION_ROUTING_BLOCK_COUNT,
LLM_KV_DECISION_HEAD_COUNT,
LLM_KV_TARGET_LAYERS,
LLM_KV_TARGET_HIDDEN_SIZE,
@@ -646,6 +649,18 @@ enum llm_tensor {
LLM_TENSOR_DEC_FFN_DOWN,
LLM_TENSOR_DEC_FFN_UP,
LLM_TENSOR_DEC_OUTPUT_NORM,
LLM_TENSOR_DEC_CROSS_ATTN_NORM_KV,
LLM_TENSOR_DECISION_HIDDEN_NORM,
LLM_TENSOR_DECISION_PROJ_MEMORY,
LLM_TENSOR_DECISION_PROJ_QUESTION,
LLM_TENSOR_DECISION_PROJ_OPTION_QUESTION,
LLM_TENSOR_DECISION_PROJ_GLOBAL,
LLM_TENSOR_DECISION_PROJ_OPTION_CONTEXT,
LLM_TENSOR_DECISION_PROJ_OPTION_LEXICAL,
LLM_TENSOR_DECISION_OPTION_SUMMARY_NORM,
LLM_TENSOR_DECISION_FIELD_NORM,
LLM_TENSOR_DECISION_OPTION_NORM,
LLM_TENSOR_DECISION_SCALES,
LLM_TENSOR_ENC_ATTN_NORM,
LLM_TENSOR_ENC_ATTN_Q,
LLM_TENSOR_ENC_ATTN_K,
+31
View File
@@ -168,6 +168,16 @@ bool llama_batch_allocr::init(
}
}
for (int32_t i = 0; i < n_tok; ++i) {
if (batch_inp.tokens[i].decision_order != 0) {
decision_order.resize(n_tok, 0);
break;
}
}
for (size_t i = 0; i < decision_order.size(); ++i) {
decision_order[i] = batch_inp.tokens[i].decision_order;
}
//
// set up the internal llama_batch to point to our owned arrays
//
@@ -765,6 +775,7 @@ void llama_batch_allocr::clear() {
seq_id .clear();
seq_id_unq .clear();
output .clear();
decision_order.clear();
for (auto & cur : seq_pos) {
cur.clear();
@@ -826,6 +837,10 @@ llama_ubatch llama_batch_allocr::ubatch_add(const std::vector<int32_t> & idxs, u
udata->n_seq_id[i] = batch.n_seq_id[idxs[i]];
udata->output[i] = batch.logits[idxs[i]];
if (!decision_order.empty()) {
udata->decision_order.push_back(decision_order[idxs[i]]);
}
for (int s = 0; s < udata->n_seq_id[i]; ++s) {
const llama_seq_id seq_id = batch.seq_id[idxs[i]][s];
@@ -870,6 +885,10 @@ llama_ubatch llama_batch_allocr::ubatch_add(const std::vector<int32_t> & idxs, u
/*.data =*/ std::move(udata),
};
if (!res.data->decision_order.empty()) {
res.decision_order = res.data->decision_order.data();
}
if (debug > 0) {
LLAMA_LOG_DEBUG("%s: added ubatch to split:\n", __func__);
@@ -1175,6 +1194,14 @@ bool llama_batch_ext::set_output(int32_t idx, bool output_last) {
return true;
}
bool llama_batch_ext::set_decision_order(int32_t idx, int32_t order) {
if (idx < 0 || idx >= (int32_t) tokens.size()) {
return false;
}
tokens[idx].decision_order = order;
return true;
}
// llama_batch_ext C API
llama_batch_ext * llama_batch_ext_init(llama_context * ctx) {
@@ -1243,6 +1270,10 @@ bool llama_batch_ext_set_output_logits(llama_batch_ext * batch, int32_t idx, boo
return batch->set_output(idx, value);
}
bool llama_batch_ext_set_decision_order(llama_batch_ext * batch, int32_t idx, int32_t order) {
return batch->set_decision_order(idx, order);
}
// llama_batch_compat
void llama_batch_compat::init(llama_batch_ext & dst, const llama_batch & batch_inp, size_t n_embd_row) {
+7
View File
@@ -63,12 +63,16 @@ struct llama_ubatch {
std::vector<int32_t> seq_idx;
std::vector<int8_t> output;
std::vector<int32_t> batch_idxs; // original batch index for each token
std::vector<int32_t> decision_order;
std::vector<llama_seq_id> seq_id_data;
};
// the llama_ubatch pointers above point to this data if set. otherwise - point to external non-owning data
std::shared_ptr<data_t> data;
// [n_tokens], see llama_batch_ext_set_decision_order(), NULL if no entry has one
int32_t * decision_order = nullptr;
};
struct llama_hparams;
@@ -96,6 +100,7 @@ struct llama_batch_ext {
bool has_embd = false; // whether embd_off is set
size_t embd_off = 0; // index offset in the embd array
bool output = false; // TODO: have dedicated output flags
int32_t decision_order = 0; // see llama_batch_ext_set_decision_order()
std::unordered_set<llama_seq_id> seq_ids;
std::array<llama_pos, GGML_MROPE_SECTIONS> pos = {0, 0, 0, 0};
};
@@ -125,6 +130,7 @@ struct llama_batch_ext {
bool set_token_embd(int32_t idx, llama_embd embd_in);
bool set_token_pos(int32_t idx, const llama_pos * pos_in);
bool set_output(int32_t idx, bool output_last);
bool set_decision_order(int32_t idx, int32_t order);
};
// a helper for sanitizing, fulfilling and splitting a batch
@@ -202,6 +208,7 @@ private:
std::vector<llama_seq_id> seq_id_unq;
std::vector<int32_t> seq_idx;
std::vector<int8_t> output;
std::vector<int32_t> decision_order; // empty if no entry has one
using pos_set_t = std::set<llama_pos>;
using seq_cpl_t = std::vector<bool>;
+1
View File
@@ -2406,6 +2406,7 @@ uint32_t llama_context::graph_max_nodes(uint32_t n_tokens) const {
model.arch == LLM_ARCH_BAILINGMOE3 ||
model.arch == LLM_ARCH_QWEN35 ||
model.arch == LLM_ARCH_QWEN35MOE ||
model.arch == LLM_ARCH_CLEF ||
model.arch == LLM_ARCH_QWEN4EXP ||
model.arch == LLM_ARCH_DEEPSEEK4 ||
(model.arch == LLM_ARCH_DFLASH && model.hparams.dsv4_hc_mult > 0) ||
+13
View File
@@ -100,6 +100,19 @@ LLAMA_API void llama_set_embeddings_nextn(struct llama_context * ctx, bool value
// chain multiple trained NextN heads. Default 0 (first head).
LLAMA_API void llama_set_nextn_layer_offset(struct llama_context * ctx, int32_t offset);
// Marks the entries that a joint decision head (clef) reads, the default is 0
// A run of entries with the same value is one span, spans must be separated by entries with value 0
// An option belongs to the last question before it
enum llama_decision_order {
LLAMA_DECISION_ORDER_NONE = 0, // not read by the head
LLAMA_DECISION_ORDER_QUESTION_NOUL = 1, // text of a question
LLAMA_DECISION_ORDER_QUESTION_CHOICE = 2,
LLAMA_DECISION_ORDER_QUESTION_SCORE = 3,
LLAMA_DECISION_ORDER_OPTION = 4, // text of an option
};
// The embeddings output has one value per entry: row i is the score of option i
LLAMA_API bool llama_batch_ext_set_decision_order(struct llama_batch_ext * batch, int32_t idx, int32_t order);
// mirrors:
// LLAMA_API float * llama_get_embeddings(struct llama_context * ctx);
LLAMA_API float * llama_get_embeddings_nextn(struct llama_context * ctx);
+1
View File
@@ -20,6 +20,7 @@ bool llama_model_saver_supports_arch(llm_arch arch) {
case LLM_ARCH_T5:
case LLM_ARCH_APERTUS:
case LLM_ARCH_STEP35:
case LLM_ARCH_CLEF: // the head tensors are not saved
return false;
default:
return true;
+12 -4
View File
@@ -326,6 +326,8 @@ static llama_model * llama_model_mapping(llm_arch arch, const llama_model_params
return new llama_model_qwen35(params);
case LLM_ARCH_QWEN35MOE:
return new llama_model_qwen35moe(params);
case LLM_ARCH_CLEF:
return new llama_model_clef(params);
case LLM_ARCH_QWEN4EXP:
return new llama_model_qwen4exp(params);
case LLM_ARCH_MISTRAL3:
@@ -607,7 +609,7 @@ struct ggml_backend_meta_split_state llama_meta_device_get_split_state(const str
auto get_split_segments = [&](int axis, uint32_t il) -> std::vector<std::pair<int64_t, uint32_t>> {
// TODO: clarify why this is necessary specifically for these models
// TODO: deduplicate condition [TAG_SPLIT_QGATE_QWEN]
if (ud->model->arch == LLM_ARCH_QWEN3NEXT || ud->model->arch == LLM_ARCH_QWEN35 || ud->model->arch == LLM_ARCH_QWEN35MOE ||
if (ud->model->arch == LLM_ARCH_QWEN3NEXT || ud->model->arch == LLM_ARCH_QWEN35 || ud->model->arch == LLM_ARCH_QWEN35MOE || ud->model->arch == LLM_ARCH_CLEF ||
ud->model->arch == LLM_ARCH_QWEN4EXP) {
// fused full attention layers with Q gate tensors that need n_embd doubled:
@@ -762,7 +764,7 @@ struct ggml_backend_meta_split_state llama_meta_device_get_split_state(const str
GGML_ASSERT(segments.size() == 1);
// some models have Q gate tensors, for those cases the granularity needs to be doubled:
// TODO: deduplicate condition [TAG_SPLIT_QGATE_QWEN]
if (ud->model->arch == LLM_ARCH_QWEN3NEXT || ud->model->arch == LLM_ARCH_QWEN35 || ud->model->arch == LLM_ARCH_QWEN35MOE ||
if (ud->model->arch == LLM_ARCH_QWEN3NEXT || ud->model->arch == LLM_ARCH_QWEN35 || ud->model->arch == LLM_ARCH_QWEN35MOE || ud->model->arch == LLM_ARCH_CLEF ||
ud->model->arch == LLM_ARCH_QWEN4EXP) {
return {std::lcm(2*n_embd_q, blck_size_perf)};
}
@@ -795,7 +797,7 @@ struct ggml_backend_meta_split_state llama_meta_device_get_split_state(const str
if (std::regex_match(tensor_name, pattern_qkv_weight) || std::regex_match(tensor_name, pattern_qkv_bias)) {
// fused full attention layers need Q gate tensors handled like above:
// TODO: deduplicate condition [TAG_SPLIT_QGATE_QWEN]
if (ud->model->arch == LLM_ARCH_QWEN3NEXT || ud->model->arch == LLM_ARCH_QWEN35 || ud->model->arch == LLM_ARCH_QWEN35MOE ||
if (ud->model->arch == LLM_ARCH_QWEN3NEXT || ud->model->arch == LLM_ARCH_QWEN35 || ud->model->arch == LLM_ARCH_QWEN35MOE || ud->model->arch == LLM_ARCH_CLEF ||
ud->model->arch == LLM_ARCH_QWEN4EXP) {
return {std::lcm(2*n_embd_q, blck_size_perf), granularity_kv};
}
@@ -1961,7 +1963,11 @@ bool llama_model_base::load_tensors(llama_model_loader & ml) {
}
ggml_tensor * llama_model_base::create_tensor(llama_model_loader & ml, const LLM_TN_IMPL & tn, const std::initializer_list<int64_t> & ne, int flags) {
const buft_list_t * buft_list_layer = tn.bid == -1 ? nullptr : pimpl->dev_layer.at(tn.bid).buft_list;
const buft_list_t * buft_list_layer = nullptr;
if (tn.bid != -1) {
// blocks that are not model layers (e.g. the blocks of a head) go with the output
buft_list_layer = (size_t) tn.bid < pimpl->dev_layer.size() ? pimpl->dev_layer.at(tn.bid).buft_list : pimpl->dev_output.buft_list;
}
return ml.create_tensor(
hparams, &pimpl->cpu_buft_list, pimpl->dev_input.buft_list, pimpl->dev_output.buft_list, buft_list_layer,
tn, ne, flags);
@@ -2369,6 +2375,7 @@ llama_memory_i * llama_model::create_memory(const llama_memory_params & params,
case LLM_ARCH_LLADA:
case LLM_ARCH_LLADA_MOE:
case LLM_ARCH_RND1:
case LLM_ARCH_CLEF:
{
res = nullptr;
} break;
@@ -3200,6 +3207,7 @@ llama_rope_type llama_model_rope_type(const llama_model * model) {
case LLM_ARCH_QWEN3VLMOE:
case LLM_ARCH_QWEN35:
case LLM_ARCH_QWEN35MOE:
case LLM_ARCH_CLEF:
case LLM_ARCH_QWEN4EXP:
case LLM_ARCH_QWEN3TTS:
return LLAMA_ROPE_TYPE_IMROPE;
+612
View File
@@ -0,0 +1,612 @@
#include "models.h"
#include <cmath>
// torch.nn.LayerNorm default
static const float CLEF_HEAD_NORM_EPS = 1e-5f;
void llama_model_clef::load_arch_hparams(llama_model_loader & ml) {
llama_model_qwen35::load_arch_hparams(ml);
ml.get_key(LLM_KV_DECISION_ROUTING_BLOCK_COUNT, n_layer_routing);
ml.get_key(LLM_KV_DECISION_BLOCK_COUNT, n_layer_joint);
ml.get_key(LLM_KV_DECISION_HEAD_COUNT, n_head_decision);
if (n_head_decision == 0) {
throw std::runtime_error("invalid number of heads in the decision head");
}
hparams.f_norm_eps = CLEF_HEAD_NORM_EPS;
// the output is one score per token, see llama_batch_ext_set_decision_order()
hparams.n_embd_out_impl = 1;
}
void llama_model_clef::load_arch_tensors(llama_model_loader & ml) {
llama_model_qwen35::load_arch_tensors(ml);
LLAMA_LOAD_LOCALS;
const auto * w_memory = ml.get_weight(tn(LLM_TENSOR_DECISION_PROJ_MEMORY, "weight").str().c_str());
const auto * w_ffn = ml.get_weight(tn(LLM_TENSOR_DEC_FFN_UP, "weight", 0).str().c_str());
if (w_memory == nullptr || w_ffn == nullptr) {
throw std::runtime_error("the decision head is missing");
}
const int64_t n_embd_h = w_memory->tensor->ne[1];
const int64_t n_ff_h = w_ffn->tensor->ne[1];
if (n_embd_h % n_head_decision != 0) {
throw std::runtime_error("invalid width of the decision head");
}
auto load_norm = [&](norm & n, llm_tensor type, int64_t size, int il = -1) {
n.w = il < 0 ? create_tensor(tn(type, "weight"), {size}, 0) : create_tensor(tn(type, "weight", il), {size}, 0);
n.b = il < 0 ? create_tensor(tn(type, "bias"), {size}, 0) : create_tensor(tn(type, "bias", il), {size}, 0);
};
auto load_attn = [&](attn & a, llm_tensor q, llm_tensor k, llm_tensor v, llm_tensor o, int il) {
a.wq = create_tensor(tn(q, "weight", il), {n_embd_h, n_embd_h}, 0);
a.bq = create_tensor(tn(q, "bias", il), {n_embd_h}, 0);
a.wk = create_tensor(tn(k, "weight", il), {n_embd_h, n_embd_h}, 0);
a.bk = create_tensor(tn(k, "bias", il), {n_embd_h}, 0);
a.wv = create_tensor(tn(v, "weight", il), {n_embd_h, n_embd_h}, 0);
a.bv = create_tensor(tn(v, "bias", il), {n_embd_h}, 0);
a.wo = create_tensor(tn(o, "weight", il), {n_embd_h, n_embd_h}, 0);
a.bo = create_tensor(tn(o, "bias", il), {n_embd_h}, 0);
};
head_layers.resize(n_layer_routing + n_layer_joint);
for (int il = 0; il < (int) head_layers.size(); ++il) {
auto & layer = head_layers[il];
if (il < (int) n_layer_routing) {
load_norm(layer.cross_norm_kv, LLM_TENSOR_DEC_CROSS_ATTN_NORM_KV, n_embd_h, il);
} else {
load_norm(layer.self_norm, LLM_TENSOR_DEC_ATTN_NORM, n_embd_h, il);
load_attn(layer.self_attn, LLM_TENSOR_DEC_ATTN_Q, LLM_TENSOR_DEC_ATTN_K, LLM_TENSOR_DEC_ATTN_V, LLM_TENSOR_DEC_ATTN_OUT, il);
}
load_norm(layer.cross_norm, LLM_TENSOR_DEC_CROSS_ATTN_NORM, n_embd_h, il);
load_attn(layer.cross_attn, LLM_TENSOR_DEC_CROSS_ATTN_Q, LLM_TENSOR_DEC_CROSS_ATTN_K, LLM_TENSOR_DEC_CROSS_ATTN_V, LLM_TENSOR_DEC_CROSS_ATTN_OUT, il);
load_norm(layer.ffn_norm, LLM_TENSOR_DEC_FFN_NORM, n_embd_h, il);
layer.ffn_up = create_tensor(tn(LLM_TENSOR_DEC_FFN_UP, "weight", il), {n_embd_h, n_ff_h}, 0);
layer.ffn_up_b = create_tensor(tn(LLM_TENSOR_DEC_FFN_UP, "bias", il), {n_ff_h}, 0);
layer.ffn_down = create_tensor(tn(LLM_TENSOR_DEC_FFN_DOWN, "weight", il), {n_ff_h, n_embd_h}, 0);
layer.ffn_down_b = create_tensor(tn(LLM_TENSOR_DEC_FFN_DOWN, "bias", il), {n_embd_h}, 0);
}
load_norm(hidden_norm, LLM_TENSOR_DECISION_HIDDEN_NORM, n_embd);
load_norm(option_summary_norm, LLM_TENSOR_DECISION_OPTION_SUMMARY_NORM, n_embd_h);
load_norm(field_norm, LLM_TENSOR_DECISION_FIELD_NORM, n_embd_h);
load_norm(option_norm, LLM_TENSOR_DECISION_OPTION_NORM, n_embd_h);
proj_memory = create_tensor(tn(LLM_TENSOR_DECISION_PROJ_MEMORY, "weight"), {n_embd, n_embd_h}, 0);
proj_question = create_tensor(tn(LLM_TENSOR_DECISION_PROJ_QUESTION, "weight"), {n_embd, n_embd_h}, 0);
proj_option_question = create_tensor(tn(LLM_TENSOR_DECISION_PROJ_OPTION_QUESTION, "weight"), {n_embd, n_embd_h}, 0);
proj_global = create_tensor(tn(LLM_TENSOR_DECISION_PROJ_GLOBAL, "weight"), {n_embd, n_embd_h}, 0);
proj_option_context = create_tensor(tn(LLM_TENSOR_DECISION_PROJ_OPTION_CONTEXT, "weight"), {n_embd, n_embd_h}, 0);
proj_option_lexical = create_tensor(tn(LLM_TENSOR_DECISION_PROJ_OPTION_LEXICAL, "weight"), {n_embd, n_embd_h}, 0);
scales = create_tensor(tn(LLM_TENSOR_DECISION_SCALES), {3}, 0);
type_embd = create_tensor(tn(LLM_TENSOR_TOKEN_TYPES, "weight"), {n_embd_h, 3}, 0);
cls = create_tensor(tn(LLM_TENSOR_CLS, "weight"), {4 * n_embd_h, n_embd_h}, 0);
cls_b = create_tensor(tn(LLM_TENSOR_CLS, "bias"), {n_embd_h}, 0);
cls_out = create_tensor(tn(LLM_TENSOR_CLS_OUT, "weight"), {n_embd_h, 1}, 0);
cls_out_b = create_tensor(tn(LLM_TENSOR_CLS_OUT, "bias"), {1}, 0);
}
std::unique_ptr<llm_graph_context> llama_model_clef::build_arch_graph(const llm_graph_params & params) const {
return std::make_unique<graph>(*this, params);
}
// spans read by the head, [start, end) in ubatch token indices
struct clef_spans {
struct question {
int32_t type; // noul, choice, score
int32_t start;
int32_t end;
};
struct option {
int32_t question;
int32_t start;
int32_t end;
};
std::vector<question> questions;
std::vector<option> options;
bool valid = false; // at least one question, and each question has an option
};
// see llama_batch_ext_set_decision_order()
static clef_spans clef_get_spans(const llama_ubatch & ubatch) {
const int32_t ORDER_QUESTION_FIRST = 1;
const int32_t ORDER_QUESTION_LAST = 3;
const int32_t ORDER_OPTION = 4;
clef_spans res;
if (ubatch.decision_order == nullptr) {
return res;
}
const int32_t n_tokens = ubatch.n_tokens;
std::vector<bool> has_option;
for (int32_t i = 0; i < n_tokens;) {
const int32_t order = ubatch.decision_order[i];
int32_t end = i + 1;
while (end < n_tokens && ubatch.decision_order[end] == order) {
end++;
}
if (order >= ORDER_QUESTION_FIRST && order <= ORDER_QUESTION_LAST) {
res.questions.push_back({ order - ORDER_QUESTION_FIRST, i, end });
has_option.push_back(false);
} else if (order == ORDER_OPTION) {
if (res.questions.empty()) {
return res;
}
res.options.push_back({ (int32_t) res.questions.size() - 1, i, end });
has_option.back() = true;
} else if (order != 0) {
return res;
}
i = end;
}
res.valid = !res.questions.empty() && std::find(has_option.begin(), has_option.end(), false) == has_option.end();
return res;
}
// pooling matrices and indices computed from the spans
class llama_model_clef::input_decision : public llm_graph_input_i {
public:
input_decision(const llama_ubatch & ubatch) : n_tokens(ubatch.n_tokens) {
const auto spans = clef_get_spans(ubatch);
if (spans.valid) {
n_questions = spans.questions.size();
n_options = spans.options.size();
}
}
void set_input(const llama_ubatch * ubatch) override {
if (tokens == nullptr) {
return;
}
GGML_ASSERT(ubatch->token);
ggml_backend_tensor_set(tokens, ubatch->token, 0, n_tokens * sizeof(llama_token));
const auto spans = clef_get_spans(*ubatch);
GGML_ASSERT(spans.valid && spans.questions.size() == n_questions && spans.options.size() == n_options);
std::vector<float> pool_q_data(n_tokens * n_questions, 0.0f);
std::vector<int32_t> types(n_questions);
for (size_t i = 0; i < n_questions; i++) {
const auto & q = spans.questions[i];
for (int32_t j = q.start; j < q.end; j++) {
pool_q_data[i * n_tokens + j] = 1.0f / (q.end - q.start);
}
types[i] = q.type;
}
std::vector<float> pool_o_data(n_tokens * n_options, 0.0f);
std::vector<int32_t> owner(n_options);
std::vector<float> mask(n_options * n_questions, -INFINITY);
for (size_t i = 0; i < n_options; i++) {
const auto & o = spans.options[i];
for (int32_t j = o.start; j < o.end; j++) {
pool_o_data[i * n_tokens + j] = 1.0f / (o.end - o.start);
}
owner[i] = o.question;
mask[o.question * n_options + i] = 0.0f;
}
ggml_backend_tensor_set(pool_q, pool_q_data.data(), 0, ggml_nbytes(pool_q));
ggml_backend_tensor_set(pool_o, pool_o_data.data(), 0, ggml_nbytes(pool_o));
ggml_backend_tensor_set(question_type, types.data(), 0, ggml_nbytes(question_type));
ggml_backend_tensor_set(option_question, owner.data(), 0, ggml_nbytes(option_question));
ggml_backend_tensor_set(option_mask, mask.data(), 0, ggml_nbytes(option_mask));
}
bool can_reuse(const llm_graph_params & params) override {
// the values are computed again in set_input(), only the shapes must match
const auto spans = clef_get_spans(params.ubatch);
const size_t n_q = spans.valid ? spans.questions.size() : 0;
const size_t n_o = spans.valid ? spans.options.size() : 0;
return n_q == n_questions && n_o == n_options;
}
ggml_tensor * tokens = nullptr; // I32 [n_tokens]
ggml_tensor * pool_q = nullptr; // F32 [n_tokens, n_questions], mean over the span of the question
ggml_tensor * pool_o = nullptr; // F32 [n_tokens, n_options], mean over the span of the option
ggml_tensor * question_type = nullptr; // I32 [n_questions]
ggml_tensor * option_question = nullptr; // I32 [n_options]
ggml_tensor * option_mask = nullptr; // F32 [n_options, n_questions], 0 if the option belongs to the question, else -inf
const int64_t n_tokens;
size_t n_questions = 0; // 0 if the batch has no valid spans
size_t n_options = 0;
};
// the backbone is copied from llama_model_qwen35::graph, without the memory module
llama_model_clef::graph::graph(const llama_model & model_base, const llm_graph_params & params) :
llm_build_delta_net_base(params), model(static_cast<const llama_model_clef &>(model_base)) {
const int64_t n_embd_head = hparams.n_embd_head_v();
GGML_ASSERT(n_embd_head == hparams.n_embd_head_k());
int sections[4];
std::copy(std::begin(hparams.rope_sections), std::begin(hparams.rope_sections) + 4, sections);
ggml_tensor * cur;
ggml_tensor * inpL;
inpL = build_inp_embd(model.tok_embd);
ggml_tensor * inp_pos = build_inp_pos();
auto * inp_attn = build_attn_inp_causal();
auto inp_decision_ptr = std::make_unique<input_decision>(ubatch);
auto * inp_decision = inp_decision_ptr.get();
res->add_input(std::move(inp_decision_ptr));
for (int il = 0; il < n_layer; ++il) {
ggml_tensor * inpSA = inpL;
cur = build_norm(inpL, model.layers[il].attn_norm, nullptr, LLM_NORM_RMS, il);
cb(cur, "attn_norm", il);
if (hparams.is_recr(il)) {
cur = build_layer_attn_linear(cur, il);
} else {
cur = build_layer_attn(inp_attn, cur, inp_pos, sections, il);
}
cur = ggml_add(ctx0, cur, inpSA);
cb(cur, "attn_residual", il);
ggml_tensor * ffn_residual = cur;
cur = build_norm(cur, model.layers[il].attn_post_norm, nullptr, LLM_NORM_RMS, il);
cb(cur, "attn_post_norm", il);
cur = build_ffn(cur,
model.layers[il].ffn_up, NULL, model.layers[il].ffn_up_s,
model.layers[il].ffn_gate, NULL, model.layers[il].ffn_gate_s,
model.layers[il].ffn_down, NULL, model.layers[il].ffn_down_s,
NULL,
LLM_FFN_SILU, LLM_FFN_PAR, il);
cb(cur, "ffn_out", il);
cur = ggml_add(ctx0, cur, ffn_residual);
cb(cur, "l_out", il);
inpL = cur;
}
cur = build_norm(inpL, model.output_norm, nullptr, LLM_NORM_RMS, -1);
cb(cur, "result_norm", -1);
// TODO: support multiple sequences
const bool is_single_seq = ubatch.n_seqs_unq == 1;
if (inp_decision->n_questions > 0 && is_single_seq) {
cur = build_head(cur, inp_decision);
// row i of the output is the score of option i
cur = ggml_pad(ctx0, cur, 0, n_tokens - cur->ne[1], 0, 0);
} else {
if (ubatch.decision_order != nullptr) {
LLAMA_LOG_ERROR("%s: %s, the head is not evaluated\n", __func__, is_single_seq ? "invalid decision order" : "more than one sequence in the batch");
}
cur = ggml_scale(ctx0, ggml_cont(ctx0, ggml_view_2d(ctx0, cur, 1, n_tokens, cur->nb[1], 0)), 0.0f);
}
cb(cur, "result_decision", -1);
res->t_embd = cur;
ggml_build_forward_expand(gf, cur);
}
// same as build_attn_inp_no_cache(), but the mask is causal even if the batch is processed by the encoder path
llm_graph_input_attn_no_cache * llama_model_clef::graph::build_attn_inp_causal() {
llama_cparams cparams_causal = cparams;
cparams_causal.causal_attn = true;
auto inp = std::make_unique<llm_graph_input_attn_no_cache>(hparams, cparams_causal);
const auto type_mask = cparams.flash_attn ? GGML_TYPE_F16 : GGML_TYPE_F32;
inp->self_kq_mask = ggml_new_tensor_4d(ctx0, type_mask, n_tokens, n_tokens, 1, 1);
ggml_set_input(inp->self_kq_mask);
cb(inp->self_kq_mask, "self_kq_mask", -1);
inp->self_kq_mask_cnv = inp->self_kq_mask;
return (llm_graph_input_attn_no_cache *) res->add_input(std::move(inp));
}
ggml_tensor * llama_model_clef::graph::build_layer_attn(
llm_graph_input_attn_no_cache * inp,
ggml_tensor * cur,
ggml_tensor * inp_pos,
int * sections,
int il) {
const int64_t n_embd_head = hparams.n_embd_head_v();
// the Q projection outputs query + gate
auto [Qcur_full, Kcur, Vcur] = build_qkv(model.layers[il], cur,
n_embd_head * 2, n_head,
n_embd_head, n_head_kv,
n_embd_head, n_head_kv,
il, false);
ggml_tensor * Qcur = ggml_view_3d(ctx0, Qcur_full, n_embd_head, n_head, n_tokens,
ggml_element_size(Qcur_full) * n_embd_head * 2,
ggml_element_size(Qcur_full) * n_embd_head * 2 * n_head, 0);
Qcur = build_norm(Qcur, model.layers[il].attn_q_norm, nullptr, LLM_NORM_RMS, il);
cb(Qcur, "Qcur_normed", il);
Kcur = ggml_reshape_3d(ctx0, Kcur, n_embd_head, n_head_kv, n_tokens);
Kcur = build_norm(Kcur, model.layers[il].attn_k_norm, nullptr, LLM_NORM_RMS, il);
cb(Kcur, "Kcur_normed", il);
ggml_tensor * gate = ggml_view_3d(ctx0, Qcur_full, n_embd_head, n_head, n_tokens,
ggml_element_size(Qcur_full) * n_embd_head * 2,
ggml_element_size(Qcur_full) * n_embd_head * 2 * n_head,
ggml_element_size(Qcur_full) * n_embd_head);
gate = ggml_cont_2d(ctx0, gate, n_embd_head * n_head, n_tokens);
Vcur = ggml_reshape_3d(ctx0, Vcur, n_embd_head, n_head_kv, n_tokens);
Qcur = ggml_rope_multi(
ctx0, Qcur, inp_pos, nullptr,
n_rot, sections, rope_type, n_ctx_orig, freq_base, freq_scale,
ext_factor, attn_factor, beta_fast, beta_slow
);
Kcur = ggml_rope_multi(
ctx0, Kcur, inp_pos, nullptr,
n_rot, sections, rope_type, n_ctx_orig, freq_base, freq_scale,
ext_factor, attn_factor, beta_fast, beta_slow
);
const float kq_scale = hparams.f_attention_scale == 0.0f ? 1.0f / sqrtf(float(n_embd_head)) : hparams.f_attention_scale;
cur = build_attn(inp,
nullptr, nullptr, nullptr,
Qcur, Kcur, Vcur, nullptr, nullptr, nullptr, kq_scale, il);
cb(cur, "attn_pregate", il);
cur = ggml_mul(ctx0, cur, ggml_sigmoid(ctx0, gate));
cur = build_lora_mm(model.layers[il].wo, cur, model.layers[il].wo_s);
cb(cur, "attn_output", il);
return cur;
}
// gated delta net over the whole batch, the conv and recurrent states start from zero and are not kept
ggml_tensor * llama_model_clef::graph::build_layer_attn_linear(
ggml_tensor * cur,
int il) {
const auto & layer = model.layers[il];
const int64_t d_inner = hparams.ssm_d_inner;
const int64_t head_k_dim = hparams.ssm_d_state;
const int64_t num_k_heads = hparams.ssm_n_group;
const int64_t num_v_heads = hparams.ssm_dt_rank;
const int64_t head_v_dim = d_inner / num_v_heads;
const int64_t n_seqs = 1;
ggml_tensor * qkv_mixed = build_lora_mm(layer.wqkv, cur, layer.wqkv_s);
qkv_mixed = ggml_reshape_3d(ctx0, qkv_mixed, qkv_mixed->ne[0], n_tokens, n_seqs);
ggml_tensor * z = build_lora_mm(layer.wqkv_gate, cur, layer.wqkv_gate_s);
ggml_tensor * beta = build_lora_mm(layer.ssm_beta, cur, layer.ssm_beta_s);
beta = ggml_reshape_4d(ctx0, beta, 1, num_v_heads, n_tokens, n_seqs);
beta = ggml_sigmoid(ctx0, beta);
ggml_tensor * alpha = build_lora_mm(layer.ssm_alpha, cur, layer.ssm_alpha_s);
alpha = ggml_reshape_3d(ctx0, alpha, num_v_heads, n_tokens, n_seqs);
ggml_tensor * gate = ggml_softplus(ctx0, ggml_add(ctx0, alpha, layer.ssm_dt));
gate = ggml_mul(ctx0, gate, layer.ssm_a);
gate = ggml_reshape_4d(ctx0, gate, 1, num_v_heads, n_tokens, n_seqs);
const int64_t conv_kernel_size = layer.ssm_conv1d->ne[0];
const int64_t conv_channels = d_inner + 2 * num_k_heads * head_k_dim;
ggml_tensor * conv_states = ggml_fill(ctx0, ggml_new_tensor_3d(ctx0, GGML_TYPE_F32, conv_kernel_size - 1, conv_channels, n_seqs), 0.0f);
ggml_tensor * conv_input = ggml_concat(ctx0, conv_states, ggml_transpose(ctx0, qkv_mixed), 0);
ggml_tensor * conv_out = ggml_silu(ctx0, ggml_ssm_conv(ctx0, conv_input, layer.ssm_conv1d));
cb(conv_out, "conv_output_silu", il);
const int64_t qkv_dim = head_k_dim * num_k_heads * 2 + head_v_dim * num_v_heads;
const int64_t nb1_qkv = ggml_row_size(conv_out->type, qkv_dim);
ggml_tensor * q_conv = ggml_view_4d(ctx0, conv_out, head_k_dim, num_k_heads, n_tokens, n_seqs,
ggml_row_size(conv_out->type, head_k_dim), nb1_qkv, nb1_qkv * n_tokens,
0);
ggml_tensor * k_conv = ggml_view_4d(ctx0, conv_out, head_k_dim, num_k_heads, n_tokens, n_seqs,
ggml_row_size(conv_out->type, head_k_dim), nb1_qkv, nb1_qkv * n_tokens,
head_k_dim * num_k_heads * ggml_element_size(conv_out));
ggml_tensor * v_conv = ggml_view_4d(ctx0, conv_out, head_v_dim, num_v_heads, n_tokens, n_seqs,
ggml_row_size(conv_out->type, head_v_dim), nb1_qkv, nb1_qkv * n_tokens,
ggml_row_size(conv_out->type, 2 * head_k_dim * num_k_heads));
q_conv = build_gdn_l2_norm(ctx0, q_conv, hparams.f_norm_rms_eps);
k_conv = build_gdn_l2_norm(ctx0, k_conv, hparams.f_norm_rms_eps);
// note: need explicit repeat only if we are not using the fused GDN
if (num_k_heads != num_v_heads && (!cparams.fused_gdn_ar || !cparams.fused_gdn_ch)) {
GGML_ASSERT(num_v_heads % num_k_heads == 0);
q_conv = ggml_repeat_4d(ctx0, q_conv, head_k_dim, num_v_heads, n_tokens, n_seqs);
k_conv = ggml_repeat_4d(ctx0, k_conv, head_k_dim, num_v_heads, n_tokens, n_seqs);
}
ggml_tensor * state = ggml_fill(ctx0, ggml_new_tensor_4d(ctx0, GGML_TYPE_F32, head_v_dim, head_v_dim, num_v_heads, n_seqs), 0.0f);
ggml_tensor * output = build_delta_net(q_conv, k_conv, v_conv, gate, beta, state, il).first;
cb(output, "attn_output", il);
// gated normalization
ggml_tensor * z_4d = ggml_reshape_4d(ctx0, z, head_v_dim, num_v_heads, n_tokens, n_seqs);
output = build_norm(output, layer.ssm_norm, nullptr, LLM_NORM_RMS, il);
output = ggml_mul(ctx0, output, ggml_silu(ctx0, z_4d));
output = ggml_reshape_3d(ctx0, output, head_v_dim * num_v_heads, n_tokens, n_seqs);
cur = build_lora_mm(layer.ssm_out, output, layer.ssm_out_s);
cb(cur, "linear_attn_out", il);
return ggml_reshape_2d(ctx0, cur, n_embd, n_tokens);
}
ggml_tensor * llama_model_clef::graph::build_head_norm(ggml_tensor * cur, const norm & n) {
return build_norm(cur, n.w, n.b, LLM_NORM, -1);
}
// q: [n_embd_h, n_q], kv: [n_embd_h, n_kv], no mask
ggml_tensor * llama_model_clef::graph::build_head_attn(ggml_tensor * q, ggml_tensor * kv, const attn & a) {
const int64_t n_embd_h = q->ne[0];
const int64_t n_head_h = model.n_head_decision;
const int64_t d_head = n_embd_h / n_head_h;
const int64_t n_q = q->ne[1];
const int64_t n_kv = kv->ne[1];
ggml_tensor * Qcur = ggml_add(ctx0, ggml_mul_mat(ctx0, a.wq, q), a.bq);
ggml_tensor * Kcur = ggml_add(ctx0, ggml_mul_mat(ctx0, a.wk, kv), a.bk);
ggml_tensor * Vcur = ggml_add(ctx0, ggml_mul_mat(ctx0, a.wv, kv), a.bv);
Qcur = ggml_permute(ctx0, ggml_reshape_3d(ctx0, Qcur, d_head, n_head_h, n_q), 0, 2, 1, 3); // [d_head, n_q, n_head]
Kcur = ggml_permute(ctx0, ggml_reshape_3d(ctx0, Kcur, d_head, n_head_h, n_kv), 0, 2, 1, 3); // [d_head, n_kv, n_head]
Vcur = ggml_permute(ctx0, ggml_reshape_3d(ctx0, Vcur, d_head, n_head_h, n_kv), 1, 2, 0, 3); // [n_kv, d_head, n_head]
Vcur = ggml_cont(ctx0, Vcur);
ggml_tensor * kq = ggml_mul_mat(ctx0, Kcur, Qcur); // [n_kv, n_q, n_head]
kq = ggml_soft_max_ext(ctx0, kq, nullptr, 1.0f / sqrtf(float(d_head)), 0.0f);
ggml_tensor * cur = ggml_mul_mat(ctx0, Vcur, kq); // [d_head, n_q, n_head]
cur = ggml_cont_2d(ctx0, ggml_permute(ctx0, cur, 0, 2, 1, 3), n_embd_h, n_q);
return ggml_add(ctx0, ggml_mul_mat(ctx0, a.wo, cur), a.bo);
}
ggml_tensor * llama_model_clef::graph::build_head_ffn(ggml_tensor * cur, const head_layer & layer) {
cur = build_head_norm(cur, layer.ffn_norm);
cur = ggml_add(ctx0, ggml_mul_mat(ctx0, layer.ffn_up, cur), layer.ffn_up_b);
cur = ggml_gelu_erf(ctx0, cur);
return ggml_add(ctx0, ggml_mul_mat(ctx0, layer.ffn_down, cur), layer.ffn_down_b);
}
// ref: JointSchemaHead in joint_schema_model.py of the model repo
ggml_tensor * llama_model_clef::graph::build_head(ggml_tensor * hidden, input_decision * inp) {
const int64_t n_questions = inp->n_questions;
const int64_t n_options = inp->n_options;
inp->tokens = ggml_new_tensor_1d(ctx0, GGML_TYPE_I32, n_tokens);
inp->pool_q = ggml_new_tensor_2d(ctx0, GGML_TYPE_F32, n_tokens, n_questions);
inp->pool_o = ggml_new_tensor_2d(ctx0, GGML_TYPE_F32, n_tokens, n_options);
inp->question_type = ggml_new_tensor_1d(ctx0, GGML_TYPE_I32, n_questions);
inp->option_question = ggml_new_tensor_1d(ctx0, GGML_TYPE_I32, n_options);
inp->option_mask = ggml_new_tensor_2d(ctx0, GGML_TYPE_F32, n_options, n_questions);
for (ggml_tensor * t : { inp->tokens, inp->pool_q, inp->pool_o, inp->question_type, inp->option_question, inp->option_mask }) {
ggml_set_input(t);
}
hidden = build_head_norm(hidden, model.hidden_norm);
ggml_tensor * memory = ggml_mul_mat(ctx0, model.proj_memory, hidden); // [n_embd_h, n_tokens]
const int64_t n_embd_h = memory->ne[0];
// hidden state of the last token
ggml_tensor * global = ggml_view_2d(ctx0, hidden, n_embd, 1, hidden->nb[1], (n_tokens - 1) * hidden->nb[1]);
// mean of the hidden states over each span
ggml_tensor * hidden_t = ggml_cont(ctx0, ggml_transpose(ctx0, hidden));
ggml_tensor * question = ggml_mul_mat(ctx0, hidden_t, inp->pool_q); // [n_embd, n_questions]
ggml_tensor * opt_ctx = ggml_mul_mat(ctx0, hidden_t, inp->pool_o); // [n_embd, n_options]
// mean of the output embeddings of the tokens of each option
ggml_tensor * lexical = ggml_get_rows(ctx0, model.output, inp->tokens);
lexical = ggml_mul_mat(ctx0, ggml_cont(ctx0, ggml_transpose(ctx0, lexical)), inp->pool_o); // [n_embd, n_options]
// one query per option
ggml_tensor * options = ggml_mul_mat(ctx0, model.proj_option_context, opt_ctx);
options = ggml_add(ctx0, options, ggml_mul_mat(ctx0, model.proj_option_lexical, lexical));
options = ggml_add(ctx0, options, ggml_get_rows(ctx0, ggml_mul_mat(ctx0, model.proj_option_question, question), inp->option_question));
// the options read the prompt
for (uint32_t il = 0; il < model.n_layer_routing; ++il) {
const auto & layer = model.head_layers[il];
ggml_tensor * cur = build_head_attn(
build_head_norm(options, layer.cross_norm),
build_head_norm(memory, layer.cross_norm_kv),
layer.cross_attn);
options = ggml_add(ctx0, options, cur);
options = ggml_add(ctx0, options, build_head_ffn(options, layer));
}
cb(options, "decision_options", -1);
// one vector per question: its text, a summary of its options, the end of the prompt and its type
ggml_tensor * fields = ggml_mul_mat(ctx0, model.proj_question, question); // [n_embd_h, n_questions]
{
// each question weights its own options
ggml_tensor * weights = ggml_mul_mat(ctx0, options, fields); // [n_options, n_questions]
weights = ggml_scale(ctx0, weights, 1.0f / sqrtf(float(n_embd_h)));
weights = ggml_soft_max(ctx0, ggml_add(ctx0, weights, inp->option_mask));
ggml_tensor * summary = ggml_mul_mat(ctx0, ggml_cont(ctx0, ggml_transpose(ctx0, options)), weights);
fields = ggml_add(ctx0, fields, build_head_norm(summary, model.option_summary_norm));
fields = ggml_add(ctx0, fields, ggml_mul_mat(ctx0, model.proj_global, global));
fields = ggml_add(ctx0, fields, ggml_get_rows(ctx0, model.type_embd, inp->question_type));
}
// the questions read each other and the prompt
for (uint32_t il = model.n_layer_routing; il < model.head_layers.size(); ++il) {
const auto & layer = model.head_layers[il];
ggml_tensor * cur = build_head_norm(fields, layer.self_norm);
fields = ggml_add(ctx0, fields, build_head_attn(cur, cur, layer.self_attn));
cur = build_head_norm(fields, layer.cross_norm);
fields = ggml_add(ctx0, fields, build_head_attn(cur, memory, layer.cross_attn));
fields = ggml_add(ctx0, fields, build_head_ffn(fields, layer));
}
fields = build_head_norm(fields, model.field_norm);
cb(fields, "decision_fields", -1);
const float eps = 1e-12f;
// prior: the output embeddings of the option against the question and the end of the prompt
ggml_tensor * anchor = ggml_l2_norm(ctx0, ggml_add(ctx0, question, global), eps);
anchor = ggml_get_rows(ctx0, anchor, inp->option_question);
ggml_tensor * prior = ggml_sum_rows(ctx0, ggml_mul(ctx0, ggml_l2_norm(ctx0, lexical, eps), anchor)); // [1, n_options]
// joint: each option against the vector of its question
options = build_head_norm(options, model.option_norm);
ggml_tensor * field = ggml_get_rows(ctx0, fields, inp->option_question); // [n_embd_h, n_options]
ggml_tensor * cosine = ggml_sum_rows(ctx0, ggml_mul(ctx0, ggml_l2_norm(ctx0, field, eps), ggml_l2_norm(ctx0, options, eps)));
ggml_tensor * features = ggml_concat(ctx0, field, options, 0);
features = ggml_concat(ctx0, features, ggml_mul(ctx0, field, options), 0);
features = ggml_concat(ctx0, features, ggml_abs(ctx0, ggml_sub(ctx0, field, options)), 0);
ggml_tensor * residual = ggml_add(ctx0, ggml_mul_mat(ctx0, model.cls, features), model.cls_b);
residual = ggml_gelu_erf(ctx0, residual);
residual = ggml_add(ctx0, ggml_mul_mat(ctx0, model.cls_out, residual), model.cls_out_b); // [1, n_options]
auto scale = [&](int i) {
return ggml_view_1d(ctx0, model.scales, 1, i * ggml_element_size(model.scales));
};
ggml_tensor * joint = ggml_add(ctx0, ggml_mul(ctx0, cosine, scale(1)), residual);
return ggml_add(ctx0, ggml_mul(ctx0, prior, scale(0)), ggml_mul(ctx0, joint, scale(2)));
}
+89
View File
@@ -2391,6 +2391,95 @@ struct llama_model_qwen35 : public llama_model_base {
};
// Qwen3.5 backbone evaluated without memory, with a joint decision head on top of its hidden states
struct llama_model_clef : public llama_model_qwen35 {
llama_model_clef(const struct llama_model_params & params) : llama_model_qwen35(params) {}
void load_arch_hparams(llama_model_loader & ml) override;
void load_arch_tensors(llama_model_loader & ml) override;
struct attn {
ggml_tensor * wq = nullptr;
ggml_tensor * bq = nullptr;
ggml_tensor * wk = nullptr;
ggml_tensor * bk = nullptr;
ggml_tensor * wv = nullptr;
ggml_tensor * bv = nullptr;
ggml_tensor * wo = nullptr;
ggml_tensor * bo = nullptr;
};
struct norm {
ggml_tensor * w = nullptr;
ggml_tensor * b = nullptr;
};
// the first n_layer_routing blocks have no self attention
struct head_layer {
norm self_norm;
attn self_attn;
norm cross_norm;
norm cross_norm_kv; // routing blocks only
attn cross_attn;
norm ffn_norm;
ggml_tensor * ffn_up = nullptr;
ggml_tensor * ffn_up_b = nullptr;
ggml_tensor * ffn_down = nullptr;
ggml_tensor * ffn_down_b = nullptr;
};
uint32_t n_layer_routing = 0;
uint32_t n_layer_joint = 0;
uint32_t n_head_decision = 0;
std::vector<head_layer> head_layers; // routing blocks, then joint blocks
norm hidden_norm;
norm option_summary_norm;
norm field_norm;
norm option_norm;
ggml_tensor * proj_memory = nullptr;
ggml_tensor * proj_question = nullptr;
ggml_tensor * proj_option_question = nullptr;
ggml_tensor * proj_global = nullptr;
ggml_tensor * proj_option_context = nullptr;
ggml_tensor * proj_option_lexical = nullptr;
ggml_tensor * scales = nullptr; // prior scale, joint scale, residual gate
class input_decision;
struct graph : public llm_build_delta_net_base {
graph(const llama_model & model, const llm_graph_params & params);
private:
llm_graph_input_attn_no_cache * build_attn_inp_causal();
ggml_tensor * build_layer_attn(
llm_graph_input_attn_no_cache * inp_attn,
ggml_tensor * cur,
ggml_tensor * inp_pos,
int * sections,
int il);
ggml_tensor * build_layer_attn_linear(
ggml_tensor * cur,
int il);
// hidden: [n_embd, n_tokens], returns one score per option: [1, n_options]
ggml_tensor * build_head(
ggml_tensor * hidden,
input_decision * inp);
ggml_tensor * build_head_norm(ggml_tensor * cur, const norm & n);
ggml_tensor * build_head_attn(ggml_tensor * q, ggml_tensor * kv, const attn & a);
ggml_tensor * build_head_ffn (ggml_tensor * cur, const head_layer & layer);
const llama_model_clef & model;
};
std::unique_ptr<llm_graph_context> build_arch_graph(const llm_graph_params & params) const override;
};
struct llama_model_qwen4exp : public llama_model_base {
llama_model_qwen4exp(const struct llama_model_params & params) : llama_model_base(params) {}
+3
View File
@@ -685,6 +685,9 @@ static bool arch_supported(const llm_arch arch) {
if (arch == LLM_ARCH_PLM) {
return false; // TODO tensor shapes
}
if (arch == LLM_ARCH_CLEF) {
return false; // TODO decision head tensors
}
if (arch == LLM_ARCH_DEEPSEEK2OCR) {
return false;
}
+5 -3
View File
@@ -1688,13 +1688,15 @@ Follows the [TypeSafe API](https://docs.typesafe.ai/api), streaming is not suppo
- `score`: An array of 2 to 10 level descriptions, lowest level first.
- `noul`: Optional. An object with the descriptions of `true` and `false`.
The questions of a request are answered independently, an answer does not depend on the other questions.
The questions of a request are answered independently, an answer does not depend on the other questions. The exception is clef: it reads all the questions in one prompt and decides them jointly.
The number of options of a `choice` question is limited by the model, for example: 52 for openjev, 255 for laya. For laya, long questions and options are truncated to the token budget the model was trained with.
The number of options of a `choice` question is limited by the model, for example: 52 for openjev, 255 for laya and clef. For laya, long questions and options are truncated to the token budget the model was trained with.
For laya and clef, the whole prompt is evaluated in one batch: it must fit in `--ubatch-size`. A server that runs clef only serves this endpoint, text generation is not available.
*Image input:*
Image input needs a model that supports it (for example: openjev) and its multimodal projector, see `--mmproj`.
Image input needs a model that supports it (for example: openjev) and its multimodal projector, see `--mmproj`. Image input is not supported yet for clef.
Images can be given in two ways, and both can be used in the same request:
+35 -2
View File
@@ -118,6 +118,7 @@ struct server_batch {
llama_pos pos;
bool output;
bool is_prompt; // for stats tracking
int32_t decision_order = 0;
};
std::vector<token> tokens;
int32_t n_tokens_alloc = 0;
@@ -177,6 +178,11 @@ struct server_batch {
tokens[idx].output = output;
}
void set_decision_order(int32_t idx, int32_t order) {
GGML_ASSERT(idx >= 0 && idx < (int32_t)tokens.size());
tokens[idx].decision_order = order;
}
// render the sub-batch [off, off + n_tokens) into view, index i in view is index off + i here
void render(int32_t off, int32_t n_tokens) {
GGML_ASSERT(off >= 0 && off < size());
@@ -192,6 +198,7 @@ struct server_batch {
} else {
view.add(t.token, t.pos, t.id_slot, t.output);
}
view.tokens.back().decision_order = t.decision_order;
}
}
};
@@ -420,6 +427,11 @@ struct server_slot {
bool can_batch_with(server_slot & other_slot) const {
GGML_ASSERT(task);
// a joint decision head reads the whole batch
if (!task->decision.order.empty() || !other_slot.task->decision.order.empty()) {
return false;
}
return task->type == other_slot.task->type
&& inp_embd.size() == other_slot.inp_embd.size()
&& are_lora_equal(lora, other_slot.lora);
@@ -3634,6 +3646,9 @@ private:
/* pos = */ slot.prompt.tokens.pos_next(),
/* output = */ slot.need_embd(),
/* is_prompt = */ true);
if (!slot.task->decision.order.empty()) {
batch.set_decision_order(batch.size() - 1, slot.task->decision.order[slot.prompt.n_tokens()]);
}
slot.prompt.tokens.push_back(cur_tok);
// break at the last user message, or at user messages at least min step past the last checkpoint
@@ -5339,12 +5354,21 @@ void server_routes::init_routes() {
return res;
}
// one task per question
// one task per question, or one task for all of them
auto & rd = res->rd;
{
std::vector<server_task> tasks;
tasks.reserve(questions.size());
if (decision.is_joint()) {
server_task task = server_task(SERVER_TASK_TYPE_DECISION);
task.id = rd.get_new_id();
decision.fill_task_joint(state, questions, task);
tasks.push_back(std::move(task));
}
for (const auto & question : questions) {
if (decision.is_joint()) {
break;
}
server_task task = server_task(SERVER_TASK_TYPE_DECISION);
task.id = rd.get_new_id();
decision.fill_task(state, question, files, ctx_server.mctx, ctx_server.init_opt, task);
@@ -5367,7 +5391,16 @@ void server_routes::init_routes() {
json answers = json::object();
int32_t n_tokens = 0;
for (size_t i = 0; i < questions.size(); i++) {
if (decision.is_joint()) {
auto * result = dynamic_cast<server_task_result_decision *>(all_results.results[0].get());
GGML_ASSERT(result != nullptr);
const auto scores = decision.split_scores(questions, result->scores);
for (size_t i = 0; i < questions.size(); i++) {
answers[questions[i].id] = decision.format_answer(questions[i], scores[i]);
}
n_tokens = result->n_tokens;
}
for (size_t i = 0; i < questions.size() && !decision.is_joint(); i++) {
auto * result = dynamic_cast<server_task_result_decision *>(all_results.results[i].get());
GGML_ASSERT(result != nullptr);
answers[questions[i].id] = decision.format_answer(questions[i], result->scores);
+182
View File
@@ -85,6 +85,9 @@ void server_decision_context::init(const llama_model * model) {
throw std::runtime_error("decision model has no valid max_head_tokens");
}
n_options_max = 255;
} else if (model_type == COMMON_DECISION_TYPE_CLEF) {
n_options_max = 255;
noul_true_first = true;
} else {
throw std::runtime_error("unsupported decision model type: " + type_name);
}
@@ -374,6 +377,185 @@ void server_decision_context::fill_task_laya(llama_tokens & tokens, const server
task.decision.column = question.type;
}
//
// joint prompt (clef)
//
// the template separates the pieces of the prompt with these, they are tokenized one by one
static const std::string CLEF_MARKER = "<<clef:";
static const std::string CLEF_PIECE_SEP = "<<clef:sep>>";
static const std::string CLEF_PIECE_STATE = "<<clef:state>>";
static const std::string CLEF_PIECE_QUESTION = "<<clef:question>>";
static const std::string CLEF_PIECE_OPTION = "<<clef:option>>";
// same value with the keys of all objects sorted
static json decision_sort_keys(const json & val) {
if (val.is_array()) {
json out = json::array();
for (const auto & item : val) {
out.push_back(decision_sort_keys(item));
}
return out;
}
if (val.is_object()) {
std::map<std::string, json> sorted;
for (const auto & [key, item] : val.items()) {
sorted[key] = decision_sort_keys(item);
}
json out = json::object();
for (const auto & [key, item] : sorted) {
out[key] = item;
}
return out;
}
return val;
}
// strings are used as is, other values are compact JSON with sorted keys
static std::string clef_render(const json & val) {
return val.is_string() ? val.get<std::string>() : decision_sort_keys(val).dump();
}
std::vector<size_t> server_decision_context::prompt_order(const server_decision_question & question) const {
std::vector<size_t> order(question.options.size());
for (size_t i = 0; i < order.size(); i++) {
order[i] = i;
}
if (type == COMMON_DECISION_TYPE_CLEF && question.type == SERVER_DECISION_QUESTION_CHOICE) {
std::sort(order.begin(), order.end(), [&](size_t a, size_t b) {
return question.options[a].key < question.options[b].key;
});
}
return order;
}
void server_decision_context::fill_task_joint(const json & state, const std::vector<server_decision_question> & questions, server_task & task) const {
auto clean = [](const json & val) {
return decision_replace_text(val, CLEF_MARKER, "<<clef ");
};
json inp_questions = json::array();
for (const auto & question : questions) {
json options = json::array();
for (const size_t i : prompt_order(question)) {
const auto & opt = question.options[i];
json description = opt.description;
if (description.is_null() && question.type == SERVER_DECISION_QUESTION_NOUL) {
description = opt.key == "true"
? "The proposition is true or the answer is yes."
: "The proposition is false or the answer is no.";
}
// keys in sorted order
json semantics = json::object();
if (!description.is_null()) {
semantics["description"] = decision_sort_keys(description);
}
semantics["option_id"] = opt.key;
options.push_back(json{{"text", semantics.dump()}});
}
inp_questions.push_back(json{
{"id", question.id},
{"type", decision_question_type_name(question.type)},
{"instructions", clef_render(question.instructions)},
{"options", options},
});
}
const json inp = clean(json{
{"state", clef_render(state)},
{"questions", inp_questions},
});
jinja::context ctx(tmpl->source());
jinja::global_from_json(ctx, inp, false);
jinja::runtime runtime(ctx);
const jinja::value results = runtime.execute(tmpl->prog);
const std::string prompt = jinja::runtime::gather_string_parts(results)->as_string().str();
// the model was trained with the pieces tokenized one by one
llama_tokens tokens;
int32_t i_question = -1;
size_t n_questions = 0;
size_t n_options = 0;
for (std::string piece : string_split(prompt, CLEF_PIECE_SEP)) {
enum { PIECE_TEXT, PIECE_QUESTION, PIECE_OPTION } kind = PIECE_TEXT;
if (string_starts_with(piece, CLEF_PIECE_QUESTION)) {
piece = piece.substr(CLEF_PIECE_QUESTION.size());
kind = PIECE_QUESTION;
} else if (string_starts_with(piece, CLEF_PIECE_OPTION)) {
piece = piece.substr(CLEF_PIECE_OPTION.size());
kind = PIECE_OPTION;
} else if (string_starts_with(piece, CLEF_PIECE_STATE)) {
piece = piece.substr(CLEF_PIECE_STATE.size());
}
const int32_t start = tokens.size();
const llama_tokens piece_tokens = common_tokenize(vocab, piece, false, true);
tokens.insert(tokens.end(), piece_tokens.begin(), piece_tokens.end());
const int32_t end = tokens.size();
if (kind == PIECE_TEXT) {
continue;
}
if (start == end) {
throw std::invalid_argument("the instructions and the options of a question must not be empty");
}
// see llama_batch_ext_set_decision_order(): question of type noul, choice, score, or option
int32_t order = 4;
if (kind == PIECE_QUESTION) {
i_question++;
switch (questions.at(i_question).type) {
case SERVER_DECISION_QUESTION_NOUL: order = 1; break;
case SERVER_DECISION_QUESTION_CHOICE: order = 2; break;
case SERVER_DECISION_QUESTION_SCORE: order = 3; break;
}
n_questions++;
} else {
// the score of option i is returned at row i
task.decision.markers.push_back(n_options);
n_options++;
}
task.decision.order.resize(tokens.size(), 0);
std::fill(task.decision.order.begin() + start, task.decision.order.end(), order);
}
task.decision.order.resize(tokens.size(), 0);
size_t n_options_exp = 0;
for (const auto & question : questions) {
n_options_exp += question.options.size();
}
if (n_questions != questions.size() || n_options != n_options_exp) {
throw std::runtime_error("unexpected layout of the decision prompt");
}
task.decision.column = 0;
task.tokens = server_tokens(tokens, false);
}
std::vector<std::vector<float>> server_decision_context::split_scores(
const std::vector<server_decision_question> & questions,
const std::vector<float> & scores) const {
std::vector<std::vector<float>> result;
size_t offset = 0;
for (const auto & question : questions) {
const auto order = prompt_order(question);
if (offset + order.size() > scores.size()) {
throw std::runtime_error("decision result does not match the number of options");
}
std::vector<float> cur(order.size());
for (size_t i = 0; i < order.size(); i++) {
cur[order[i]] = scores[offset + i];
}
offset += order.size();
result.push_back(std::move(cur));
}
return result;
}
//
// answer
//
+15
View File
@@ -45,7 +45,13 @@ struct server_decision_context {
}
}
// true if all the questions of a request go in one prompt, see fill_task_joint()
bool is_joint() const {
return type == COMMON_DECISION_TYPE_CLEF;
}
// true if the prompt of the model has a place for images
// TODO: clef needs token and embedding entries in the same batch, see https://github.com/ggml-org/llama.cpp/pull/29622
bool can_use_images() const {
switch (type) {
case COMMON_DECISION_TYPE_OPENJEV:
@@ -72,6 +78,12 @@ struct server_decision_context {
const mtmd_helper_init_opt & init_opt,
server_task & task) const;
// set the prompt of all the questions, and where to read their results
void fill_task_joint(const json & state, const std::vector<server_decision_question> & questions, server_task & task) const;
// result of fill_task_joint() -> scores of each question, in the order of its options
std::vector<std::vector<float>> split_scores(const std::vector<server_decision_question> & questions, const std::vector<float> & scores) const;
// scores: one raw model output per option
json format_answer(const server_decision_question & question, const std::vector<float> & scores) const;
@@ -96,6 +108,9 @@ private:
std::string render(const json & state, const server_decision_question & question, size_t n_images) const;
void fill_task_laya(llama_tokens & tokens, const server_decision_question & question, server_task & task) const;
// indices of the options of a question, in the order they have in the prompt
std::vector<size_t> prompt_order(const server_decision_question & question) const;
float get_temperature(const server_decision_question & question) const;
};
+4
View File
@@ -181,6 +181,10 @@ struct server_task {
std::vector<llama_token> labels; // logits of these tokens, at the last prompt token
std::vector<int32_t> markers; // embeddings[column] at these prompt positions
int32_t column = 0;
// for a joint head: one value per prompt token, see llama_batch_ext_set_decision_order()
std::vector<int32_t> order;
};
decision decision;