#include "models.h" #include // torch.nn.LayerNorm default static const float CLEF_HEAD_NORM_EPS = 1e-5f; void llama_model_clef::load_arch_hparams(llama_model_loader & ml) { llama_model_qwen35::load_arch_hparams(ml); ml.get_key(LLM_KV_DECISION_ROUTING_BLOCK_COUNT, n_layer_routing); ml.get_key(LLM_KV_DECISION_BLOCK_COUNT, n_layer_joint); ml.get_key(LLM_KV_DECISION_HEAD_COUNT, n_head_decision); if (n_head_decision == 0) { throw std::runtime_error("invalid number of heads in the decision head"); } hparams.f_norm_eps = CLEF_HEAD_NORM_EPS; // the output is one score per token, see llama_batch_ext_set_decision_order() hparams.n_embd_out_impl = 1; } void llama_model_clef::load_arch_tensors(llama_model_loader & ml) { llama_model_qwen35::load_arch_tensors(ml); LLAMA_LOAD_LOCALS; const auto * w_memory = ml.get_weight(tn(LLM_TENSOR_DECISION_PROJ_MEMORY, "weight").str().c_str()); const auto * w_ffn = ml.get_weight(tn(LLM_TENSOR_DEC_FFN_UP, "weight", 0).str().c_str()); if (w_memory == nullptr || w_ffn == nullptr) { throw std::runtime_error("the decision head is missing"); } const int64_t n_embd_h = w_memory->tensor->ne[1]; const int64_t n_ff_h = w_ffn->tensor->ne[1]; if (n_embd_h % n_head_decision != 0) { throw std::runtime_error("invalid width of the decision head"); } auto load_norm = [&](norm & n, llm_tensor type, int64_t size, int il = -1) { n.w = il < 0 ? create_tensor(tn(type, "weight"), {size}, 0) : create_tensor(tn(type, "weight", il), {size}, 0); n.b = il < 0 ? create_tensor(tn(type, "bias"), {size}, 0) : create_tensor(tn(type, "bias", il), {size}, 0); }; auto load_attn = [&](attn & a, llm_tensor q, llm_tensor k, llm_tensor v, llm_tensor o, int il) { a.wq = create_tensor(tn(q, "weight", il), {n_embd_h, n_embd_h}, 0); a.bq = create_tensor(tn(q, "bias", il), {n_embd_h}, 0); a.wk = create_tensor(tn(k, "weight", il), {n_embd_h, n_embd_h}, 0); a.bk = create_tensor(tn(k, "bias", il), {n_embd_h}, 0); a.wv = create_tensor(tn(v, "weight", il), {n_embd_h, n_embd_h}, 0); a.bv = create_tensor(tn(v, "bias", il), {n_embd_h}, 0); a.wo = create_tensor(tn(o, "weight", il), {n_embd_h, n_embd_h}, 0); a.bo = create_tensor(tn(o, "bias", il), {n_embd_h}, 0); }; head_layers.resize(n_layer_routing + n_layer_joint); for (int il = 0; il < (int) head_layers.size(); ++il) { auto & layer = head_layers[il]; if (il < (int) n_layer_routing) { load_norm(layer.cross_norm_kv, LLM_TENSOR_DEC_CROSS_ATTN_NORM_KV, n_embd_h, il); } else { load_norm(layer.self_norm, LLM_TENSOR_DEC_ATTN_NORM, n_embd_h, il); load_attn(layer.self_attn, LLM_TENSOR_DEC_ATTN_Q, LLM_TENSOR_DEC_ATTN_K, LLM_TENSOR_DEC_ATTN_V, LLM_TENSOR_DEC_ATTN_OUT, il); } load_norm(layer.cross_norm, LLM_TENSOR_DEC_CROSS_ATTN_NORM, n_embd_h, il); load_attn(layer.cross_attn, LLM_TENSOR_DEC_CROSS_ATTN_Q, LLM_TENSOR_DEC_CROSS_ATTN_K, LLM_TENSOR_DEC_CROSS_ATTN_V, LLM_TENSOR_DEC_CROSS_ATTN_OUT, il); load_norm(layer.ffn_norm, LLM_TENSOR_DEC_FFN_NORM, n_embd_h, il); layer.ffn_up = create_tensor(tn(LLM_TENSOR_DEC_FFN_UP, "weight", il), {n_embd_h, n_ff_h}, 0); layer.ffn_up_b = create_tensor(tn(LLM_TENSOR_DEC_FFN_UP, "bias", il), {n_ff_h}, 0); layer.ffn_down = create_tensor(tn(LLM_TENSOR_DEC_FFN_DOWN, "weight", il), {n_ff_h, n_embd_h}, 0); layer.ffn_down_b = create_tensor(tn(LLM_TENSOR_DEC_FFN_DOWN, "bias", il), {n_embd_h}, 0); } load_norm(hidden_norm, LLM_TENSOR_DECISION_HIDDEN_NORM, n_embd); load_norm(option_summary_norm, LLM_TENSOR_DECISION_OPTION_SUMMARY_NORM, n_embd_h); load_norm(field_norm, LLM_TENSOR_DECISION_FIELD_NORM, n_embd_h); load_norm(option_norm, LLM_TENSOR_DECISION_OPTION_NORM, n_embd_h); proj_memory = create_tensor(tn(LLM_TENSOR_DECISION_PROJ_MEMORY, "weight"), {n_embd, n_embd_h}, 0); proj_question = create_tensor(tn(LLM_TENSOR_DECISION_PROJ_QUESTION, "weight"), {n_embd, n_embd_h}, 0); proj_option_question = create_tensor(tn(LLM_TENSOR_DECISION_PROJ_OPTION_QUESTION, "weight"), {n_embd, n_embd_h}, 0); proj_global = create_tensor(tn(LLM_TENSOR_DECISION_PROJ_GLOBAL, "weight"), {n_embd, n_embd_h}, 0); proj_option_context = create_tensor(tn(LLM_TENSOR_DECISION_PROJ_OPTION_CONTEXT, "weight"), {n_embd, n_embd_h}, 0); proj_option_lexical = create_tensor(tn(LLM_TENSOR_DECISION_PROJ_OPTION_LEXICAL, "weight"), {n_embd, n_embd_h}, 0); scales = create_tensor(tn(LLM_TENSOR_DECISION_SCALES), {3}, 0); type_embd = create_tensor(tn(LLM_TENSOR_TOKEN_TYPES, "weight"), {n_embd_h, 3}, 0); scorer = create_tensor(tn(LLM_TENSOR_DECISION_SCORER, "weight"), {4 * n_embd_h, n_embd_h}, 0); scorer_b = create_tensor(tn(LLM_TENSOR_DECISION_SCORER, "bias"), {n_embd_h}, 0); scorer_out = create_tensor(tn(LLM_TENSOR_DECISION_SCORER_OUT, "weight"), {n_embd_h, 1}, 0); scorer_out_b = create_tensor(tn(LLM_TENSOR_DECISION_SCORER_OUT, "bias"), {1}, 0); } std::unique_ptr llama_model_clef::build_arch_graph(const llm_graph_params & params) const { return std::make_unique(*this, params); } // spans read by the head, [start, end) in ubatch token indices struct clef_spans { struct question { int32_t type; // noul, choice, score int32_t start; int32_t end; }; struct option { int32_t question; int32_t start; int32_t end; }; std::vector questions; std::vector