diff --git a/gguf-py/gguf/tensor_mapping.py b/gguf-py/gguf/tensor_mapping.py index d59e248357..cef04d0827 100644 --- a/gguf-py/gguf/tensor_mapping.py +++ b/gguf-py/gguf/tensor_mapping.py @@ -1314,6 +1314,14 @@ class TensorNameMap: "joint_head.option_norm", # clef ), + MODEL_TENSOR.DECISION_SCORER: ( + "joint_head.residual_scorer.0", # clef + ), + + MODEL_TENSOR.DECISION_SCORER_OUT: ( + "joint_head.residual_scorer.3", # clef + ), + MODEL_TENSOR.ENC_ATTN_NORM: ( "encoder.block.{bid}.layer.0.layer_norm", # t5 ), @@ -1515,13 +1523,11 @@ class TensorNameMap: "dense", # neobert "head.dense", # modern-bert "scorer.1", # laya - "joint_head.residual_scorer.0", # clef ), MODEL_TENSOR.CLS_OUT: ( "classifier.out_proj", # roberta "scorer.3", # laya - "joint_head.residual_scorer.3", # clef ), MODEL_TENSOR.CLS_NORM: ( diff --git a/src/llama-arch.cpp b/src/llama-arch.cpp index 1b1dac0ff3..a3f9697466 100644 --- a/src/llama-arch.cpp +++ b/src/llama-arch.cpp @@ -614,6 +614,8 @@ static const std::map LLM_TENSOR_NAMES = { { LLM_TENSOR_DECISION_FIELD_NORM, "decision.field_norm" }, { LLM_TENSOR_DECISION_OPTION_NORM, "decision.option_norm" }, { LLM_TENSOR_DECISION_SCALES, "decision.scales" }, + { LLM_TENSOR_DECISION_SCORER, "decision.scorer" }, + { LLM_TENSOR_DECISION_SCORER_OUT, "decision.scorer_out" }, { LLM_TENSOR_DEC_ATTN_NORM, "dec.blk.%d.attn_norm" }, { LLM_TENSOR_DEC_ATTN_Q, "dec.blk.%d.attn_q" }, { LLM_TENSOR_DEC_ATTN_K, "dec.blk.%d.attn_k" }, @@ -771,6 +773,8 @@ static const std::map LLM_TENSOR_INFOS = { {LLM_TENSOR_DECISION_FIELD_NORM, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_MUL}}, {LLM_TENSOR_DECISION_OPTION_NORM, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_MUL}}, {LLM_TENSOR_DECISION_SCALES, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_MUL}}, + {LLM_TENSOR_DECISION_SCORER, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_MUL_MAT}}, + {LLM_TENSOR_DECISION_SCORER_OUT, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_MUL_MAT}}, {LLM_TENSOR_ENC_OUTPUT_NORM, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_MUL}}, {LLM_TENSOR_ROPE_FREQS, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_ROPE}}, {LLM_TENSOR_ROPE_FACTORS_LONG, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_ROPE}}, diff --git a/src/llama-arch.h b/src/llama-arch.h index 4106027675..eddded280f 100644 --- a/src/llama-arch.h +++ b/src/llama-arch.h @@ -661,6 +661,8 @@ enum llm_tensor { LLM_TENSOR_DECISION_FIELD_NORM, LLM_TENSOR_DECISION_OPTION_NORM, LLM_TENSOR_DECISION_SCALES, + LLM_TENSOR_DECISION_SCORER, + LLM_TENSOR_DECISION_SCORER_OUT, LLM_TENSOR_ENC_ATTN_NORM, LLM_TENSOR_ENC_ATTN_Q, LLM_TENSOR_ENC_ATTN_K, diff --git a/src/models/clef.cpp b/src/models/clef.cpp index 7bb846db16..b1ce283b98 100644 --- a/src/models/clef.cpp +++ b/src/models/clef.cpp @@ -90,10 +90,11 @@ void llama_model_clef::load_arch_tensors(llama_model_loader & ml) { scales = create_tensor(tn(LLM_TENSOR_DECISION_SCALES), {3}, 0); type_embd = create_tensor(tn(LLM_TENSOR_TOKEN_TYPES, "weight"), {n_embd_h, 3}, 0); - cls = create_tensor(tn(LLM_TENSOR_CLS, "weight"), {4 * n_embd_h, n_embd_h}, 0); - cls_b = create_tensor(tn(LLM_TENSOR_CLS, "bias"), {n_embd_h}, 0); - cls_out = create_tensor(tn(LLM_TENSOR_CLS_OUT, "weight"), {n_embd_h, 1}, 0); - cls_out_b = create_tensor(tn(LLM_TENSOR_CLS_OUT, "bias"), {1}, 0); + + scorer = create_tensor(tn(LLM_TENSOR_DECISION_SCORER, "weight"), {4 * n_embd_h, n_embd_h}, 0); + scorer_b = create_tensor(tn(LLM_TENSOR_DECISION_SCORER, "bias"), {n_embd_h}, 0); + scorer_out = create_tensor(tn(LLM_TENSOR_DECISION_SCORER_OUT, "weight"), {n_embd_h, 1}, 0); + scorer_out_b = create_tensor(tn(LLM_TENSOR_DECISION_SCORER_OUT, "bias"), {1}, 0); } std::unique_ptr llama_model_clef::build_arch_graph(const llm_graph_params & params) const { @@ -598,9 +599,9 @@ ggml_tensor * llama_model_clef::graph::build_head(ggml_tensor * hidden, input_de features = ggml_concat(ctx0, features, ggml_mul(ctx0, field, options), 0); features = ggml_concat(ctx0, features, ggml_abs(ctx0, ggml_sub(ctx0, field, options)), 0); - ggml_tensor * residual = ggml_add(ctx0, ggml_mul_mat(ctx0, model.cls, features), model.cls_b); + ggml_tensor * residual = ggml_add(ctx0, ggml_mul_mat(ctx0, model.scorer, features), model.scorer_b); residual = ggml_gelu_erf(ctx0, residual); - residual = ggml_add(ctx0, ggml_mul_mat(ctx0, model.cls_out, residual), model.cls_out_b); // [1, n_options] + residual = ggml_add(ctx0, ggml_mul_mat(ctx0, model.scorer_out, residual), model.scorer_out_b); // [1, n_options] auto scale = [&](int i) { return ggml_view_1d(ctx0, model.scales, 1, i * ggml_element_size(model.scales)); diff --git a/src/models/models.h b/src/models/models.h index f23419e592..387a4adcb2 100644 --- a/src/models/models.h +++ b/src/models/models.h @@ -2445,6 +2445,10 @@ struct llama_model_clef : public llama_model_qwen35 { ggml_tensor * proj_option_context = nullptr; ggml_tensor * proj_option_lexical = nullptr; ggml_tensor * scales = nullptr; // prior scale, joint scale, residual gate + ggml_tensor * scorer = nullptr; + ggml_tensor * scorer_b = nullptr; + ggml_tensor * scorer_out = nullptr; + ggml_tensor * scorer_out_b = nullptr; class input_decision; diff --git a/tools/server/server-decision.cpp b/tools/server/server-decision.cpp index 4e51910632..da370ee6d9 100644 --- a/tools/server/server-decision.cpp +++ b/tools/server/server-decision.cpp @@ -534,29 +534,6 @@ static const std::string CLEF_PIECE_STATE = "<>"; static const std::string CLEF_PIECE_QUESTION = "<>"; static const std::string CLEF_PIECE_OPTION = "<>"; -// same value with the keys of all objects sorted -static json decision_sort_keys(const json & val) { - if (val.is_array()) { - json out = json::array(); - for (const auto & item : val) { - out.push_back(decision_sort_keys(item)); - } - return out; - } - if (val.is_object()) { - std::map sorted; - for (const auto & [key, item] : val.items()) { - sorted[key] = decision_sort_keys(item); - } - json out = json::object(); - for (const auto & [key, item] : sorted) { - out[key] = item; - } - return out; - } - return val; -} - // strings are used as is, other values are compact JSON with sorted keys static std::string clef_render(const json & val) { return val.is_string() ? val.get() : decision_sort_keys(val).dump();