mirror of
https://github.com/ggml-org/llama.cpp.git
synced 2026-10-02 02:47:26 -05:00
fix merge
This commit is contained in:
@@ -1314,6 +1314,14 @@ class TensorNameMap:
|
||||
"joint_head.option_norm", # clef
|
||||
),
|
||||
|
||||
MODEL_TENSOR.DECISION_SCORER: (
|
||||
"joint_head.residual_scorer.0", # clef
|
||||
),
|
||||
|
||||
MODEL_TENSOR.DECISION_SCORER_OUT: (
|
||||
"joint_head.residual_scorer.3", # clef
|
||||
),
|
||||
|
||||
MODEL_TENSOR.ENC_ATTN_NORM: (
|
||||
"encoder.block.{bid}.layer.0.layer_norm", # t5
|
||||
),
|
||||
@@ -1515,13 +1523,11 @@ class TensorNameMap:
|
||||
"dense", # neobert
|
||||
"head.dense", # modern-bert
|
||||
"scorer.1", # laya
|
||||
"joint_head.residual_scorer.0", # clef
|
||||
),
|
||||
|
||||
MODEL_TENSOR.CLS_OUT: (
|
||||
"classifier.out_proj", # roberta
|
||||
"scorer.3", # laya
|
||||
"joint_head.residual_scorer.3", # clef
|
||||
),
|
||||
|
||||
MODEL_TENSOR.CLS_NORM: (
|
||||
|
||||
@@ -614,6 +614,8 @@ static const std::map<llm_tensor, const char *> LLM_TENSOR_NAMES = {
|
||||
{ LLM_TENSOR_DECISION_FIELD_NORM, "decision.field_norm" },
|
||||
{ LLM_TENSOR_DECISION_OPTION_NORM, "decision.option_norm" },
|
||||
{ LLM_TENSOR_DECISION_SCALES, "decision.scales" },
|
||||
{ LLM_TENSOR_DECISION_SCORER, "decision.scorer" },
|
||||
{ LLM_TENSOR_DECISION_SCORER_OUT, "decision.scorer_out" },
|
||||
{ LLM_TENSOR_DEC_ATTN_NORM, "dec.blk.%d.attn_norm" },
|
||||
{ LLM_TENSOR_DEC_ATTN_Q, "dec.blk.%d.attn_q" },
|
||||
{ LLM_TENSOR_DEC_ATTN_K, "dec.blk.%d.attn_k" },
|
||||
@@ -771,6 +773,8 @@ static const std::map<llm_tensor, llm_tensor_info> LLM_TENSOR_INFOS = {
|
||||
{LLM_TENSOR_DECISION_FIELD_NORM, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_MUL}},
|
||||
{LLM_TENSOR_DECISION_OPTION_NORM, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_MUL}},
|
||||
{LLM_TENSOR_DECISION_SCALES, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_MUL}},
|
||||
{LLM_TENSOR_DECISION_SCORER, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_MUL_MAT}},
|
||||
{LLM_TENSOR_DECISION_SCORER_OUT, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_MUL_MAT}},
|
||||
{LLM_TENSOR_ENC_OUTPUT_NORM, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_MUL}},
|
||||
{LLM_TENSOR_ROPE_FREQS, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_ROPE}},
|
||||
{LLM_TENSOR_ROPE_FACTORS_LONG, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_ROPE}},
|
||||
|
||||
@@ -661,6 +661,8 @@ enum llm_tensor {
|
||||
LLM_TENSOR_DECISION_FIELD_NORM,
|
||||
LLM_TENSOR_DECISION_OPTION_NORM,
|
||||
LLM_TENSOR_DECISION_SCALES,
|
||||
LLM_TENSOR_DECISION_SCORER,
|
||||
LLM_TENSOR_DECISION_SCORER_OUT,
|
||||
LLM_TENSOR_ENC_ATTN_NORM,
|
||||
LLM_TENSOR_ENC_ATTN_Q,
|
||||
LLM_TENSOR_ENC_ATTN_K,
|
||||
|
||||
+7
-6
@@ -90,10 +90,11 @@ void llama_model_clef::load_arch_tensors(llama_model_loader & ml) {
|
||||
|
||||
scales = create_tensor(tn(LLM_TENSOR_DECISION_SCALES), {3}, 0);
|
||||
type_embd = create_tensor(tn(LLM_TENSOR_TOKEN_TYPES, "weight"), {n_embd_h, 3}, 0);
|
||||
cls = create_tensor(tn(LLM_TENSOR_CLS, "weight"), {4 * n_embd_h, n_embd_h}, 0);
|
||||
cls_b = create_tensor(tn(LLM_TENSOR_CLS, "bias"), {n_embd_h}, 0);
|
||||
cls_out = create_tensor(tn(LLM_TENSOR_CLS_OUT, "weight"), {n_embd_h, 1}, 0);
|
||||
cls_out_b = create_tensor(tn(LLM_TENSOR_CLS_OUT, "bias"), {1}, 0);
|
||||
|
||||
scorer = create_tensor(tn(LLM_TENSOR_DECISION_SCORER, "weight"), {4 * n_embd_h, n_embd_h}, 0);
|
||||
scorer_b = create_tensor(tn(LLM_TENSOR_DECISION_SCORER, "bias"), {n_embd_h}, 0);
|
||||
scorer_out = create_tensor(tn(LLM_TENSOR_DECISION_SCORER_OUT, "weight"), {n_embd_h, 1}, 0);
|
||||
scorer_out_b = create_tensor(tn(LLM_TENSOR_DECISION_SCORER_OUT, "bias"), {1}, 0);
|
||||
}
|
||||
|
||||
std::unique_ptr<llm_graph_context> llama_model_clef::build_arch_graph(const llm_graph_params & params) const {
|
||||
@@ -598,9 +599,9 @@ ggml_tensor * llama_model_clef::graph::build_head(ggml_tensor * hidden, input_de
|
||||
features = ggml_concat(ctx0, features, ggml_mul(ctx0, field, options), 0);
|
||||
features = ggml_concat(ctx0, features, ggml_abs(ctx0, ggml_sub(ctx0, field, options)), 0);
|
||||
|
||||
ggml_tensor * residual = ggml_add(ctx0, ggml_mul_mat(ctx0, model.cls, features), model.cls_b);
|
||||
ggml_tensor * residual = ggml_add(ctx0, ggml_mul_mat(ctx0, model.scorer, features), model.scorer_b);
|
||||
residual = ggml_gelu_erf(ctx0, residual);
|
||||
residual = ggml_add(ctx0, ggml_mul_mat(ctx0, model.cls_out, residual), model.cls_out_b); // [1, n_options]
|
||||
residual = ggml_add(ctx0, ggml_mul_mat(ctx0, model.scorer_out, residual), model.scorer_out_b); // [1, n_options]
|
||||
|
||||
auto scale = [&](int i) {
|
||||
return ggml_view_1d(ctx0, model.scales, 1, i * ggml_element_size(model.scales));
|
||||
|
||||
@@ -2445,6 +2445,10 @@ struct llama_model_clef : public llama_model_qwen35 {
|
||||
ggml_tensor * proj_option_context = nullptr;
|
||||
ggml_tensor * proj_option_lexical = nullptr;
|
||||
ggml_tensor * scales = nullptr; // prior scale, joint scale, residual gate
|
||||
ggml_tensor * scorer = nullptr;
|
||||
ggml_tensor * scorer_b = nullptr;
|
||||
ggml_tensor * scorer_out = nullptr;
|
||||
ggml_tensor * scorer_out_b = nullptr;
|
||||
|
||||
class input_decision;
|
||||
|
||||
|
||||
@@ -534,29 +534,6 @@ static const std::string CLEF_PIECE_STATE = "<<clef:state>>";
|
||||
static const std::string CLEF_PIECE_QUESTION = "<<clef:question>>";
|
||||
static const std::string CLEF_PIECE_OPTION = "<<clef:option>>";
|
||||
|
||||
// same value with the keys of all objects sorted
|
||||
static json decision_sort_keys(const json & val) {
|
||||
if (val.is_array()) {
|
||||
json out = json::array();
|
||||
for (const auto & item : val) {
|
||||
out.push_back(decision_sort_keys(item));
|
||||
}
|
||||
return out;
|
||||
}
|
||||
if (val.is_object()) {
|
||||
std::map<std::string, json> sorted;
|
||||
for (const auto & [key, item] : val.items()) {
|
||||
sorted[key] = decision_sort_keys(item);
|
||||
}
|
||||
json out = json::object();
|
||||
for (const auto & [key, item] : sorted) {
|
||||
out[key] = item;
|
||||
}
|
||||
return out;
|
||||
}
|
||||
return val;
|
||||
}
|
||||
|
||||
// strings are used as is, other values are compact JSON with sorted keys
|
||||
static std::string clef_render(const json & val) {
|
||||
return val.is_string() ? val.get<std::string>() : decision_sort_keys(val).dump();
|
||||
|
||||
Reference in New Issue
Block a user