Hardcode post_norm_rms_eps instead of new param

This commit is contained in:
Pedro Cuenca
2026-08-08 23:07:53 +02:00
parent 84b932497f
commit d0eca60285
8 changed files with 5 additions and 14 deletions
-1
View File
@@ -45,7 +45,6 @@ class OnyxModel(TextModel):
self.gguf_writer.add_final_logit_softcapping(hparams["final_logit_softcapping"])
self.gguf_writer.add_logit_scale(hparams["output_multiplier"])
self.gguf_writer.add_post_norm_rms_eps(hparams["post_norm_eps"])
# SWA + NoPE: [SW, SW, SW, Full], NoPE used on Full layers. References:
# https://huggingface.co/someorgtoo-hf/onyx-hf-converted/blob/main/config.json#L19
-1
View File
@@ -172,7 +172,6 @@ class Keys:
VALUE_LENGTH = "{arch}.attention.value_length"
LAYERNORM_EPS = "{arch}.attention.layer_norm_epsilon"
LAYERNORM_RMS_EPS = "{arch}.attention.layer_norm_rms_epsilon"
POST_NORM_RMS_EPS = "{arch}.attention.post_norm_rms_epsilon"
GROUPNORM_EPS = "{arch}.attention.group_norm_epsilon"
GROUPNORM_GROUPS = "{arch}.attention.group_norm_groups"
CAUSAL = "{arch}.attention.causal"
-3
View File
@@ -923,9 +923,6 @@ class GGUFWriter:
def add_layer_norm_rms_eps(self, value: float) -> None:
self.add_float32(Keys.Attention.LAYERNORM_RMS_EPS.format(arch=self.arch), value)
def add_post_norm_rms_eps(self, value: float) -> None:
self.add_float32(Keys.Attention.POST_NORM_RMS_EPS.format(arch=self.arch), value)
def add_group_norm_eps(self, value: float) -> None:
self.add_float32(Keys.Attention.GROUPNORM_EPS.format(arch=self.arch), value)
-1
View File
@@ -234,7 +234,6 @@ static const std::map<llm_kv, const char *> LLM_KV_NAMES = {
{ LLM_KV_ATTENTION_VALUE_LENGTH, "%s.attention.value_length" },
{ LLM_KV_ATTENTION_LAYERNORM_EPS, "%s.attention.layer_norm_epsilon" },
{ LLM_KV_ATTENTION_LAYERNORM_RMS_EPS, "%s.attention.layer_norm_rms_epsilon" },
{ LLM_KV_ATTENTION_POST_NORM_RMS_EPS, "%s.attention.post_norm_rms_epsilon" },
{ LLM_KV_ATTENTION_GROUPNORM_EPS, "%s.attention.group_norm_epsilon" },
{ LLM_KV_ATTENTION_GROUPNORM_GROUPS, "%s.attention.group_norm_groups" },
{ LLM_KV_ATTENTION_CAUSAL, "%s.attention.causal" },
-1
View File
@@ -239,7 +239,6 @@ enum llm_kv {
LLM_KV_ATTENTION_VALUE_LENGTH,
LLM_KV_ATTENTION_LAYERNORM_EPS,
LLM_KV_ATTENTION_LAYERNORM_RMS_EPS,
LLM_KV_ATTENTION_POST_NORM_RMS_EPS,
LLM_KV_ATTENTION_GROUPNORM_EPS,
LLM_KV_ATTENTION_GROUPNORM_GROUPS,
LLM_KV_ATTENTION_CAUSAL,
-1
View File
@@ -106,7 +106,6 @@ struct llama_hparams {
float f_norm_eps;
float f_norm_rms_eps;
float f_norm_group_eps;
float f_post_norm_rms_eps;
float f_attn_logit_softcapping = 50.0f;
float f_router_logit_softcapping = 30.0f;
-1
View File
@@ -1793,7 +1793,6 @@ void llama_model::print_info() const {
LLAMA_LOG_INFO("%s: n_embd_v_gqa = %s\n", __func__, print_f([&](uint32_t il) { return hparams.n_embd_v_gqa(il); }, hparams.n_layer_all).c_str());
LLAMA_LOG_INFO("%s: f_norm_eps = %.1e\n", __func__, hparams.f_norm_eps);
LLAMA_LOG_INFO("%s: f_norm_rms_eps = %.1e\n", __func__, hparams.f_norm_rms_eps);
LLAMA_LOG_INFO("%s: f_post_norm_rms_eps = %.1e\n", __func__, hparams.f_post_norm_rms_eps);
LLAMA_LOG_INFO("%s: f_clamp_kqv = %.1e\n", __func__, hparams.f_clamp_kqv);
LLAMA_LOG_INFO("%s: f_max_alibi_bias = %.1e\n", __func__, hparams.f_max_alibi_bias);
LLAMA_LOG_INFO("%s: f_logit_scale = %.1e\n", __func__, hparams.f_logit_scale);
+5 -5
View File
@@ -5,7 +5,6 @@ void llama_model_onyx::load_arch_hparams(llama_model_loader & ml) {
ml.get_key(LLM_KV_ATTENTION_SLIDING_WINDOW, hparams.n_swa, false);
ml.get_key(LLM_KV_FINAL_LOGIT_SOFTCAPPING, hparams.f_final_logit_softcapping, false);
ml.get_key(LLM_KV_LOGIT_SCALE, hparams.f_logit_scale);
ml.get_key(LLM_KV_ATTENTION_POST_NORM_RMS_EPS, hparams.f_post_norm_rms_eps);
// SWA layers share the model rope theta; they are also the only layers that use rope
// here (global layers are NoPE), so the 10000.0 default would apply to all of them.
@@ -67,6 +66,9 @@ llama_model_onyx::graph::graph(const llama_model & model, const llm_graph_params
const int64_t n_embd_head = hparams.n_embd_head_v();
GGML_ASSERT(n_embd_head == hparams.n_embd_head_k());
// Different to f_norm_rms_eps for post-attn / post-FFN norms
const float post_norm_eps = 1e-8f;
ggml_tensor * cur;
ggml_tensor * inpL;
@@ -143,8 +145,7 @@ llama_model_onyx::graph::graph(const llama_model & model, const llm_graph_params
cb(cur, "attn_o_proj", il);
}
// post-attention norm (uses f_post_norm_rms_eps, not the general f_norm_rms_eps).
cur = ggml_rms_norm(ctx0, cur, hparams.f_post_norm_rms_eps);
cur = ggml_rms_norm(ctx0, cur, post_norm_eps);
cur = ggml_mul(ctx0, cur, model.layers[il].attn_post_norm);
cb(cur, "attn_post_norm", il);
@@ -169,8 +170,7 @@ llama_model_onyx::graph::graph(const llama_model & model, const llm_graph_params
LLM_FFN_SILU, LLM_FFN_PAR, il);
cb(cur, "ffn_out", il);
// post-FFN norm (uses f_post_norm_rms_eps, not the general f_norm_rms_eps).
cur = ggml_rms_norm(ctx0, cur, hparams.f_post_norm_rms_eps);
cur = ggml_rms_norm(ctx0, cur, post_norm_eps);
cur = ggml_mul(ctx0, cur, model.layers[il].ffn_post_norm);
cb(cur, "ffn_post_norm", il);