diff --git a/conversion/onyx.py b/conversion/onyx.py index 4b0e14f4ed..c31e61c8ca 100644 --- a/conversion/onyx.py +++ b/conversion/onyx.py @@ -45,7 +45,6 @@ class OnyxModel(TextModel): self.gguf_writer.add_final_logit_softcapping(hparams["final_logit_softcapping"]) self.gguf_writer.add_logit_scale(hparams["output_multiplier"]) - self.gguf_writer.add_post_norm_rms_eps(hparams["post_norm_eps"]) # SWA + NoPE: [SW, SW, SW, Full], NoPE used on Full layers. References: # https://huggingface.co/someorgtoo-hf/onyx-hf-converted/blob/main/config.json#L19 diff --git a/gguf-py/gguf/constants.py b/gguf-py/gguf/constants.py index 7a2ba560a8..7dbf63f831 100644 --- a/gguf-py/gguf/constants.py +++ b/gguf-py/gguf/constants.py @@ -172,7 +172,6 @@ class Keys: VALUE_LENGTH = "{arch}.attention.value_length" LAYERNORM_EPS = "{arch}.attention.layer_norm_epsilon" LAYERNORM_RMS_EPS = "{arch}.attention.layer_norm_rms_epsilon" - POST_NORM_RMS_EPS = "{arch}.attention.post_norm_rms_epsilon" GROUPNORM_EPS = "{arch}.attention.group_norm_epsilon" GROUPNORM_GROUPS = "{arch}.attention.group_norm_groups" CAUSAL = "{arch}.attention.causal" diff --git a/gguf-py/gguf/gguf_writer.py b/gguf-py/gguf/gguf_writer.py index af14e24fe4..c5905164c3 100644 --- a/gguf-py/gguf/gguf_writer.py +++ b/gguf-py/gguf/gguf_writer.py @@ -923,9 +923,6 @@ class GGUFWriter: def add_layer_norm_rms_eps(self, value: float) -> None: self.add_float32(Keys.Attention.LAYERNORM_RMS_EPS.format(arch=self.arch), value) - def add_post_norm_rms_eps(self, value: float) -> None: - self.add_float32(Keys.Attention.POST_NORM_RMS_EPS.format(arch=self.arch), value) - def add_group_norm_eps(self, value: float) -> None: self.add_float32(Keys.Attention.GROUPNORM_EPS.format(arch=self.arch), value) diff --git a/src/llama-arch.cpp b/src/llama-arch.cpp index 661409a7ca..2ce0a7752d 100644 --- a/src/llama-arch.cpp +++ b/src/llama-arch.cpp @@ -234,7 +234,6 @@ static const std::map LLM_KV_NAMES = { { LLM_KV_ATTENTION_VALUE_LENGTH, "%s.attention.value_length" }, { LLM_KV_ATTENTION_LAYERNORM_EPS, "%s.attention.layer_norm_epsilon" }, { LLM_KV_ATTENTION_LAYERNORM_RMS_EPS, "%s.attention.layer_norm_rms_epsilon" }, - { LLM_KV_ATTENTION_POST_NORM_RMS_EPS, "%s.attention.post_norm_rms_epsilon" }, { LLM_KV_ATTENTION_GROUPNORM_EPS, "%s.attention.group_norm_epsilon" }, { LLM_KV_ATTENTION_GROUPNORM_GROUPS, "%s.attention.group_norm_groups" }, { LLM_KV_ATTENTION_CAUSAL, "%s.attention.causal" }, diff --git a/src/llama-arch.h b/src/llama-arch.h index 7e97292b25..24d5d97c48 100644 --- a/src/llama-arch.h +++ b/src/llama-arch.h @@ -239,7 +239,6 @@ enum llm_kv { LLM_KV_ATTENTION_VALUE_LENGTH, LLM_KV_ATTENTION_LAYERNORM_EPS, LLM_KV_ATTENTION_LAYERNORM_RMS_EPS, - LLM_KV_ATTENTION_POST_NORM_RMS_EPS, LLM_KV_ATTENTION_GROUPNORM_EPS, LLM_KV_ATTENTION_GROUPNORM_GROUPS, LLM_KV_ATTENTION_CAUSAL, diff --git a/src/llama-hparams.h b/src/llama-hparams.h index ce430d4364..fc770bf003 100644 --- a/src/llama-hparams.h +++ b/src/llama-hparams.h @@ -106,7 +106,6 @@ struct llama_hparams { float f_norm_eps; float f_norm_rms_eps; float f_norm_group_eps; - float f_post_norm_rms_eps; float f_attn_logit_softcapping = 50.0f; float f_router_logit_softcapping = 30.0f; diff --git a/src/llama-model.cpp b/src/llama-model.cpp index 2180d36b2a..578e423953 100644 --- a/src/llama-model.cpp +++ b/src/llama-model.cpp @@ -1793,7 +1793,6 @@ void llama_model::print_info() const { LLAMA_LOG_INFO("%s: n_embd_v_gqa = %s\n", __func__, print_f([&](uint32_t il) { return hparams.n_embd_v_gqa(il); }, hparams.n_layer_all).c_str()); LLAMA_LOG_INFO("%s: f_norm_eps = %.1e\n", __func__, hparams.f_norm_eps); LLAMA_LOG_INFO("%s: f_norm_rms_eps = %.1e\n", __func__, hparams.f_norm_rms_eps); - LLAMA_LOG_INFO("%s: f_post_norm_rms_eps = %.1e\n", __func__, hparams.f_post_norm_rms_eps); LLAMA_LOG_INFO("%s: f_clamp_kqv = %.1e\n", __func__, hparams.f_clamp_kqv); LLAMA_LOG_INFO("%s: f_max_alibi_bias = %.1e\n", __func__, hparams.f_max_alibi_bias); LLAMA_LOG_INFO("%s: f_logit_scale = %.1e\n", __func__, hparams.f_logit_scale); diff --git a/src/models/onyx.cpp b/src/models/onyx.cpp index 636f71ca8c..2d950e2d82 100644 --- a/src/models/onyx.cpp +++ b/src/models/onyx.cpp @@ -5,7 +5,6 @@ void llama_model_onyx::load_arch_hparams(llama_model_loader & ml) { ml.get_key(LLM_KV_ATTENTION_SLIDING_WINDOW, hparams.n_swa, false); ml.get_key(LLM_KV_FINAL_LOGIT_SOFTCAPPING, hparams.f_final_logit_softcapping, false); ml.get_key(LLM_KV_LOGIT_SCALE, hparams.f_logit_scale); - ml.get_key(LLM_KV_ATTENTION_POST_NORM_RMS_EPS, hparams.f_post_norm_rms_eps); // SWA layers share the model rope theta; they are also the only layers that use rope // here (global layers are NoPE), so the 10000.0 default would apply to all of them. @@ -67,6 +66,9 @@ llama_model_onyx::graph::graph(const llama_model & model, const llm_graph_params const int64_t n_embd_head = hparams.n_embd_head_v(); GGML_ASSERT(n_embd_head == hparams.n_embd_head_k()); + // Different to f_norm_rms_eps for post-attn / post-FFN norms + const float post_norm_eps = 1e-8f; + ggml_tensor * cur; ggml_tensor * inpL; @@ -143,8 +145,7 @@ llama_model_onyx::graph::graph(const llama_model & model, const llm_graph_params cb(cur, "attn_o_proj", il); } - // post-attention norm (uses f_post_norm_rms_eps, not the general f_norm_rms_eps). - cur = ggml_rms_norm(ctx0, cur, hparams.f_post_norm_rms_eps); + cur = ggml_rms_norm(ctx0, cur, post_norm_eps); cur = ggml_mul(ctx0, cur, model.layers[il].attn_post_norm); cb(cur, "attn_post_norm", il); @@ -169,8 +170,7 @@ llama_model_onyx::graph::graph(const llama_model & model, const llm_graph_params LLM_FFN_SILU, LLM_FFN_PAR, il); cb(cur, "ffn_out", il); - // post-FFN norm (uses f_post_norm_rms_eps, not the general f_norm_rms_eps). - cur = ggml_rms_norm(ctx0, cur, hparams.f_post_norm_rms_eps); + cur = ggml_rms_norm(ctx0, cur, post_norm_eps); cur = ggml_mul(ctx0, cur, model.layers[il].ffn_post_norm); cb(cur, "ffn_post_norm", il);