This commit is contained in:
Pedro Cuenca
2026-07-27 10:31:19 +02:00
parent 73aad674b2
commit c557a81ead
+4 -8
View File
@@ -5,10 +5,7 @@ void llama_model_onyx::load_arch_hparams(llama_model_loader & ml) {
ml.get_key(LLM_KV_ATTENTION_SLIDING_WINDOW, hparams.n_swa, false);
ml.get_key(LLM_KV_FINAL_LOGIT_SOFTCAPPING, hparams.f_final_logit_softcapping, false);
// ISWA period + NoPE tie: Onyx's [SW, SW, SW, Full] pattern has NoPE on the
// full-attention layers, so `n_no_rope_layer_step` shares the SWA period.
// (afmoe.cpp:13-19 sets up the SWA pattern the same way; the NoPE tie is
// Onyx-specific — afmoe leaves `n_no_rope_layer_step` at its default.)
// SWA + NoPE: [SW, SW, SW, Full], NoPE used on Full layers.
if (hparams.n_swa > 0) {
hparams.swa_type = LLAMA_SWA_TYPE_STANDARD;
uint32_t swa_period = 4;
@@ -32,12 +29,11 @@ void llama_model_onyx::load_arch_tensors(llama_model_loader &) {
for (int i = 0; i < n_layer; ++i) {
auto & layer = layers[i];
// Pre/post-attention norms (Onyx's `weight + 1` fold applied at conversion time).
// Pre/post-attention norms (Onyx's `weight + 1` applied at conversion time).
layer.attn_norm = create_tensor(tn(LLM_TENSOR_ATTN_NORM, "weight", i), {n_embd}, 0);
layer.attn_post_norm = create_tensor(tn(LLM_TENSOR_ATTN_POST_NORM, "weight", i), {n_embd}, 0);
// Q/K/V/O projections. `create_tensor_qkv` handles the split-vs-merged layout
// and optional biases (Onyx has no biases; helper skips them cleanly).
// Q/K/V/O projections.
create_tensor_qkv(layer, i, n_embd, n_embd_head_k * n_head, n_embd_k_gqa, n_embd_v_gqa, 0);
layer.wo = create_tensor(tn(LLM_TENSOR_ATTN_OUT, "weight", i), {n_embd_head_k * n_head, n_embd}, 0);
@@ -45,7 +41,7 @@ void llama_model_onyx::load_arch_tensors(llama_model_loader &) {
layer.attn_q_norm = create_tensor(tn(LLM_TENSOR_ATTN_Q_NORM, "weight", i), {n_embd_head_k}, 0);
layer.attn_k_norm = create_tensor(tn(LLM_TENSOR_ATTN_K_NORM, "weight", i), {n_embd_head_k}, 0);
// Attention output gate: sigmoid(gate) * attn_out before o_proj (afmoe.cpp:73).
// Attention output gate: sigmoid(gate) * attn_out before o_proj (same as afmoe).
layer.wqkv_gate = create_tensor(tn(LLM_TENSOR_ATTN_GATE, "weight", i), {n_embd, n_embd_head_k * n_head}, 0);
// Pre/post-FFN norms (FFN_PRE_NORM is aliased to LLM_TENSOR_FFN_NORM).