From 73aad674b2938d0fe4fbe6afc932f87a0ffef032 Mon Sep 17 00:00:00 2001 From: Pedro Cuenca Date: Mon, 27 Jul 2026 10:21:38 +0200 Subject: [PATCH] Loading tensors --- src/llama-arch.cpp | 1 + src/llama-arch.h | 1 + src/llama-model.cpp | 3 ++ src/models/models.h | 13 +++++++++ src/models/onyx.cpp | 69 +++++++++++++++++++++++++++++++++++++++++++++ 5 files changed, 87 insertions(+) create mode 100644 src/models/onyx.cpp diff --git a/src/llama-arch.cpp b/src/llama-arch.cpp index 39bf2c7959..1ee5f123fe 100644 --- a/src/llama-arch.cpp +++ b/src/llama-arch.cpp @@ -71,6 +71,7 @@ static const std::map LLM_ARCH_NAMES = { { LLM_ARCH_OLMO, "olmo" }, { LLM_ARCH_OLMO2, "olmo2" }, { LLM_ARCH_OLMOE, "olmoe" }, + { LLM_ARCH_ONYX, "onyx" }, { LLM_ARCH_OPENELM, "openelm" }, { LLM_ARCH_ARCTIC, "arctic" }, { LLM_ARCH_DEEPSEEK, "deepseek" }, diff --git a/src/llama-arch.h b/src/llama-arch.h index 2e3916a0be..9a2f790c10 100644 --- a/src/llama-arch.h +++ b/src/llama-arch.h @@ -76,6 +76,7 @@ enum llm_arch { LLM_ARCH_OLMO, LLM_ARCH_OLMO2, LLM_ARCH_OLMOE, + LLM_ARCH_ONYX, LLM_ARCH_OPENELM, LLM_ARCH_ARCTIC, LLM_ARCH_DEEPSEEK, diff --git a/src/llama-model.cpp b/src/llama-model.cpp index 5179692108..575b9a1bf2 100644 --- a/src/llama-model.cpp +++ b/src/llama-model.cpp @@ -169,6 +169,8 @@ static llama_model * llama_model_mapping(llm_arch arch, const llama_model_params return new llama_model_olmo2(params); case LLM_ARCH_OLMOE: return new llama_model_olmoe(params); + case LLM_ARCH_ONYX: + return new llama_model_onyx(params); case LLM_ARCH_OPENELM: return new llama_model_openelm(params); case LLM_ARCH_GPTNEOX: @@ -2473,6 +2475,7 @@ llama_rope_type llama_model_rope_type(const llama_model * model) { case LLM_ARCH_DEEPSEEK2OCR: case LLM_ARCH_DEEPSEEK32: case LLM_ARCH_DEEPSEEK4: + case LLM_ARCH_ONYX: case LLM_ARCH_PLM: case LLM_ARCH_CHATGLM: case LLM_ARCH_GRANITE: diff --git a/src/models/models.h b/src/models/models.h index 916459e127..7c7e092e0e 100644 --- a/src/models/models.h +++ b/src/models/models.h @@ -1007,6 +1007,19 @@ struct llama_model_olmoe : public llama_model_base { }; +struct llama_model_onyx : public llama_model_base { + llama_model_onyx(const struct llama_model_params & params) : llama_model_base(params) {} + void load_arch_hparams(llama_model_loader & ml) override; + void load_arch_tensors(llama_model_loader & ml) override; + + struct graph : public llm_graph_context { + graph(const llama_model & model, const llm_graph_params & params); + }; + + std::unique_ptr build_arch_graph(const llm_graph_params & params) const override; +}; + + struct llama_model_openelm : public llama_model_base { llama_model_openelm(const struct llama_model_params & params) : llama_model_base(params) {} void load_arch_hparams(llama_model_loader & ml) override; diff --git a/src/models/onyx.cpp b/src/models/onyx.cpp new file mode 100644 index 0000000000..a304049116 --- /dev/null +++ b/src/models/onyx.cpp @@ -0,0 +1,69 @@ +#include "models.h" + +void llama_model_onyx::load_arch_hparams(llama_model_loader & ml) { + ml.get_key(LLM_KV_ATTENTION_LAYERNORM_RMS_EPS, hparams.f_norm_rms_eps); + ml.get_key(LLM_KV_ATTENTION_SLIDING_WINDOW, hparams.n_swa, false); + ml.get_key(LLM_KV_FINAL_LOGIT_SOFTCAPPING, hparams.f_final_logit_softcapping, false); + + // ISWA period + NoPE tie: Onyx's [SW, SW, SW, Full] pattern has NoPE on the + // full-attention layers, so `n_no_rope_layer_step` shares the SWA period. + // (afmoe.cpp:13-19 sets up the SWA pattern the same way; the NoPE tie is + // Onyx-specific — afmoe leaves `n_no_rope_layer_step` at its default.) + if (hparams.n_swa > 0) { + hparams.swa_type = LLAMA_SWA_TYPE_STANDARD; + uint32_t swa_period = 4; + ml.get_key_or_arr(LLM_KV_ATTENTION_SLIDING_WINDOW_PATTERN, swa_period, false); + hparams.set_swa_pattern(swa_period); + hparams.n_no_rope_layer_step = swa_period; + } else { + hparams.swa_type = LLAMA_SWA_TYPE_NONE; + } + + type = LLM_TYPE_UNKNOWN; +} + +void llama_model_onyx::load_arch_tensors(llama_model_loader &) { + LLAMA_LOAD_LOCALS; + + tok_embd = create_tensor(tn(LLM_TENSOR_TOKEN_EMBD, "weight"), {n_embd, n_vocab}, 0); + output_norm = create_tensor(tn(LLM_TENSOR_OUTPUT_NORM, "weight"), {n_embd}, 0); + output = create_tensor(tn(LLM_TENSOR_OUTPUT, "weight"), {n_embd, n_vocab}, 0); + + for (int i = 0; i < n_layer; ++i) { + auto & layer = layers[i]; + + // Pre/post-attention norms (Onyx's `weight + 1` fold applied at conversion time). + layer.attn_norm = create_tensor(tn(LLM_TENSOR_ATTN_NORM, "weight", i), {n_embd}, 0); + layer.attn_post_norm = create_tensor(tn(LLM_TENSOR_ATTN_POST_NORM, "weight", i), {n_embd}, 0); + + // Q/K/V/O projections. `create_tensor_qkv` handles the split-vs-merged layout + // and optional biases (Onyx has no biases; helper skips them cleanly). + create_tensor_qkv(layer, i, n_embd, n_embd_head_k * n_head, n_embd_k_gqa, n_embd_v_gqa, 0); + layer.wo = create_tensor(tn(LLM_TENSOR_ATTN_OUT, "weight", i), {n_embd_head_k * n_head, n_embd}, 0); + + // QK-norm. Weights are synthesized at conversion time to absorb `qk_scale_factor`. + layer.attn_q_norm = create_tensor(tn(LLM_TENSOR_ATTN_Q_NORM, "weight", i), {n_embd_head_k}, 0); + layer.attn_k_norm = create_tensor(tn(LLM_TENSOR_ATTN_K_NORM, "weight", i), {n_embd_head_k}, 0); + + // Attention output gate: sigmoid(gate) * attn_out before o_proj (afmoe.cpp:73). + layer.wqkv_gate = create_tensor(tn(LLM_TENSOR_ATTN_GATE, "weight", i), {n_embd, n_embd_head_k * n_head}, 0); + + // Pre/post-FFN norms (FFN_PRE_NORM is aliased to LLM_TENSOR_FFN_NORM). + layer.ffn_norm = create_tensor(tn(LLM_TENSOR_FFN_NORM, "weight", i), {n_embd}, 0); + layer.ffn_post_norm = create_tensor(tn(LLM_TENSOR_FFN_POST_NORM, "weight", i), {n_embd}, 0); + + // Dense FFN (unlike afmoe, no MoE branches). + layer.ffn_gate = create_tensor(tn(LLM_TENSOR_FFN_GATE, "weight", i), {n_embd, n_ff}, 0); + layer.ffn_down = create_tensor(tn(LLM_TENSOR_FFN_DOWN, "weight", i), {n_ff, n_embd}, 0); + layer.ffn_up = create_tensor(tn(LLM_TENSOR_FFN_UP, "weight", i), {n_embd, n_ff}, 0); + } +} + +llama_model_onyx::graph::graph(const llama_model & /*model*/, const llm_graph_params & params) + : llm_graph_context(params) { + GGML_ABORT("onyx: build_arch_graph not implemented yet"); +} + +std::unique_ptr llama_model_onyx::build_arch_graph(const llm_graph_params & params) const { + return std::make_unique(*this, params); +}