Files
llama.cpp/src/models/qwen4exp.cpp
T
Daniel Han ddf0980e52 llama: fix the qwen4exp PLE conv state and unblock test-llama-archs
build_rs writes into the state tensor in place, zeroing one row and copying the
carried-over states, so calling it twice for the same layer let the second call
clobber the first write-back. The PLE layer is also a delta-net layer, so that
is exactly what happened: both convolutions gathered the same row. They now
share a single gather per layer.

The earlier claim that the conv state was carried correctly was tested on a
fixture whose conv weights are zero, where the branch contributes nothing and
chunking matches trivially. Re-running with non-zero conv weights showed the
divergence, growing with the number of ubatch boundaries: 97.1% top-1 at one
boundary down to 90.2% at seven. With the shared gather it is bit-identical to
the single-shot run at every chunk size tried, 512, 128 and 64, with a maximum
logprob deviation of exactly zero over 1023 positions. The delta-net-only model
stays bit-identical too, so nothing regressed there.

Also derive the delta-net conv channel count the way load_arch_tensors sizes
wqkv instead of from ssm_d_inner. The two agree for this model, but n_embd_r()
only bounds the row and the convolution has to match the tensor feeding it.

test-llama-archs previously aborted on this architecture and took every later
architecture with it. qwen4exp is marked MoE-only, given the hyper-connection
keys and an ssm_d_inner consistent with its tensor derivation, and skipped for
now: the hyper-connection keys written by get_gguf_ctx are not reaching the
synthesised file, which needs a separate look. The suite completes again, 124
architectures at 0.00e+00.
2026-08-26 13:40:52 +00:00

925 lines
40 KiB
C++

#include "models.h"
#include "llama-memory-recurrent.h"
void llama_model_qwen4exp::load_arch_hparams(llama_model_loader & ml) {
ml.get_key(LLM_KV_EXPERT_FEED_FORWARD_LENGTH, hparams.n_ff_exp, false);
ml.get_key(LLM_KV_EXPERT_SHARED_FEED_FORWARD_LENGTH, hparams.n_ff_shexp, false);
ml.get_key(LLM_KV_ATTENTION_LAYERNORM_RMS_EPS, hparams.f_norm_rms_eps);
ml.get_key_or_arr(LLM_KV_ROPE_DIMENSION_SECTIONS, hparams.rope_sections, 4, true);
ml.get_key(LLM_KV_SSM_CONV_KERNEL, hparams.ssm_d_conv);
ml.get_key(LLM_KV_SSM_INNER_SIZE, hparams.ssm_d_inner);
ml.get_key(LLM_KV_SSM_STATE_SIZE, hparams.ssm_d_state);
ml.get_key(LLM_KV_SSM_TIME_STEP_RANK, hparams.ssm_dt_rank);
ml.get_key(LLM_KV_SSM_GROUP_COUNT, hparams.ssm_n_group);
// HC; low_rank is qwen4exp-specific, DeepSeek-V4 leaves it absent (full rank)
ml.get_key(LLM_KV_HYPER_CONNECTION_COUNT, hparams.dsv4_hc_mult);
ml.get_key(LLM_KV_HYPER_CONNECTION_LOW_RANK, hparams.hc_low_rank);
GGML_ASSERT(hparams.dsv4_hc_mult > 0 && "qwen4exp needs a hyper-connection count");
GGML_ASSERT(hparams.hc_low_rank > 0 && "qwen4exp needs a hyper-connection low rank");
hparams.n_embd_out_impl = hparams.dsv4_hc_mult * hparams.n_embd;
ml.get_key(LLM_KV_ATTENTION_INDEXER_HEAD_COUNT, hparams.indexer_n_head);
ml.get_key(LLM_KV_ATTENTION_INDEXER_KEY_LENGTH, hparams.indexer_head_size);
ml.get_key(LLM_KV_ATTENTION_INDEXER_TOP_K, hparams.indexer_top_k);
ml.get_key_or_arr(LLM_KV_ATTENTION_COMPRESS_RATIOS, hparams.dsv4_compress_ratios, hparams.n_layer_all, false);
// PLE n-gram hash embeddings; if the key group is absent every field stays zero
std::fill(hparams.is_ple_impl.begin(), hparams.is_ple_impl.end(), 0);
hparams.ple_n_heads = 0;
uint32_t n_ple = 0;
ml.get_arr_n(LLM_KV_PLE_LAYERS, n_ple, false);
if (n_ple > 0) {
std::vector<uint32_t> ple_layers;
ml.get_arr(LLM_KV_PLE_LAYERS, ple_layers);
for (uint32_t il : ple_layers) {
GGML_ASSERT(il < hparams.n_layer_all);
hparams.is_ple_impl[il] = 1;
}
ml.get_key(LLM_KV_PLE_NGRAM_SIZE, hparams.ple_ngram_size);
ml.get_key(LLM_KV_PLE_HEADS_PER_NGRAM, hparams.ple_heads_per_ngram);
ml.get_key(LLM_KV_PLE_CONV_KERNEL, hparams.ple_conv_kernel);
ml.get_key(LLM_KV_PLE_EOS_TOKEN_ID, hparams.ple_eos_token_id);
ml.get_key(LLM_KV_EMBEDDING_LENGTH_PER_LAYER, hparams.n_embd_per_layer);
hparams.ple_n_heads = (hparams.ple_ngram_size - 1) * hparams.ple_heads_per_ngram;
hparams.ple_head_dim = hparams.n_embd_per_layer;
GGML_ASSERT(hparams.ple_ngram_size >= 2 && hparams.ple_ngram_size <= LLAMA_MAX_PLE_NGRAM);
GGML_ASSERT(hparams.ple_n_heads > 0 && hparams.ple_n_heads <= LLAMA_MAX_PLE_HEADS);
ml.get_arr(LLM_KV_PLE_LAYER_MULTIPLIERS, hparams.ple_layer_multipliers);
ml.get_arr(LLM_KV_PLE_HEAD_OFFSETS, hparams.ple_head_offsets);
ml.get_arr(LLM_KV_PLE_HEAD_VOCAB_SIZES, hparams.ple_head_vocab_sizes);
}
// linear attention everywhere except every full_attention_interval-th layer
if (!ml.get_key_or_arr(LLM_KV_ATTENTION_RECURRENT_LAYERS, hparams.is_recr_impl, hparams.n_layer_all, false)) {
uint32_t full_attn_interval = 4;
ml.get_key(LLM_KV_FULL_ATTENTION_INTERVAL, full_attn_interval, false);
for (uint32_t i = 0; i < hparams.n_layer_all; ++i) {
hparams.is_recr_impl[i] = (i < hparams.n_layer()) && ((i + 1) % full_attn_interval != 0);
}
}
switch (hparams.n_layer()) {
case 48: type = LLM_TYPE_A3B; break;
default: type = LLM_TYPE_UNKNOWN;
}
}
void llama_model_qwen4exp::load_arch_tensors(llama_model_loader & ml) {
LLAMA_LOAD_LOCALS;
const int64_t hc = hparams.dsv4_hc_mult;
const int64_t hc_dim = hc * n_embd;
const int64_t hc_lr = hparams.hc_low_rank;
tok_embd = create_tensor(tn(LLM_TENSOR_TOKEN_EMBD, "weight"), { n_embd, n_vocab }, 0);
// there is no output_norm: the final hyper-connection mixer carries it
hc_head_norm = create_tensor(tn(LLM_TENSOR_HC_HEAD_NORM, "weight"), { hc_dim }, 0);
hc_head_down = create_tensor(tn(LLM_TENSOR_HC_HEAD_DOWN, "weight"), { hc_dim, hc_lr }, 0);
hc_head_up = create_tensor(tn(LLM_TENSOR_HC_HEAD_UP, "weight"), { hc_lr, hc_dim }, 0);
output = create_tensor(tn(LLM_TENSOR_OUTPUT, "weight"), { n_embd, n_vocab }, TENSOR_NOT_REQUIRED);
if (output == NULL) {
output = create_tensor(tn(LLM_TENSOR_TOKEN_EMBD, "weight"), { n_embd, n_vocab }, TENSOR_DUPLICATED);
}
// flat [ple_head_dim, n_rows] gather target; n_rows is padded, so read it back
if (hparams.ple_n_heads > 0) {
const std::string ple_name = tn(LLM_TENSOR_PER_LAYER_TOKEN_EMBD, "weight").str();
const auto * ple_w = ml.get_weight(ple_name.c_str());
GGML_ASSERT(ple_w != nullptr && "qwen4exp is missing the PLE n-gram table");
const int64_t ple_rows = ple_w->tensor->ne[1];
per_layer_tok_embd = create_tensor(tn(LLM_TENSOR_PER_LAYER_TOKEN_EMBD, "weight"),
{ hparams.ple_head_dim, ple_rows }, 0);
}
for (int il = 0; il < n_layer; ++il) {
auto & layer = layers[il];
const int64_t n_ff_exp = hparams.n_ff_exp ? hparams.n_ff_exp : n_ff / n_expert_used;
const int64_t n_ff_shexp = hparams.n_ff_shexp ? hparams.n_ff_shexp : n_ff;
const int64_t head_k_dim = hparams.ssm_d_state;
const int64_t head_v_dim = hparams.ssm_d_state;
const int64_t n_k_heads = hparams.ssm_n_group;
const int64_t n_v_heads = hparams.ssm_dt_rank;
const int64_t key_dim = head_k_dim * n_k_heads;
const int64_t value_dim = head_v_dim * n_v_heads;
const int64_t conv_dim = key_dim * 2 + value_dim;
// two HC modules per layer: before the token mixer, before the MoE
layer.hc_attn_norm = create_tensor(tn(LLM_TENSOR_HC_ATTN_NORM, "weight", il), { hc_dim }, 0);
layer.hc_attn_down = create_tensor(tn(LLM_TENSOR_HC_ATTN_DOWN, "weight", il), { hc_dim, hc_lr }, 0);
layer.hc_attn_up = create_tensor(tn(LLM_TENSOR_HC_ATTN_UP, "weight", il), { hc_lr, hc_dim }, 0);
layer.hc_attn_inject = create_tensor(tn(LLM_TENSOR_HC_ATTN_INJECT, "weight", il), { hc_dim, hc }, 0);
layer.hc_ffn_norm = create_tensor(tn(LLM_TENSOR_HC_FFN_NORM, "weight", il), { hc_dim }, 0);
layer.hc_ffn_down = create_tensor(tn(LLM_TENSOR_HC_FFN_DOWN, "weight", il), { hc_dim, hc_lr }, 0);
layer.hc_ffn_up = create_tensor(tn(LLM_TENSOR_HC_FFN_UP, "weight", il), { hc_lr, hc_dim }, 0);
layer.hc_ffn_inject = create_tensor(tn(LLM_TENSOR_HC_FFN_INJECT, "weight", il), { hc_dim, hc }, 0);
if (!hparams.is_recr(il)) {
// full attention: wq holds [q|gate] interleaved per head
create_tensor_qkv(layer, il, n_embd, n_embd_head_k * n_head * 2, n_embd_k_gqa, n_embd_v_gqa, 0);
layer.wo = create_tensor(tn(LLM_TENSOR_ATTN_OUT, "weight", il), { n_embd_head_k * n_head, n_embd }, 0);
layer.attn_q_norm = create_tensor(tn(LLM_TENSOR_ATTN_Q_NORM, "weight", il), { n_embd_head_k }, 0);
layer.attn_k_norm = create_tensor(tn(LLM_TENSOR_ATTN_K_NORM, "weight", il), { n_embd_head_k }, 0);
const int64_t idx_dim = hparams.indexer_head_size;
layer.index_q_proj = create_tensor(tn(LLM_TENSOR_INDEXER_Q_PROJ, "weight", il), { n_embd, hparams.indexer_n_head * idx_dim }, 0);
layer.index_k_proj = create_tensor(tn(LLM_TENSOR_INDEXER_K_PROJ, "weight", il), { n_embd, idx_dim }, 0);
layer.index_q_norm = create_tensor(tn(LLM_TENSOR_INDEXER_Q_NORM, "weight", il), { idx_dim }, 0);
layer.index_k_norm = create_tensor(tn(LLM_TENSOR_INDEXER_K_NORM, "weight", il), { idx_dim }, 0);
} else {
layer.wqkv = create_tensor(tn(LLM_TENSOR_ATTN_QKV, "weight", il), { n_embd, key_dim * 2 + value_dim }, 0);
layer.wqkv_gate = create_tensor(tn(LLM_TENSOR_ATTN_GATE, "weight", il), { n_embd, value_dim }, 0);
layer.ssm_conv1d = create_tensor(tn(LLM_TENSOR_SSM_CONV1D, "weight", il), { hparams.ssm_d_conv, conv_dim }, 0);
layer.ssm_dt = create_tensor(tn(LLM_TENSOR_SSM_DT, "bias", il), { hparams.ssm_dt_rank }, 0);
layer.ssm_a = create_tensor(tn(LLM_TENSOR_SSM_A_NOSCAN, il), { hparams.ssm_dt_rank }, 0);
layer.ssm_beta = create_tensor(tn(LLM_TENSOR_SSM_BETA, "weight", il), { n_embd, n_v_heads }, 0);
layer.ssm_alpha = create_tensor(tn(LLM_TENSOR_SSM_ALPHA, "weight", il), { n_embd, n_v_heads }, 0);
layer.ssm_norm = create_tensor(tn(LLM_TENSOR_SSM_NORM, "weight", il), { head_v_dim }, 0);
layer.ssm_out = create_tensor(tn(LLM_TENSOR_SSM_OUT, "weight", il), { value_dim, n_embd }, 0);
}
if (hparams.is_ple(il)) {
layer.ple_key = create_tensor(tn(LLM_TENSOR_PLE_KEY, "weight", il), { n_embd, hc_dim }, 0);
layer.ple_value = create_tensor(tn(LLM_TENSOR_PLE_VALUE, "weight", il), { n_embd, n_embd }, 0);
layer.ple_norm_key = create_tensor(tn(LLM_TENSOR_PLE_NORM_KEY, "weight", il), { hc_dim }, 0);
layer.ple_norm_query = create_tensor(tn(LLM_TENSOR_PLE_NORM_QUERY, "weight", il), { hc_dim }, 0);
layer.ple_norm_conv = create_tensor(tn(LLM_TENSOR_PLE_NORM_CONV, "weight", il), { hc_dim }, 0);
layer.ple_conv1d = create_tensor(tn(LLM_TENSOR_PLE_CONV1D, "weight", il), { hparams.ple_conv_kernel, hc_dim }, 0);
}
layer.ffn_gate_inp = create_tensor(tn(LLM_TENSOR_FFN_GATE_INP, "weight", il), { n_embd, n_expert }, 0);
layer.ffn_down_exps = create_tensor(tn(LLM_TENSOR_FFN_DOWN_EXPS, "weight", il), { n_ff_exp, n_embd, n_expert }, 0);
create_tensor_gate_up_exps(layer, il, n_embd, n_ff_exp, n_expert, 0);
layer.ffn_gate_inp_shexp = create_tensor(tn(LLM_TENSOR_FFN_GATE_INP_SHEXP, "weight", il), { n_embd }, 0);
layer.ffn_gate_shexp = create_tensor(tn(LLM_TENSOR_FFN_GATE_SHEXP, "weight", il), { n_embd, n_ff_shexp }, 0);
layer.ffn_up_shexp = create_tensor(tn(LLM_TENSOR_FFN_UP_SHEXP, "weight", il), { n_embd, n_ff_shexp }, 0);
layer.ffn_down_shexp = create_tensor(tn(LLM_TENSOR_FFN_DOWN_SHEXP, "weight", il), { n_ff_shexp, n_embd }, 0);
}
}
std::unique_ptr<llm_graph_context> llama_model_qwen4exp::build_arch_graph(const llm_graph_params & params) const {
return std::make_unique<graph>(*this, params);
}
// Hyper-connections replace every layer norm: the state between blocks is `hc`
// parallel residual streams [n_embd, hc, T]; each block reads one mixed [n_embd, T]
// view and writes back through per-stream injection weights.
// Not shared with deepseek4.cpp: DSV4 mixes full-rank + Sinkhorn, this is a
// low-rank down/silu/up gate with a plain mean collapse.
// The mix output is [n_embd, T]; `inject` receives the [hc, T] scatter weights.
ggml_tensor * llama_model_qwen4exp::graph::build_hc_mix(
ggml_tensor * x,
ggml_tensor * w_norm,
ggml_tensor * w_down,
ggml_tensor * w_up,
ggml_tensor * w_inject,
ggml_tensor ** inject,
int il) {
const int64_t hc = hparams.dsv4_hc_mult;
const int64_t hc_dim = hc * n_embd;
const int64_t nt = x->ne[2];
// grouped RMSNorm: rms_norm reduces over one residual stream, then the [hc_dim]
// gamma scales all streams. Gammas were folded to (1 + w) by the converter.
ggml_tensor * xn = ggml_rms_norm(ctx0, x, hparams.f_norm_rms_eps);
xn = ggml_reshape_2d(ctx0, xn, hc_dim, nt);
xn = ggml_mul(ctx0, xn, w_norm);
cb(xn, "hc_norm", il);
ggml_tensor * lo = build_lora_mm(w_down, xn);
lo = ggml_silu(ctx0, ggml_scale(ctx0, lo, 1.0f / (float) hc));
ggml_tensor * gate = ggml_sigmoid(ctx0, build_lora_mm(w_up, lo));
cb(gate, "hc_gate", il);
ggml_tensor * gated = ggml_mul(ctx0, xn, gate);
gated = ggml_reshape_3d(ctx0, gated, n_embd, hc, nt);
// collapse the streams by their mean
ggml_tensor * mixed = ggml_view_2d(ctx0, gated, n_embd, nt,
ggml_row_size(gated->type, n_embd) * hc, 0);
mixed = ggml_cont(ctx0, mixed);
for (int64_t c = 1; c < hc; ++c) {
ggml_tensor * s = ggml_view_2d(ctx0, gated, n_embd, nt,
ggml_row_size(gated->type, n_embd) * hc,
ggml_row_size(gated->type, n_embd) * c);
mixed = ggml_add(ctx0, mixed, s);
}
mixed = ggml_scale(ctx0, mixed, 1.0f / (float) hc);
cb(mixed, "hc_mixed", il);
if (inject) {
*inject = build_lora_mm(w_inject, xn);
cb(*inject, "hc_inject", il);
}
return mixed;
}
ggml_tensor * llama_model_qwen4exp::graph::build_hc_combine(
ggml_tensor * residual,
ggml_tensor * block_out,
ggml_tensor * inject,
int il) {
const int64_t hc = hparams.dsv4_hc_mult;
const int64_t nt = residual->ne[2];
// 2*sigmoid centres the scatter weights on 1, so an untrained injection matrix
// reproduces the plain residual add
ggml_tensor * w = ggml_sigmoid(ctx0, ggml_scale(ctx0, inject, 1.0f / (float) hc));
w = ggml_scale(ctx0, w, 2.0f);
w = ggml_reshape_3d(ctx0, w, 1, hc, nt);
ggml_tensor * b = ggml_reshape_3d(ctx0, block_out, n_embd, 1, nt);
b = ggml_repeat_4d(ctx0, b, n_embd, hc, nt, 1);
ggml_tensor * cur = ggml_add(ctx0, residual, ggml_mul(ctx0, b, w));
cb(cur, "hc_combine", il);
return cur;
}
llama_model_qwen4exp::graph::graph(const llama_model & model, const llm_graph_params & params) :
llm_build_delta_net_base(params), model(model) {
const int64_t hc = hparams.dsv4_hc_mult;
GGML_ASSERT(hparams.n_embd_head_v() == hparams.n_embd_head_k());
int sections[4];
std::copy(std::begin(hparams.rope_sections), std::begin(hparams.rope_sections) + 4, sections);
ggml_tensor * inpL = build_inp_embd(model.tok_embd);
cb(inpL, "model.input_embed", -1);
auto * inp = build_inp_mem_hybrid();
ggml_tensor * inp_pos = build_inp_pos();
ggml_tensor * inp_out_ids = build_inp_out_ids();
// the wide residual starts as hc identical copies of the embedding
ggml_tensor * res_hc = ggml_repeat_4d(ctx0,
ggml_reshape_3d(ctx0, inpL, n_embd, 1, n_tokens),
n_embd, hc, n_tokens, 1);
cb(res_hc, "hc_init", -1);
for (int il = 0; il < n_layer; ++il) {
res->t_layer_inp[il] = res_hc;
if (hparams.is_ple(il)) {
res_hc = build_ple(inp->get_recr(), res_hc, il);
}
ggml_tensor * inject = nullptr;
ggml_tensor * cur = build_hc_mix(res_hc,
model.layers[il].hc_attn_norm,
model.layers[il].hc_attn_down,
model.layers[il].hc_attn_up,
model.layers[il].hc_attn_inject,
&inject, il);
ggml_build_forward_expand(gf, cur);
if (hparams.is_recr(il)) {
cur = build_layer_attn_linear(inp->get_recr(), cur, il);
} else {
cur = build_layer_attn(inp->get_attn(), cur, inp_pos, sections, il);
}
res_hc = build_hc_combine(res_hc, cur, inject, il);
cur = build_hc_mix(res_hc,
model.layers[il].hc_ffn_norm,
model.layers[il].hc_ffn_down,
model.layers[il].hc_ffn_up,
model.layers[il].hc_ffn_inject,
&inject, il);
cur = build_layer_ffn(cur, il);
cb(cur, "ffn_out", il);
res_hc = build_hc_combine(res_hc, cur, inject, il);
// build_cvec expects [n_embd, T], so steer the stream mean and let the next mix
// carry it. Tagged "l_last": the layer-output name imatrix_FIXED.cpp parses.
cb(res_hc, "l_last", il);
}
// the final mixer is the output norm: there is no separate one
ggml_tensor * cur = build_hc_mix(res_hc,
model.hc_head_norm, model.hc_head_down, model.hc_head_up,
nullptr, nullptr, -1);
if (inp_out_ids) {
cur = ggml_get_rows(ctx0, cur, inp_out_ids);
}
cb(cur, "result_norm", -1);
res->t_embd = cur;
cur = build_lora_mm(model.output, cur, model.output_s);
cb(cur, "result_output", -1);
res->t_logits = cur;
ggml_build_forward_expand(gf, cur);
}
std::pair<ggml_tensor *, ggml_tensor *> llama_model_qwen4exp::graph::build_qkvz(
ggml_tensor * input,
int il) {
const int64_t n_seqs = ubatch.n_seqs;
const int64_t n_seq_tokens = ubatch.n_seq_tokens;
ggml_tensor * qkv_mixed = build_lora_mm(model.layers[il].wqkv, input, model.layers[il].wqkv_s);
qkv_mixed = ggml_reshape_3d(ctx0, qkv_mixed, qkv_mixed->ne[0], n_seq_tokens, n_seqs);
cb(qkv_mixed, "linear_attn_qkv_mixed", il);
ggml_tensor * z = build_lora_mm(model.layers[il].wqkv_gate, input, model.layers[il].wqkv_gate_s);
cb(z, "z", il);
return { qkv_mixed, z };
}
ggml_tensor * llama_model_qwen4exp::graph::build_norm_gated(
ggml_tensor * input,
ggml_tensor * weights,
ggml_tensor * gate,
int layer) {
// the one numerical difference from Qwen3.5's GDN: sigmoid output gate, not silu
ggml_tensor * normalized = build_norm(input, weights, nullptr, LLM_NORM_RMS, layer);
ggml_tensor * gated = ggml_sigmoid(ctx0, gate);
return ggml_mul(ctx0, normalized, gated);
}
ggml_tensor * llama_model_qwen4exp::graph::build_layer_attn(
llm_graph_input_attn_kv * inp,
ggml_tensor * cur,
ggml_tensor * inp_pos,
int * sections,
int il) {
const int64_t n_embd_head = hparams.n_embd_head_v();
GGML_ASSERT(n_embd_head == hparams.n_embd_head_k());
// Order: joint QG projection, QG split, Q norm, KV projection, K norm, RoPE, attention
// Qwen3Next uses a single Q projection that outputs query + gate
ggml_tensor * Qcur_full = build_lora_mm(model.layers[il].wq, cur, model.layers[il].wq_s); // [ (n_embd_head * 2) * n_head, n_tokens ]
cb(Qcur_full, "Qcur_full", il);
ggml_tensor * Qcur = ggml_view_3d(ctx0, Qcur_full, n_embd_head, n_head, n_tokens,
ggml_element_size(Qcur_full) * n_embd_head * 2,
ggml_element_size(Qcur_full) * n_embd_head * 2 * n_head, 0);
cb(Qcur, "Qcur_reshaped", il);
// Apply Q normalization
Qcur = build_norm(Qcur, model.layers[il].attn_q_norm, nullptr, LLM_NORM_RMS, il);
cb(Qcur, "Qcur_normed", il);
ggml_tensor * Kcur = build_lora_mm(model.layers[il].wk, cur, model.layers[il].wk_s);
cb(Kcur, "Kcur", il);
ggml_tensor * Vcur = build_lora_mm(model.layers[il].wv, cur, model.layers[il].wv_s);
cb(Vcur, "Vcur", il);
// Apply K normalization
Kcur = ggml_reshape_3d(ctx0, Kcur, n_embd_head, n_head_kv, n_tokens);
Kcur = build_norm(Kcur, model.layers[il].attn_k_norm, nullptr, LLM_NORM_RMS, il);
cb(Kcur, "Kcur_normed", il);
ggml_tensor * gate = ggml_view_3d(ctx0, Qcur_full, n_embd_head, n_head, n_tokens,
ggml_element_size(Qcur_full) * n_embd_head * 2,
ggml_element_size(Qcur_full) * n_embd_head * 2 * n_head,
ggml_element_size(Qcur_full) * n_embd_head);
gate = ggml_cont_2d(ctx0, gate, n_embd_head * n_head, n_tokens);
cb(gate, "gate_reshaped", il);
Vcur = ggml_reshape_3d(ctx0, Vcur, n_embd_head, n_head_kv, n_tokens);
// Apply IMRoPE
Qcur = ggml_rope_multi(
ctx0, Qcur, inp_pos, nullptr,
n_rot, sections, rope_type, n_ctx_orig, freq_base, freq_scale,
ext_factor, attn_factor, beta_fast, beta_slow
);
Kcur = ggml_rope_multi(
ctx0, Kcur, inp_pos, nullptr,
n_rot, sections, rope_type, n_ctx_orig, freq_base, freq_scale,
ext_factor, attn_factor, beta_fast, beta_slow
);
cb(Qcur, "Qcur", il);
cb(Kcur, "Kcur", il);
cb(Vcur, "Vcur", il);
// Attention computation
const float kq_scale = hparams.f_attention_scale == 0.0f ? 1.0f / sqrtf(float(n_embd_head)) : hparams.f_attention_scale;
cur = build_attn(inp,
nullptr, nullptr, nullptr,
Qcur, Kcur, Vcur, nullptr, nullptr, nullptr, kq_scale, il);
cb(cur, "attn_pregate", il);
ggml_tensor * gate_sigmoid = ggml_sigmoid(ctx0, gate);
cb(gate_sigmoid, "gate_sigmoid", il);
cur = ggml_mul(ctx0, cur, gate_sigmoid);
cb(cur, "attn_gated", il);
cur = build_lora_mm(model.layers[il].wo, cur, model.layers[il].wo_s);
cb(cur, "attn_output", il);
return cur;
}
ggml_tensor * llama_model_qwen4exp::graph::build_layer_attn_linear(
llm_graph_input_rs * inp,
ggml_tensor * cur,
int il) {
const auto * mctx_cur = inp->mctx;
const int64_t d_inner = hparams.ssm_d_inner;
const int64_t n_seqs = ubatch.n_seqs;
const int64_t head_k_dim = hparams.ssm_d_state;
const int64_t num_k_heads = hparams.ssm_n_group;
const int64_t num_v_heads = hparams.ssm_dt_rank;
const int64_t head_v_dim = d_inner / num_v_heads;
const int64_t n_seq_tokens = ubatch.n_seq_tokens;
GGML_ASSERT(n_seqs != 0);
GGML_ASSERT(ubatch.equal_seqs());
GGML_ASSERT(ubatch.n_tokens == n_seq_tokens * n_seqs);
// Input projections
auto qkvz = build_qkvz(cur, il);
ggml_tensor * qkv_mixed = qkvz.first;
ggml_tensor * z = qkvz.second;
ggml_tensor * beta = build_lora_mm(model.layers[il].ssm_beta, cur, model.layers[il].ssm_beta_s);
beta = ggml_reshape_4d(ctx0, beta, 1, num_v_heads, n_seq_tokens, n_seqs);
cb(beta, "beta", il);
beta = ggml_sigmoid(ctx0, beta);
cb(beta, "beta_sigmoid", il);
ggml_tensor * alpha = build_lora_mm(model.layers[il].ssm_alpha, cur, model.layers[il].ssm_alpha_s);
alpha = ggml_reshape_3d(ctx0, alpha, num_v_heads, n_seq_tokens, n_seqs);
cb(alpha, "alpha", il);
ggml_tensor * alpha_biased = ggml_add(ctx0, alpha, model.layers[il].ssm_dt);
ggml_tensor * alpha_softplus = ggml_softplus(ctx0, alpha_biased);
cb(alpha_softplus, "a_softplus", il);
ggml_tensor * gate = ggml_mul(ctx0, alpha_softplus, model.layers[il].ssm_a); // -A_log.exp() * softplus
cb(gate, "gate", il);
gate = ggml_reshape_4d(ctx0, gate, 1, num_v_heads, n_seq_tokens, n_seqs);
ggml_tensor * conv_states_all = mctx_cur->get_r_l(il);
ggml_tensor * ssm_states_all = mctx_cur->get_s_l(il);
ggml_tensor * conv_kernel = model.layers[il].ssm_conv1d;
const int64_t conv_kernel_size = conv_kernel->ne[0];
// channel count from how load_arch_tensors sizes wqkv, not ssm_d_inner: n_embd_r()
// only bounds the row, and the convolution must match the tensor feeding it
const int64_t conv_channels = head_k_dim * num_k_heads * 2 + head_v_dim * num_v_heads;
// offset 0: delta-net history first, PLE history (if any) after it
ggml_tensor * conv_input = build_conv_state_at(inp, conv_states_all, qkv_mixed,
conv_kernel_size - 1, conv_channels, 0, il);
ggml_tensor * state = build_rs(inp, ssm_states_all, hparams.n_embd_s(), n_seqs);
state = ggml_reshape_4d(ctx0, state, head_v_dim, head_v_dim, num_v_heads, n_seqs);
cb(state, "state_predelta", il);
ggml_tensor * conv_output_proper = ggml_ssm_conv(ctx0, conv_input, conv_kernel);
cb(conv_output_proper, "conv_output_raw", il);
ggml_tensor * conv_output_silu = ggml_silu(ctx0, conv_output_proper);
cb(conv_output_silu, "conv_output_silu", il);
ggml_tensor * conv_qkv_mix = conv_output_silu;
// Calculate the total conv dimension
int64_t qkv_dim = head_k_dim * num_k_heads * 2 + head_v_dim * num_v_heads;
int64_t nb1_qkv = ggml_row_size(conv_qkv_mix->type, qkv_dim);
// Extract the convolved Q, K, V from conv_output
ggml_tensor * q_conv = ggml_view_4d(ctx0, conv_qkv_mix, head_k_dim, num_k_heads, n_seq_tokens, n_seqs,
ggml_row_size(conv_qkv_mix->type, head_k_dim),
nb1_qkv,
nb1_qkv * n_seq_tokens,
0);
ggml_tensor * k_conv = ggml_view_4d(ctx0, conv_qkv_mix, head_k_dim, num_k_heads, n_seq_tokens, n_seqs,
ggml_row_size(conv_qkv_mix->type, head_k_dim),
nb1_qkv,
nb1_qkv * n_seq_tokens,
head_k_dim * num_k_heads * ggml_element_size(conv_qkv_mix));
ggml_tensor * v_conv = ggml_view_4d(ctx0, conv_qkv_mix, head_v_dim, num_v_heads, n_seq_tokens, n_seqs,
ggml_row_size(conv_qkv_mix->type, head_v_dim),
nb1_qkv,
nb1_qkv * n_seq_tokens,
ggml_row_size(conv_qkv_mix->type, 2 * head_k_dim * num_k_heads));
cb(q_conv, "q_conv", il);
cb(k_conv, "k_conv", il);
cb(v_conv, "v_conv", il);
const float eps_norm = hparams.f_norm_rms_eps;
q_conv = ggml_l2_norm(ctx0, q_conv, eps_norm);
k_conv = ggml_l2_norm(ctx0, k_conv, eps_norm);
// repeat to match shapes when head keys != value keys; unneeded with the fused GDN
if (num_k_heads != num_v_heads && (!cparams.fused_gdn_ar || !cparams.fused_gdn_ch)) {
GGML_ASSERT(num_v_heads % num_k_heads == 0);
q_conv = ggml_repeat_4d(ctx0, q_conv, head_k_dim, num_v_heads, n_seq_tokens, n_seqs);
k_conv = ggml_repeat_4d(ctx0, k_conv, head_k_dim, num_v_heads, n_seq_tokens, n_seqs);
}
cb(q_conv, "q_conv_predelta", il);
cb(k_conv, "k_conv_predelta", il);
cb(v_conv, "v_conv_predelta", il);
ggml_tensor * output = build_recurrent_attn(inp, ssm_states_all, q_conv, k_conv, v_conv, gate, beta, state, il);
// z: [head_dim, n_heads, n_tokens, n_seqs] -> [n_heads * n_tokens * n_seqs, head_dim]
ggml_tensor * z_2d = ggml_reshape_4d(ctx0, z, head_v_dim, num_v_heads, n_seq_tokens, n_seqs);
// Apply gated normalization: self.norm(core_attn_out, z)
ggml_tensor * attn_out_norm = build_norm_gated(output, model.layers[il].ssm_norm, z_2d, il);
// Final reshape: [head_dim, n_heads, n_tokens, n_seqs] -> [n_tokens, n_seqs, n_heads * head_dim]
ggml_tensor * final_output = ggml_reshape_3d(ctx0, attn_out_norm, head_v_dim * num_v_heads, n_seq_tokens, n_seqs);
cb(final_output, "final_output", il);
// Output projection
cur = build_lora_mm(model.layers[il].ssm_out, final_output, model.layers[il].ssm_out_s);
cb(cur, "linear_attn_out", il);
// Reshape back to original dimensions
cur = ggml_reshape_2d(ctx0, cur, n_embd, n_seq_tokens * n_seqs);
return cur;
}
ggml_tensor * llama_model_qwen4exp::graph::build_layer_ffn(ggml_tensor * cur, const int il) {
// Check if this is an MoE layer
GGML_ASSERT(model.layers[il].ffn_gate_inp != nullptr);
ggml_tensor * moe_out =
build_moe_ffn(cur,
model.layers[il].ffn_gate_inp,
model.layers[il].ffn_up_exps,
model.layers[il].ffn_gate_exps,
model.layers[il].ffn_down_exps,
nullptr,
n_expert, n_expert_used,
LLM_FFN_SILU, true,
hparams.expert_weights_scale,
LLAMA_EXPERT_GATING_FUNC_TYPE_SOFTMAX, il,
nullptr, model.layers[il].ffn_gate_up_exps,
model.layers[il].ffn_up_exps_s,
model.layers[il].ffn_gate_exps_s,
model.layers[il].ffn_down_exps_s);
cb(moe_out, "ffn_moe_out", il);
// Add shared experts if present - following Qwen3Next reference implementation
if (model.layers[il].ffn_up_shexp != nullptr) {
ggml_tensor * ffn_shexp =
build_ffn(cur,
model.layers[il].ffn_up_shexp, NULL, model.layers[il].ffn_up_shexp_s,
model.layers[il].ffn_gate_shexp, NULL, model.layers[il].ffn_gate_shexp_s,
model.layers[il].ffn_down_shexp, NULL, model.layers[il].ffn_down_shexp_s,
NULL,
LLM_FFN_SILU, LLM_FFN_PAR, il);
cb(ffn_shexp, "ffn_shexp", il);
// shared expert has its own sigmoided gate (ffn_gate_inp_shexp, one value per token)
ggml_tensor * shared_gate = build_lora_mm(model.layers[il].ffn_gate_inp_shexp, cur);
cb(shared_gate, "shared_expert_gate", il);
// Apply sigmoid to the gate
shared_gate = ggml_sigmoid(ctx0, shared_gate);
cb(shared_gate, "shared_expert_gate_sigmoid", il);
// Apply the gate to the shared expert output
ffn_shexp = ggml_mul(ctx0, ffn_shexp, shared_gate);
cb(ffn_shexp, "ffn_shexp_gated", il);
cur = ggml_add(ctx0, moe_out, ffn_shexp);
cb(cur, "ffn_out", il);
} else {
cur = moe_out;
}
return cur;
}
// PLE n-gram hash embedding: each token gathers ple_n_heads rows of a shared table.
// mixed_n = (t[p]*m[0]) ^ ... ^ (t[p-n+1]*m[n-1]); row = mixed_n % vocab[h] + offset[h]
// Multipliers reach ~2^45, so the hash runs host-side: ggml has no int64 and no xor.
// Predecessors reset at EOS; positions before the sequence start read as EOS.
class llm_graph_input_ple : public llm_graph_input_i {
public:
llm_graph_input_ple(const llama_model_qwen4exp & pmodel) : pmodel(pmodel) {}
virtual ~llm_graph_input_ple() = default;
void set_input(const llama_ubatch * ubatch) override;
ggml_tensor * rows = nullptr; // I32 [ple_n_heads * n_tokens]
const llama_model_qwen4exp & pmodel;
};
void llm_graph_input_ple::set_input(const llama_ubatch * ubatch) {
if (!ubatch->token) {
return;
}
const auto & hp = pmodel.hparams;
const int64_t n_tokens = ubatch->n_tokens;
const int64_t n_gram = hp.ple_ngram_size;
const int64_t n_heads = hp.ple_n_heads;
const int64_t per_gram = hp.ple_heads_per_ngram;
const int64_t eos = hp.ple_eos_token_id;
std::vector<int32_t> idx(n_heads * n_tokens);
// Missing predecessors come from per-sequence history (vLLM's ngram_context),
// trusted only when contiguous with the incoming position, else EOS padding.
auto & hist_map = pmodel.ple_hist;
// Snapshot the incoming history before touching it. Reading and updating in
// the same pass would let a token near the start of the ubatch pick up an
// earlier token of this same ubatch as if it were prior context.
std::unordered_map<llama_seq_id, std::vector<llama_token>> snap;
for (int64_t i = 0; i < n_tokens; ++i) {
const llama_seq_id seq = ubatch->seq_id[i][0];
if (snap.count(seq)) {
continue;
}
auto & h = hist_map[seq];
if (h.next_pos != ubatch->pos[i]) {
h.toks.assign(n_gram - 1, eos);
}
h.toks.resize(n_gram - 1, eos);
snap[seq] = h.toks;
}
for (int64_t i = 0; i < n_tokens; ++i) {
const llama_seq_id seq = ubatch->seq_id[i][0];
const llama_pos pos = ubatch->pos[i];
const auto & hist = snap[seq];
// predecessor s (1-based) of this token, EOS past a segment boundary
auto prev = [&](int64_t s) -> int64_t {
const int64_t j = i - s;
if (j >= 0 && ubatch->seq_id[j][0] == seq && ubatch->pos[j] == pos - s) {
return ubatch->token[j];
}
// s - i positions before this ubatch started, most recent last
const int64_t back = s - i;
const int64_t k = (int64_t) hist.size() - back;
if (back > 0 && k >= 0 && k < (int64_t) hist.size() && pos - s >= 0) {
return hist[k];
}
return eos;
};
// an EOS in the window resets everything at or before it
// Note the token's own EOS does not cut its context: the reference
// takes the last EOS strictly *before* this position, so a segment
// boundary only hides tokens from the positions that follow it.
std::vector<int64_t> ctx(n_gram);
ctx[0] = ubatch->token[i];
bool cut = false;
for (int64_t s = 1; s < n_gram; ++s) {
ctx[s] = cut ? eos : prev(s);
if (ctx[s] == eos) {
cut = true;
}
}
for (int64_t n = 2; n <= n_gram; ++n) {
uint64_t mixed = (uint64_t) ctx[0] * hp.ple_layer_multipliers[0];
for (int64_t j = 1; j < n; ++j) {
mixed ^= (uint64_t) ctx[j] * hp.ple_layer_multipliers[j];
}
const int64_t base = (n - 2) * per_gram;
for (int64_t g = 0; g < per_gram; ++g) {
const int64_t h_i = base + g;
idx[i * n_heads + h_i] =
(int32_t) (mixed % hp.ple_head_vocab_sizes[h_i] + hp.ple_head_offsets[h_i]);
}
}
auto & h = hist_map[seq];
h.toks.push_back(ubatch->token[i]);
if ((int64_t) h.toks.size() > n_gram - 1) {
h.toks.erase(h.toks.begin(), h.toks.end() - (n_gram - 1));
}
h.next_pos = pos + 1;
}
ggml_backend_tensor_set(rows, idx.data(), 0, idx.size()*ggml_element_size(rows));
}
// Fetch one conv history out of the recurrent row and write the updated tail
// back, at an explicit offset within that row.
//
// The shared build_conv_state assumes the whole row belongs to one convolution.
// Here the row carries the delta-net conv history followed by the PLE one, so
// each caller addresses its own slice. Same structure as the shared helper,
// only with an offset and an explicit dilation.
ggml_tensor * llama_model_qwen4exp::graph::build_conv_state_at(
llm_graph_input_rs * inp,
ggml_tensor * conv_states_all,
ggml_tensor * x,
int64_t state_cols,
int64_t channels,
int64_t row_offset,
int il) {
const auto * mctx_cur = inp->mctx;
const auto kv_head = mctx_cur->get_head();
const auto mem_size = mctx_cur->get_size();
const int64_t n_seqs = ubatch.n_seqs;
const int64_t row_total = hparams.n_embd_r();
// the gather needs the whole row, then this convolution takes its slice
auto it = rs_rows.find(il);
if (it == rs_rows.end()) {
it = rs_rows.emplace(il, build_rs(inp, conv_states_all, row_total, n_seqs)).first;
}
ggml_tensor * rows = it->second;
const size_t esz = ggml_element_size(rows);
ggml_tensor * state = ggml_cont(ctx0,
ggml_view_2d(ctx0, rows, state_cols * channels, n_seqs,
rows->nb[1], row_offset * esz));
state = ggml_reshape_3d(ctx0, state, state_cols, channels, n_seqs);
cb(state, "conv_state_at", il);
ggml_tensor * conv_input = ggml_concat(ctx0, state, ggml_transpose(ctx0, x), 0);
// keep the last state_cols columns for the next ubatch
const size_t row_size = ggml_row_size(conv_states_all->type, row_total);
ggml_tensor * tail = ggml_view_3d(ctx0, conv_input,
state_cols, channels, n_seqs,
conv_input->nb[1], conv_input->nb[2],
ggml_row_size(conv_input->type, conv_input->ne[0] - state_cols));
ggml_tensor * dst = ggml_view_2d(ctx0, conv_states_all,
state_cols * channels, n_seqs,
conv_states_all->nb[1],
kv_head * row_size + row_offset * ggml_element_size(conv_states_all));
ggml_build_forward_expand(gf, ggml_cpy(ctx0, ggml_cont(ctx0, tail), dst));
return conv_input;
}
ggml_tensor * llama_model_qwen4exp::graph::build_ple(
llm_graph_input_rs * inp,
ggml_tensor * hidden,
int il) {
GGML_UNUSED(inp);
const int64_t hc = hparams.dsv4_hc_mult;
const int64_t hc_dim = hc * n_embd;
const int64_t n_heads = hparams.ple_n_heads;
auto ple_inp = std::make_unique<llm_graph_input_ple>(
static_cast<const llama_model_qwen4exp &>(model));
ple_inp->rows = ggml_new_tensor_1d(ctx0, GGML_TYPE_I32, n_heads * n_tokens);
ggml_set_input(ple_inp->rows);
ggml_tensor * rows = ple_inp->rows;
res->add_input(std::move(ple_inp));
// gather then flatten the heads: get_rows already lays the head dimension
// out slowest, matching the reference's flatten over the head axis
ggml_tensor * emb = ggml_get_rows(ctx0, model.per_layer_tok_embd, rows);
emb = ggml_reshape_2d(ctx0, emb, hparams.ple_head_dim * n_heads, n_tokens);
cb(emb, "ple_embd", il);
ggml_tensor * key = build_lora_mm(model.layers[il].ple_key, emb);
ggml_tensor * value = build_lora_mm(model.layers[il].ple_value, emb);
// both norms are grouped over one hc stream, with an affine weight that
// spans the whole hc*n_embd layout, exactly as in build_hc_mix
auto grouped_norm = [&](ggml_tensor * x, ggml_tensor * w) {
ggml_tensor * t = ggml_reshape_3d(ctx0, x, n_embd, hc, n_tokens);
t = ggml_rms_norm(ctx0, t, hparams.f_norm_rms_eps);
t = ggml_reshape_2d(ctx0, t, hc_dim, n_tokens);
t = ggml_mul(ctx0, t, w);
return ggml_reshape_3d(ctx0, t, n_embd, hc, n_tokens);
};
key = grouped_norm(key, model.layers[il].ple_norm_key);
ggml_tensor * query = grouped_norm(hidden, model.layers[il].ple_norm_query);
// per-stream dot product, then a signed square root before the sigmoid
ggml_tensor * s = ggml_sum_rows(ctx0, ggml_mul(ctx0, key, query));
s = ggml_scale(ctx0, s, 1.0f / sqrtf((float) n_embd));
ggml_tensor * mag = ggml_sqrt(ctx0, ggml_clamp(ctx0, ggml_abs(ctx0, s), 1e-6f, INFINITY));
ggml_tensor * gate = ggml_sigmoid(ctx0, ggml_mul(ctx0, ggml_sgn(ctx0, s), mag));
cb(gate, "ple_gate", il);
// [n_embd, 1, T] value broadcast across the hc streams, scaled by the gate
ggml_tensor * v3 = ggml_reshape_3d(ctx0, value, n_embd, 1, n_tokens);
v3 = ggml_repeat_4d(ctx0, v3, n_embd, hc, n_tokens, 1);
ggml_tensor * gated = ggml_mul(ctx0, v3, gate);
cb(gated, "ple_gated_value", il);
ggml_tensor * normalized = grouped_norm(
ggml_reshape_2d(ctx0, gated, hc_dim, n_tokens),
model.layers[il].ple_norm_conv);
normalized = ggml_reshape_2d(ctx0, normalized, hc_dim, n_tokens);
// Depthwise causal conv, dilated by the n-gram size. Written out as a sum
// of shifted, per-channel-scaled copies rather than via ggml_conv_1d_dw:
// that op carries a "very likely wrong for some cases" warning upstream,
// and this form is a handful of ops on a tensor this small.
//
// out[c, t] = sum_k w[k, c] * x[c, t - (K-1-k)*dilation]
//
// History from earlier ubatches is prepended, so decode and chunked prefill
// see the same context a single-shot prefill would. A fresh sequence starts
// with a zeroed state, which is what the reference's zero-padded nn.Conv1d
// gives at a sequence start.
const int64_t kern = hparams.ple_conv_kernel;
const int64_t dil = hparams.ple_ngram_size;
const int64_t hist = (kern - 1) * dil;
// the conv history is per sequence, so the input has to carry the sequence
// axis too rather than relying on it being one
const int64_t n_seqs = ubatch.n_seqs;
const int64_t n_seq_tokens = ubatch.n_seq_tokens;
// [hist + n_seq_tokens, hc_dim, n_seqs], tokens on ne[0]
ggml_tensor * padded = build_conv_state_at(inp, inp->mctx->get_r_l(il),
ggml_reshape_3d(ctx0, normalized, hc_dim, n_seq_tokens, n_seqs),
hist, hc_dim,
hparams.n_embd_r() - hparams.ple_conv_state(), il);
ggml_tensor * conv_out = nullptr;
for (int64_t k = 0; k < kern; ++k) {
// tap k reads (kern-1-k)*dilation positions back
const int64_t start = hist - (kern - 1 - k) * dil;
ggml_tensor * shifted = ggml_cont(ctx0,
ggml_transpose(ctx0,
ggml_view_3d(ctx0, padded, n_seq_tokens, hc_dim, n_seqs,
padded->nb[1], padded->nb[2],
ggml_row_size(padded->type, start))));
// column k of the [kern, hc_dim] kernel is one weight per channel
ggml_tensor * wk = ggml_cont(ctx0,
ggml_view_2d(ctx0, model.layers[il].ple_conv1d, 1, hc_dim,
model.layers[il].ple_conv1d->nb[1],
k * model.layers[il].ple_conv1d->nb[0]));
// unlike the 1-D norm gammas, this kernel keeps the file's type, so it
// needs an explicit cast before multiplying an f32 activation
wk = ggml_reshape_1d(ctx0, wk, hc_dim);
if (wk->type != GGML_TYPE_F32) {
wk = ggml_cast(ctx0, wk, GGML_TYPE_F32);
}
ggml_tensor * term = ggml_mul(ctx0, shifted, wk);
conv_out = conv_out ? ggml_add(ctx0, conv_out, term) : term;
}
conv_out = ggml_silu(ctx0, conv_out);
conv_out = ggml_reshape_3d(ctx0, ggml_cont(ctx0, conv_out), n_embd, hc, n_tokens);
cb(conv_out, "ple_conv_out", il);
return ggml_add(ctx0, hidden, ggml_add(ctx0, gated, conv_out));
}