From 10bad7ad186e1a79e32a11677ff227a92dca0ad5 Mon Sep 17 00:00:00 2001 From: Pedro Cuenca Date: Tue, 4 Aug 2026 16:20:24 +0200 Subject: [PATCH 01/26] mmproj conversion Note: some fields to be renamed after the implementation works. We are keeping compatibility with the reference Meta gguf for testing purposes. --- conversion/__init__.py | 1 + conversion/onyx.py | 62 ++++++++++++++++++++++++++++++++-- gguf-py/gguf/constants.py | 14 ++++++-- gguf-py/gguf/tensor_mapping.py | 29 ++++++++++++++-- 4 files changed, 99 insertions(+), 7 deletions(-) diff --git a/conversion/__init__.py b/conversion/__init__.py index 88e7bbe1ab..52c6dbb63c 100644 --- a/conversion/__init__.py +++ b/conversion/__init__.py @@ -295,6 +295,7 @@ MMPROJ_MODEL_MAP: dict[str, str] = { "MiniCPMV4_6ForConditionalGeneration": "minicpm", "Mistral3ForConditionalGeneration": "llava", "NemotronH_Nano_VL_V2": "nemotron", + "OnyxForConditionalGeneration": "onyx", "PaddleOCRVisionModel": "ernie", "Phi4ForCausalLMV": "phi", "Qwen2AudioForConditionalGeneration": "ultravox", diff --git a/conversion/onyx.py b/conversion/onyx.py index e1805118b4..3da3da167d 100644 --- a/conversion/onyx.py +++ b/conversion/onyx.py @@ -1,13 +1,13 @@ from __future__ import annotations -from typing import Iterable, TYPE_CHECKING +from typing import Any, Iterable, TYPE_CHECKING import torch if TYPE_CHECKING: from torch import Tensor -from .base import ModelBase, TextModel, gguf +from .base import MmprojModel, ModelBase, TextModel, gguf @ModelBase.register("OnyxForConditionalGeneration") @@ -61,3 +61,61 @@ class OnyxModel(TextModel): ) yield from super().modify_tensors(data_torch, name, bid) + + +@ModelBase.register("OnyxForConditionalGeneration") +class OnyxVisionModel(MmprojModel): + # fallback for rope_parameters.rope_theta + ROPE_THETA = 10000.0 + + def get_vision_config(self) -> dict[str, Any] | None: + c = self.global_config.get("vision_config") + if not c: + return None + # Onyx actually uses dynamic size, initialize with nominal size + image_size = c["pos_emb_height"] * c["patch_size"] * c["merge_size"] + # Derive sparse_attention_factor from layer_types + fulls = [i for i, t in enumerate(c["layer_types"]) if t == "full_attention"] + if not fulls: + raise ValueError("vision_config.layer_types has no full_attention layer") + sparse_factor = fulls[0] + 1 + return {**c, "image_size": image_size, "sparse_attention_factor": sparse_factor} + + def set_gguf_parameters(self): + super().set_gguf_parameters() + c = self.hparams_vision # enriched vision_config from get_vision_config() + + self.gguf_writer.add_clip_projector_type(gguf.VisionProjectorType.ONYX) + self.gguf_writer.add_vision_attention_layernorm_eps(float(c["layer_norm_eps"])) + + rope_theta = float(c.get("rope_parameters", {}).get("rope_theta", self.ROPE_THETA)) + self.gguf_writer.add_uint32 ("clip.vision.onyx.patch_temporal", int(c["patch_temporal"])) + self.gguf_writer.add_uint32 ("clip.vision.onyx.downsample_factor", int(c["merge_size"])) + self.gguf_writer.add_uint32 ("clip.vision.onyx.sparse_attention_factor", int(c["sparse_attention_factor"])) + self.gguf_writer.add_uint32 ("clip.vision.onyx.pos_emb_grid", int(c["pos_emb_height"])) + self.gguf_writer.add_float32("clip.vision.onyx.rope_theta", rope_theta) + + @classmethod + def filter_tensors(cls, item): + name, gen = item + keep = ("model.vision_tower.", "model.vision_adapter.", "model.vision_projection.") + if not any(name.startswith(k) for k in keep): + return None + return super().filter_tensors((name, gen)) + + @staticmethod + def _unpermute_for_rope(tensor: "Tensor", n_heads: int) -> "Tensor": + """clip.cpp uses the interleaved convention, so we invert the permutation here.""" + if tensor.ndim == 2: + dim1, dim2 = tensor.shape + return tensor.view(n_heads, 2, dim1 // n_heads // 2, dim2).transpose(1, 2).reshape(dim1, dim2) + if tensor.ndim == 1: + (dim1,) = tensor.shape + return tensor.view(n_heads, 2, dim1 // n_heads // 2).transpose(1, 2).reshape(dim1) + raise ValueError(f"_unpermute_for_rope: unexpected shape {tuple(tensor.shape)}") + + def modify_tensors(self, data_torch, name, bid): + if ".attn.q_proj." in name or ".attn.k_proj." in name: + n_heads = int(self.hparams_vision["num_attention_heads"]) + data_torch = self._unpermute_for_rope(data_torch, n_heads) + yield (self.map_tensor_name(name), data_torch) diff --git a/gguf-py/gguf/constants.py b/gguf-py/gguf/constants.py index 7a2ba560a8..0fec76ec69 100644 --- a/gguf-py/gguf/constants.py +++ b/gguf-py/gguf/constants.py @@ -873,6 +873,9 @@ class MODEL_TENSOR(IntEnum): V_MM_GATE = auto() # cogvlm V_MM_MERGER_FC1 = auto() # minimax-m3 (patch-merge MLP) V_MM_MERGER_FC2 = auto() # minimax-m3 (patch-merge MLP) + V_MM_ADAPTER_FC = auto() # onyx + V_MM_ADAPTER_PROJ = auto() # onyx + V_MM_VISION_PROJ = auto() # onyx V_TOK_BOI = auto() # cogvlm V_TOK_EOI = auto() # cogvlm V_TOK_IMG_BEGIN = auto() # hunyuanvl @@ -1481,8 +1484,11 @@ TENSOR_NAMES: dict[MODEL_TENSOR, str] = { MODEL_TENSOR.V_MM_UP: "mm.up", MODEL_TENSOR.V_MM_DOWN: "mm.down", MODEL_TENSOR.V_MM_GATE: "mm.gate", - MODEL_TENSOR.V_MM_MERGER_FC1: "mm.merger.fc1", - MODEL_TENSOR.V_MM_MERGER_FC2: "mm.merger.fc2", + MODEL_TENSOR.V_MM_MERGER_FC1: "mm.merger.fc1", + MODEL_TENSOR.V_MM_MERGER_FC2: "mm.merger.fc2", + MODEL_TENSOR.V_MM_ADAPTER_FC: "mm.adapter_fc", # onyx + MODEL_TENSOR.V_MM_ADAPTER_PROJ: "mm.adapter_proj", # onyx + MODEL_TENSOR.V_MM_VISION_PROJ: "mm.vision_proj", # onyx MODEL_TENSOR.V_TOK_BOI: "v.boi", MODEL_TENSOR.V_TOK_EOI: "v.eoi", MODEL_TENSOR.V_MM_PRE_NORM: "mm.pre_norm", @@ -1717,6 +1723,9 @@ MODEL_TENSORS: dict[MODEL_ARCH, list[MODEL_TENSOR]] = { MODEL_TENSOR.V_MM_UP, MODEL_TENSOR.V_MM_DOWN, MODEL_TENSOR.V_MM_GATE, + MODEL_TENSOR.V_MM_ADAPTER_FC, + MODEL_TENSOR.V_MM_ADAPTER_PROJ, + MODEL_TENSOR.V_MM_VISION_PROJ, MODEL_TENSOR.V_TOK_BOI, MODEL_TENSOR.V_TOK_EOI, MODEL_TENSOR.V_MM_PRE_NORM, @@ -4907,6 +4916,7 @@ class VisionProjectorType: MIMOVL = "mimovl" MIMO_AUDIO = "mimo_audio" GRANITE4_VISION = "granite4_vision" + ONYX = "onyx" # Items here are (block size, type size) diff --git a/gguf-py/gguf/tensor_mapping.py b/gguf-py/gguf/tensor_mapping.py index 79539c46d1..44a85d2960 100644 --- a/gguf-py/gguf/tensor_mapping.py +++ b/gguf-py/gguf/tensor_mapping.py @@ -1467,6 +1467,7 @@ class TensorNameMap: "vision_tower.patch_embed.patchifier.proj", # dots.ocr "vision_model.conv1", # Step3-VL "model.vision_embedder.patch_dense", # gemma4 unified + "model.vision_tower.patch_embedder.patch_embedding", # onyx ), MODEL_TENSOR.V_ENC_EMBD_NORM: ( @@ -1534,7 +1535,8 @@ class TensorNameMap: "siglip2.vision_model.encoder.layers.{bid}.self_attn.q_proj", # youtuvl "model.vision_model.transformer.layers.{bid}.self_attn.q_proj", # Deepseek-OCR CLIP, generated "vision_model.model.layers.{bid}.self_attn.q_proj.linear", # gemma4 - "model.qwen2_model.model.model.layers.{bid}.self_attn.q_proj" # Deepseek-OCR-2 qwen2 + "model.qwen2_model.model.model.layers.{bid}.self_attn.q_proj", # Deepseek-OCR-2 qwen2 + "model.vision_tower.layers.{bid}.attn.q_proj", # onyx ), MODEL_TENSOR.V_ENC_ATTN_Q_NORM: ( @@ -1560,7 +1562,8 @@ class TensorNameMap: "model.vision_model.transformer.layers.{bid}.self_attn.k_proj", # Deepseek-OCR CLIP, generated "siglip2.vision_model.encoder.layers.{bid}.self_attn.k_proj", "vision_model.model.layers.{bid}.self_attn.k_proj.linear", # gemma4 - "model.qwen2_model.model.model.layers.{bid}.self_attn.k_proj" # Deepseek-OCR-2 qwen2 + "model.qwen2_model.model.model.layers.{bid}.self_attn.k_proj", # Deepseek-OCR-2 qwen2 + "model.vision_tower.layers.{bid}.attn.k_proj", # onyx ), MODEL_TENSOR.V_ENC_ATTN_K_NORM: ( @@ -1586,7 +1589,8 @@ class TensorNameMap: "siglip2.vision_model.encoder.layers.{bid}.self_attn.v_proj", "model.vision_model.transformer.layers.{bid}.self_attn.v_proj", # Deepseek-OCR CLIP, generated "vision_model.model.layers.{bid}.self_attn.v_proj.linear", # gemma4 - "model.qwen2_model.model.model.layers.{bid}.self_attn.v_proj" # Deepseek-OCR-2 qwen2 + "model.qwen2_model.model.model.layers.{bid}.self_attn.v_proj", # Deepseek-OCR-2 qwen2 + "model.vision_tower.layers.{bid}.attn.v_proj", # onyx ), MODEL_TENSOR.V_ENC_INPUT_NORM: ( @@ -1610,6 +1614,7 @@ class TensorNameMap: "vision_tower.blocks.{bid}.norm1", # dots.ocr "vision_model.transformer.resblocks.{bid}.ln_1", # Step3-VL "model.qwen2_model.model.model.layers.{bid}.input_layernorm", # Deepseek-OCR-2 qwen2 + "model.vision_tower.layers.{bid}.norm1", # onyx ), MODEL_TENSOR.V_ENC_ATTN_O: ( @@ -1635,6 +1640,7 @@ class TensorNameMap: "vision_model.model.layers.{bid}.self_attn.o_proj.linear", # gemma4 "vision_tower.blocks.{bid}.attn.proj", # dots.ocr "vision_model.transformer.resblocks.{bid}.attn.out_proj", # Step3-VL + "model.vision_tower.layers.{bid}.attn.proj", # onyx ), MODEL_TENSOR.V_ENC_ATTN_SINKS: ( @@ -1663,6 +1669,7 @@ class TensorNameMap: "vision_tower.blocks.{bid}.norm2", # dots.ocr "vision_model.transformer.resblocks.{bid}.ln_2", # Step3-VL "model.qwen2_model.model.model.layers.{bid}.post_attention_layernorm", # Deepseek-OCR-2 qwen2 + "model.vision_tower.layers.{bid}.norm2", # onyx ), MODEL_TENSOR.V_ENC_FFN_UP: ( @@ -1687,6 +1694,7 @@ class TensorNameMap: "vision_model.model.layers.{bid}.mlp.up_proj", # gemma4 "vision_model.transformer.resblocks.{bid}.mlp.c_fc", # Step3-VL "model.qwen2_model.model.model.layers.{bid}.mlp.up_proj", # Deepseek-OCR-2 qwen2 + "model.vision_tower.layers.{bid}.mlp.fc1", # onyx ), MODEL_TENSOR.V_ENC_FFN_GATE: ( @@ -1719,6 +1727,7 @@ class TensorNameMap: "model.qwen2_model.model.model.layers.{bid}.mlp.down_proj" , # Deepseek-OCR-2 qwen2 "vision_model.model.layers.{bid}.mlp.down_proj", # gemma4 "vision_model.transformer.resblocks.{bid}.mlp.c_proj", # Step3-VL + "model.vision_tower.layers.{bid}.mlp.fc2", # onyx ), MODEL_TENSOR.V_ENC_ATTN_POST_NORM: ( @@ -1753,6 +1762,7 @@ class TensorNameMap: "model.vision_model.pre_layrnorm", # Deepseek-OCR CLIP "vision_tower.patch_embed.patchifier.norm", # dots.ocr "vision_model.ln_pre", # Step3-VL + "model.vision_tower.ln_pre", # onyx ), MODEL_TENSOR.V_POST_NORM: ( @@ -1766,6 +1776,7 @@ class TensorNameMap: "visual.post_layernorm", # glm4v "siglip2.vision_model.post_layernorm", "model.qwen2_model.model.model.norm", # Deepseek-OCR-2 qwen2 + "model.vision_tower.ln_post", # onyx ), MODEL_TENSOR.V_MM_POST_NORM: ( @@ -1858,6 +1869,18 @@ class TensorNameMap: "patch_merge_mlp.linear_2", # minimax-m3 ), + MODEL_TENSOR.V_MM_ADAPTER_FC: ( + "model.vision_adapter.fc1", # onyx + ), + + MODEL_TENSOR.V_MM_ADAPTER_PROJ: ( + "model.vision_adapter.fc2", # onyx + ), + + MODEL_TENSOR.V_MM_VISION_PROJ: ( + "model.vision_projection", # onyx + ), + MODEL_TENSOR.V_DS_NORM: ( "model.visual.deepstack_merger_list.{bid}.norm", # deepstack in qwen3vl ), From efed93383d249209adcd29250815c5b70ba4a412 Mon Sep 17 00:00:00 2001 From: Pedro Cuenca Date: Tue, 4 Aug 2026 16:32:54 +0200 Subject: [PATCH 02/26] "clip" header declarations --- tools/mtmd/clip-impl.h | 11 +++++++++++ tools/mtmd/clip-model.h | 10 ++++++++++ 2 files changed, 21 insertions(+) diff --git a/tools/mtmd/clip-impl.h b/tools/mtmd/clip-impl.h index d42b38222c..0a2326ff6d 100644 --- a/tools/mtmd/clip-impl.h +++ b/tools/mtmd/clip-impl.h @@ -91,6 +91,12 @@ #define KEY_A_LOCAL_GROUP_SIZE "clip.audio.local_group_size" // mimo-v2.5: input_local_transformer grouping size #define KEY_AUDIO_SUBSAMPLING_FACTOR "clip.audio.subsampling_factor" +#define KEY_ONYX_PATCH_TEMPORAL "clip.vision.onyx.patch_temporal" +#define KEY_ONYX_DOWNSAMPLE "clip.vision.onyx.downsample_factor" +#define KEY_ONYX_SPARSE_FACTOR "clip.vision.onyx.sparse_attention_factor" +#define KEY_ONYX_POS_GRID "clip.vision.onyx.pos_emb_grid" +#define KEY_ONYX_ROPE_THETA "clip.vision.onyx.rope_theta" + // // tensor name constants // @@ -128,6 +134,9 @@ #define TN_MM_GATE "mm.gate.%s" #define TN_MM_DOWN "mm.down.%s" #define TN_MM_POST_NORM "mm.post_norm.%s" +#define TN_MM_ADAPTER_FC "mm.adapter_fc.%s" // onyx +#define TN_MM_ADAPTER_PROJ "mm.adapter_proj.%s" // onyx +#define TN_MM_VISION_PROJ "mm.vision_proj.%s" // onyx #define TN_MVLM_PROJ_MLP "mm.model.mlp.%d.%s" #define TN_MVLM_PROJ_BLOCK "mm.model.mb_block.%d.block.%d.%s" #define TN_MVLM_PROJ_PEG "mm.model.peg.%d.%s" @@ -408,6 +417,7 @@ enum projector_type { PROJECTOR_TYPE_MINIMAX_M3, PROJECTOR_TYPE_GRANITE4_VISION, PROJECTOR_TYPE_MIMO_AUDIO, + PROJECTOR_TYPE_ONYX, PROJECTOR_TYPE_UNKNOWN, }; @@ -465,6 +475,7 @@ static std::map PROJECTOR_TYPE_NAMES = { { PROJECTOR_TYPE_GRANITE4_VISION, "granite4_vision"}, { PROJECTOR_TYPE_MIMO_AUDIO, "mimo_audio"}, { PROJECTOR_TYPE_PARAKEET, "parakeet"}, + { PROJECTOR_TYPE_ONYX, "onyx"}, }; static projector_type clip_projector_type_from_string(const std::string & str) { diff --git a/tools/mtmd/clip-model.h b/tools/mtmd/clip-model.h index 8b9db5101d..8214d3271b 100644 --- a/tools/mtmd/clip-model.h +++ b/tools/mtmd/clip-model.h @@ -109,6 +109,11 @@ struct clip_hparams { int32_t downsample_query_side; int32_t downsample_window_side; + // Onyx vision (per-block sparse-window pattern, learned pos-emb, patch-temporal) + int32_t onyx_patch_temporal = 0; + int32_t onyx_sparse_factor = 0; + int32_t onyx_pos_grid = 0; + // audio int32_t n_mel_bins = 0; // whisper preprocessor int32_t proj_stack_factor = 0; // ultravox @@ -408,6 +413,11 @@ struct clip_model { ggml_tensor * mm_post_norm_w = nullptr; ggml_tensor * mm_post_norm_b = nullptr; + // Onyx adapter + final vision projection (3-linear MLP with erf-GELU) + ggml_tensor * mm_adapter_fc = nullptr; + ggml_tensor * mm_adapter_proj = nullptr; + ggml_tensor * mm_vision_proj = nullptr; + // LLaVA projection ggml_tensor * mm_input_norm_w = nullptr; ggml_tensor * mm_input_norm_b = nullptr; From 68d766eb0d8b9b4cb1035979851f15806897eeab Mon Sep 17 00:00:00 2001 From: Pedro Cuenca Date: Tue, 4 Aug 2026 16:57:04 +0200 Subject: [PATCH 03/26] Load mmproj --- tools/mtmd/clip.cpp | 23 +++++++++++++++++++++++ 1 file changed, 23 insertions(+) diff --git a/tools/mtmd/clip.cpp b/tools/mtmd/clip.cpp index 5f0d00b660..dbcd0dc4e1 100644 --- a/tools/mtmd/clip.cpp +++ b/tools/mtmd/clip.cpp @@ -1516,6 +1516,19 @@ struct clip_model_loader { hparams.set_limit_image_tokens(8, 576); hparams.set_warmup_n_tokens(16*16); } break; + case PROJECTOR_TYPE_ONYX: + { + hparams.n_merge = 2; // pixel-shuffle downsample after the ViT + hparams.image_resize_algo = RESIZE_ALGO_LANCZOS; + hparams.rope_theta = 10000.0f; + get_u32(KEY_ONYX_DOWNSAMPLE, hparams.n_merge, false); + get_f32(KEY_ONYX_ROPE_THETA, hparams.rope_theta, false); + get_u32(KEY_ONYX_PATCH_TEMPORAL, hparams.onyx_patch_temporal, false); + get_u32(KEY_ONYX_SPARSE_FACTOR, hparams.onyx_sparse_factor, false); + get_u32(KEY_ONYX_POS_GRID, hparams.onyx_pos_grid, false); + hparams.set_limit_image_tokens(1, 4096); + hparams.set_warmup_n_tokens(32*32); + } break; case PROJECTOR_TYPE_MIMOVL: { hparams.n_merge = 2; // spatial_merge_size @@ -2218,6 +2231,13 @@ struct clip_model_loader { model.mm_merger_fc2_w = get_tensor(string_format(TN_MM_MERGER_FC2, "weight")); model.mm_merger_fc2_b = get_tensor(string_format(TN_MM_MERGER_FC2, "bias")); } break; + case PROJECTOR_TYPE_ONYX: + { + // 3-linear MLP: fc -> erf-GELU -> proj -> erf-GELU -> vision_proj (into LLM residual dim) + model.mm_adapter_fc = get_tensor(string_format(TN_MM_ADAPTER_FC, "weight")); + model.mm_adapter_proj = get_tensor(string_format(TN_MM_ADAPTER_PROJ, "weight")); + model.mm_vision_proj = get_tensor(string_format(TN_MM_VISION_PROJ, "weight")); + } break; case PROJECTOR_TYPE_STEP3VL: { model.mm_0_w = get_tensor(string_format(TN_LLAVA_PROJ, 0, "weight")); @@ -3498,6 +3518,7 @@ int clip_n_output_tokens_x(const clip_ctx * ctx, const clip_image_f32 * img) { case PROJECTOR_TYPE_PADDLEOCR: case PROJECTOR_TYPE_HUNYUANVL: case PROJECTOR_TYPE_YOUTUVL: + case PROJECTOR_TYPE_ONYX: return (img->nx() / params.patch_size) / 2; case PROJECTOR_TYPE_STEP3VL: return img->nx() / (params.patch_size * params.n_merge); @@ -3523,6 +3544,7 @@ int clip_n_output_tokens_y(const clip_ctx * ctx, const clip_image_f32 * img) { case PROJECTOR_TYPE_PADDLEOCR: case PROJECTOR_TYPE_HUNYUANVL: case PROJECTOR_TYPE_YOUTUVL: + case PROJECTOR_TYPE_ONYX: return (img->ny() / params.patch_size) / 2; case PROJECTOR_TYPE_STEP3VL: return img->ny() / (params.patch_size * params.n_merge); @@ -3602,6 +3624,7 @@ int clip_n_output_tokens(const clip_ctx * ctx, const clip_image_f32 * img) { case PROJECTOR_TYPE_MINIMAX_M3: case PROJECTOR_TYPE_GLM4V: case PROJECTOR_TYPE_YOUTUVL: + case PROJECTOR_TYPE_ONYX: { // dynamic size (2 conv, so double patch size) int x_patch = img->nx() / (params.patch_size * 2); From f215b1655d204bf32e9cdc4c4fb0d5e110f1b429 Mon Sep 17 00:00:00 2001 From: Pedro Cuenca Date: Tue, 4 Aug 2026 17:24:42 +0200 Subject: [PATCH 04/26] Pre-processing --- tools/mtmd/mtmd-image.cpp | 62 +++++++++++++++++++++++++++++++++++++++ tools/mtmd/mtmd-image.h | 6 ++++ tools/mtmd/mtmd.cpp | 5 ++++ 3 files changed, 73 insertions(+) diff --git a/tools/mtmd/mtmd-image.cpp b/tools/mtmd/mtmd-image.cpp index 72d35fce69..7c3c585ad0 100644 --- a/tools/mtmd/mtmd-image.cpp +++ b/tools/mtmd/mtmd-image.cpp @@ -1591,3 +1591,65 @@ mtmd_image_preproc_out mtmd_image_preprocessor_granite::preprocess(const clip_im } return output; } + +// +// mtmd_image_preprocessor_onyx +// + +// Replicates transformers' get_aspect_ratio_preserving_size (image_processing_onyx.py) +static clip_image_size onyx_grid_size(int img_w, int img_h, int patch_hw, int max_tokens) { + double i_nph = (double) img_h / patch_hw; + double i_npw = (double) img_w / patch_hw; + const double ratio = i_nph > 0.0 ? i_npw / i_nph : 1.0; + if (i_nph * i_npw > (double) max_tokens) { + i_nph = std::sqrt((double) max_tokens / ratio); + i_npw = i_nph * ratio; + } + const int hs[2] = { (int) std::floor(i_nph), (int) std::ceil(i_nph) }; + const int ws[2] = { (int) std::floor(i_npw), (int) std::ceil(i_npw) }; + const double target_ar = (double) img_h / (double) img_w; + int best_nph = -1; + int best_npw = -1; + double best_d = 0.0; + for (int a = 0; a < 2; ++a) { + for (int b = 0; b < 2; ++b) { + const int nph = hs[a]; + const int npw = ws[b]; + if (nph < 1 || npw < 1 || nph * npw > max_tokens) { + continue; + } + const double d = std::fabs((double) nph / (double) npw - target_ar); + const int n_tokens = nph * npw; + const int best_n_tokens = best_nph * best_npw; + if (best_nph < 0 || d < best_d || (d == best_d && n_tokens > best_n_tokens)) { + best_nph = nph; + best_npw = npw; + best_d = d; + } + } + } + if (best_nph < 0) { // no candidate fit under the cap: round and clamp + best_nph = std::max(1, (int) std::lround(i_nph)); + best_npw = std::max(1, (int) std::lround(i_npw)); + } + return clip_image_size{ best_npw * patch_hw, best_nph * patch_hw }; +} + +mtmd_image_preproc_out mtmd_image_preprocessor_onyx::preprocess(const clip_image_u8 & img) { + const int patch_hw = hparams.patch_size * hparams.n_merge; + const int patch_area = hparams.patch_size * hparams.patch_size * hparams.n_merge * hparams.n_merge; + GGML_ASSERT(patch_area > 0 && hparams.image_max_pixels > 0); + const int max_tokens = hparams.image_max_pixels / patch_area; + + const clip_image_size original_size = img.get_size(); + const clip_image_size target_size = onyx_grid_size( + original_size.width, original_size.height, patch_hw, max_tokens); + + // PIL resizes directly to (target_w, target_h) -- a stretch, no padding. + clip_image_u8 resized_image; + img_tool::resize(img, resized_image, target_size, hparams.image_resize_algo, PAD_NONE); + + mtmd_image_preproc_out output; + output.append(hparams, resized_image, true); + return output; +} diff --git a/tools/mtmd/mtmd-image.h b/tools/mtmd/mtmd-image.h index 115cba51e8..26fd84d485 100644 --- a/tools/mtmd/mtmd-image.h +++ b/tools/mtmd/mtmd-image.h @@ -224,3 +224,9 @@ struct mtmd_image_preprocessor_granite : mtmd_image_preprocessor_llava_uhd { mtmd_image_preprocessor_granite(const clip_ctx * ctx) : mtmd_image_preprocessor_llava_uhd(ctx) {} mtmd_image_preproc_out preprocess(const clip_image_u8 & img) override; }; + +// pick the patch grid closest to the input aspect ratio under the per-image token cap, stretch-resize. +struct mtmd_image_preprocessor_onyx : mtmd_image_preprocessor { + mtmd_image_preprocessor_onyx(const clip_ctx * ctx) : mtmd_image_preprocessor(ctx) {} + mtmd_image_preproc_out preprocess(const clip_image_u8 & img) override; +}; diff --git a/tools/mtmd/mtmd.cpp b/tools/mtmd/mtmd.cpp index 93ca8cbcf8..5c49b35408 100644 --- a/tools/mtmd/mtmd.cpp +++ b/tools/mtmd/mtmd.cpp @@ -470,6 +470,11 @@ struct mtmd_context { img_end = "]<]end of image[>["; image_preproc = std::make_unique(ctx_v); } break; + case PROJECTOR_TYPE_ONYX: + { + // Follow transformers Onyx processing: <|patch|>*N, no delimiters + image_preproc = std::make_unique(ctx_v); + } break; case PROJECTOR_TYPE_YOUTUVL: { // <|vision_start|> ... (image embeddings) ... <|vision_end|> From 1ca6e2ab3655144851d07a669c0070fc7eba30c2 Mon Sep 17 00:00:00 2001 From: Pedro Cuenca Date: Tue, 4 Aug 2026 17:41:40 +0200 Subject: [PATCH 05/26] Graph --- tools/mtmd/CMakeLists.txt | 1 + tools/mtmd/clip.cpp | 148 ++++++++++++++++++++++++++++++++++++- tools/mtmd/models/models.h | 5 ++ 3 files changed, 153 insertions(+), 1 deletion(-) diff --git a/tools/mtmd/CMakeLists.txt b/tools/mtmd/CMakeLists.txt index 15040e4af5..82f7752487 100644 --- a/tools/mtmd/CMakeLists.txt +++ b/tools/mtmd/CMakeLists.txt @@ -41,6 +41,7 @@ add_library(mtmd models/kimivl.cpp models/kimik25.cpp models/nemotron-v2-vl.cpp + models/onyx.cpp models/llama4.cpp models/llava.cpp models/minicpmv.cpp diff --git a/tools/mtmd/clip.cpp b/tools/mtmd/clip.cpp index dbcd0dc4e1..3a6411c888 100644 --- a/tools/mtmd/clip.cpp +++ b/tools/mtmd/clip.cpp @@ -928,6 +928,10 @@ static std::unique_ptr clip_get_graph_builder(clip_ctx * ctx, const { builder = std::make_unique(ctx, img); } break; + case PROJECTOR_TYPE_ONYX: + { + builder = std::make_unique(ctx, img); + } break; case PROJECTOR_TYPE_STEP3VL: { builder = std::make_unique(ctx, img); @@ -3894,7 +3898,10 @@ bool clip_image_batch_encode(clip_ctx * ctx, int n_threads, const clip_image_f32 }; // set input pixel values - if (!imgs.is_audio) { + // onyx feeds host-patchified "onyx_patches" instead of raw pixels (handled in the switch below) + if (ctx->model.proj_type == PROJECTOR_TYPE_ONYX) { + // no generic pixel input + } else if (!imgs.is_audio) { size_t nelem = 0; for (const auto & img : imgs.entries) { nelem += img.nx() * img.ny() * 3; @@ -3951,6 +3958,143 @@ bool clip_image_batch_encode(clip_ctx * ctx, int n_threads, const clip_image_f32 // set input per projector switch (ctx->model.proj_type) { + case PROJECTOR_TYPE_ONYX: + { + const int grid_w = pos_w; // image_size_width / patch_size + const int grid_h = pos_h; // image_size_height / patch_size + const int n_tok = grid_w * grid_h; + const int ps = patch_size; + const int nx = image_size_width; + const int pt = hparams.onyx_patch_temporal; + const int pgrid = hparams.onyx_pos_grid; // 32 + const int nemb = hparams.n_embd; // 1536 + const int f = hparams.n_merge; // downsample 2 + const int patch_dim = pt * 3 * ps * ps; + const auto & buf = imgs.entries[0].get_ro_buf(); // interleaved pixels (3ch image / 6ch video) + // channel count: 3 = image (duplicate frame across patch_temporal), + // 3*pt = video frame-pair (distinct frames per temporal slot). + const int nchan = (int) (buf.size() / ((size_t) nx * imgs.entries[0].ny())); + + // --- patchify: [pt,c,ps,ps] per token; token = gy*grid_w + gx --- + std::vector patches((size_t) patch_dim * n_tok, 0.0f); + for (int gy = 0; gy < grid_h; gy++) { + for (int gx = 0; gx < grid_w; gx++) { + const int tok = gy * grid_w + gx; + for (int fr = 0; fr < pt; fr++) { + for (int c = 0; c < 3; c++) { + // image: same RGB for all temporal slots; video: distinct frame per slot + const int src_c = (nchan == 3) ? c : (fr * 3 + c); + for (int py = 0; py < ps; py++) { + for (int px = 0; px < ps; px++) { + const int iy = gy * ps + py; + const int ix = gx * ps + px; + const int d = ((fr * 3 + c) * ps + py) * ps + px; + patches[(size_t) tok * patch_dim + d] = + buf[(size_t) nchan * ((size_t) iy * nx + ix) + src_c]; + } + } + } + } + } + } + set_input_f32("onyx_patches", patches); + + // --- learned pos-emb bilinear interpolation (grid_sample align_corners=False, + // zeros padding) from pgrid x pgrid to grid_h x grid_w --- + ggml_tensor * pe = ctx->model.position_embeddings; // [nemb, pgrid*pgrid] + std::vector pe_host((size_t) nemb * pgrid * pgrid); + { + std::vector raw(ggml_nbytes(pe)); + ggml_backend_tensor_get(pe, raw.data(), 0, ggml_nbytes(pe)); + if (pe->type == GGML_TYPE_F32) { + std::memcpy(pe_host.data(), raw.data(), ggml_nbytes(pe)); + } else { + const auto * tt = ggml_get_type_traits(pe->type); + tt->to_float(raw.data(), pe_host.data(), (int64_t) nemb * pgrid * pgrid); + } + } + // NOTE: the reference uses meshgrid(ys, xs, indexing="xy") which yields a + // [grid_w, grid_h] grid (width-outer) and grid_sample maps coord0=ys->W axis, + // coord1=xs->H axis (a transposed sampling). Token t decomposes as + // i=t/grid_h (width), j=t%grid_h (height); sample row a from i (grid_w scale), + // col b from j (grid_h scale). This matches the trained convention exactly. + std::vector pos_emb((size_t) nemb * n_tok, 0.0f); + for (int t = 0; t < n_tok; t++) { + const int i = t / grid_h; // width index (0..grid_w-1) + const int j = t % grid_h; // height index (0..grid_h-1) + const float af = (i + 0.5f) * pgrid / grid_w - 0.5f; // row / H axis + const float bf = (j + 0.5f) * pgrid / grid_h - 0.5f; // col / W axis + const int a0 = (int) std::floor(af); const float wa = af - a0; + const int b0 = (int) std::floor(bf); const float wb = bf - b0; + const int as[2] = { a0, a0 + 1 }; const float was[2] = { 1.0f - wa, wa }; + const int bs[2] = { b0, b0 + 1 }; const float wbs[2] = { 1.0f - wb, wb }; + float * dst = pos_emb.data() + (size_t) t * nemb; + for (int ka = 0; ka < 2; ka++) { + if (as[ka] < 0 || as[ka] >= pgrid) continue; + for (int kb = 0; kb < 2; kb++) { + if (bs[kb] < 0 || bs[kb] >= pgrid) continue; + const float w = was[ka] * wbs[kb]; + if (w == 0.0f) continue; + const float * src = pe_host.data() + (size_t) (as[ka] * pgrid + bs[kb]) * nemb; + for (int c = 0; c < nemb; c++) dst[c] += w * src[c]; + } + } + } + set_input_f32("onyx_pos_emb", pos_emb); + + // --- sparse window grouping (pgrid x pgrid windows) --- + const int win = pgrid; + const int nwin_h = (grid_h + win - 1) / win; + const int nwin_w = (grid_w + win - 1) / win; + std::vector sp_perm; sp_perm.reserve(n_tok); + std::vector sp_slens; + for (int wy = 0; wy < nwin_h; wy++) { + for (int wx = 0; wx < nwin_w; wx++) { + int cnt = 0; + for (int hh = 0; hh < win; hh++) { + for (int ww = 0; ww < win; ww++) { + const int gy = wy * win + hh; + const int gx = wx * win + ww; + if (gy < grid_h && gx < grid_w) { sp_perm.push_back(gy * grid_w + gx); cnt++; } + } + } + if (cnt > 0) sp_slens.push_back(cnt); + } + } + std::vector rpos_w(n_tok), rpos_h(n_tok), inv_perm(n_tok); + for (int i = 0; i < n_tok; i++) { + const int orig = sp_perm[i]; + rpos_w[i] = (orig % grid_w) + 1; // 1-indexed + rpos_h[i] = (orig / grid_w) + 1; + inv_perm[orig] = i; + } + set_input_i32("onyx_sp_perm", sp_perm); + set_input_i32("onyx_inv_perm", inv_perm); + set_input_i32("onyx_pos_w", rpos_w); + set_input_i32("onyx_pos_h", rpos_h); + + // block-diagonal window mask (permuted order) + std::vector sp_mask((size_t) n_tok * n_tok, -INFINITY); + { + int off = 0; + for (int s : sp_slens) { + for (int a = 0; a < s; a++) + for (int b = 0; b < s; b++) + sp_mask[(size_t) (off + a) * n_tok + (off + b)] = 0.0f; + off += s; + } + } + set_input_f32("onyx_sp_mask", sp_mask); + + // pixel-shuffle gather (original order): f*f spatial neighbours grouped + std::vector dsp; dsp.reserve(n_tok); + for (int oy = 0; oy < grid_h / f; oy++) + for (int ox = 0; ox < grid_w / f; ox++) + for (int ry = 0; ry < f; ry++) + for (int rx = 0; rx < f; rx++) + dsp.push_back((oy * f + ry) * grid_w + (ox * f + rx)); + set_input_i32("onyx_ds_perm", dsp); + } break; case PROJECTOR_TYPE_MINICPMV: { // inspired from siglip: @@ -4999,6 +5143,8 @@ int clip_n_mmproj_embd(const struct clip_ctx * ctx) { return ctx->model.mm_model_mlp_3_w->ne[1]; case PROJECTOR_TYPE_MINIMAX_M3: return ctx->model.mm_merger_fc2_b->ne[0]; + case PROJECTOR_TYPE_ONYX: + return ctx->model.mm_vision_proj->ne[1]; case PROJECTOR_TYPE_QWEN2VL: case PROJECTOR_TYPE_QWEN25VL: case PROJECTOR_TYPE_EXAONE4_5: diff --git a/tools/mtmd/models/models.h b/tools/mtmd/models/models.h index e54366a086..614d66f507 100644 --- a/tools/mtmd/models/models.h +++ b/tools/mtmd/models/models.h @@ -254,3 +254,8 @@ private: ggml_tensor * build_newline_row(ggml_context * ctx0); ggml_tensor * append_rowwise_newlines(ggml_context * ctx0, ggml_tensor * tile_output); }; + +struct clip_graph_onyx : clip_graph { + clip_graph_onyx(clip_ctx * ctx, const clip_image_f32 & img) : clip_graph(ctx, img) {} + ggml_cgraph * build() override; +}; From cd5ba86fe9395cdc7527d3a57d1f852031cbdbe4 Mon Sep 17 00:00:00 2001 From: Pedro Cuenca Date: Tue, 4 Aug 2026 23:04:06 +0200 Subject: [PATCH 06/26] Go back to using delimiters. Otherwise our generations are worse. Transformers does not use them. We need to trace inputs to verify whether they are equivalent. --- tools/mtmd/mtmd.cpp | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/tools/mtmd/mtmd.cpp b/tools/mtmd/mtmd.cpp index 5c49b35408..ea8f744e5f 100644 --- a/tools/mtmd/mtmd.cpp +++ b/tools/mtmd/mtmd.cpp @@ -472,7 +472,10 @@ struct mtmd_context { } break; case PROJECTOR_TYPE_ONYX: { - // Follow transformers Onyx processing: <|patch|>*N, no delimiters + // NOTE: transformers no longer uses delimiters (just <|patch|>*N), + // but we get subpar generations if we don't. + img_beg = "<|image_start|>"; + img_end = "<|image_end|>"; image_preproc = std::make_unique(ctx_v); } break; case PROJECTOR_TYPE_YOUTUVL: From ef9e715b8c264d652baafd3149feec0f564395da Mon Sep 17 00:00:00 2001 From: Pedro Cuenca Date: Tue, 4 Aug 2026 23:13:01 +0200 Subject: [PATCH 07/26] downsample_factor -> merge_size --- conversion/onyx.py | 2 +- tools/mtmd/clip-impl.h | 1 - tools/mtmd/clip.cpp | 2 +- 3 files changed, 2 insertions(+), 3 deletions(-) diff --git a/conversion/onyx.py b/conversion/onyx.py index 3da3da167d..a2eaea4615 100644 --- a/conversion/onyx.py +++ b/conversion/onyx.py @@ -87,10 +87,10 @@ class OnyxVisionModel(MmprojModel): self.gguf_writer.add_clip_projector_type(gguf.VisionProjectorType.ONYX) self.gguf_writer.add_vision_attention_layernorm_eps(float(c["layer_norm_eps"])) + self.gguf_writer.add_vision_spatial_merge_size(int(c["merge_size"])) rope_theta = float(c.get("rope_parameters", {}).get("rope_theta", self.ROPE_THETA)) self.gguf_writer.add_uint32 ("clip.vision.onyx.patch_temporal", int(c["patch_temporal"])) - self.gguf_writer.add_uint32 ("clip.vision.onyx.downsample_factor", int(c["merge_size"])) self.gguf_writer.add_uint32 ("clip.vision.onyx.sparse_attention_factor", int(c["sparse_attention_factor"])) self.gguf_writer.add_uint32 ("clip.vision.onyx.pos_emb_grid", int(c["pos_emb_height"])) self.gguf_writer.add_float32("clip.vision.onyx.rope_theta", rope_theta) diff --git a/tools/mtmd/clip-impl.h b/tools/mtmd/clip-impl.h index 0a2326ff6d..e3466852ba 100644 --- a/tools/mtmd/clip-impl.h +++ b/tools/mtmd/clip-impl.h @@ -92,7 +92,6 @@ #define KEY_AUDIO_SUBSAMPLING_FACTOR "clip.audio.subsampling_factor" #define KEY_ONYX_PATCH_TEMPORAL "clip.vision.onyx.patch_temporal" -#define KEY_ONYX_DOWNSAMPLE "clip.vision.onyx.downsample_factor" #define KEY_ONYX_SPARSE_FACTOR "clip.vision.onyx.sparse_attention_factor" #define KEY_ONYX_POS_GRID "clip.vision.onyx.pos_emb_grid" #define KEY_ONYX_ROPE_THETA "clip.vision.onyx.rope_theta" diff --git a/tools/mtmd/clip.cpp b/tools/mtmd/clip.cpp index 3a6411c888..aad3ffbbdf 100644 --- a/tools/mtmd/clip.cpp +++ b/tools/mtmd/clip.cpp @@ -1525,7 +1525,7 @@ struct clip_model_loader { hparams.n_merge = 2; // pixel-shuffle downsample after the ViT hparams.image_resize_algo = RESIZE_ALGO_LANCZOS; hparams.rope_theta = 10000.0f; - get_u32(KEY_ONYX_DOWNSAMPLE, hparams.n_merge, false); + get_u32(KEY_SPATIAL_MERGE_SIZE, hparams.n_merge, false); get_f32(KEY_ONYX_ROPE_THETA, hparams.rope_theta, false); get_u32(KEY_ONYX_PATCH_TEMPORAL, hparams.onyx_patch_temporal, false); get_u32(KEY_ONYX_SPARSE_FACTOR, hparams.onyx_sparse_factor, false); From 05b8f43f7c55f025b8178fe5879644382820d7db Mon Sep 17 00:00:00 2001 From: Pedro Cuenca Date: Tue, 4 Aug 2026 23:16:42 +0200 Subject: [PATCH 08/26] Add vision graph lol, forgot from a previous commit --- tools/mtmd/models/onyx.cpp | 121 +++++++++++++++++++++++++++++++++++++ 1 file changed, 121 insertions(+) create mode 100644 tools/mtmd/models/onyx.cpp diff --git a/tools/mtmd/models/onyx.cpp b/tools/mtmd/models/onyx.cpp new file mode 100644 index 0000000000..73325e3597 --- /dev/null +++ b/tools/mtmd/models/onyx.cpp @@ -0,0 +1,121 @@ +#include "models.h" + +// Onyx vision encoder: 50-layer ViT with 2D RoPE, sparse block-diagonal +// window attention (every 4th + last layer global), pixel-shuffle downsample, then +// adapter MLP + LLM's vision_projection. Output dim = 6656 (onyx n_embd), +// injected via llama_batch.embd; the onyx LLM graph applies the scaleless rms_norm +// (== reference perception_emb_norm). +// +// Several quantities are precomputed on host and fed as named graph inputs (filled in +// clip.cpp set_input, PROJECTOR_TYPE_ONYX branch): +// onyx_patches [patch_dim, n_tok] : patchified pixels ([pt,c,ps,ps] layout) +// onyx_pos_emb [n_embd, n_tok] : bilinear-interpolated learned pos-emb (orig order) +// onyx_pos_w/_h [n_tok] i32 : 1-indexed RoPE positions (sparse-permuted order) +// onyx_sp_perm [n_tok] i32 : window grouping permutation (applied after ln_pre) +// onyx_inv_perm [n_tok] i32 : inverse of sp_perm (applied after blocks) +// onyx_ds_perm [n_tok] i32 : pixel-shuffle gather (original order) +// onyx_sp_mask [n_tok, n_tok] f32 : block-diagonal window mask (sparse layers) +ggml_cgraph * clip_graph_onyx::build() { + const int ds = hparams.n_merge; // downsample factor (2) + const int pt = hparams.onyx_patch_temporal; // 2 + const int sf = hparams.onyx_sparse_factor; // 4 + const int n_tok = n_patches; + const int patch_dim = pt * 3 * patch_size * patch_size; // 1176 + const int n_out = (n_patches_x / ds) * (n_patches_y / ds); + const float rope_base = hparams.rope_theta; // 10000 + const float attn_scale = 1.0f / sqrtf((float) d_head); // SDPA default + + auto inp_i32 = [&](const char * name, int64_t n) { + ggml_tensor * t = ggml_new_tensor_1d(ctx0, GGML_TYPE_I32, n); + ggml_set_name(t, name); ggml_set_input(t); + return t; + }; + + ggml_tensor * patches = ggml_new_tensor_2d(ctx0, GGML_TYPE_F32, patch_dim, n_tok); + ggml_set_name(patches, "onyx_patches"); ggml_set_input(patches); + + ggml_tensor * pos_emb = ggml_new_tensor_2d(ctx0, GGML_TYPE_F32, n_embd, n_tok); + ggml_set_name(pos_emb, "onyx_pos_emb"); ggml_set_input(pos_emb); + + ggml_tensor * pos_w = inp_i32("onyx_pos_w", n_tok); + ggml_tensor * pos_h = inp_i32("onyx_pos_h", n_tok); + ggml_tensor * sp_perm = inp_i32("onyx_sp_perm", n_tok); + ggml_tensor * inv_perm = inp_i32("onyx_inv_perm", n_tok); + ggml_tensor * ds_perm = inp_i32("onyx_ds_perm", n_tok); + + ggml_tensor * sp_mask = ggml_new_tensor_2d(ctx0, GGML_TYPE_F32, n_tok, n_tok); + ggml_set_name(sp_mask, "onyx_sp_mask"); ggml_set_input(sp_mask); + + // patchify (conv1_linear as a matmul, no bias) + learned pos-emb + ggml_tensor * x = build_mm(model.patch_embeddings_0, patches); // [n_embd, n_tok] + x = ggml_add(ctx0, x, pos_emb); + cb(x, "after_posemb", -1); + + // ln_pre (LayerNorm) + x = build_norm(x, model.pre_ln_w, model.pre_ln_b, NORM_TYPE_NORMAL, eps, -1); + + // group patches into 32x32 windows (sparse attention order) + x = ggml_get_rows(ctx0, x, sp_perm); + cb(x, "after_ln_pre", -1); + + for (int il = 0; il < n_layer; il++) { + const auto & layer = model.layers[il]; + const bool is_global = (il == n_layer - 1) || ((il + 1) % sf == 0); + + ggml_tensor * inpL = x; + + ggml_tensor * cur = build_norm(x, layer.ln_1_w, layer.ln_1_b, NORM_TYPE_NORMAL, eps, il); + + ggml_tensor * Q = ggml_add(ctx0, build_mm(layer.q_w, cur), layer.q_b); + ggml_tensor * K = ggml_add(ctx0, build_mm(layer.k_w, cur), layer.k_b); + ggml_tensor * V = ggml_add(ctx0, build_mm(layer.v_w, cur), layer.v_b); + + Q = ggml_reshape_3d(ctx0, Q, d_head, n_head, n_tok); + K = ggml_reshape_3d(ctx0, K, d_head, n_head, n_tok); + V = ggml_reshape_3d(ctx0, V, d_head, n_head, n_tok); + + // 2D RoPE: first half of head_dim uses width pos, second half uses height pos + Q = build_rope_2d(ctx0, Q, pos_w, pos_h, rope_base, false); + K = build_rope_2d(ctx0, K, pos_w, pos_h, rope_base, false); + + ggml_tensor * mask = is_global ? nullptr : sp_mask; + cur = build_attn(layer.o_w, layer.o_b, Q, K, V, mask, attn_scale, il); + + x = ggml_add(ctx0, inpL, cur); // residual 1 + inpL = x; + + cur = build_norm(x, layer.ln_2_w, layer.ln_2_b, NORM_TYPE_NORMAL, eps, il); + cur = build_ffn(cur, + layer.ff_up_w, layer.ff_up_b, + nullptr, nullptr, + layer.ff_down_w, layer.ff_down_b, + FFN_GELU_ERF, il); // reference uses exact (erf) GELU + x = ggml_add(ctx0, inpL, cur); // residual 2 + cb(x, "layer_out", il); + } + + // un-permute back to original grid order, then ln_post + x = ggml_get_rows(ctx0, x, inv_perm); + x = build_norm(x, model.post_ln_w, model.post_ln_b, NORM_TYPE_NORMAL, eps, -1); + cb(x, "after_ln_post", -1); + + // pixel-shuffle downsample: gather f*f spatial neighbors then concat channel-outer. + // out[c*(ds*ds)+s, o] = x[ds_perm gathered][o*(ds*ds)+s, c] + x = ggml_get_rows(ctx0, x, ds_perm); // [n_embd, n_tok], grouped + x = ggml_reshape_3d(ctx0, x, n_embd, ds * ds, n_out);// [c, s, o] + x = ggml_permute(ctx0, x, 1, 0, 2, 3); // [s, c, o] + x = ggml_cont(ctx0, x); + x = ggml_reshape_2d(ctx0, x, n_embd * ds * ds, n_out); // [6144, n_out] + cb(x, "encoder_out", -1); + + // adapter (6144->4096->4096, exact GELU each) + LLM vision_projection (4096->6656) + x = build_mm(model.mm_adapter_fc, x); + x = ggml_gelu_erf(ctx0, x); + x = build_mm(model.mm_adapter_proj, x); + x = ggml_gelu_erf(ctx0, x); + x = build_mm(model.mm_vision_proj, x); // [6656, n_out] + cb(x, "projected", -1); + + ggml_build_forward_expand(gf, x); + return gf; +} From 4db53f43bf543e18c31106d67f0a99a756d8b1ea Mon Sep 17 00:00:00 2001 From: Pedro Cuenca Date: Tue, 4 Aug 2026 23:31:30 +0200 Subject: [PATCH 09/26] Additional renames, align with llama.cpp / transformers --- conversion/onyx.py | 20 +++++++++---- gguf-py/gguf/constants.py | 10 +------ gguf-py/gguf/gguf_writer.py | 3 ++ gguf-py/gguf/tensor_mapping.py | 12 -------- tools/mtmd/clip-impl.h | 8 ++---- tools/mtmd/clip-model.h | 8 ++---- tools/mtmd/clip.cpp | 51 ++++++++++++++++++---------------- tools/mtmd/models/onyx.cpp | 6 ++-- 8 files changed, 54 insertions(+), 64 deletions(-) diff --git a/conversion/onyx.py b/conversion/onyx.py index a2eaea4615..bc95d899f2 100644 --- a/conversion/onyx.py +++ b/conversion/onyx.py @@ -88,12 +88,12 @@ class OnyxVisionModel(MmprojModel): self.gguf_writer.add_clip_projector_type(gguf.VisionProjectorType.ONYX) self.gguf_writer.add_vision_attention_layernorm_eps(float(c["layer_norm_eps"])) self.gguf_writer.add_vision_spatial_merge_size(int(c["merge_size"])) + self.gguf_writer.add_vision_rope_theta(float(c.get("rope_parameters", {}).get("rope_theta", self.ROPE_THETA))) - rope_theta = float(c.get("rope_parameters", {}).get("rope_theta", self.ROPE_THETA)) - self.gguf_writer.add_uint32 ("clip.vision.onyx.patch_temporal", int(c["patch_temporal"])) - self.gguf_writer.add_uint32 ("clip.vision.onyx.sparse_attention_factor", int(c["sparse_attention_factor"])) - self.gguf_writer.add_uint32 ("clip.vision.onyx.pos_emb_grid", int(c["pos_emb_height"])) - self.gguf_writer.add_float32("clip.vision.onyx.rope_theta", rope_theta) + self.gguf_writer.add_uint32("clip.vision.onyx.patch_temporal", int(c["patch_temporal"])) + self.gguf_writer.add_uint32("clip.vision.onyx.sparse_attention_factor", int(c["sparse_attention_factor"])) + self.gguf_writer.add_uint32("clip.vision.onyx.pos_emb_height", int(c["pos_emb_height"])) + self.gguf_writer.add_uint32("clip.vision.onyx.pos_emb_width", int(c["pos_emb_width"])) @classmethod def filter_tensors(cls, item): @@ -114,8 +114,18 @@ class OnyxVisionModel(MmprojModel): return tensor.view(n_heads, 2, dim1 // n_heads // 2).transpose(1, 2).reshape(dim1) raise ValueError(f"_unpermute_for_rope: unexpected shape {tuple(tensor.shape)}") + # 3-layer projector MLP: emit as the standard V_MMPROJ numbered slots (mm.0/mm.1/mm.2) + _MM_MLP_MAP = { + "model.vision_adapter.fc1.weight": "mm.0.weight", + "model.vision_adapter.fc2.weight": "mm.1.weight", + "model.vision_projection.weight": "mm.2.weight", + } + def modify_tensors(self, data_torch, name, bid): if ".attn.q_proj." in name or ".attn.k_proj." in name: n_heads = int(self.hparams_vision["num_attention_heads"]) data_torch = self._unpermute_for_rope(data_torch, n_heads) + if name in self._MM_MLP_MAP: + yield (self._MM_MLP_MAP[name], data_torch) + return yield (self.map_tensor_name(name), data_torch) diff --git a/gguf-py/gguf/constants.py b/gguf-py/gguf/constants.py index 0fec76ec69..f675ba699f 100644 --- a/gguf-py/gguf/constants.py +++ b/gguf-py/gguf/constants.py @@ -341,6 +341,7 @@ class Keys: IMAGE_MEAN = "clip.vision.image_mean" IMAGE_STD = "clip.vision.image_std" SPATIAL_MERGE_SIZE = "clip.vision.spatial_merge_size" + ROPE_THETA = "clip.vision.rope_theta" USE_GELU = "clip.use_gelu" USE_SILU = "clip.use_silu" N_WA_PATTERN = "clip.vision.n_wa_pattern" # used by qwen2.5vl @@ -873,9 +874,6 @@ class MODEL_TENSOR(IntEnum): V_MM_GATE = auto() # cogvlm V_MM_MERGER_FC1 = auto() # minimax-m3 (patch-merge MLP) V_MM_MERGER_FC2 = auto() # minimax-m3 (patch-merge MLP) - V_MM_ADAPTER_FC = auto() # onyx - V_MM_ADAPTER_PROJ = auto() # onyx - V_MM_VISION_PROJ = auto() # onyx V_TOK_BOI = auto() # cogvlm V_TOK_EOI = auto() # cogvlm V_TOK_IMG_BEGIN = auto() # hunyuanvl @@ -1486,9 +1484,6 @@ TENSOR_NAMES: dict[MODEL_TENSOR, str] = { MODEL_TENSOR.V_MM_GATE: "mm.gate", MODEL_TENSOR.V_MM_MERGER_FC1: "mm.merger.fc1", MODEL_TENSOR.V_MM_MERGER_FC2: "mm.merger.fc2", - MODEL_TENSOR.V_MM_ADAPTER_FC: "mm.adapter_fc", # onyx - MODEL_TENSOR.V_MM_ADAPTER_PROJ: "mm.adapter_proj", # onyx - MODEL_TENSOR.V_MM_VISION_PROJ: "mm.vision_proj", # onyx MODEL_TENSOR.V_TOK_BOI: "v.boi", MODEL_TENSOR.V_TOK_EOI: "v.eoi", MODEL_TENSOR.V_MM_PRE_NORM: "mm.pre_norm", @@ -1723,9 +1718,6 @@ MODEL_TENSORS: dict[MODEL_ARCH, list[MODEL_TENSOR]] = { MODEL_TENSOR.V_MM_UP, MODEL_TENSOR.V_MM_DOWN, MODEL_TENSOR.V_MM_GATE, - MODEL_TENSOR.V_MM_ADAPTER_FC, - MODEL_TENSOR.V_MM_ADAPTER_PROJ, - MODEL_TENSOR.V_MM_VISION_PROJ, MODEL_TENSOR.V_TOK_BOI, MODEL_TENSOR.V_TOK_EOI, MODEL_TENSOR.V_MM_PRE_NORM, diff --git a/gguf-py/gguf/gguf_writer.py b/gguf-py/gguf/gguf_writer.py index af14e24fe4..73377f9491 100644 --- a/gguf-py/gguf/gguf_writer.py +++ b/gguf-py/gguf/gguf_writer.py @@ -1271,6 +1271,9 @@ class GGUFWriter: def add_vision_spatial_merge_size(self, value: int) -> None: self.add_uint32(Keys.ClipVision.SPATIAL_MERGE_SIZE, value) + def add_vision_rope_theta(self, value: float) -> None: + self.add_float32(Keys.ClipVision.ROPE_THETA, value) + def add_vision_use_gelu(self, value: bool) -> None: self.add_bool(Keys.ClipVision.USE_GELU, value) diff --git a/gguf-py/gguf/tensor_mapping.py b/gguf-py/gguf/tensor_mapping.py index 44a85d2960..1bcc726159 100644 --- a/gguf-py/gguf/tensor_mapping.py +++ b/gguf-py/gguf/tensor_mapping.py @@ -1869,18 +1869,6 @@ class TensorNameMap: "patch_merge_mlp.linear_2", # minimax-m3 ), - MODEL_TENSOR.V_MM_ADAPTER_FC: ( - "model.vision_adapter.fc1", # onyx - ), - - MODEL_TENSOR.V_MM_ADAPTER_PROJ: ( - "model.vision_adapter.fc2", # onyx - ), - - MODEL_TENSOR.V_MM_VISION_PROJ: ( - "model.vision_projection", # onyx - ), - MODEL_TENSOR.V_DS_NORM: ( "model.visual.deepstack_merger_list.{bid}.norm", # deepstack in qwen3vl ), diff --git a/tools/mtmd/clip-impl.h b/tools/mtmd/clip-impl.h index e3466852ba..3346e946fa 100644 --- a/tools/mtmd/clip-impl.h +++ b/tools/mtmd/clip-impl.h @@ -61,6 +61,7 @@ #define KEY_PROJ_SAMPLE_WINDOW_SIDE "clip.vision.projector.window_side" #define KEY_PROJ_SPATIAL_OFFSETS "clip.vision.projector.spatial_offsets" #define KEY_SPATIAL_MERGE_SIZE "clip.vision.spatial_merge_size" +#define KEY_V_ROPE_THETA "clip.vision.rope_theta" #define KEY_MM_PATCH_MERGE_TYPE "clip.vision.mm_patch_merge_type" #define KEY_IMAGE_GRID_PINPOINTS "clip.vision.image_grid_pinpoints" @@ -93,8 +94,8 @@ #define KEY_ONYX_PATCH_TEMPORAL "clip.vision.onyx.patch_temporal" #define KEY_ONYX_SPARSE_FACTOR "clip.vision.onyx.sparse_attention_factor" -#define KEY_ONYX_POS_GRID "clip.vision.onyx.pos_emb_grid" -#define KEY_ONYX_ROPE_THETA "clip.vision.onyx.rope_theta" +#define KEY_ONYX_POS_EMB_H "clip.vision.onyx.pos_emb_height" +#define KEY_ONYX_POS_EMB_W "clip.vision.onyx.pos_emb_width" // // tensor name constants @@ -133,9 +134,6 @@ #define TN_MM_GATE "mm.gate.%s" #define TN_MM_DOWN "mm.down.%s" #define TN_MM_POST_NORM "mm.post_norm.%s" -#define TN_MM_ADAPTER_FC "mm.adapter_fc.%s" // onyx -#define TN_MM_ADAPTER_PROJ "mm.adapter_proj.%s" // onyx -#define TN_MM_VISION_PROJ "mm.vision_proj.%s" // onyx #define TN_MVLM_PROJ_MLP "mm.model.mlp.%d.%s" #define TN_MVLM_PROJ_BLOCK "mm.model.mb_block.%d.block.%d.%s" #define TN_MVLM_PROJ_PEG "mm.model.peg.%d.%s" diff --git a/tools/mtmd/clip-model.h b/tools/mtmd/clip-model.h index 8214d3271b..1c71a04681 100644 --- a/tools/mtmd/clip-model.h +++ b/tools/mtmd/clip-model.h @@ -112,7 +112,8 @@ struct clip_hparams { // Onyx vision (per-block sparse-window pattern, learned pos-emb, patch-temporal) int32_t onyx_patch_temporal = 0; int32_t onyx_sparse_factor = 0; - int32_t onyx_pos_grid = 0; + int32_t onyx_pos_emb_h = 0; + int32_t onyx_pos_emb_w = 0; // audio int32_t n_mel_bins = 0; // whisper preprocessor @@ -413,11 +414,6 @@ struct clip_model { ggml_tensor * mm_post_norm_w = nullptr; ggml_tensor * mm_post_norm_b = nullptr; - // Onyx adapter + final vision projection (3-linear MLP with erf-GELU) - ggml_tensor * mm_adapter_fc = nullptr; - ggml_tensor * mm_adapter_proj = nullptr; - ggml_tensor * mm_vision_proj = nullptr; - // LLaVA projection ggml_tensor * mm_input_norm_w = nullptr; ggml_tensor * mm_input_norm_b = nullptr; diff --git a/tools/mtmd/clip.cpp b/tools/mtmd/clip.cpp index aad3ffbbdf..ec89043077 100644 --- a/tools/mtmd/clip.cpp +++ b/tools/mtmd/clip.cpp @@ -1526,10 +1526,11 @@ struct clip_model_loader { hparams.image_resize_algo = RESIZE_ALGO_LANCZOS; hparams.rope_theta = 10000.0f; get_u32(KEY_SPATIAL_MERGE_SIZE, hparams.n_merge, false); - get_f32(KEY_ONYX_ROPE_THETA, hparams.rope_theta, false); + get_f32(KEY_V_ROPE_THETA, hparams.rope_theta, false); get_u32(KEY_ONYX_PATCH_TEMPORAL, hparams.onyx_patch_temporal, false); get_u32(KEY_ONYX_SPARSE_FACTOR, hparams.onyx_sparse_factor, false); - get_u32(KEY_ONYX_POS_GRID, hparams.onyx_pos_grid, false); + get_u32(KEY_ONYX_POS_EMB_H, hparams.onyx_pos_emb_h, false); + get_u32(KEY_ONYX_POS_EMB_W, hparams.onyx_pos_emb_w, false); hparams.set_limit_image_tokens(1, 4096); hparams.set_warmup_n_tokens(32*32); } break; @@ -2238,9 +2239,9 @@ struct clip_model_loader { case PROJECTOR_TYPE_ONYX: { // 3-linear MLP: fc -> erf-GELU -> proj -> erf-GELU -> vision_proj (into LLM residual dim) - model.mm_adapter_fc = get_tensor(string_format(TN_MM_ADAPTER_FC, "weight")); - model.mm_adapter_proj = get_tensor(string_format(TN_MM_ADAPTER_PROJ, "weight")); - model.mm_vision_proj = get_tensor(string_format(TN_MM_VISION_PROJ, "weight")); + model.mm_0_w = get_tensor(string_format(TN_LLAVA_PROJ, 0, "weight")); + model.mm_1_w = get_tensor(string_format(TN_LLAVA_PROJ, 1, "weight")); + model.mm_2_w = get_tensor(string_format(TN_LLAVA_PROJ, 2, "weight")); } break; case PROJECTOR_TYPE_STEP3VL: { @@ -3966,7 +3967,8 @@ bool clip_image_batch_encode(clip_ctx * ctx, int n_threads, const clip_image_f32 const int ps = patch_size; const int nx = image_size_width; const int pt = hparams.onyx_patch_temporal; - const int pgrid = hparams.onyx_pos_grid; // 32 + const int pos_h = hparams.onyx_pos_emb_h; // 32 + const int pos_w = hparams.onyx_pos_emb_w; // 32 const int nemb = hparams.n_embd; // 1536 const int f = hparams.n_merge; // downsample 2 const int patch_dim = pt * 3 * ps * ps; @@ -4000,9 +4002,9 @@ bool clip_image_batch_encode(clip_ctx * ctx, int n_threads, const clip_image_f32 set_input_f32("onyx_patches", patches); // --- learned pos-emb bilinear interpolation (grid_sample align_corners=False, - // zeros padding) from pgrid x pgrid to grid_h x grid_w --- - ggml_tensor * pe = ctx->model.position_embeddings; // [nemb, pgrid*pgrid] - std::vector pe_host((size_t) nemb * pgrid * pgrid); + // zeros padding) from pos_h x pos_w to grid_h x grid_w --- + ggml_tensor * pe = ctx->model.position_embeddings; // [nemb, pos_h*pos_w] + std::vector pe_host((size_t) nemb * pos_h * pos_w); { std::vector raw(ggml_nbytes(pe)); ggml_backend_tensor_get(pe, raw.data(), 0, ggml_nbytes(pe)); @@ -4010,7 +4012,7 @@ bool clip_image_batch_encode(clip_ctx * ctx, int n_threads, const clip_image_f32 std::memcpy(pe_host.data(), raw.data(), ggml_nbytes(pe)); } else { const auto * tt = ggml_get_type_traits(pe->type); - tt->to_float(raw.data(), pe_host.data(), (int64_t) nemb * pgrid * pgrid); + tt->to_float(raw.data(), pe_host.data(), (int64_t) nemb * pos_h * pos_w); } } // NOTE: the reference uses meshgrid(ys, xs, indexing="xy") which yields a @@ -4022,39 +4024,40 @@ bool clip_image_batch_encode(clip_ctx * ctx, int n_threads, const clip_image_f32 for (int t = 0; t < n_tok; t++) { const int i = t / grid_h; // width index (0..grid_w-1) const int j = t % grid_h; // height index (0..grid_h-1) - const float af = (i + 0.5f) * pgrid / grid_w - 0.5f; // row / H axis - const float bf = (j + 0.5f) * pgrid / grid_h - 0.5f; // col / W axis + const float af = (i + 0.5f) * pos_h / grid_w - 0.5f; // row axis (pos_h) + const float bf = (j + 0.5f) * pos_w / grid_h - 0.5f; // col axis (pos_w) const int a0 = (int) std::floor(af); const float wa = af - a0; const int b0 = (int) std::floor(bf); const float wb = bf - b0; const int as[2] = { a0, a0 + 1 }; const float was[2] = { 1.0f - wa, wa }; const int bs[2] = { b0, b0 + 1 }; const float wbs[2] = { 1.0f - wb, wb }; float * dst = pos_emb.data() + (size_t) t * nemb; for (int ka = 0; ka < 2; ka++) { - if (as[ka] < 0 || as[ka] >= pgrid) continue; + if (as[ka] < 0 || as[ka] >= pos_h) continue; for (int kb = 0; kb < 2; kb++) { - if (bs[kb] < 0 || bs[kb] >= pgrid) continue; + if (bs[kb] < 0 || bs[kb] >= pos_w) continue; const float w = was[ka] * wbs[kb]; if (w == 0.0f) continue; - const float * src = pe_host.data() + (size_t) (as[ka] * pgrid + bs[kb]) * nemb; + const float * src = pe_host.data() + (size_t) (as[ka] * pos_w + bs[kb]) * nemb; for (int c = 0; c < nemb; c++) dst[c] += w * src[c]; } } } set_input_f32("onyx_pos_emb", pos_emb); - // --- sparse window grouping (pgrid x pgrid windows) --- - const int win = pgrid; - const int nwin_h = (grid_h + win - 1) / win; - const int nwin_w = (grid_w + win - 1) / win; + // --- sparse window grouping (pos_h x pos_w windows; square in the trained model) --- + const int win_h = pos_h; + const int win_w = pos_w; + const int nwin_h = (grid_h + win_h - 1) / win_h; + const int nwin_w = (grid_w + win_w - 1) / win_w; std::vector sp_perm; sp_perm.reserve(n_tok); std::vector sp_slens; for (int wy = 0; wy < nwin_h; wy++) { for (int wx = 0; wx < nwin_w; wx++) { int cnt = 0; - for (int hh = 0; hh < win; hh++) { - for (int ww = 0; ww < win; ww++) { - const int gy = wy * win + hh; - const int gx = wx * win + ww; + for (int hh = 0; hh < win_h; hh++) { + for (int ww = 0; ww < win_w; ww++) { + const int gy = wy * win_h + hh; + const int gx = wx * win_w + ww; if (gy < grid_h && gx < grid_w) { sp_perm.push_back(gy * grid_w + gx); cnt++; } } } @@ -5144,7 +5147,7 @@ int clip_n_mmproj_embd(const struct clip_ctx * ctx) { case PROJECTOR_TYPE_MINIMAX_M3: return ctx->model.mm_merger_fc2_b->ne[0]; case PROJECTOR_TYPE_ONYX: - return ctx->model.mm_vision_proj->ne[1]; + return ctx->model.mm_2_w->ne[1]; case PROJECTOR_TYPE_QWEN2VL: case PROJECTOR_TYPE_QWEN25VL: case PROJECTOR_TYPE_EXAONE4_5: diff --git a/tools/mtmd/models/onyx.cpp b/tools/mtmd/models/onyx.cpp index 73325e3597..fced3317ea 100644 --- a/tools/mtmd/models/onyx.cpp +++ b/tools/mtmd/models/onyx.cpp @@ -109,11 +109,11 @@ ggml_cgraph * clip_graph_onyx::build() { cb(x, "encoder_out", -1); // adapter (6144->4096->4096, exact GELU each) + LLM vision_projection (4096->6656) - x = build_mm(model.mm_adapter_fc, x); + x = build_mm(model.mm_0_w, x); x = ggml_gelu_erf(ctx0, x); - x = build_mm(model.mm_adapter_proj, x); + x = build_mm(model.mm_1_w, x); x = ggml_gelu_erf(ctx0, x); - x = build_mm(model.mm_vision_proj, x); // [6656, n_out] + x = build_mm(model.mm_2_w, x); // [6656, n_out] cb(x, "projected", -1); ggml_build_forward_expand(gf, x); From 18acf25487c455ee0b0def74acd2228a3bf8e9ef Mon Sep 17 00:00:00 2001 From: Pedro Cuenca Date: Tue, 4 Aug 2026 23:39:00 +0200 Subject: [PATCH 10/26] Prefer _size instead of independent _h and _w --- conversion/onyx.py | 5 +++-- tools/mtmd/clip-impl.h | 3 +-- tools/mtmd/clip-model.h | 3 +-- tools/mtmd/clip.cpp | 43 +++++++++++++++++++---------------------- 4 files changed, 25 insertions(+), 29 deletions(-) diff --git a/conversion/onyx.py b/conversion/onyx.py index bc95d899f2..255f347c9c 100644 --- a/conversion/onyx.py +++ b/conversion/onyx.py @@ -90,10 +90,11 @@ class OnyxVisionModel(MmprojModel): self.gguf_writer.add_vision_spatial_merge_size(int(c["merge_size"])) self.gguf_writer.add_vision_rope_theta(float(c.get("rope_parameters", {}).get("rope_theta", self.ROPE_THETA))) + pos_h, pos_w = int(c["pos_emb_height"]), int(c["pos_emb_width"]) + assert pos_h == pos_w, f"Onyx assumes square pos-emb grid; got {pos_h}x{pos_w}" self.gguf_writer.add_uint32("clip.vision.onyx.patch_temporal", int(c["patch_temporal"])) self.gguf_writer.add_uint32("clip.vision.onyx.sparse_attention_factor", int(c["sparse_attention_factor"])) - self.gguf_writer.add_uint32("clip.vision.onyx.pos_emb_height", int(c["pos_emb_height"])) - self.gguf_writer.add_uint32("clip.vision.onyx.pos_emb_width", int(c["pos_emb_width"])) + self.gguf_writer.add_uint32("clip.vision.onyx.pos_emb_size", pos_h) @classmethod def filter_tensors(cls, item): diff --git a/tools/mtmd/clip-impl.h b/tools/mtmd/clip-impl.h index 3346e946fa..b66359845f 100644 --- a/tools/mtmd/clip-impl.h +++ b/tools/mtmd/clip-impl.h @@ -94,8 +94,7 @@ #define KEY_ONYX_PATCH_TEMPORAL "clip.vision.onyx.patch_temporal" #define KEY_ONYX_SPARSE_FACTOR "clip.vision.onyx.sparse_attention_factor" -#define KEY_ONYX_POS_EMB_H "clip.vision.onyx.pos_emb_height" -#define KEY_ONYX_POS_EMB_W "clip.vision.onyx.pos_emb_width" +#define KEY_ONYX_POS_EMB_SIZE "clip.vision.onyx.pos_emb_size" // // tensor name constants diff --git a/tools/mtmd/clip-model.h b/tools/mtmd/clip-model.h index 1c71a04681..b4fe606005 100644 --- a/tools/mtmd/clip-model.h +++ b/tools/mtmd/clip-model.h @@ -112,8 +112,7 @@ struct clip_hparams { // Onyx vision (per-block sparse-window pattern, learned pos-emb, patch-temporal) int32_t onyx_patch_temporal = 0; int32_t onyx_sparse_factor = 0; - int32_t onyx_pos_emb_h = 0; - int32_t onyx_pos_emb_w = 0; + int32_t onyx_pos_emb_size = 0; // audio int32_t n_mel_bins = 0; // whisper preprocessor diff --git a/tools/mtmd/clip.cpp b/tools/mtmd/clip.cpp index ec89043077..a6aa24303f 100644 --- a/tools/mtmd/clip.cpp +++ b/tools/mtmd/clip.cpp @@ -1529,8 +1529,7 @@ struct clip_model_loader { get_f32(KEY_V_ROPE_THETA, hparams.rope_theta, false); get_u32(KEY_ONYX_PATCH_TEMPORAL, hparams.onyx_patch_temporal, false); get_u32(KEY_ONYX_SPARSE_FACTOR, hparams.onyx_sparse_factor, false); - get_u32(KEY_ONYX_POS_EMB_H, hparams.onyx_pos_emb_h, false); - get_u32(KEY_ONYX_POS_EMB_W, hparams.onyx_pos_emb_w, false); + get_u32(KEY_ONYX_POS_EMB_SIZE, hparams.onyx_pos_emb_size, false); hparams.set_limit_image_tokens(1, 4096); hparams.set_warmup_n_tokens(32*32); } break; @@ -3967,9 +3966,8 @@ bool clip_image_batch_encode(clip_ctx * ctx, int n_threads, const clip_image_f32 const int ps = patch_size; const int nx = image_size_width; const int pt = hparams.onyx_patch_temporal; - const int pos_h = hparams.onyx_pos_emb_h; // 32 - const int pos_w = hparams.onyx_pos_emb_w; // 32 - const int nemb = hparams.n_embd; // 1536 + const int pgrid = hparams.onyx_pos_emb_size; // 32 + const int nemb = hparams.n_embd; // 1536 const int f = hparams.n_merge; // downsample 2 const int patch_dim = pt * 3 * ps * ps; const auto & buf = imgs.entries[0].get_ro_buf(); // interleaved pixels (3ch image / 6ch video) @@ -4002,9 +4000,9 @@ bool clip_image_batch_encode(clip_ctx * ctx, int n_threads, const clip_image_f32 set_input_f32("onyx_patches", patches); // --- learned pos-emb bilinear interpolation (grid_sample align_corners=False, - // zeros padding) from pos_h x pos_w to grid_h x grid_w --- - ggml_tensor * pe = ctx->model.position_embeddings; // [nemb, pos_h*pos_w] - std::vector pe_host((size_t) nemb * pos_h * pos_w); + // zeros padding) from pgrid x pgrid to grid_h x grid_w --- + ggml_tensor * pe = ctx->model.position_embeddings; // [nemb, pgrid*pgrid] + std::vector pe_host((size_t) nemb * pgrid * pgrid); { std::vector raw(ggml_nbytes(pe)); ggml_backend_tensor_get(pe, raw.data(), 0, ggml_nbytes(pe)); @@ -4012,7 +4010,7 @@ bool clip_image_batch_encode(clip_ctx * ctx, int n_threads, const clip_image_f32 std::memcpy(pe_host.data(), raw.data(), ggml_nbytes(pe)); } else { const auto * tt = ggml_get_type_traits(pe->type); - tt->to_float(raw.data(), pe_host.data(), (int64_t) nemb * pos_h * pos_w); + tt->to_float(raw.data(), pe_host.data(), (int64_t) nemb * pgrid * pgrid); } } // NOTE: the reference uses meshgrid(ys, xs, indexing="xy") which yields a @@ -4024,40 +4022,39 @@ bool clip_image_batch_encode(clip_ctx * ctx, int n_threads, const clip_image_f32 for (int t = 0; t < n_tok; t++) { const int i = t / grid_h; // width index (0..grid_w-1) const int j = t % grid_h; // height index (0..grid_h-1) - const float af = (i + 0.5f) * pos_h / grid_w - 0.5f; // row axis (pos_h) - const float bf = (j + 0.5f) * pos_w / grid_h - 0.5f; // col axis (pos_w) + const float af = (i + 0.5f) * pgrid / grid_w - 0.5f; // row / H axis + const float bf = (j + 0.5f) * pgrid / grid_h - 0.5f; // col / W axis const int a0 = (int) std::floor(af); const float wa = af - a0; const int b0 = (int) std::floor(bf); const float wb = bf - b0; const int as[2] = { a0, a0 + 1 }; const float was[2] = { 1.0f - wa, wa }; const int bs[2] = { b0, b0 + 1 }; const float wbs[2] = { 1.0f - wb, wb }; float * dst = pos_emb.data() + (size_t) t * nemb; for (int ka = 0; ka < 2; ka++) { - if (as[ka] < 0 || as[ka] >= pos_h) continue; + if (as[ka] < 0 || as[ka] >= pgrid) continue; for (int kb = 0; kb < 2; kb++) { - if (bs[kb] < 0 || bs[kb] >= pos_w) continue; + if (bs[kb] < 0 || bs[kb] >= pgrid) continue; const float w = was[ka] * wbs[kb]; if (w == 0.0f) continue; - const float * src = pe_host.data() + (size_t) (as[ka] * pos_w + bs[kb]) * nemb; + const float * src = pe_host.data() + (size_t) (as[ka] * pgrid + bs[kb]) * nemb; for (int c = 0; c < nemb; c++) dst[c] += w * src[c]; } } } set_input_f32("onyx_pos_emb", pos_emb); - // --- sparse window grouping (pos_h x pos_w windows; square in the trained model) --- - const int win_h = pos_h; - const int win_w = pos_w; - const int nwin_h = (grid_h + win_h - 1) / win_h; - const int nwin_w = (grid_w + win_w - 1) / win_w; + // --- sparse window grouping (pgrid x pgrid windows) --- + const int win = pgrid; + const int nwin_h = (grid_h + win - 1) / win; + const int nwin_w = (grid_w + win - 1) / win; std::vector sp_perm; sp_perm.reserve(n_tok); std::vector sp_slens; for (int wy = 0; wy < nwin_h; wy++) { for (int wx = 0; wx < nwin_w; wx++) { int cnt = 0; - for (int hh = 0; hh < win_h; hh++) { - for (int ww = 0; ww < win_w; ww++) { - const int gy = wy * win_h + hh; - const int gx = wx * win_w + ww; + for (int hh = 0; hh < win; hh++) { + for (int ww = 0; ww < win; ww++) { + const int gy = wy * win + hh; + const int gx = wx * win + ww; if (gy < grid_h && gx < grid_w) { sp_perm.push_back(gy * grid_w + gx); cnt++; } } } From 6af293185390b97c88cfc109c1e2ba899b2316e5 Mon Sep 17 00:00:00 2001 From: Pedro Cuenca Date: Wed, 5 Aug 2026 14:33:31 +0200 Subject: [PATCH 11/26] Fix token layout Co-authored-by: Young Han --- tools/mtmd/clip.cpp | 16 +++++++--------- 1 file changed, 7 insertions(+), 9 deletions(-) diff --git a/tools/mtmd/clip.cpp b/tools/mtmd/clip.cpp index a6aa24303f..c2019b2d5c 100644 --- a/tools/mtmd/clip.cpp +++ b/tools/mtmd/clip.cpp @@ -4013,17 +4013,15 @@ bool clip_image_batch_encode(clip_ctx * ctx, int n_threads, const clip_image_f32 tt->to_float(raw.data(), pe_host.data(), (int64_t) nemb * pgrid * pgrid); } } - // NOTE: the reference uses meshgrid(ys, xs, indexing="xy") which yields a - // [grid_w, grid_h] grid (width-outer) and grid_sample maps coord0=ys->W axis, - // coord1=xs->H axis (a transposed sampling). Token t decomposes as - // i=t/grid_h (width), j=t%grid_h (height); sample row a from i (grid_w scale), - // col b from j (grid_h scale). This matches the trained convention exactly. + // Token layout is row-major with grid_w as the inner stride (matches the sparse + // window loop below). H axis samples pe rows scaled by grid_h; W axis samples + // pe cols scaled by grid_w. Self-cancelling on square grids. std::vector pos_emb((size_t) nemb * n_tok, 0.0f); for (int t = 0; t < n_tok; t++) { - const int i = t / grid_h; // width index (0..grid_w-1) - const int j = t % grid_h; // height index (0..grid_h-1) - const float af = (i + 0.5f) * pgrid / grid_w - 0.5f; // row / H axis - const float bf = (j + 0.5f) * pgrid / grid_h - 0.5f; // col / W axis + const int gy = t / grid_w; // height index (0..grid_h-1) + const int gx = t % grid_w; // width index (0..grid_w-1) + const float af = (gy + 0.5f) * pgrid / grid_h - 0.5f; // H axis + const float bf = (gx + 0.5f) * pgrid / grid_w - 0.5f; // W axis const int a0 = (int) std::floor(af); const float wa = af - a0; const int b0 = (int) std::floor(bf); const float wb = bf - b0; const int as[2] = { a0, a0 + 1 }; const float was[2] = { 1.0f - wa, wa }; From d6e2e02d75978e9026b8ca577f04ce129f79336b Mon Sep 17 00:00:00 2001 From: Pedro Cuenca Date: Thu, 6 Aug 2026 12:10:55 +0200 Subject: [PATCH 12/26] Less params, bilinear pos-emb interpolation as a graph op instead of CPU --- conversion/onyx.py | 3 --- tools/mtmd/clip-impl.h | 1 - tools/mtmd/clip-model.h | 1 - tools/mtmd/clip.cpp | 45 +------------------------------------- tools/mtmd/models/onyx.cpp | 6 +---- 5 files changed, 2 insertions(+), 54 deletions(-) diff --git a/conversion/onyx.py b/conversion/onyx.py index 255f347c9c..1e7257dccd 100644 --- a/conversion/onyx.py +++ b/conversion/onyx.py @@ -90,11 +90,8 @@ class OnyxVisionModel(MmprojModel): self.gguf_writer.add_vision_spatial_merge_size(int(c["merge_size"])) self.gguf_writer.add_vision_rope_theta(float(c.get("rope_parameters", {}).get("rope_theta", self.ROPE_THETA))) - pos_h, pos_w = int(c["pos_emb_height"]), int(c["pos_emb_width"]) - assert pos_h == pos_w, f"Onyx assumes square pos-emb grid; got {pos_h}x{pos_w}" self.gguf_writer.add_uint32("clip.vision.onyx.patch_temporal", int(c["patch_temporal"])) self.gguf_writer.add_uint32("clip.vision.onyx.sparse_attention_factor", int(c["sparse_attention_factor"])) - self.gguf_writer.add_uint32("clip.vision.onyx.pos_emb_size", pos_h) @classmethod def filter_tensors(cls, item): diff --git a/tools/mtmd/clip-impl.h b/tools/mtmd/clip-impl.h index b66359845f..9b0f486e39 100644 --- a/tools/mtmd/clip-impl.h +++ b/tools/mtmd/clip-impl.h @@ -94,7 +94,6 @@ #define KEY_ONYX_PATCH_TEMPORAL "clip.vision.onyx.patch_temporal" #define KEY_ONYX_SPARSE_FACTOR "clip.vision.onyx.sparse_attention_factor" -#define KEY_ONYX_POS_EMB_SIZE "clip.vision.onyx.pos_emb_size" // // tensor name constants diff --git a/tools/mtmd/clip-model.h b/tools/mtmd/clip-model.h index b4fe606005..5dc706c26e 100644 --- a/tools/mtmd/clip-model.h +++ b/tools/mtmd/clip-model.h @@ -112,7 +112,6 @@ struct clip_hparams { // Onyx vision (per-block sparse-window pattern, learned pos-emb, patch-temporal) int32_t onyx_patch_temporal = 0; int32_t onyx_sparse_factor = 0; - int32_t onyx_pos_emb_size = 0; // audio int32_t n_mel_bins = 0; // whisper preprocessor diff --git a/tools/mtmd/clip.cpp b/tools/mtmd/clip.cpp index c2019b2d5c..abef8a1272 100644 --- a/tools/mtmd/clip.cpp +++ b/tools/mtmd/clip.cpp @@ -1529,7 +1529,6 @@ struct clip_model_loader { get_f32(KEY_V_ROPE_THETA, hparams.rope_theta, false); get_u32(KEY_ONYX_PATCH_TEMPORAL, hparams.onyx_patch_temporal, false); get_u32(KEY_ONYX_SPARSE_FACTOR, hparams.onyx_sparse_factor, false); - get_u32(KEY_ONYX_POS_EMB_SIZE, hparams.onyx_pos_emb_size, false); hparams.set_limit_image_tokens(1, 4096); hparams.set_warmup_n_tokens(32*32); } break; @@ -3966,8 +3965,7 @@ bool clip_image_batch_encode(clip_ctx * ctx, int n_threads, const clip_image_f32 const int ps = patch_size; const int nx = image_size_width; const int pt = hparams.onyx_patch_temporal; - const int pgrid = hparams.onyx_pos_emb_size; // 32 - const int nemb = hparams.n_embd; // 1536 + const int pgrid = (int) std::sqrt((double) ctx->model.position_embeddings->ne[1]); // 32 const int f = hparams.n_merge; // downsample 2 const int patch_dim = pt * 3 * ps * ps; const auto & buf = imgs.entries[0].get_ro_buf(); // interleaved pixels (3ch image / 6ch video) @@ -3999,47 +3997,6 @@ bool clip_image_batch_encode(clip_ctx * ctx, int n_threads, const clip_image_f32 } set_input_f32("onyx_patches", patches); - // --- learned pos-emb bilinear interpolation (grid_sample align_corners=False, - // zeros padding) from pgrid x pgrid to grid_h x grid_w --- - ggml_tensor * pe = ctx->model.position_embeddings; // [nemb, pgrid*pgrid] - std::vector pe_host((size_t) nemb * pgrid * pgrid); - { - std::vector raw(ggml_nbytes(pe)); - ggml_backend_tensor_get(pe, raw.data(), 0, ggml_nbytes(pe)); - if (pe->type == GGML_TYPE_F32) { - std::memcpy(pe_host.data(), raw.data(), ggml_nbytes(pe)); - } else { - const auto * tt = ggml_get_type_traits(pe->type); - tt->to_float(raw.data(), pe_host.data(), (int64_t) nemb * pgrid * pgrid); - } - } - // Token layout is row-major with grid_w as the inner stride (matches the sparse - // window loop below). H axis samples pe rows scaled by grid_h; W axis samples - // pe cols scaled by grid_w. Self-cancelling on square grids. - std::vector pos_emb((size_t) nemb * n_tok, 0.0f); - for (int t = 0; t < n_tok; t++) { - const int gy = t / grid_w; // height index (0..grid_h-1) - const int gx = t % grid_w; // width index (0..grid_w-1) - const float af = (gy + 0.5f) * pgrid / grid_h - 0.5f; // H axis - const float bf = (gx + 0.5f) * pgrid / grid_w - 0.5f; // W axis - const int a0 = (int) std::floor(af); const float wa = af - a0; - const int b0 = (int) std::floor(bf); const float wb = bf - b0; - const int as[2] = { a0, a0 + 1 }; const float was[2] = { 1.0f - wa, wa }; - const int bs[2] = { b0, b0 + 1 }; const float wbs[2] = { 1.0f - wb, wb }; - float * dst = pos_emb.data() + (size_t) t * nemb; - for (int ka = 0; ka < 2; ka++) { - if (as[ka] < 0 || as[ka] >= pgrid) continue; - for (int kb = 0; kb < 2; kb++) { - if (bs[kb] < 0 || bs[kb] >= pgrid) continue; - const float w = was[ka] * wbs[kb]; - if (w == 0.0f) continue; - const float * src = pe_host.data() + (size_t) (as[ka] * pgrid + bs[kb]) * nemb; - for (int c = 0; c < nemb; c++) dst[c] += w * src[c]; - } - } - } - set_input_f32("onyx_pos_emb", pos_emb); - // --- sparse window grouping (pgrid x pgrid windows) --- const int win = pgrid; const int nwin_h = (grid_h + win - 1) / win; diff --git a/tools/mtmd/models/onyx.cpp b/tools/mtmd/models/onyx.cpp index fced3317ea..7daa825787 100644 --- a/tools/mtmd/models/onyx.cpp +++ b/tools/mtmd/models/onyx.cpp @@ -9,7 +9,6 @@ // Several quantities are precomputed on host and fed as named graph inputs (filled in // clip.cpp set_input, PROJECTOR_TYPE_ONYX branch): // onyx_patches [patch_dim, n_tok] : patchified pixels ([pt,c,ps,ps] layout) -// onyx_pos_emb [n_embd, n_tok] : bilinear-interpolated learned pos-emb (orig order) // onyx_pos_w/_h [n_tok] i32 : 1-indexed RoPE positions (sparse-permuted order) // onyx_sp_perm [n_tok] i32 : window grouping permutation (applied after ln_pre) // onyx_inv_perm [n_tok] i32 : inverse of sp_perm (applied after blocks) @@ -34,9 +33,6 @@ ggml_cgraph * clip_graph_onyx::build() { ggml_tensor * patches = ggml_new_tensor_2d(ctx0, GGML_TYPE_F32, patch_dim, n_tok); ggml_set_name(patches, "onyx_patches"); ggml_set_input(patches); - ggml_tensor * pos_emb = ggml_new_tensor_2d(ctx0, GGML_TYPE_F32, n_embd, n_tok); - ggml_set_name(pos_emb, "onyx_pos_emb"); ggml_set_input(pos_emb); - ggml_tensor * pos_w = inp_i32("onyx_pos_w", n_tok); ggml_tensor * pos_h = inp_i32("onyx_pos_h", n_tok); ggml_tensor * sp_perm = inp_i32("onyx_sp_perm", n_tok); @@ -48,7 +44,7 @@ ggml_cgraph * clip_graph_onyx::build() { // patchify (conv1_linear as a matmul, no bias) + learned pos-emb ggml_tensor * x = build_mm(model.patch_embeddings_0, patches); // [n_embd, n_tok] - x = ggml_add(ctx0, x, pos_emb); + x = ggml_add(ctx0, x, resize_position_embeddings(GGML_SCALE_MODE_BILINEAR)); cb(x, "after_posemb", -1); // ln_pre (LayerNorm) From 06a8f4eee72d97650deaae722bd5b60f83c49e43 Mon Sep 17 00:00:00 2001 From: Pedro Cuenca Date: Thu, 6 Aug 2026 12:43:04 +0200 Subject: [PATCH 13/26] Map to symbolic V_MMPROJ instead of strings --- conversion/onyx.py | 14 ++++++++------ 1 file changed, 8 insertions(+), 6 deletions(-) diff --git a/conversion/onyx.py b/conversion/onyx.py index 1e7257dccd..962d7352a6 100644 --- a/conversion/onyx.py +++ b/conversion/onyx.py @@ -112,18 +112,20 @@ class OnyxVisionModel(MmprojModel): return tensor.view(n_heads, 2, dim1 // n_heads // 2).transpose(1, 2).reshape(dim1) raise ValueError(f"_unpermute_for_rope: unexpected shape {tuple(tensor.shape)}") - # 3-layer projector MLP: emit as the standard V_MMPROJ numbered slots (mm.0/mm.1/mm.2) + # 3-layer projector MLP _MM_MLP_MAP = { - "model.vision_adapter.fc1.weight": "mm.0.weight", - "model.vision_adapter.fc2.weight": "mm.1.weight", - "model.vision_projection.weight": "mm.2.weight", + "model.vision_adapter.fc1": (gguf.MODEL_TENSOR.V_MMPROJ, 0), + "model.vision_adapter.fc2": (gguf.MODEL_TENSOR.V_MMPROJ, 1), + "model.vision_projection": (gguf.MODEL_TENSOR.V_MMPROJ, 2), } def modify_tensors(self, data_torch, name, bid): if ".attn.q_proj." in name or ".attn.k_proj." in name: n_heads = int(self.hparams_vision["num_attention_heads"]) data_torch = self._unpermute_for_rope(data_torch, n_heads) - if name in self._MM_MLP_MAP: - yield (self._MM_MLP_MAP[name], data_torch) + stem, _, suffix = name.rpartition(".") + if stem in self._MM_MLP_MAP: + tensor_key, idx = self._MM_MLP_MAP[stem] + yield (self.format_tensor_name(tensor_key, bid=idx, suffix="." + suffix), data_torch) return yield (self.map_tensor_name(name), data_torch) From 30d81565cb5a73ff896db0a5f5f7d81ba8086621 Mon Sep 17 00:00:00 2001 From: Pedro Cuenca Date: Thu, 6 Aug 2026 12:51:41 +0200 Subject: [PATCH 14/26] Make a couple params explicit --- conversion/onyx.py | 10 +--------- tools/mtmd/clip-impl.h | 2 -- tools/mtmd/clip.cpp | 4 ++-- 3 files changed, 3 insertions(+), 13 deletions(-) diff --git a/conversion/onyx.py b/conversion/onyx.py index 962d7352a6..26d7c7015a 100644 --- a/conversion/onyx.py +++ b/conversion/onyx.py @@ -74,12 +74,7 @@ class OnyxVisionModel(MmprojModel): return None # Onyx actually uses dynamic size, initialize with nominal size image_size = c["pos_emb_height"] * c["patch_size"] * c["merge_size"] - # Derive sparse_attention_factor from layer_types - fulls = [i for i, t in enumerate(c["layer_types"]) if t == "full_attention"] - if not fulls: - raise ValueError("vision_config.layer_types has no full_attention layer") - sparse_factor = fulls[0] + 1 - return {**c, "image_size": image_size, "sparse_attention_factor": sparse_factor} + return {**c, "image_size": image_size} def set_gguf_parameters(self): super().set_gguf_parameters() @@ -90,9 +85,6 @@ class OnyxVisionModel(MmprojModel): self.gguf_writer.add_vision_spatial_merge_size(int(c["merge_size"])) self.gguf_writer.add_vision_rope_theta(float(c.get("rope_parameters", {}).get("rope_theta", self.ROPE_THETA))) - self.gguf_writer.add_uint32("clip.vision.onyx.patch_temporal", int(c["patch_temporal"])) - self.gguf_writer.add_uint32("clip.vision.onyx.sparse_attention_factor", int(c["sparse_attention_factor"])) - @classmethod def filter_tensors(cls, item): name, gen = item diff --git a/tools/mtmd/clip-impl.h b/tools/mtmd/clip-impl.h index 9b0f486e39..67012ff491 100644 --- a/tools/mtmd/clip-impl.h +++ b/tools/mtmd/clip-impl.h @@ -92,8 +92,6 @@ #define KEY_A_LOCAL_GROUP_SIZE "clip.audio.local_group_size" // mimo-v2.5: input_local_transformer grouping size #define KEY_AUDIO_SUBSAMPLING_FACTOR "clip.audio.subsampling_factor" -#define KEY_ONYX_PATCH_TEMPORAL "clip.vision.onyx.patch_temporal" -#define KEY_ONYX_SPARSE_FACTOR "clip.vision.onyx.sparse_attention_factor" // // tensor name constants diff --git a/tools/mtmd/clip.cpp b/tools/mtmd/clip.cpp index abef8a1272..7abbc139ed 100644 --- a/tools/mtmd/clip.cpp +++ b/tools/mtmd/clip.cpp @@ -1525,10 +1525,10 @@ struct clip_model_loader { hparams.n_merge = 2; // pixel-shuffle downsample after the ViT hparams.image_resize_algo = RESIZE_ALGO_LANCZOS; hparams.rope_theta = 10000.0f; + hparams.onyx_patch_temporal = 2; // Onyx-arch invariant + hparams.onyx_sparse_factor = 4; // 3 sparse layers + 1 global, repeating get_u32(KEY_SPATIAL_MERGE_SIZE, hparams.n_merge, false); get_f32(KEY_V_ROPE_THETA, hparams.rope_theta, false); - get_u32(KEY_ONYX_PATCH_TEMPORAL, hparams.onyx_patch_temporal, false); - get_u32(KEY_ONYX_SPARSE_FACTOR, hparams.onyx_sparse_factor, false); hparams.set_limit_image_tokens(1, 4096); hparams.set_warmup_n_tokens(32*32); } break; From df6fe72a7ef6fdbc0399f4e1cc2baf5b0858f8bb Mon Sep 17 00:00:00 2001 From: Pedro Cuenca Date: Thu, 6 Aug 2026 13:05:32 +0200 Subject: [PATCH 15/26] Patchify via build_inp() --- conversion/onyx.py | 6 ++++++ tools/mtmd/clip.cpp | 40 ++++---------------------------------- tools/mtmd/models/onyx.cpp | 10 ++-------- 3 files changed, 12 insertions(+), 44 deletions(-) diff --git a/conversion/onyx.py b/conversion/onyx.py index 26d7c7015a..c5327fa7f8 100644 --- a/conversion/onyx.py +++ b/conversion/onyx.py @@ -115,6 +115,12 @@ class OnyxVisionModel(MmprojModel): if ".attn.q_proj." in name or ".attn.k_proj." in name: n_heads = int(self.hparams_vision["num_attention_heads"]) data_torch = self._unpermute_for_rope(data_torch, n_heads) + # Lay out the pt=2 temporal slabs of the patch embedding as a conv2d for build_inp() + if name.endswith("patch_embedder.patch_embedding.weight"): + n_embd = data_torch.shape[0] + pt = int(self.hparams_vision["patch_temporal"]) + ps = int(self.hparams_vision["patch_size"]) + data_torch = data_torch.view(n_embd, pt, 3, ps, ps).sum(dim=1) # (n_embd, 3, ps, ps) stem, _, suffix = name.rpartition(".") if stem in self._MM_MLP_MAP: tensor_key, idx = self._MM_MLP_MAP[stem] diff --git a/tools/mtmd/clip.cpp b/tools/mtmd/clip.cpp index 7abbc139ed..77f5b75c0a 100644 --- a/tools/mtmd/clip.cpp +++ b/tools/mtmd/clip.cpp @@ -3897,10 +3897,7 @@ bool clip_image_batch_encode(clip_ctx * ctx, int n_threads, const clip_image_f32 }; // set input pixel values - // onyx feeds host-patchified "onyx_patches" instead of raw pixels (handled in the switch below) - if (ctx->model.proj_type == PROJECTOR_TYPE_ONYX) { - // no generic pixel input - } else if (!imgs.is_audio) { + if (!imgs.is_audio) { size_t nelem = 0; for (const auto & img : imgs.entries) { nelem += img.nx() * img.ny() * 3; @@ -3962,40 +3959,11 @@ bool clip_image_batch_encode(clip_ctx * ctx, int n_threads, const clip_image_f32 const int grid_w = pos_w; // image_size_width / patch_size const int grid_h = pos_h; // image_size_height / patch_size const int n_tok = grid_w * grid_h; - const int ps = patch_size; - const int nx = image_size_width; - const int pt = hparams.onyx_patch_temporal; const int pgrid = (int) std::sqrt((double) ctx->model.position_embeddings->ne[1]); // 32 - const int f = hparams.n_merge; // downsample 2 - const int patch_dim = pt * 3 * ps * ps; - const auto & buf = imgs.entries[0].get_ro_buf(); // interleaved pixels (3ch image / 6ch video) - // channel count: 3 = image (duplicate frame across patch_temporal), - // 3*pt = video frame-pair (distinct frames per temporal slot). - const int nchan = (int) (buf.size() / ((size_t) nx * imgs.entries[0].ny())); + const int f = hparams.n_merge; // downsample 2 - // --- patchify: [pt,c,ps,ps] per token; token = gy*grid_w + gx --- - std::vector patches((size_t) patch_dim * n_tok, 0.0f); - for (int gy = 0; gy < grid_h; gy++) { - for (int gx = 0; gx < grid_w; gx++) { - const int tok = gy * grid_w + gx; - for (int fr = 0; fr < pt; fr++) { - for (int c = 0; c < 3; c++) { - // image: same RGB for all temporal slots; video: distinct frame per slot - const int src_c = (nchan == 3) ? c : (fr * 3 + c); - for (int py = 0; py < ps; py++) { - for (int px = 0; px < ps; px++) { - const int iy = gy * ps + py; - const int ix = gx * ps + px; - const int d = ((fr * 3 + c) * ps + py) * ps + px; - patches[(size_t) tok * patch_dim + d] = - buf[(size_t) nchan * ((size_t) iy * nx + ix) + src_c]; - } - } - } - } - } - } - set_input_f32("onyx_patches", patches); + // pixel patchify runs inside the graph via build_inp() (ggml_conv_2d); + // pos-emb bilinear interp via resize_position_embeddings(). // --- sparse window grouping (pgrid x pgrid windows) --- const int win = pgrid; diff --git a/tools/mtmd/models/onyx.cpp b/tools/mtmd/models/onyx.cpp index 7daa825787..1949fb34b6 100644 --- a/tools/mtmd/models/onyx.cpp +++ b/tools/mtmd/models/onyx.cpp @@ -8,7 +8,6 @@ // // Several quantities are precomputed on host and fed as named graph inputs (filled in // clip.cpp set_input, PROJECTOR_TYPE_ONYX branch): -// onyx_patches [patch_dim, n_tok] : patchified pixels ([pt,c,ps,ps] layout) // onyx_pos_w/_h [n_tok] i32 : 1-indexed RoPE positions (sparse-permuted order) // onyx_sp_perm [n_tok] i32 : window grouping permutation (applied after ln_pre) // onyx_inv_perm [n_tok] i32 : inverse of sp_perm (applied after blocks) @@ -16,10 +15,8 @@ // onyx_sp_mask [n_tok, n_tok] f32 : block-diagonal window mask (sparse layers) ggml_cgraph * clip_graph_onyx::build() { const int ds = hparams.n_merge; // downsample factor (2) - const int pt = hparams.onyx_patch_temporal; // 2 const int sf = hparams.onyx_sparse_factor; // 4 const int n_tok = n_patches; - const int patch_dim = pt * 3 * patch_size * patch_size; // 1176 const int n_out = (n_patches_x / ds) * (n_patches_y / ds); const float rope_base = hparams.rope_theta; // 10000 const float attn_scale = 1.0f / sqrtf((float) d_head); // SDPA default @@ -30,9 +27,6 @@ ggml_cgraph * clip_graph_onyx::build() { return t; }; - ggml_tensor * patches = ggml_new_tensor_2d(ctx0, GGML_TYPE_F32, patch_dim, n_tok); - ggml_set_name(patches, "onyx_patches"); ggml_set_input(patches); - ggml_tensor * pos_w = inp_i32("onyx_pos_w", n_tok); ggml_tensor * pos_h = inp_i32("onyx_pos_h", n_tok); ggml_tensor * sp_perm = inp_i32("onyx_sp_perm", n_tok); @@ -42,8 +36,8 @@ ggml_cgraph * clip_graph_onyx::build() { ggml_tensor * sp_mask = ggml_new_tensor_2d(ctx0, GGML_TYPE_F32, n_tok, n_tok); ggml_set_name(sp_mask, "onyx_sp_mask"); ggml_set_input(sp_mask); - // patchify (conv1_linear as a matmul, no bias) + learned pos-emb - ggml_tensor * x = build_mm(model.patch_embeddings_0, patches); // [n_embd, n_tok] + // patchify via build_inp (conv2d over raw pixels) + bilinear-resized learned pos-emb + ggml_tensor * x = build_inp(); // [n_embd, n_tok, 1] x = ggml_add(ctx0, x, resize_position_embeddings(GGML_SCALE_MODE_BILINEAR)); cb(x, "after_posemb", -1); From 26ed5924e78b25472440ea152ec5d935204399f8 Mon Sep 17 00:00:00 2001 From: Pedro Cuenca Date: Thu, 6 Aug 2026 13:52:29 +0200 Subject: [PATCH 16/26] No param for rope_theta --- conversion/onyx.py | 4 ---- gguf-py/gguf/constants.py | 2 -- gguf-py/gguf/gguf_writer.py | 3 --- tools/mtmd/clip-impl.h | 2 -- tools/mtmd/clip.cpp | 1 - 5 files changed, 12 deletions(-) diff --git a/conversion/onyx.py b/conversion/onyx.py index c5327fa7f8..e3acbc305e 100644 --- a/conversion/onyx.py +++ b/conversion/onyx.py @@ -65,9 +65,6 @@ class OnyxModel(TextModel): @ModelBase.register("OnyxForConditionalGeneration") class OnyxVisionModel(MmprojModel): - # fallback for rope_parameters.rope_theta - ROPE_THETA = 10000.0 - def get_vision_config(self) -> dict[str, Any] | None: c = self.global_config.get("vision_config") if not c: @@ -83,7 +80,6 @@ class OnyxVisionModel(MmprojModel): self.gguf_writer.add_clip_projector_type(gguf.VisionProjectorType.ONYX) self.gguf_writer.add_vision_attention_layernorm_eps(float(c["layer_norm_eps"])) self.gguf_writer.add_vision_spatial_merge_size(int(c["merge_size"])) - self.gguf_writer.add_vision_rope_theta(float(c.get("rope_parameters", {}).get("rope_theta", self.ROPE_THETA))) @classmethod def filter_tensors(cls, item): diff --git a/gguf-py/gguf/constants.py b/gguf-py/gguf/constants.py index f675ba699f..4be161131d 100644 --- a/gguf-py/gguf/constants.py +++ b/gguf-py/gguf/constants.py @@ -341,7 +341,6 @@ class Keys: IMAGE_MEAN = "clip.vision.image_mean" IMAGE_STD = "clip.vision.image_std" SPATIAL_MERGE_SIZE = "clip.vision.spatial_merge_size" - ROPE_THETA = "clip.vision.rope_theta" USE_GELU = "clip.use_gelu" USE_SILU = "clip.use_silu" N_WA_PATTERN = "clip.vision.n_wa_pattern" # used by qwen2.5vl @@ -1531,7 +1530,6 @@ TENSOR_NAMES: dict[MODEL_TENSOR, str] = { MODEL_TENSOR.V_MULTI_PROJ_NORM: "v.proj_blk.{bid}.norm", MODEL_TENSOR.V_MULTI_PROJ_LINEAR: "v.proj_blk.{bid}.linear", MODEL_TENSOR.V_MULTI_PROJ_POST_NORM: "v.proj_blk.{bid}.post_norm", - # audio (mtmd) # note: all audio tensor names must use prefix "a." or "mm.a." MODEL_TENSOR.A_ENC_EMBD_POS: "a.position_embd", diff --git a/gguf-py/gguf/gguf_writer.py b/gguf-py/gguf/gguf_writer.py index 73377f9491..af14e24fe4 100644 --- a/gguf-py/gguf/gguf_writer.py +++ b/gguf-py/gguf/gguf_writer.py @@ -1271,9 +1271,6 @@ class GGUFWriter: def add_vision_spatial_merge_size(self, value: int) -> None: self.add_uint32(Keys.ClipVision.SPATIAL_MERGE_SIZE, value) - def add_vision_rope_theta(self, value: float) -> None: - self.add_float32(Keys.ClipVision.ROPE_THETA, value) - def add_vision_use_gelu(self, value: bool) -> None: self.add_bool(Keys.ClipVision.USE_GELU, value) diff --git a/tools/mtmd/clip-impl.h b/tools/mtmd/clip-impl.h index 67012ff491..c159e249e2 100644 --- a/tools/mtmd/clip-impl.h +++ b/tools/mtmd/clip-impl.h @@ -61,7 +61,6 @@ #define KEY_PROJ_SAMPLE_WINDOW_SIDE "clip.vision.projector.window_side" #define KEY_PROJ_SPATIAL_OFFSETS "clip.vision.projector.spatial_offsets" #define KEY_SPATIAL_MERGE_SIZE "clip.vision.spatial_merge_size" -#define KEY_V_ROPE_THETA "clip.vision.rope_theta" #define KEY_MM_PATCH_MERGE_TYPE "clip.vision.mm_patch_merge_type" #define KEY_IMAGE_GRID_PINPOINTS "clip.vision.image_grid_pinpoints" @@ -92,7 +91,6 @@ #define KEY_A_LOCAL_GROUP_SIZE "clip.audio.local_group_size" // mimo-v2.5: input_local_transformer grouping size #define KEY_AUDIO_SUBSAMPLING_FACTOR "clip.audio.subsampling_factor" - // // tensor name constants // diff --git a/tools/mtmd/clip.cpp b/tools/mtmd/clip.cpp index 77f5b75c0a..a204e7fb25 100644 --- a/tools/mtmd/clip.cpp +++ b/tools/mtmd/clip.cpp @@ -1528,7 +1528,6 @@ struct clip_model_loader { hparams.onyx_patch_temporal = 2; // Onyx-arch invariant hparams.onyx_sparse_factor = 4; // 3 sparse layers + 1 global, repeating get_u32(KEY_SPATIAL_MERGE_SIZE, hparams.n_merge, false); - get_f32(KEY_V_ROPE_THETA, hparams.rope_theta, false); hparams.set_limit_image_tokens(1, 4096); hparams.set_warmup_n_tokens(32*32); } break; From ebac3dd7ea482f7fa1cbe12517c33e397760ebee Mon Sep 17 00:00:00 2001 From: Pedro Cuenca Date: Thu, 6 Aug 2026 13:57:31 +0200 Subject: [PATCH 17/26] Small cleanup --- tools/mtmd/models/onyx.cpp | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/tools/mtmd/models/onyx.cpp b/tools/mtmd/models/onyx.cpp index 1949fb34b6..12e63cff25 100644 --- a/tools/mtmd/models/onyx.cpp +++ b/tools/mtmd/models/onyx.cpp @@ -2,9 +2,7 @@ // Onyx vision encoder: 50-layer ViT with 2D RoPE, sparse block-diagonal // window attention (every 4th + last layer global), pixel-shuffle downsample, then -// adapter MLP + LLM's vision_projection. Output dim = 6656 (onyx n_embd), -// injected via llama_batch.embd; the onyx LLM graph applies the scaleless rms_norm -// (== reference perception_emb_norm). +// adapter MLP + LLM's vision_projection. // // Several quantities are precomputed on host and fed as named graph inputs (filled in // clip.cpp set_input, PROJECTOR_TYPE_ONYX branch): @@ -23,7 +21,8 @@ ggml_cgraph * clip_graph_onyx::build() { auto inp_i32 = [&](const char * name, int64_t n) { ggml_tensor * t = ggml_new_tensor_1d(ctx0, GGML_TYPE_I32, n); - ggml_set_name(t, name); ggml_set_input(t); + ggml_set_name(t, name); + ggml_set_input(t); return t; }; @@ -34,7 +33,8 @@ ggml_cgraph * clip_graph_onyx::build() { ggml_tensor * ds_perm = inp_i32("onyx_ds_perm", n_tok); ggml_tensor * sp_mask = ggml_new_tensor_2d(ctx0, GGML_TYPE_F32, n_tok, n_tok); - ggml_set_name(sp_mask, "onyx_sp_mask"); ggml_set_input(sp_mask); + ggml_set_name(sp_mask, "onyx_sp_mask"); + ggml_set_input(sp_mask); // patchify via build_inp (conv2d over raw pixels) + bilinear-resized learned pos-emb ggml_tensor * x = build_inp(); // [n_embd, n_tok, 1] From c784520acbdaf9204a46f1c567fff3d8edb53bb3 Mon Sep 17 00:00:00 2001 From: Pedro Cuenca Date: Thu, 6 Aug 2026 14:02:21 +0200 Subject: [PATCH 18/26] Restore blank line --- gguf-py/gguf/constants.py | 1 + 1 file changed, 1 insertion(+) diff --git a/gguf-py/gguf/constants.py b/gguf-py/gguf/constants.py index 4be161131d..3bea1145ac 100644 --- a/gguf-py/gguf/constants.py +++ b/gguf-py/gguf/constants.py @@ -1530,6 +1530,7 @@ TENSOR_NAMES: dict[MODEL_TENSOR, str] = { MODEL_TENSOR.V_MULTI_PROJ_NORM: "v.proj_blk.{bid}.norm", MODEL_TENSOR.V_MULTI_PROJ_LINEAR: "v.proj_blk.{bid}.linear", MODEL_TENSOR.V_MULTI_PROJ_POST_NORM: "v.proj_blk.{bid}.post_norm", + # audio (mtmd) # note: all audio tensor names must use prefix "a." or "mm.a." MODEL_TENSOR.A_ENC_EMBD_POS: "a.position_embd", From ead38d271f80891615a0d8356eba9e68c85a3614 Mon Sep 17 00:00:00 2001 From: Pedro Cuenca Date: Thu, 6 Aug 2026 20:40:05 +0200 Subject: [PATCH 19/26] Remove duplicated function --- conversion/onyx.py | 13 +------------ 1 file changed, 1 insertion(+), 12 deletions(-) diff --git a/conversion/onyx.py b/conversion/onyx.py index 01aae88b16..94a9d3d10b 100644 --- a/conversion/onyx.py +++ b/conversion/onyx.py @@ -107,17 +107,6 @@ class OnyxVisionModel(MmprojModel): return None return super().filter_tensors((name, gen)) - @staticmethod - def _unpermute_for_rope(tensor: "Tensor", n_heads: int) -> "Tensor": - """clip.cpp uses the interleaved convention, so we invert the permutation here.""" - if tensor.ndim == 2: - dim1, dim2 = tensor.shape - return tensor.view(n_heads, 2, dim1 // n_heads // 2, dim2).transpose(1, 2).reshape(dim1, dim2) - if tensor.ndim == 1: - (dim1,) = tensor.shape - return tensor.view(n_heads, 2, dim1 // n_heads // 2).transpose(1, 2).reshape(dim1) - raise ValueError(f"_unpermute_for_rope: unexpected shape {tuple(tensor.shape)}") - # 3-layer projector MLP _MM_MLP_MAP = { "model.vision_adapter.fc1": (gguf.MODEL_TENSOR.V_MMPROJ, 0), @@ -128,7 +117,7 @@ class OnyxVisionModel(MmprojModel): def modify_tensors(self, data_torch, name, bid): if ".attn.q_proj." in name or ".attn.k_proj." in name: n_heads = int(self.hparams_vision["num_attention_heads"]) - data_torch = self._unpermute_for_rope(data_torch, n_heads) + data_torch = _unpermute_for_rope(data_torch, n_heads) # Lay out the pt=2 temporal slabs of the patch embedding as a conv2d for build_inp() if name.endswith("patch_embedder.patch_embedding.weight"): n_embd = data_torch.shape[0] From 1f2a51f268cad3b56dfacef9f20fcfdc22164268 Mon Sep 17 00:00:00 2001 From: Pedro Cuenca Date: Thu, 6 Aug 2026 23:49:35 +0200 Subject: [PATCH 20/26] build_vit --- tools/mtmd/models/onyx.cpp | 71 +++++++++++++------------------------- 1 file changed, 24 insertions(+), 47 deletions(-) diff --git a/tools/mtmd/models/onyx.cpp b/tools/mtmd/models/onyx.cpp index 12e63cff25..20308c8353 100644 --- a/tools/mtmd/models/onyx.cpp +++ b/tools/mtmd/models/onyx.cpp @@ -16,8 +16,7 @@ ggml_cgraph * clip_graph_onyx::build() { const int sf = hparams.onyx_sparse_factor; // 4 const int n_tok = n_patches; const int n_out = (n_patches_x / ds) * (n_patches_y / ds); - const float rope_base = hparams.rope_theta; // 10000 - const float attn_scale = 1.0f / sqrtf((float) d_head); // SDPA default + const float rope_base = hparams.rope_theta; // 10000 auto inp_i32 = [&](const char * name, int64_t n) { ggml_tensor * t = ggml_new_tensor_1d(ctx0, GGML_TYPE_I32, n); @@ -26,11 +25,11 @@ ggml_cgraph * clip_graph_onyx::build() { return t; }; - ggml_tensor * pos_w = inp_i32("onyx_pos_w", n_tok); - ggml_tensor * pos_h = inp_i32("onyx_pos_h", n_tok); - ggml_tensor * sp_perm = inp_i32("onyx_sp_perm", n_tok); + ggml_tensor * pos_w = inp_i32("onyx_pos_w", n_tok); + ggml_tensor * pos_h = inp_i32("onyx_pos_h", n_tok); + ggml_tensor * sp_perm = inp_i32("onyx_sp_perm", n_tok); ggml_tensor * inv_perm = inp_i32("onyx_inv_perm", n_tok); - ggml_tensor * ds_perm = inp_i32("onyx_ds_perm", n_tok); + ggml_tensor * ds_perm = inp_i32("onyx_ds_perm", n_tok); ggml_tensor * sp_mask = ggml_new_tensor_2d(ctx0, GGML_TYPE_F32, n_tok, n_tok); ggml_set_name(sp_mask, "onyx_sp_mask"); @@ -41,53 +40,31 @@ ggml_cgraph * clip_graph_onyx::build() { x = ggml_add(ctx0, x, resize_position_embeddings(GGML_SCALE_MODE_BILINEAR)); cb(x, "after_posemb", -1); - // ln_pre (LayerNorm) - x = build_norm(x, model.pre_ln_w, model.pre_ln_b, NORM_TYPE_NORMAL, eps, -1); - - // group patches into 32x32 windows (sparse attention order) + // group patches into pgrid x pgrid windows (sparse attention order) x = ggml_get_rows(ctx0, x, sp_perm); - cb(x, "after_ln_pre", -1); + cb(x, "after_sp_perm", -1); - for (int il = 0; il < n_layer; il++) { - const auto & layer = model.layers[il]; + // per-layer mask: sparse layers get sp_mask, global layers (every sf-th and last) get none + std::vector attn_mask_layers(n_layer); + for (int il = 0; il < n_layer; ++il) { const bool is_global = (il == n_layer - 1) || ((il + 1) % sf == 0); - - ggml_tensor * inpL = x; - - ggml_tensor * cur = build_norm(x, layer.ln_1_w, layer.ln_1_b, NORM_TYPE_NORMAL, eps, il); - - ggml_tensor * Q = ggml_add(ctx0, build_mm(layer.q_w, cur), layer.q_b); - ggml_tensor * K = ggml_add(ctx0, build_mm(layer.k_w, cur), layer.k_b); - ggml_tensor * V = ggml_add(ctx0, build_mm(layer.v_w, cur), layer.v_b); - - Q = ggml_reshape_3d(ctx0, Q, d_head, n_head, n_tok); - K = ggml_reshape_3d(ctx0, K, d_head, n_head, n_tok); - V = ggml_reshape_3d(ctx0, V, d_head, n_head, n_tok); - - // 2D RoPE: first half of head_dim uses width pos, second half uses height pos - Q = build_rope_2d(ctx0, Q, pos_w, pos_h, rope_base, false); - K = build_rope_2d(ctx0, K, pos_w, pos_h, rope_base, false); - - ggml_tensor * mask = is_global ? nullptr : sp_mask; - cur = build_attn(layer.o_w, layer.o_b, Q, K, V, mask, attn_scale, il); - - x = ggml_add(ctx0, inpL, cur); // residual 1 - inpL = x; - - cur = build_norm(x, layer.ln_2_w, layer.ln_2_b, NORM_TYPE_NORMAL, eps, il); - cur = build_ffn(cur, - layer.ff_up_w, layer.ff_up_b, - nullptr, nullptr, - layer.ff_down_w, layer.ff_down_b, - FFN_GELU_ERF, il); // reference uses exact (erf) GELU - x = ggml_add(ctx0, inpL, cur); // residual 2 - cb(x, "layer_out", il); + attn_mask_layers[il] = is_global ? nullptr : sp_mask; } - // un-permute back to original grid order, then ln_post + // 2D RoPE: first half of head_dim uses width pos, second half uses height pos + auto add_pos = [&](ggml_tensor * cur, const clip_layer &) { + return build_rope_2d(ctx0, cur, pos_w, pos_h, rope_base, false); + }; + + build_vit_opts opts; + opts.attn_mask_layers = std::move(attn_mask_layers); + + // pre_ln, per-layer transformer, post_ln (all inside build_vit); reference uses exact (erf) GELU + x = build_vit(x, n_tok, NORM_TYPE_NORMAL, FFN_GELU_ERF, nullptr, add_pos, opts); + + // un-permute back to original grid order x = ggml_get_rows(ctx0, x, inv_perm); - x = build_norm(x, model.post_ln_w, model.post_ln_b, NORM_TYPE_NORMAL, eps, -1); - cb(x, "after_ln_post", -1); + cb(x, "after_inv_perm", -1); // pixel-shuffle downsample: gather f*f spatial neighbors then concat channel-outer. // out[c*(ds*ds)+s, o] = x[ds_perm gathered][o*(ds*ds)+s, c] From 9e8f4dabdeb57b26ae4d0797fe4b5e2a012a6daa Mon Sep 17 00:00:00 2001 From: Pedro Cuenca Date: Sat, 8 Aug 2026 21:41:21 +0200 Subject: [PATCH 21/26] Apply suggestion from @pcuenca --- tools/mtmd/mtmd.cpp | 2 -- 1 file changed, 2 deletions(-) diff --git a/tools/mtmd/mtmd.cpp b/tools/mtmd/mtmd.cpp index ea8f744e5f..bd81780538 100644 --- a/tools/mtmd/mtmd.cpp +++ b/tools/mtmd/mtmd.cpp @@ -472,8 +472,6 @@ struct mtmd_context { } break; case PROJECTOR_TYPE_ONYX: { - // NOTE: transformers no longer uses delimiters (just <|patch|>*N), - // but we get subpar generations if we don't. img_beg = "<|image_start|>"; img_end = "<|image_end|>"; image_preproc = std::make_unique(ctx_v); From 2ba2385aa21dfce9087e7338e98446df88975d8f Mon Sep 17 00:00:00 2001 From: Pedro Cuenca Date: Sat, 8 Aug 2026 22:23:21 +0200 Subject: [PATCH 22/26] Set model type --- src/models/onyx.cpp | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/models/onyx.cpp b/src/models/onyx.cpp index b2897b6935..636f71ca8c 100644 --- a/src/models/onyx.cpp +++ b/src/models/onyx.cpp @@ -23,7 +23,7 @@ void llama_model_onyx::load_arch_hparams(llama_model_loader & ml) { hparams.swa_type = LLAMA_SWA_TYPE_NONE; } - type = LLM_TYPE_UNKNOWN; + type = LLM_TYPE_30B; } void llama_model_onyx::load_arch_tensors(llama_model_loader &) { From 84b932497f3a0baec3d896110c2c8adbb7505331 Mon Sep 17 00:00:00 2001 From: Pedro Cuenca Date: Sat, 8 Aug 2026 22:27:36 +0200 Subject: [PATCH 23/26] Remove comment that will become obsolete --- src/llama-model.cpp | 1 - 1 file changed, 1 deletion(-) diff --git a/src/llama-model.cpp b/src/llama-model.cpp index 1b9bc6f9fa..2180d36b2a 100644 --- a/src/llama-model.cpp +++ b/src/llama-model.cpp @@ -2623,7 +2623,6 @@ llama_rope_type llama_model_rope_type(const llama_model * model) { case LLM_ARCH_STEP35: case LLM_ARCH_TALKIE: case LLM_ARCH_MELLUM: - // fallback, dflash rope is inherited from the linked target case LLM_ARCH_DFLASH: return LLAMA_ROPE_TYPE_NEOX; From d0eca60285a4fe3708d387cd50f4f2264f44c73a Mon Sep 17 00:00:00 2001 From: Pedro Cuenca Date: Sat, 8 Aug 2026 23:07:53 +0200 Subject: [PATCH 24/26] Hardcode post_norm_rms_eps instead of new param --- conversion/onyx.py | 1 - gguf-py/gguf/constants.py | 1 - gguf-py/gguf/gguf_writer.py | 3 --- src/llama-arch.cpp | 1 - src/llama-arch.h | 1 - src/llama-hparams.h | 1 - src/llama-model.cpp | 1 - src/models/onyx.cpp | 10 +++++----- 8 files changed, 5 insertions(+), 14 deletions(-) diff --git a/conversion/onyx.py b/conversion/onyx.py index 4b0e14f4ed..c31e61c8ca 100644 --- a/conversion/onyx.py +++ b/conversion/onyx.py @@ -45,7 +45,6 @@ class OnyxModel(TextModel): self.gguf_writer.add_final_logit_softcapping(hparams["final_logit_softcapping"]) self.gguf_writer.add_logit_scale(hparams["output_multiplier"]) - self.gguf_writer.add_post_norm_rms_eps(hparams["post_norm_eps"]) # SWA + NoPE: [SW, SW, SW, Full], NoPE used on Full layers. References: # https://huggingface.co/someorgtoo-hf/onyx-hf-converted/blob/main/config.json#L19 diff --git a/gguf-py/gguf/constants.py b/gguf-py/gguf/constants.py index 7a2ba560a8..7dbf63f831 100644 --- a/gguf-py/gguf/constants.py +++ b/gguf-py/gguf/constants.py @@ -172,7 +172,6 @@ class Keys: VALUE_LENGTH = "{arch}.attention.value_length" LAYERNORM_EPS = "{arch}.attention.layer_norm_epsilon" LAYERNORM_RMS_EPS = "{arch}.attention.layer_norm_rms_epsilon" - POST_NORM_RMS_EPS = "{arch}.attention.post_norm_rms_epsilon" GROUPNORM_EPS = "{arch}.attention.group_norm_epsilon" GROUPNORM_GROUPS = "{arch}.attention.group_norm_groups" CAUSAL = "{arch}.attention.causal" diff --git a/gguf-py/gguf/gguf_writer.py b/gguf-py/gguf/gguf_writer.py index af14e24fe4..c5905164c3 100644 --- a/gguf-py/gguf/gguf_writer.py +++ b/gguf-py/gguf/gguf_writer.py @@ -923,9 +923,6 @@ class GGUFWriter: def add_layer_norm_rms_eps(self, value: float) -> None: self.add_float32(Keys.Attention.LAYERNORM_RMS_EPS.format(arch=self.arch), value) - def add_post_norm_rms_eps(self, value: float) -> None: - self.add_float32(Keys.Attention.POST_NORM_RMS_EPS.format(arch=self.arch), value) - def add_group_norm_eps(self, value: float) -> None: self.add_float32(Keys.Attention.GROUPNORM_EPS.format(arch=self.arch), value) diff --git a/src/llama-arch.cpp b/src/llama-arch.cpp index 661409a7ca..2ce0a7752d 100644 --- a/src/llama-arch.cpp +++ b/src/llama-arch.cpp @@ -234,7 +234,6 @@ static const std::map LLM_KV_NAMES = { { LLM_KV_ATTENTION_VALUE_LENGTH, "%s.attention.value_length" }, { LLM_KV_ATTENTION_LAYERNORM_EPS, "%s.attention.layer_norm_epsilon" }, { LLM_KV_ATTENTION_LAYERNORM_RMS_EPS, "%s.attention.layer_norm_rms_epsilon" }, - { LLM_KV_ATTENTION_POST_NORM_RMS_EPS, "%s.attention.post_norm_rms_epsilon" }, { LLM_KV_ATTENTION_GROUPNORM_EPS, "%s.attention.group_norm_epsilon" }, { LLM_KV_ATTENTION_GROUPNORM_GROUPS, "%s.attention.group_norm_groups" }, { LLM_KV_ATTENTION_CAUSAL, "%s.attention.causal" }, diff --git a/src/llama-arch.h b/src/llama-arch.h index 7e97292b25..24d5d97c48 100644 --- a/src/llama-arch.h +++ b/src/llama-arch.h @@ -239,7 +239,6 @@ enum llm_kv { LLM_KV_ATTENTION_VALUE_LENGTH, LLM_KV_ATTENTION_LAYERNORM_EPS, LLM_KV_ATTENTION_LAYERNORM_RMS_EPS, - LLM_KV_ATTENTION_POST_NORM_RMS_EPS, LLM_KV_ATTENTION_GROUPNORM_EPS, LLM_KV_ATTENTION_GROUPNORM_GROUPS, LLM_KV_ATTENTION_CAUSAL, diff --git a/src/llama-hparams.h b/src/llama-hparams.h index ce430d4364..fc770bf003 100644 --- a/src/llama-hparams.h +++ b/src/llama-hparams.h @@ -106,7 +106,6 @@ struct llama_hparams { float f_norm_eps; float f_norm_rms_eps; float f_norm_group_eps; - float f_post_norm_rms_eps; float f_attn_logit_softcapping = 50.0f; float f_router_logit_softcapping = 30.0f; diff --git a/src/llama-model.cpp b/src/llama-model.cpp index 2180d36b2a..578e423953 100644 --- a/src/llama-model.cpp +++ b/src/llama-model.cpp @@ -1793,7 +1793,6 @@ void llama_model::print_info() const { LLAMA_LOG_INFO("%s: n_embd_v_gqa = %s\n", __func__, print_f([&](uint32_t il) { return hparams.n_embd_v_gqa(il); }, hparams.n_layer_all).c_str()); LLAMA_LOG_INFO("%s: f_norm_eps = %.1e\n", __func__, hparams.f_norm_eps); LLAMA_LOG_INFO("%s: f_norm_rms_eps = %.1e\n", __func__, hparams.f_norm_rms_eps); - LLAMA_LOG_INFO("%s: f_post_norm_rms_eps = %.1e\n", __func__, hparams.f_post_norm_rms_eps); LLAMA_LOG_INFO("%s: f_clamp_kqv = %.1e\n", __func__, hparams.f_clamp_kqv); LLAMA_LOG_INFO("%s: f_max_alibi_bias = %.1e\n", __func__, hparams.f_max_alibi_bias); LLAMA_LOG_INFO("%s: f_logit_scale = %.1e\n", __func__, hparams.f_logit_scale); diff --git a/src/models/onyx.cpp b/src/models/onyx.cpp index 636f71ca8c..2d950e2d82 100644 --- a/src/models/onyx.cpp +++ b/src/models/onyx.cpp @@ -5,7 +5,6 @@ void llama_model_onyx::load_arch_hparams(llama_model_loader & ml) { ml.get_key(LLM_KV_ATTENTION_SLIDING_WINDOW, hparams.n_swa, false); ml.get_key(LLM_KV_FINAL_LOGIT_SOFTCAPPING, hparams.f_final_logit_softcapping, false); ml.get_key(LLM_KV_LOGIT_SCALE, hparams.f_logit_scale); - ml.get_key(LLM_KV_ATTENTION_POST_NORM_RMS_EPS, hparams.f_post_norm_rms_eps); // SWA layers share the model rope theta; they are also the only layers that use rope // here (global layers are NoPE), so the 10000.0 default would apply to all of them. @@ -67,6 +66,9 @@ llama_model_onyx::graph::graph(const llama_model & model, const llm_graph_params const int64_t n_embd_head = hparams.n_embd_head_v(); GGML_ASSERT(n_embd_head == hparams.n_embd_head_k()); + // Different to f_norm_rms_eps for post-attn / post-FFN norms + const float post_norm_eps = 1e-8f; + ggml_tensor * cur; ggml_tensor * inpL; @@ -143,8 +145,7 @@ llama_model_onyx::graph::graph(const llama_model & model, const llm_graph_params cb(cur, "attn_o_proj", il); } - // post-attention norm (uses f_post_norm_rms_eps, not the general f_norm_rms_eps). - cur = ggml_rms_norm(ctx0, cur, hparams.f_post_norm_rms_eps); + cur = ggml_rms_norm(ctx0, cur, post_norm_eps); cur = ggml_mul(ctx0, cur, model.layers[il].attn_post_norm); cb(cur, "attn_post_norm", il); @@ -169,8 +170,7 @@ llama_model_onyx::graph::graph(const llama_model & model, const llm_graph_params LLM_FFN_SILU, LLM_FFN_PAR, il); cb(cur, "ffn_out", il); - // post-FFN norm (uses f_post_norm_rms_eps, not the general f_norm_rms_eps). - cur = ggml_rms_norm(ctx0, cur, hparams.f_post_norm_rms_eps); + cur = ggml_rms_norm(ctx0, cur, post_norm_eps); cur = ggml_mul(ctx0, cur, model.layers[il].ffn_post_norm); cb(cur, "ffn_post_norm", il); From 5b3f24594b7e5ab41f5f660658fd1f96f991e258 Mon Sep 17 00:00:00 2001 From: Pedro Cuenca Date: Sun, 9 Aug 2026 00:18:00 +0200 Subject: [PATCH 25/26] Derive SWA+RoPE pattern from gguf array or scalar --- conversion/onyx.py | 6 +----- src/models/onyx.cpp | 17 ++++++----------- 2 files changed, 7 insertions(+), 16 deletions(-) diff --git a/conversion/onyx.py b/conversion/onyx.py index c31e61c8ca..bf7577146c 100644 --- a/conversion/onyx.py +++ b/conversion/onyx.py @@ -45,12 +45,8 @@ class OnyxModel(TextModel): self.gguf_writer.add_final_logit_softcapping(hparams["final_logit_softcapping"]) self.gguf_writer.add_logit_scale(hparams["output_multiplier"]) - - # SWA + NoPE: [SW, SW, SW, Full], NoPE used on Full layers. References: - # https://huggingface.co/someorgtoo-hf/onyx-hf-converted/blob/main/config.json#L19 - # https://huggingface.co/someorgtoo-hf/onyx-hf-converted/blob/main/config.json#L73 self.gguf_writer.add_sliding_window(hparams["sliding_window"]) - self.gguf_writer.add_sliding_window_pattern(4) + self.gguf_writer.add_sliding_window_pattern([t == "sliding_attention" for t in hparams["layer_types"]]) def modify_tensors(self, data_torch: Tensor, name: str, bid: int | None) -> Iterable[tuple[str, Tensor]]: shift = self.norm_shift(name) diff --git a/src/models/onyx.cpp b/src/models/onyx.cpp index 2d950e2d82..f23dd8a1fd 100644 --- a/src/models/onyx.cpp +++ b/src/models/onyx.cpp @@ -6,20 +6,15 @@ void llama_model_onyx::load_arch_hparams(llama_model_loader & ml) { ml.get_key(LLM_KV_FINAL_LOGIT_SOFTCAPPING, hparams.f_final_logit_softcapping, false); ml.get_key(LLM_KV_LOGIT_SCALE, hparams.f_logit_scale); - // SWA layers share the model rope theta; they are also the only layers that use rope - // here (global layers are NoPE), so the 10000.0 default would apply to all of them. hparams.rope_freq_base_train_swa = hparams.rope_freq_base_train; ml.get_key(LLM_KV_ROPE_FREQ_BASE_SWA, hparams.rope_freq_base_train_swa, false); - // SWA + NoPE: [SW, SW, SW, Full], NoPE used on Full layers. - if (hparams.n_swa > 0) { - hparams.swa_type = LLAMA_SWA_TYPE_STANDARD; - uint32_t swa_period = 4; - ml.get_key_or_arr(LLM_KV_ATTENTION_SLIDING_WINDOW_PATTERN, swa_period, false); + hparams.swa_type = LLAMA_SWA_TYPE_STANDARD; + uint32_t swa_period = 4; + if (ml.get_key_or_arr(LLM_KV_ATTENTION_SLIDING_WINDOW_PATTERN, swa_period, false)) { hparams.set_swa_pattern(swa_period); - hparams.n_no_rope_layer_step = swa_period; } else { - hparams.swa_type = LLAMA_SWA_TYPE_NONE; + ml.get_key_or_arr(LLM_KV_ATTENTION_SLIDING_WINDOW_PATTERN, hparams.is_swa_impl, hparams.n_layer()); } type = LLM_TYPE_30B; @@ -91,8 +86,8 @@ llama_model_onyx::graph::graph(const llama_model & model, const llm_graph_params ggml_tensor * inpSA = inpL; - const bool use_rope = hparams.n_no_rope_layer_step > 0 && - (il + 1) % hparams.n_no_rope_layer_step != 0; + // RoPE runs on the SWA layers, NoPE on full ones. + const bool use_rope = hparams.is_swa(il); // pre-attention norm (weight+1 folded at conversion time) cur = build_norm(inpL, model.layers[il].attn_norm, NULL, LLM_NORM_RMS, il); From 31c515f427d249733459fcfa18af23a66061d75d Mon Sep 17 00:00:00 2001 From: Pedro Cuenca Date: Sun, 9 Aug 2026 00:56:53 +0200 Subject: [PATCH 26/26] Fix model type <-> number of layers --- src/models/onyx.cpp | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/src/models/onyx.cpp b/src/models/onyx.cpp index f23dd8a1fd..8aef05197f 100644 --- a/src/models/onyx.cpp +++ b/src/models/onyx.cpp @@ -17,7 +17,10 @@ void llama_model_onyx::load_arch_hparams(llama_model_loader & ml) { ml.get_key_or_arr(LLM_KV_ATTENTION_SLIDING_WINDOW_PATTERN, hparams.is_swa_impl, hparams.n_layer()); } - type = LLM_TYPE_30B; + switch (hparams.n_layer()) { + case 52: type = LLM_TYPE_30B; break; + default: type = LLM_TYPE_UNKNOWN; + } } void llama_model_onyx::load_arch_tensors(llama_model_loader &) {