mmproj conversion

Note: some fields to be renamed after the implementation works. We are
keeping compatibility with the reference Meta gguf for testing purposes.
This commit is contained in:
Pedro Cuenca
2026-08-04 16:20:24 +02:00
parent 7c5abdae0f
commit 10bad7ad18
4 changed files with 99 additions and 7 deletions
+1
View File
@@ -295,6 +295,7 @@ MMPROJ_MODEL_MAP: dict[str, str] = {
"MiniCPMV4_6ForConditionalGeneration": "minicpm",
"Mistral3ForConditionalGeneration": "llava",
"NemotronH_Nano_VL_V2": "nemotron",
"OnyxForConditionalGeneration": "onyx",
"PaddleOCRVisionModel": "ernie",
"Phi4ForCausalLMV": "phi",
"Qwen2AudioForConditionalGeneration": "ultravox",
+60 -2
View File
@@ -1,13 +1,13 @@
from __future__ import annotations
from typing import Iterable, TYPE_CHECKING
from typing import Any, Iterable, TYPE_CHECKING
import torch
if TYPE_CHECKING:
from torch import Tensor
from .base import ModelBase, TextModel, gguf
from .base import MmprojModel, ModelBase, TextModel, gguf
@ModelBase.register("OnyxForConditionalGeneration")
@@ -61,3 +61,61 @@ class OnyxModel(TextModel):
)
yield from super().modify_tensors(data_torch, name, bid)
@ModelBase.register("OnyxForConditionalGeneration")
class OnyxVisionModel(MmprojModel):
# fallback for rope_parameters.rope_theta
ROPE_THETA = 10000.0
def get_vision_config(self) -> dict[str, Any] | None:
c = self.global_config.get("vision_config")
if not c:
return None
# Onyx actually uses dynamic size, initialize with nominal size
image_size = c["pos_emb_height"] * c["patch_size"] * c["merge_size"]
# Derive sparse_attention_factor from layer_types
fulls = [i for i, t in enumerate(c["layer_types"]) if t == "full_attention"]
if not fulls:
raise ValueError("vision_config.layer_types has no full_attention layer")
sparse_factor = fulls[0] + 1
return {**c, "image_size": image_size, "sparse_attention_factor": sparse_factor}
def set_gguf_parameters(self):
super().set_gguf_parameters()
c = self.hparams_vision # enriched vision_config from get_vision_config()
self.gguf_writer.add_clip_projector_type(gguf.VisionProjectorType.ONYX)
self.gguf_writer.add_vision_attention_layernorm_eps(float(c["layer_norm_eps"]))
rope_theta = float(c.get("rope_parameters", {}).get("rope_theta", self.ROPE_THETA))
self.gguf_writer.add_uint32 ("clip.vision.onyx.patch_temporal", int(c["patch_temporal"]))
self.gguf_writer.add_uint32 ("clip.vision.onyx.downsample_factor", int(c["merge_size"]))
self.gguf_writer.add_uint32 ("clip.vision.onyx.sparse_attention_factor", int(c["sparse_attention_factor"]))
self.gguf_writer.add_uint32 ("clip.vision.onyx.pos_emb_grid", int(c["pos_emb_height"]))
self.gguf_writer.add_float32("clip.vision.onyx.rope_theta", rope_theta)
@classmethod
def filter_tensors(cls, item):
name, gen = item
keep = ("model.vision_tower.", "model.vision_adapter.", "model.vision_projection.")
if not any(name.startswith(k) for k in keep):
return None
return super().filter_tensors((name, gen))
@staticmethod
def _unpermute_for_rope(tensor: "Tensor", n_heads: int) -> "Tensor":
"""clip.cpp uses the interleaved convention, so we invert the permutation here."""
if tensor.ndim == 2:
dim1, dim2 = tensor.shape
return tensor.view(n_heads, 2, dim1 // n_heads // 2, dim2).transpose(1, 2).reshape(dim1, dim2)
if tensor.ndim == 1:
(dim1,) = tensor.shape
return tensor.view(n_heads, 2, dim1 // n_heads // 2).transpose(1, 2).reshape(dim1)
raise ValueError(f"_unpermute_for_rope: unexpected shape {tuple(tensor.shape)}")
def modify_tensors(self, data_torch, name, bid):
if ".attn.q_proj." in name or ".attn.k_proj." in name:
n_heads = int(self.hparams_vision["num_attention_heads"])
data_torch = self._unpermute_for_rope(data_torch, n_heads)
yield (self.map_tensor_name(name), data_torch)
+12 -2
View File
@@ -873,6 +873,9 @@ class MODEL_TENSOR(IntEnum):
V_MM_GATE = auto() # cogvlm
V_MM_MERGER_FC1 = auto() # minimax-m3 (patch-merge MLP)
V_MM_MERGER_FC2 = auto() # minimax-m3 (patch-merge MLP)
V_MM_ADAPTER_FC = auto() # onyx
V_MM_ADAPTER_PROJ = auto() # onyx
V_MM_VISION_PROJ = auto() # onyx
V_TOK_BOI = auto() # cogvlm
V_TOK_EOI = auto() # cogvlm
V_TOK_IMG_BEGIN = auto() # hunyuanvl
@@ -1481,8 +1484,11 @@ TENSOR_NAMES: dict[MODEL_TENSOR, str] = {
MODEL_TENSOR.V_MM_UP: "mm.up",
MODEL_TENSOR.V_MM_DOWN: "mm.down",
MODEL_TENSOR.V_MM_GATE: "mm.gate",
MODEL_TENSOR.V_MM_MERGER_FC1: "mm.merger.fc1",
MODEL_TENSOR.V_MM_MERGER_FC2: "mm.merger.fc2",
MODEL_TENSOR.V_MM_MERGER_FC1: "mm.merger.fc1",
MODEL_TENSOR.V_MM_MERGER_FC2: "mm.merger.fc2",
MODEL_TENSOR.V_MM_ADAPTER_FC: "mm.adapter_fc", # onyx
MODEL_TENSOR.V_MM_ADAPTER_PROJ: "mm.adapter_proj", # onyx
MODEL_TENSOR.V_MM_VISION_PROJ: "mm.vision_proj", # onyx
MODEL_TENSOR.V_TOK_BOI: "v.boi",
MODEL_TENSOR.V_TOK_EOI: "v.eoi",
MODEL_TENSOR.V_MM_PRE_NORM: "mm.pre_norm",
@@ -1717,6 +1723,9 @@ MODEL_TENSORS: dict[MODEL_ARCH, list[MODEL_TENSOR]] = {
MODEL_TENSOR.V_MM_UP,
MODEL_TENSOR.V_MM_DOWN,
MODEL_TENSOR.V_MM_GATE,
MODEL_TENSOR.V_MM_ADAPTER_FC,
MODEL_TENSOR.V_MM_ADAPTER_PROJ,
MODEL_TENSOR.V_MM_VISION_PROJ,
MODEL_TENSOR.V_TOK_BOI,
MODEL_TENSOR.V_TOK_EOI,
MODEL_TENSOR.V_MM_PRE_NORM,
@@ -4907,6 +4916,7 @@ class VisionProjectorType:
MIMOVL = "mimovl"
MIMO_AUDIO = "mimo_audio"
GRANITE4_VISION = "granite4_vision"
ONYX = "onyx"
# Items here are (block size, type size)
+26 -3
View File
@@ -1467,6 +1467,7 @@ class TensorNameMap:
"vision_tower.patch_embed.patchifier.proj", # dots.ocr
"vision_model.conv1", # Step3-VL
"model.vision_embedder.patch_dense", # gemma4 unified
"model.vision_tower.patch_embedder.patch_embedding", # onyx
),
MODEL_TENSOR.V_ENC_EMBD_NORM: (
@@ -1534,7 +1535,8 @@ class TensorNameMap:
"siglip2.vision_model.encoder.layers.{bid}.self_attn.q_proj", # youtuvl
"model.vision_model.transformer.layers.{bid}.self_attn.q_proj", # Deepseek-OCR CLIP, generated
"vision_model.model.layers.{bid}.self_attn.q_proj.linear", # gemma4
"model.qwen2_model.model.model.layers.{bid}.self_attn.q_proj" # Deepseek-OCR-2 qwen2
"model.qwen2_model.model.model.layers.{bid}.self_attn.q_proj", # Deepseek-OCR-2 qwen2
"model.vision_tower.layers.{bid}.attn.q_proj", # onyx
),
MODEL_TENSOR.V_ENC_ATTN_Q_NORM: (
@@ -1560,7 +1562,8 @@ class TensorNameMap:
"model.vision_model.transformer.layers.{bid}.self_attn.k_proj", # Deepseek-OCR CLIP, generated
"siglip2.vision_model.encoder.layers.{bid}.self_attn.k_proj",
"vision_model.model.layers.{bid}.self_attn.k_proj.linear", # gemma4
"model.qwen2_model.model.model.layers.{bid}.self_attn.k_proj" # Deepseek-OCR-2 qwen2
"model.qwen2_model.model.model.layers.{bid}.self_attn.k_proj", # Deepseek-OCR-2 qwen2
"model.vision_tower.layers.{bid}.attn.k_proj", # onyx
),
MODEL_TENSOR.V_ENC_ATTN_K_NORM: (
@@ -1586,7 +1589,8 @@ class TensorNameMap:
"siglip2.vision_model.encoder.layers.{bid}.self_attn.v_proj",
"model.vision_model.transformer.layers.{bid}.self_attn.v_proj", # Deepseek-OCR CLIP, generated
"vision_model.model.layers.{bid}.self_attn.v_proj.linear", # gemma4
"model.qwen2_model.model.model.layers.{bid}.self_attn.v_proj" # Deepseek-OCR-2 qwen2
"model.qwen2_model.model.model.layers.{bid}.self_attn.v_proj", # Deepseek-OCR-2 qwen2
"model.vision_tower.layers.{bid}.attn.v_proj", # onyx
),
MODEL_TENSOR.V_ENC_INPUT_NORM: (
@@ -1610,6 +1614,7 @@ class TensorNameMap:
"vision_tower.blocks.{bid}.norm1", # dots.ocr
"vision_model.transformer.resblocks.{bid}.ln_1", # Step3-VL
"model.qwen2_model.model.model.layers.{bid}.input_layernorm", # Deepseek-OCR-2 qwen2
"model.vision_tower.layers.{bid}.norm1", # onyx
),
MODEL_TENSOR.V_ENC_ATTN_O: (
@@ -1635,6 +1640,7 @@ class TensorNameMap:
"vision_model.model.layers.{bid}.self_attn.o_proj.linear", # gemma4
"vision_tower.blocks.{bid}.attn.proj", # dots.ocr
"vision_model.transformer.resblocks.{bid}.attn.out_proj", # Step3-VL
"model.vision_tower.layers.{bid}.attn.proj", # onyx
),
MODEL_TENSOR.V_ENC_ATTN_SINKS: (
@@ -1663,6 +1669,7 @@ class TensorNameMap:
"vision_tower.blocks.{bid}.norm2", # dots.ocr
"vision_model.transformer.resblocks.{bid}.ln_2", # Step3-VL
"model.qwen2_model.model.model.layers.{bid}.post_attention_layernorm", # Deepseek-OCR-2 qwen2
"model.vision_tower.layers.{bid}.norm2", # onyx
),
MODEL_TENSOR.V_ENC_FFN_UP: (
@@ -1687,6 +1694,7 @@ class TensorNameMap:
"vision_model.model.layers.{bid}.mlp.up_proj", # gemma4
"vision_model.transformer.resblocks.{bid}.mlp.c_fc", # Step3-VL
"model.qwen2_model.model.model.layers.{bid}.mlp.up_proj", # Deepseek-OCR-2 qwen2
"model.vision_tower.layers.{bid}.mlp.fc1", # onyx
),
MODEL_TENSOR.V_ENC_FFN_GATE: (
@@ -1719,6 +1727,7 @@ class TensorNameMap:
"model.qwen2_model.model.model.layers.{bid}.mlp.down_proj" , # Deepseek-OCR-2 qwen2
"vision_model.model.layers.{bid}.mlp.down_proj", # gemma4
"vision_model.transformer.resblocks.{bid}.mlp.c_proj", # Step3-VL
"model.vision_tower.layers.{bid}.mlp.fc2", # onyx
),
MODEL_TENSOR.V_ENC_ATTN_POST_NORM: (
@@ -1753,6 +1762,7 @@ class TensorNameMap:
"model.vision_model.pre_layrnorm", # Deepseek-OCR CLIP
"vision_tower.patch_embed.patchifier.norm", # dots.ocr
"vision_model.ln_pre", # Step3-VL
"model.vision_tower.ln_pre", # onyx
),
MODEL_TENSOR.V_POST_NORM: (
@@ -1766,6 +1776,7 @@ class TensorNameMap:
"visual.post_layernorm", # glm4v
"siglip2.vision_model.post_layernorm",
"model.qwen2_model.model.model.norm", # Deepseek-OCR-2 qwen2
"model.vision_tower.ln_post", # onyx
),
MODEL_TENSOR.V_MM_POST_NORM: (
@@ -1858,6 +1869,18 @@ class TensorNameMap:
"patch_merge_mlp.linear_2", # minimax-m3
),
MODEL_TENSOR.V_MM_ADAPTER_FC: (
"model.vision_adapter.fc1", # onyx
),
MODEL_TENSOR.V_MM_ADAPTER_PROJ: (
"model.vision_adapter.fc2", # onyx
),
MODEL_TENSOR.V_MM_VISION_PROJ: (
"model.vision_projection", # onyx
),
MODEL_TENSOR.V_DS_NORM: (
"model.visual.deepstack_merger_list.{bid}.norm", # deepstack in qwen3vl
),