From e286e9b8c4088aa427d12799fbde9dbb2cccba7a Mon Sep 17 00:00:00 2001 From: Pedro Cuenca Date: Sun, 9 Aug 2026 16:27:34 +0200 Subject: [PATCH] Rename --- common/chat.cpp | 67 ++----------------- conversion/__init__.py | 6 +- conversion/{onyx.py => muse_glimmer.py} | 24 +++---- gguf-py/gguf/constants.py | 8 +-- gguf-py/gguf/tensor_mapping.py | 28 ++++---- src/llama-arch.cpp | 2 +- src/llama-arch.h | 2 +- src/llama-model.cpp | 6 +- src/models/models.h | 4 +- src/models/{onyx.cpp => muse-glimmer.cpp} | 10 +-- tools/mtmd/CMakeLists.txt | 2 +- tools/mtmd/clip-impl.h | 4 +- tools/mtmd/clip-model.h | 7 +- tools/mtmd/clip.cpp | 34 +++++----- tools/mtmd/models/models.h | 4 +- .../models/{onyx.cpp => muse-glimmer.cpp} | 30 ++++----- tools/mtmd/mtmd-image.cpp | 10 +-- tools/mtmd/mtmd-image.h | 4 +- tools/mtmd/mtmd.cpp | 4 +- 19 files changed, 101 insertions(+), 155 deletions(-) rename conversion/{onyx.py => muse_glimmer.py} (91%) rename src/models/{onyx.cpp => muse-glimmer.cpp} (94%) rename tools/mtmd/models/{onyx.cpp => muse-glimmer.cpp} (73%) diff --git a/common/chat.cpp b/common/chat.cpp index 653a9e6e60..be67832204 100644 --- a/common/chat.cpp +++ b/common/chat.cpp @@ -2902,16 +2902,11 @@ static common_chat_params common_chat_params_init_minicpm5(const common_chat_tem // - final answer: to=user, terminated by <|eot|> // The generation prompt is just "<|start|>assistant"; the model emits its own // " to=...<|message|>". -static common_chat_params common_chat_params_init_onyx(const common_chat_template & tmpl, - const autoparser::generation_params & inputs) { +static common_chat_params common_chat_params_init_muse_glimmer(const common_chat_template & tmpl, + const autoparser::generation_params & inputs) { common_chat_params data; data.prompt = common_chat_template_direct_apply_impl(tmpl, inputs); - // generation_prompt is the exact assistant-turn prefix the server strips when recovering - // the model's output from the full decoded text. Hardcode it to "<|start|>assistant" (the - // model emits its own " to=<|message|>" after it). Re-rendering the template - // with add_generation_prompt would also inject a default system message, which must not - // become part of this stripped prefix. data.generation_prompt = "<|start|>assistant"; data.format = COMMON_CHAT_FORMAT_PEG_NATIVE; data.supports_thinking = true; @@ -2944,27 +2939,16 @@ static common_chat_params common_chat_params_init_onyx(const common_chat_templat auto extract_reasoning = inputs.reasoning_format != COMMON_REASONING_FORMAT_NONE; auto has_tools = inputs.tools.is_array() && !inputs.tools.empty(); - // Constrained grammar whenever tools are offered. For tool_choice=REQUIRED the grammar is - // non-lazy (forces a valid ATEM tool call from the first token). For the auto path it is lazy - // and only activates once the model commits to a tool recipient (" to=<|message|>"), so - // the free-text reasoning (to=self) and final (to=user) turns generate unconstrained. See the - // grammar_triggers below. + // Constrained grammar whenever tools are offered. auto include_grammar = has_tools && inputs.tool_choice != COMMON_CHAT_TOOL_CHOICE_NONE; auto parser = build_chat_peg_parser([&](common_chat_peg_builder & p) { auto start = p.rule("start", p.literal("<|start|>assistant")); - // No reasoning to split out and no tools: emit the raw stream as content. - // Tool parsing does not depend on reasoning_format. include_grammar (above) forces a tool - // call whenever tools are offered, so the parser runs the tool rules even under NONE. - // reasoning_format only decides whether the to=self block is tagged reasoning or content. if (!extract_reasoning && !include_grammar) { return start + p.content(p.rest()); } - // chain-of-thought message: " to=self<|message|>{body}<|eom|>". The literals stay - // structural so a partial " to=" prefix does not emit a content delta before it resolves - // to a tool recipient. Under NONE the body is tagged content instead of reasoning. if (extract_reasoning) { p.rule("analysis", p.literal(" to=self<|message|>") + p.reasoning(p.until("<|eom|>")) + p.literal("<|eom|>")); } else { @@ -2972,29 +2956,10 @@ static common_chat_params common_chat_params_init_onyx(const common_chat_templat } auto analysis = p.ref("analysis"); - // final answer: "[ to=user]<|message|>{content}<|eot|>". The recipient is kept fixed - // (not a wildcard) so a partial chain-of-thought (" to=self...") never matches this - // branch and leaks into content before its <|eom|> terminator arrives. auto recipient = p.optional(p.literal(" to=user")); auto final_msg = p.rule("final", recipient + p.literal("<|message|>") + p.content(p.until("<|eot|>"))); - // Tool-call turn: " to=<|message|>...{term}". - // Ported from the ATEM tool-call parser. Only added when tools are offered; otherwise the - // grammar and top-level rule are exactly the reasoning/final structure above. - // - // These tool-call rules assume grammar-constrained input, which is what the - // server produces during generation: canonical whitespace and in-schema params, - // with no inline preamble because onyx puts preamble in the to=self message. - // Parsing unconstrained or historical assistant text is strict on purpose and - // will reject the whole tool call rather than guess. if (has_tools && inputs.tool_choice != COMMON_CHAT_TOOL_CHOICE_NONE) { - // A tool-call message closes with "", terminated by <|eom|> - // (more calls this turn) or <|eot|> (end of turn). We do NOT consume the trailing - // <|eot|> -- it fires as the server's end-of-turn EOG stop; the <|eom|> between - // parallel calls IS consumed (see tool_calls below). - - // Schema-declared string args are captured verbatim up to the closing tag; anything - // else is parsed as JSON (via the per-arg schema rule) before its closing tag. auto string_value = p.ac( p.tool_arg_string_value(p.until("")) + p.tool_arg_close(p.literal("")), ""); @@ -3030,13 +2995,6 @@ static common_chat_params common_chat_params_init_onyx(const common_chat_templat args = p.zero_or_more(arg_choice + p.space()); } - // The tool identity is keyed on the (the - // actual call), NOT the " to=". Onyx models sometimes emit - // a recipient that disagrees with the invoke name (e.g. " to= - // container.exec" with ); accept any - // recipient here so such calls are parsed rather than dropped. The - // "" literal disambiguates this from the to=user - // final turn. auto tool_parser = p.tool( p.tool_open(p.literal(" to=") + p.until("<|message|>") + p.literal("<|message|>") + p.space() + @@ -3047,12 +3005,6 @@ static common_chat_params common_chat_params_init_onyx(const common_chat_templat tool_choice |= p.rule("tool-" + name, tool_parser); }); - // Parallel tool calls are consecutive assistant messages separated by <|eom|>, the turn - // ending with <|eot|> after the last. Gate the loop on parallel_tool_calls like the other - // formats: disabled -> single call; enabled -> multiple. NOTE: this model is unreliable at - // single-turn parallel (it often ends the turn after the first call), so - // parallel_tool_calls=false (sequential, driven by the agent loop) is recommended; true is - // honored but may under-emit. auto tool_calls = inputs.parallel_tool_calls ? p.trigger_rule("tool-call", tool_choice + p.zero_or_more(p.literal("<|eom|>") + start + tool_choice)) : p.trigger_rule("tool-call", tool_choice); @@ -3079,11 +3031,6 @@ static common_chat_params common_chat_params_init_onyx(const common_chat_templat }); parser.build_grammar(builder, data.grammar_lazy); }); - // Trigger only on a TOOL recipient, not the to=self (reasoning) or to=user (final) turns, - // which are free text and would empty the grammar stack. The capture group starts at " to=" - // (matching where the grammar's tool rule begins) so the replayed input lines up with the - // grammar; requiring the trailing "<|message|>" (and excluding self/user before it) means - // the pattern only fires once the full tool recipient is known. data.grammar_triggers = { { COMMON_GRAMMAR_TRIGGER_TYPE_PATTERN, "<\\|start\\|>assistant( to=(?!self<\\|message\\|>)(?!user<\\|message\\|>)[^<]*?<\\|message\\|>)" }, @@ -3121,12 +3068,10 @@ std::optional common_chat_try_specialized_template( return common_chat_params_init_gpt_oss(tmpl, params); } - // Onyx format using " to=" recipients and <|eom|>/<|eot|> message - // terminators. The ATEM tool-call markup "" is unique to - // this template. + // Muse Glimmer format using " to=" recipients and <|eom|>/<|eot|> message terminators. if (src.find("") != std::string::npos && src.find("<|eom|>") != std::string::npos) { - LOG_DBG("Using specialized template: Onyx\n"); - return common_chat_params_init_onyx(tmpl, params); + LOG_DBG("Using specialized template: Muse Glimmer\n"); + return common_chat_params_init_muse_glimmer(tmpl, params); } // Functionary v3.2 - uses recipient-based format with >>>recipient\n{content} diff --git a/conversion/__init__.py b/conversion/__init__.py index 9fffe82c07..4c55085f8a 100644 --- a/conversion/__init__.py +++ b/conversion/__init__.py @@ -180,8 +180,8 @@ TEXT_MODEL_MAP: dict[str, str] = { "Olmo3ForCausalLM": "olmo", "OlmoForCausalLM": "olmo", "OlmoeForCausalLM": "olmo", - "OnyxAssistantModel": "onyx", - "OnyxForConditionalGeneration": "onyx", + "MuseGlimmerAssistantModel": "muse-glimmer", + "MuseGlimmerForConditionalGeneration": "muse-glimmer", "OpenELMForCausalLM": "openelm", "OrionForCausalLM": "orion", "PLMForCausalLM": "plm", @@ -296,7 +296,7 @@ MMPROJ_MODEL_MAP: dict[str, str] = { "MiniCPMV4_6ForConditionalGeneration": "minicpm", "Mistral3ForConditionalGeneration": "llava", "NemotronH_Nano_VL_V2": "nemotron", - "OnyxForConditionalGeneration": "onyx", + "MuseGlimmerForConditionalGeneration": "muse-glimmer", "PaddleOCRVisionModel": "ernie", "Phi4ForCausalLMV": "phi", "Qwen2AudioForConditionalGeneration": "ultravox", diff --git a/conversion/onyx.py b/conversion/muse_glimmer.py similarity index 91% rename from conversion/onyx.py rename to conversion/muse_glimmer.py index 004879317b..54a3adc36e 100644 --- a/conversion/onyx.py +++ b/conversion/muse_glimmer.py @@ -23,9 +23,9 @@ def _unpermute_for_rope(tensor: "Tensor", n_heads: int) -> "Tensor": raise ValueError(f"_unpermute_for_rope: unexpected shape {tuple(tensor.shape)}") -@ModelBase.register("OnyxForConditionalGeneration") -class OnyxModel(TextModel): - model_arch = gguf.MODEL_ARCH.ONYX +@ModelBase.register("MuseGlimmerForConditionalGeneration") +class MuseGlimmerModel(TextModel): + model_arch = gguf.MODEL_ARCH.MUSE_GLIMMER def norm_shift(self, name: str) -> float: # All four layer norms use 1, the final norm uses 0. @@ -61,7 +61,7 @@ class OnyxModel(TextModel): data_torch = _unpermute_for_rope(data_torch, int(self.hparams["num_key_value_heads"])) # Synthesize QK-norm weights to absorb qk_scale_factor. - # Onyx implementation: scaleless RMSNorm followed by qk_scale_factor.. + # MuseGlimmer implementation: scaleless RMSNorm followed by qk_scale_factor.. if bid is not None and name.endswith(f"model.layers.{bid}.self_attn.q_proj.weight"): head_dim = self.hparams["head_dim"] q_scale = float(self.hparams["qk_scale_factor"]) @@ -77,13 +77,13 @@ class OnyxModel(TextModel): yield from super().modify_tensors(data_torch, name, bid) -@ModelBase.register("OnyxForConditionalGeneration") -class OnyxVisionModel(MmprojModel): +@ModelBase.register("MuseGlimmerForConditionalGeneration") +class MuseGlimmerVisionModel(MmprojModel): def get_vision_config(self) -> dict[str, Any] | None: c = self.global_config.get("vision_config") if not c: return None - # Onyx actually uses dynamic size, initialize with nominal size + # MuseGlimmer actually uses dynamic size, initialize with nominal size image_size = c["pos_emb_height"] * c["patch_size"] * c["merge_size"] return {**c, "image_size": image_size} @@ -91,7 +91,7 @@ class OnyxVisionModel(MmprojModel): super().set_gguf_parameters() c = self.hparams_vision # enriched vision_config from get_vision_config() - self.gguf_writer.add_clip_projector_type(gguf.VisionProjectorType.ONYX) + self.gguf_writer.add_clip_projector_type(gguf.VisionProjectorType.MUSE_GLIMMER) self.gguf_writer.add_vision_attention_layernorm_eps(float(c["layer_norm_eps"])) self.gguf_writer.add_vision_spatial_merge_size(int(c["merge_size"])) @@ -128,15 +128,15 @@ class OnyxVisionModel(MmprojModel): yield (self.map_tensor_name(name), data_torch) -@ModelBase.register("OnyxAssistantModel") -class OnyxAssistantModel(TextModel): +@ModelBase.register("MuseGlimmerAssistantModel") +class MuseGlimmerAssistantModel(TextModel): model_arch = gguf.MODEL_ARCH.DFLASH def set_vocab(self): if self.target_model_dir is None: raise ValueError( - "OnyxAssistant (DFlash drafter) requires --target-model-dir pointing to the " - "target Onyx HF directory" + "MuseGlimmerAssistant (DFlash drafter) requires --target-model-dir pointing to the " + "target MuseGlimmer HF directory" ) original_dir = self.dir_model diff --git a/gguf-py/gguf/constants.py b/gguf-py/gguf/constants.py index 1ae9deb3b9..f511feb977 100644 --- a/gguf-py/gguf/constants.py +++ b/gguf-py/gguf/constants.py @@ -488,7 +488,7 @@ class MODEL_ARCH(IntEnum): OLMO = auto() OLMO2 = auto() OLMOE = auto() - ONYX = auto() + MUSE_GLIMMER = auto() OPENELM = auto() ARCTIC = auto() DEEPSEEK = auto() @@ -1100,7 +1100,7 @@ MODEL_ARCH_NAMES: dict[MODEL_ARCH, str] = { MODEL_ARCH.OLMO: "olmo", MODEL_ARCH.OLMO2: "olmo2", MODEL_ARCH.OLMOE: "olmoe", - MODEL_ARCH.ONYX: "onyx", + MODEL_ARCH.MUSE_GLIMMER: "muse-glimmer", MODEL_ARCH.OPENELM: "openelm", MODEL_ARCH.ARCTIC: "arctic", MODEL_ARCH.DEEPSEEK: "deepseek", @@ -3129,7 +3129,7 @@ MODEL_TENSORS: dict[MODEL_ARCH, list[MODEL_TENSOR]] = { MODEL_TENSOR.FFN_UP_EXP, MODEL_TENSOR.FFN_DOWN_EXP, ], - MODEL_ARCH.ONYX: [ + MODEL_ARCH.MUSE_GLIMMER: [ MODEL_TENSOR.TOKEN_EMBD, MODEL_TENSOR.OUTPUT, MODEL_TENSOR.OUTPUT_NORM, @@ -4906,7 +4906,7 @@ class VisionProjectorType: MIMOVL = "mimovl" MIMO_AUDIO = "mimo_audio" GRANITE4_VISION = "granite4_vision" - ONYX = "onyx" + MUSE_GLIMMER = "muse-glimmer" # Items here are (block size, type size) diff --git a/gguf-py/gguf/tensor_mapping.py b/gguf-py/gguf/tensor_mapping.py index 28228d9f3b..50c9ac7abc 100644 --- a/gguf-py/gguf/tensor_mapping.py +++ b/gguf-py/gguf/tensor_mapping.py @@ -382,7 +382,7 @@ class TensorNameMap: ), MODEL_TENSOR.ATTN_GATE: ( - "model.layers.{bid}.self_attn.gate_proj", # afmoe onyx + "model.layers.{bid}.self_attn.gate_proj", # afmoe muse-glimmer "model.layers.{bid}.linear_attn.in_proj_z", # qwen3.5 "model.layers.{bid}.self_attn.g_proj", # step3.5 head-wise attention gate ), @@ -1298,12 +1298,12 @@ class TensorNameMap: "encoder.final_layer_norm", # t5 "layer_norm", # neobert "model.hidden_norm", # dflash - "encoder.output_norm_enc", # dflash (transformers OnyxAssistant) + "encoder.output_norm_enc", # dflash (transformers MuseGlimmerAssistant) ), MODEL_TENSOR.FC: ( "model.fc", # dflash - "encoder.fc", # dflash (transformers OnyxAssistant) + "encoder.fc", # dflash (transformers MuseGlimmerAssistant) ), MODEL_TENSOR.DSPARK_MARKOV_W1: ( @@ -1469,7 +1469,7 @@ class TensorNameMap: "vision_tower.patch_embed.patchifier.proj", # dots.ocr "vision_model.conv1", # Step3-VL "model.vision_embedder.patch_dense", # gemma4 unified - "model.vision_tower.patch_embedder.patch_embedding", # onyx + "model.vision_tower.patch_embedder.patch_embedding", # muse-glimmer ), MODEL_TENSOR.V_ENC_EMBD_NORM: ( @@ -1538,7 +1538,7 @@ class TensorNameMap: "model.vision_model.transformer.layers.{bid}.self_attn.q_proj", # Deepseek-OCR CLIP, generated "vision_model.model.layers.{bid}.self_attn.q_proj.linear", # gemma4 "model.qwen2_model.model.model.layers.{bid}.self_attn.q_proj", # Deepseek-OCR-2 qwen2 - "model.vision_tower.layers.{bid}.attn.q_proj", # onyx + "model.vision_tower.layers.{bid}.attn.q_proj", # muse-glimmer ), MODEL_TENSOR.V_ENC_ATTN_Q_NORM: ( @@ -1565,7 +1565,7 @@ class TensorNameMap: "siglip2.vision_model.encoder.layers.{bid}.self_attn.k_proj", "vision_model.model.layers.{bid}.self_attn.k_proj.linear", # gemma4 "model.qwen2_model.model.model.layers.{bid}.self_attn.k_proj", # Deepseek-OCR-2 qwen2 - "model.vision_tower.layers.{bid}.attn.k_proj", # onyx + "model.vision_tower.layers.{bid}.attn.k_proj", # muse-glimmer ), MODEL_TENSOR.V_ENC_ATTN_K_NORM: ( @@ -1592,7 +1592,7 @@ class TensorNameMap: "model.vision_model.transformer.layers.{bid}.self_attn.v_proj", # Deepseek-OCR CLIP, generated "vision_model.model.layers.{bid}.self_attn.v_proj.linear", # gemma4 "model.qwen2_model.model.model.layers.{bid}.self_attn.v_proj", # Deepseek-OCR-2 qwen2 - "model.vision_tower.layers.{bid}.attn.v_proj", # onyx + "model.vision_tower.layers.{bid}.attn.v_proj", # muse-glimmer ), MODEL_TENSOR.V_ENC_INPUT_NORM: ( @@ -1616,7 +1616,7 @@ class TensorNameMap: "vision_tower.blocks.{bid}.norm1", # dots.ocr "vision_model.transformer.resblocks.{bid}.ln_1", # Step3-VL "model.qwen2_model.model.model.layers.{bid}.input_layernorm", # Deepseek-OCR-2 qwen2 - "model.vision_tower.layers.{bid}.norm1", # onyx + "model.vision_tower.layers.{bid}.norm1", # muse-glimmer ), MODEL_TENSOR.V_ENC_ATTN_O: ( @@ -1642,7 +1642,7 @@ class TensorNameMap: "vision_model.model.layers.{bid}.self_attn.o_proj.linear", # gemma4 "vision_tower.blocks.{bid}.attn.proj", # dots.ocr "vision_model.transformer.resblocks.{bid}.attn.out_proj", # Step3-VL - "model.vision_tower.layers.{bid}.attn.proj", # onyx + "model.vision_tower.layers.{bid}.attn.proj", # muse-glimmer ), MODEL_TENSOR.V_ENC_ATTN_SINKS: ( @@ -1671,7 +1671,7 @@ class TensorNameMap: "vision_tower.blocks.{bid}.norm2", # dots.ocr "vision_model.transformer.resblocks.{bid}.ln_2", # Step3-VL "model.qwen2_model.model.model.layers.{bid}.post_attention_layernorm", # Deepseek-OCR-2 qwen2 - "model.vision_tower.layers.{bid}.norm2", # onyx + "model.vision_tower.layers.{bid}.norm2", # muse-glimmer ), MODEL_TENSOR.V_ENC_FFN_UP: ( @@ -1696,7 +1696,7 @@ class TensorNameMap: "vision_model.model.layers.{bid}.mlp.up_proj", # gemma4 "vision_model.transformer.resblocks.{bid}.mlp.c_fc", # Step3-VL "model.qwen2_model.model.model.layers.{bid}.mlp.up_proj", # Deepseek-OCR-2 qwen2 - "model.vision_tower.layers.{bid}.mlp.fc1", # onyx + "model.vision_tower.layers.{bid}.mlp.fc1", # muse-glimmer ), MODEL_TENSOR.V_ENC_FFN_GATE: ( @@ -1729,7 +1729,7 @@ class TensorNameMap: "model.qwen2_model.model.model.layers.{bid}.mlp.down_proj" , # Deepseek-OCR-2 qwen2 "vision_model.model.layers.{bid}.mlp.down_proj", # gemma4 "vision_model.transformer.resblocks.{bid}.mlp.c_proj", # Step3-VL - "model.vision_tower.layers.{bid}.mlp.fc2", # onyx + "model.vision_tower.layers.{bid}.mlp.fc2", # muse-glimmer ), MODEL_TENSOR.V_ENC_ATTN_POST_NORM: ( @@ -1764,7 +1764,7 @@ class TensorNameMap: "model.vision_model.pre_layrnorm", # Deepseek-OCR CLIP "vision_tower.patch_embed.patchifier.norm", # dots.ocr "vision_model.ln_pre", # Step3-VL - "model.vision_tower.ln_pre", # onyx + "model.vision_tower.ln_pre", # muse-glimmer ), MODEL_TENSOR.V_POST_NORM: ( @@ -1778,7 +1778,7 @@ class TensorNameMap: "visual.post_layernorm", # glm4v "siglip2.vision_model.post_layernorm", "model.qwen2_model.model.model.norm", # Deepseek-OCR-2 qwen2 - "model.vision_tower.ln_post", # onyx + "model.vision_tower.ln_post", # muse-glimmer ), MODEL_TENSOR.V_MM_POST_NORM: ( diff --git a/src/llama-arch.cpp b/src/llama-arch.cpp index 2ce0a7752d..f7dfe4af0d 100644 --- a/src/llama-arch.cpp +++ b/src/llama-arch.cpp @@ -71,7 +71,7 @@ static const std::map LLM_ARCH_NAMES = { { LLM_ARCH_OLMO, "olmo" }, { LLM_ARCH_OLMO2, "olmo2" }, { LLM_ARCH_OLMOE, "olmoe" }, - { LLM_ARCH_ONYX, "onyx" }, + { LLM_ARCH_MUSE_GLIMMER, "muse-glimmer" }, { LLM_ARCH_OPENELM, "openelm" }, { LLM_ARCH_ARCTIC, "arctic" }, { LLM_ARCH_DEEPSEEK, "deepseek" }, diff --git a/src/llama-arch.h b/src/llama-arch.h index 24d5d97c48..71a485b47f 100644 --- a/src/llama-arch.h +++ b/src/llama-arch.h @@ -76,7 +76,7 @@ enum llm_arch { LLM_ARCH_OLMO, LLM_ARCH_OLMO2, LLM_ARCH_OLMOE, - LLM_ARCH_ONYX, + LLM_ARCH_MUSE_GLIMMER, LLM_ARCH_OPENELM, LLM_ARCH_ARCTIC, LLM_ARCH_DEEPSEEK, diff --git a/src/llama-model.cpp b/src/llama-model.cpp index 578e423953..9a7a0b20de 100644 --- a/src/llama-model.cpp +++ b/src/llama-model.cpp @@ -171,8 +171,8 @@ static llama_model * llama_model_mapping(llm_arch arch, const llama_model_params return new llama_model_olmo2(params); case LLM_ARCH_OLMOE: return new llama_model_olmoe(params); - case LLM_ARCH_ONYX: - return new llama_model_onyx(params); + case LLM_ARCH_MUSE_GLIMMER: + return new llama_model_muse_glimmer(params); case LLM_ARCH_OPENELM: return new llama_model_openelm(params); case LLM_ARCH_GPTNEOX: @@ -2530,7 +2530,7 @@ llama_rope_type llama_model_rope_type(const llama_model * model) { case LLM_ARCH_DEEPSEEK2OCR: case LLM_ARCH_DEEPSEEK32: case LLM_ARCH_DEEPSEEK4: - case LLM_ARCH_ONYX: + case LLM_ARCH_MUSE_GLIMMER: case LLM_ARCH_PLM: case LLM_ARCH_CHATGLM: case LLM_ARCH_GRANITE: diff --git a/src/models/models.h b/src/models/models.h index 6cdd92bde1..235674047b 100644 --- a/src/models/models.h +++ b/src/models/models.h @@ -1023,8 +1023,8 @@ struct llama_model_olmoe : public llama_model_base { }; -struct llama_model_onyx : public llama_model_base { - llama_model_onyx(const struct llama_model_params & params) : llama_model_base(params) {} +struct llama_model_muse_glimmer : public llama_model_base { + llama_model_muse_glimmer(const struct llama_model_params & params) : llama_model_base(params) {} void load_arch_hparams(llama_model_loader & ml) override; void load_arch_tensors(llama_model_loader & ml) override; diff --git a/src/models/onyx.cpp b/src/models/muse-glimmer.cpp similarity index 94% rename from src/models/onyx.cpp rename to src/models/muse-glimmer.cpp index 8aef05197f..78470c5222 100644 --- a/src/models/onyx.cpp +++ b/src/models/muse-glimmer.cpp @@ -1,6 +1,6 @@ #include "models.h" -void llama_model_onyx::load_arch_hparams(llama_model_loader & ml) { +void llama_model_muse_glimmer::load_arch_hparams(llama_model_loader & ml) { ml.get_key(LLM_KV_ATTENTION_LAYERNORM_RMS_EPS, hparams.f_norm_rms_eps); ml.get_key(LLM_KV_ATTENTION_SLIDING_WINDOW, hparams.n_swa, false); ml.get_key(LLM_KV_FINAL_LOGIT_SOFTCAPPING, hparams.f_final_logit_softcapping, false); @@ -23,7 +23,7 @@ void llama_model_onyx::load_arch_hparams(llama_model_loader & ml) { } } -void llama_model_onyx::load_arch_tensors(llama_model_loader &) { +void llama_model_muse_glimmer::load_arch_tensors(llama_model_loader &) { LLAMA_LOAD_LOCALS; tok_embd = create_tensor(tn(LLM_TENSOR_TOKEN_EMBD, "weight"), {n_embd, n_vocab}, 0); @@ -33,7 +33,7 @@ void llama_model_onyx::load_arch_tensors(llama_model_loader &) { for (int i = 0; i < n_layer; ++i) { auto & layer = layers[i]; - // Pre/post-attention norms (Onyx's `weight + 1` applied at conversion time). + // Pre/post-attention norms (Muse Glimmer's `weight + 1` applied at conversion time). layer.attn_norm = create_tensor(tn(LLM_TENSOR_ATTN_NORM, "weight", i), {n_embd}, 0); layer.attn_post_norm = create_tensor(tn(LLM_TENSOR_ATTN_POST_NORM, "weight", i), {n_embd}, 0); @@ -59,7 +59,7 @@ void llama_model_onyx::load_arch_tensors(llama_model_loader &) { } } -llama_model_onyx::graph::graph(const llama_model & model, const llm_graph_params & params) +llama_model_muse_glimmer::graph::graph(const llama_model & model, const llm_graph_params & params) : llm_graph_context(params) { const int64_t n_embd_head = hparams.n_embd_head_v(); GGML_ASSERT(n_embd_head == hparams.n_embd_head_k()); @@ -203,6 +203,6 @@ llama_model_onyx::graph::graph(const llama_model & model, const llm_graph_params ggml_build_forward_expand(gf, cur); } -std::unique_ptr llama_model_onyx::build_arch_graph(const llm_graph_params & params) const { +std::unique_ptr llama_model_muse_glimmer::build_arch_graph(const llm_graph_params & params) const { return std::make_unique(*this, params); } diff --git a/tools/mtmd/CMakeLists.txt b/tools/mtmd/CMakeLists.txt index 82f7752487..be866dc738 100644 --- a/tools/mtmd/CMakeLists.txt +++ b/tools/mtmd/CMakeLists.txt @@ -41,7 +41,7 @@ add_library(mtmd models/kimivl.cpp models/kimik25.cpp models/nemotron-v2-vl.cpp - models/onyx.cpp + models/muse-glimmer.cpp models/llama4.cpp models/llava.cpp models/minicpmv.cpp diff --git a/tools/mtmd/clip-impl.h b/tools/mtmd/clip-impl.h index c159e249e2..3f770d4128 100644 --- a/tools/mtmd/clip-impl.h +++ b/tools/mtmd/clip-impl.h @@ -408,7 +408,7 @@ enum projector_type { PROJECTOR_TYPE_MINIMAX_M3, PROJECTOR_TYPE_GRANITE4_VISION, PROJECTOR_TYPE_MIMO_AUDIO, - PROJECTOR_TYPE_ONYX, + PROJECTOR_TYPE_MUSE_GLIMMER, PROJECTOR_TYPE_UNKNOWN, }; @@ -466,7 +466,7 @@ static std::map PROJECTOR_TYPE_NAMES = { { PROJECTOR_TYPE_GRANITE4_VISION, "granite4_vision"}, { PROJECTOR_TYPE_MIMO_AUDIO, "mimo_audio"}, { PROJECTOR_TYPE_PARAKEET, "parakeet"}, - { PROJECTOR_TYPE_ONYX, "onyx"}, + { PROJECTOR_TYPE_MUSE_GLIMMER, "muse-glimmer"}, }; static projector_type clip_projector_type_from_string(const std::string & str) { diff --git a/tools/mtmd/clip-model.h b/tools/mtmd/clip-model.h index 5dc706c26e..23ff9b9d28 100644 --- a/tools/mtmd/clip-model.h +++ b/tools/mtmd/clip-model.h @@ -109,9 +109,10 @@ struct clip_hparams { int32_t downsample_query_side; int32_t downsample_window_side; - // Onyx vision (per-block sparse-window pattern, learned pos-emb, patch-temporal) - int32_t onyx_patch_temporal = 0; - int32_t onyx_sparse_factor = 0; + // Muse Glimmer vision (per-block sparse-window pattern, learned pos-emb, patch-temporal) + // NOTE: these perhaps shouldn't have the architecture prefix + int32_t muse_glimmer_patch_temporal = 0; + int32_t muse_glimmer_sparse_factor = 0; // audio int32_t n_mel_bins = 0; // whisper preprocessor diff --git a/tools/mtmd/clip.cpp b/tools/mtmd/clip.cpp index a204e7fb25..b300647d4b 100644 --- a/tools/mtmd/clip.cpp +++ b/tools/mtmd/clip.cpp @@ -928,9 +928,9 @@ static std::unique_ptr clip_get_graph_builder(clip_ctx * ctx, const { builder = std::make_unique(ctx, img); } break; - case PROJECTOR_TYPE_ONYX: + case PROJECTOR_TYPE_MUSE_GLIMMER: { - builder = std::make_unique(ctx, img); + builder = std::make_unique(ctx, img); } break; case PROJECTOR_TYPE_STEP3VL: { @@ -1520,13 +1520,13 @@ struct clip_model_loader { hparams.set_limit_image_tokens(8, 576); hparams.set_warmup_n_tokens(16*16); } break; - case PROJECTOR_TYPE_ONYX: + case PROJECTOR_TYPE_MUSE_GLIMMER: { hparams.n_merge = 2; // pixel-shuffle downsample after the ViT hparams.image_resize_algo = RESIZE_ALGO_LANCZOS; hparams.rope_theta = 10000.0f; - hparams.onyx_patch_temporal = 2; // Onyx-arch invariant - hparams.onyx_sparse_factor = 4; // 3 sparse layers + 1 global, repeating + hparams.muse_glimmer_patch_temporal = 2; + hparams.muse_glimmer_sparse_factor = 4; // 3 sparse layers + 1 global, repeating get_u32(KEY_SPATIAL_MERGE_SIZE, hparams.n_merge, false); hparams.set_limit_image_tokens(1, 4096); hparams.set_warmup_n_tokens(32*32); @@ -2233,7 +2233,7 @@ struct clip_model_loader { model.mm_merger_fc2_w = get_tensor(string_format(TN_MM_MERGER_FC2, "weight")); model.mm_merger_fc2_b = get_tensor(string_format(TN_MM_MERGER_FC2, "bias")); } break; - case PROJECTOR_TYPE_ONYX: + case PROJECTOR_TYPE_MUSE_GLIMMER: { // 3-linear MLP: fc -> erf-GELU -> proj -> erf-GELU -> vision_proj (into LLM residual dim) model.mm_0_w = get_tensor(string_format(TN_LLAVA_PROJ, 0, "weight")); @@ -3520,7 +3520,7 @@ int clip_n_output_tokens_x(const clip_ctx * ctx, const clip_image_f32 * img) { case PROJECTOR_TYPE_PADDLEOCR: case PROJECTOR_TYPE_HUNYUANVL: case PROJECTOR_TYPE_YOUTUVL: - case PROJECTOR_TYPE_ONYX: + case PROJECTOR_TYPE_MUSE_GLIMMER: return (img->nx() / params.patch_size) / 2; case PROJECTOR_TYPE_STEP3VL: return img->nx() / (params.patch_size * params.n_merge); @@ -3546,7 +3546,7 @@ int clip_n_output_tokens_y(const clip_ctx * ctx, const clip_image_f32 * img) { case PROJECTOR_TYPE_PADDLEOCR: case PROJECTOR_TYPE_HUNYUANVL: case PROJECTOR_TYPE_YOUTUVL: - case PROJECTOR_TYPE_ONYX: + case PROJECTOR_TYPE_MUSE_GLIMMER: return (img->ny() / params.patch_size) / 2; case PROJECTOR_TYPE_STEP3VL: return img->ny() / (params.patch_size * params.n_merge); @@ -3626,7 +3626,7 @@ int clip_n_output_tokens(const clip_ctx * ctx, const clip_image_f32 * img) { case PROJECTOR_TYPE_MINIMAX_M3: case PROJECTOR_TYPE_GLM4V: case PROJECTOR_TYPE_YOUTUVL: - case PROJECTOR_TYPE_ONYX: + case PROJECTOR_TYPE_MUSE_GLIMMER: { // dynamic size (2 conv, so double patch size) int x_patch = img->nx() / (params.patch_size * 2); @@ -3953,7 +3953,7 @@ bool clip_image_batch_encode(clip_ctx * ctx, int n_threads, const clip_image_f32 // set input per projector switch (ctx->model.proj_type) { - case PROJECTOR_TYPE_ONYX: + case PROJECTOR_TYPE_MUSE_GLIMMER: { const int grid_w = pos_w; // image_size_width / patch_size const int grid_h = pos_h; // image_size_height / patch_size @@ -3990,10 +3990,10 @@ bool clip_image_batch_encode(clip_ctx * ctx, int n_threads, const clip_image_f32 rpos_h[i] = (orig / grid_w) + 1; inv_perm[orig] = i; } - set_input_i32("onyx_sp_perm", sp_perm); - set_input_i32("onyx_inv_perm", inv_perm); - set_input_i32("onyx_pos_w", rpos_w); - set_input_i32("onyx_pos_h", rpos_h); + set_input_i32("muse_glimmer_sp_perm", sp_perm); + set_input_i32("muse_glimmer_inv_perm", inv_perm); + set_input_i32("muse_glimmer_pos_w", rpos_w); + set_input_i32("muse_glimmer_pos_h", rpos_h); // block-diagonal window mask (permuted order) std::vector sp_mask((size_t) n_tok * n_tok, -INFINITY); @@ -4006,7 +4006,7 @@ bool clip_image_batch_encode(clip_ctx * ctx, int n_threads, const clip_image_f32 off += s; } } - set_input_f32("onyx_sp_mask", sp_mask); + set_input_f32("muse_glimmer_sp_mask", sp_mask); // pixel-shuffle gather (original order): f*f spatial neighbours grouped std::vector dsp; dsp.reserve(n_tok); @@ -4015,7 +4015,7 @@ bool clip_image_batch_encode(clip_ctx * ctx, int n_threads, const clip_image_f32 for (int ry = 0; ry < f; ry++) for (int rx = 0; rx < f; rx++) dsp.push_back((oy * f + ry) * grid_w + (ox * f + rx)); - set_input_i32("onyx_ds_perm", dsp); + set_input_i32("muse_glimmer_ds_perm", dsp); } break; case PROJECTOR_TYPE_MINICPMV: { @@ -5065,7 +5065,7 @@ int clip_n_mmproj_embd(const struct clip_ctx * ctx) { return ctx->model.mm_model_mlp_3_w->ne[1]; case PROJECTOR_TYPE_MINIMAX_M3: return ctx->model.mm_merger_fc2_b->ne[0]; - case PROJECTOR_TYPE_ONYX: + case PROJECTOR_TYPE_MUSE_GLIMMER: return ctx->model.mm_2_w->ne[1]; case PROJECTOR_TYPE_QWEN2VL: case PROJECTOR_TYPE_QWEN25VL: diff --git a/tools/mtmd/models/models.h b/tools/mtmd/models/models.h index 614d66f507..51e840ee7c 100644 --- a/tools/mtmd/models/models.h +++ b/tools/mtmd/models/models.h @@ -255,7 +255,7 @@ private: ggml_tensor * append_rowwise_newlines(ggml_context * ctx0, ggml_tensor * tile_output); }; -struct clip_graph_onyx : clip_graph { - clip_graph_onyx(clip_ctx * ctx, const clip_image_f32 & img) : clip_graph(ctx, img) {} +struct clip_graph_muse_glimmer : clip_graph { + clip_graph_muse_glimmer(clip_ctx * ctx, const clip_image_f32 & img) : clip_graph(ctx, img) {} ggml_cgraph * build() override; }; diff --git a/tools/mtmd/models/onyx.cpp b/tools/mtmd/models/muse-glimmer.cpp similarity index 73% rename from tools/mtmd/models/onyx.cpp rename to tools/mtmd/models/muse-glimmer.cpp index 20308c8353..b201536f58 100644 --- a/tools/mtmd/models/onyx.cpp +++ b/tools/mtmd/models/muse-glimmer.cpp @@ -1,19 +1,19 @@ #include "models.h" -// Onyx vision encoder: 50-layer ViT with 2D RoPE, sparse block-diagonal +// MuseGlimmer vision encoder: 50-layer ViT with 2D RoPE, sparse block-diagonal // window attention (every 4th + last layer global), pixel-shuffle downsample, then // adapter MLP + LLM's vision_projection. // // Several quantities are precomputed on host and fed as named graph inputs (filled in -// clip.cpp set_input, PROJECTOR_TYPE_ONYX branch): -// onyx_pos_w/_h [n_tok] i32 : 1-indexed RoPE positions (sparse-permuted order) -// onyx_sp_perm [n_tok] i32 : window grouping permutation (applied after ln_pre) -// onyx_inv_perm [n_tok] i32 : inverse of sp_perm (applied after blocks) -// onyx_ds_perm [n_tok] i32 : pixel-shuffle gather (original order) -// onyx_sp_mask [n_tok, n_tok] f32 : block-diagonal window mask (sparse layers) -ggml_cgraph * clip_graph_onyx::build() { +// clip.cpp set_input, PROJECTOR_TYPE_MUSE_GLIMMER branch): +// muse_glimmer_pos_w/_h [n_tok] i32 : 1-indexed RoPE positions (sparse-permuted order) +// muse_glimmer_sp_perm [n_tok] i32 : window grouping permutation (applied after ln_pre) +// muse_glimmer_inv_perm [n_tok] i32 : inverse of sp_perm (applied after blocks) +// muse_glimmer_ds_perm [n_tok] i32 : pixel-shuffle gather (original order) +// muse_glimmer_sp_mask [n_tok, n_tok] f32 : block-diagonal window mask (sparse layers) +ggml_cgraph * clip_graph_muse_glimmer::build() { const int ds = hparams.n_merge; // downsample factor (2) - const int sf = hparams.onyx_sparse_factor; // 4 + const int sf = hparams.muse_glimmer_sparse_factor; // 4 const int n_tok = n_patches; const int n_out = (n_patches_x / ds) * (n_patches_y / ds); const float rope_base = hparams.rope_theta; // 10000 @@ -25,14 +25,14 @@ ggml_cgraph * clip_graph_onyx::build() { return t; }; - ggml_tensor * pos_w = inp_i32("onyx_pos_w", n_tok); - ggml_tensor * pos_h = inp_i32("onyx_pos_h", n_tok); - ggml_tensor * sp_perm = inp_i32("onyx_sp_perm", n_tok); - ggml_tensor * inv_perm = inp_i32("onyx_inv_perm", n_tok); - ggml_tensor * ds_perm = inp_i32("onyx_ds_perm", n_tok); + ggml_tensor * pos_w = inp_i32("muse_glimmer_pos_w", n_tok); + ggml_tensor * pos_h = inp_i32("muse_glimmer_pos_h", n_tok); + ggml_tensor * sp_perm = inp_i32("muse_glimmer_sp_perm", n_tok); + ggml_tensor * inv_perm = inp_i32("muse_glimmer_inv_perm", n_tok); + ggml_tensor * ds_perm = inp_i32("muse_glimmer_ds_perm", n_tok); ggml_tensor * sp_mask = ggml_new_tensor_2d(ctx0, GGML_TYPE_F32, n_tok, n_tok); - ggml_set_name(sp_mask, "onyx_sp_mask"); + ggml_set_name(sp_mask, "muse_glimmer_sp_mask"); ggml_set_input(sp_mask); // patchify via build_inp (conv2d over raw pixels) + bilinear-resized learned pos-emb diff --git a/tools/mtmd/mtmd-image.cpp b/tools/mtmd/mtmd-image.cpp index 7c3c585ad0..6814d333d6 100644 --- a/tools/mtmd/mtmd-image.cpp +++ b/tools/mtmd/mtmd-image.cpp @@ -1593,11 +1593,11 @@ mtmd_image_preproc_out mtmd_image_preprocessor_granite::preprocess(const clip_im } // -// mtmd_image_preprocessor_onyx +// mtmd_image_preprocessor_muse_glimmer // -// Replicates transformers' get_aspect_ratio_preserving_size (image_processing_onyx.py) -static clip_image_size onyx_grid_size(int img_w, int img_h, int patch_hw, int max_tokens) { +// Replicates transformers' get_aspect_ratio_preserving_size +static clip_image_size muse_glimmer_grid_size(int img_w, int img_h, int patch_hw, int max_tokens) { double i_nph = (double) img_h / patch_hw; double i_npw = (double) img_w / patch_hw; const double ratio = i_nph > 0.0 ? i_npw / i_nph : 1.0; @@ -1635,14 +1635,14 @@ static clip_image_size onyx_grid_size(int img_w, int img_h, int patch_hw, int ma return clip_image_size{ best_npw * patch_hw, best_nph * patch_hw }; } -mtmd_image_preproc_out mtmd_image_preprocessor_onyx::preprocess(const clip_image_u8 & img) { +mtmd_image_preproc_out mtmd_image_preprocessor_muse_glimmer::preprocess(const clip_image_u8 & img) { const int patch_hw = hparams.patch_size * hparams.n_merge; const int patch_area = hparams.patch_size * hparams.patch_size * hparams.n_merge * hparams.n_merge; GGML_ASSERT(patch_area > 0 && hparams.image_max_pixels > 0); const int max_tokens = hparams.image_max_pixels / patch_area; const clip_image_size original_size = img.get_size(); - const clip_image_size target_size = onyx_grid_size( + const clip_image_size target_size = muse_glimmer_grid_size( original_size.width, original_size.height, patch_hw, max_tokens); // PIL resizes directly to (target_w, target_h) -- a stretch, no padding. diff --git a/tools/mtmd/mtmd-image.h b/tools/mtmd/mtmd-image.h index 26fd84d485..2467f85511 100644 --- a/tools/mtmd/mtmd-image.h +++ b/tools/mtmd/mtmd-image.h @@ -226,7 +226,7 @@ struct mtmd_image_preprocessor_granite : mtmd_image_preprocessor_llava_uhd { }; // pick the patch grid closest to the input aspect ratio under the per-image token cap, stretch-resize. -struct mtmd_image_preprocessor_onyx : mtmd_image_preprocessor { - mtmd_image_preprocessor_onyx(const clip_ctx * ctx) : mtmd_image_preprocessor(ctx) {} +struct mtmd_image_preprocessor_muse_glimmer : mtmd_image_preprocessor { + mtmd_image_preprocessor_muse_glimmer(const clip_ctx * ctx) : mtmd_image_preprocessor(ctx) {} mtmd_image_preproc_out preprocess(const clip_image_u8 & img) override; }; diff --git a/tools/mtmd/mtmd.cpp b/tools/mtmd/mtmd.cpp index bd81780538..66b63b68f0 100644 --- a/tools/mtmd/mtmd.cpp +++ b/tools/mtmd/mtmd.cpp @@ -470,11 +470,11 @@ struct mtmd_context { img_end = "]<]end of image[>["; image_preproc = std::make_unique(ctx_v); } break; - case PROJECTOR_TYPE_ONYX: + case PROJECTOR_TYPE_MUSE_GLIMMER: { img_beg = "<|image_start|>"; img_end = "<|image_end|>"; - image_preproc = std::make_unique(ctx_v); + image_preproc = std::make_unique(ctx_v); } break; case PROJECTOR_TYPE_YOUTUVL: {