mirror of
https://github.com/ggml-org/llama.cpp.git
synced 2026-10-02 19:07:25 -05:00
Rename
This commit is contained in:
+6
-61
@@ -2902,16 +2902,11 @@ static common_chat_params common_chat_params_init_minicpm5(const common_chat_tem
|
||||
// - final answer: to=user, terminated by <|eot|>
|
||||
// The generation prompt is just "<|start|>assistant"; the model emits its own
|
||||
// " to=...<|message|>".
|
||||
static common_chat_params common_chat_params_init_onyx(const common_chat_template & tmpl,
|
||||
const autoparser::generation_params & inputs) {
|
||||
static common_chat_params common_chat_params_init_muse_glimmer(const common_chat_template & tmpl,
|
||||
const autoparser::generation_params & inputs) {
|
||||
common_chat_params data;
|
||||
|
||||
data.prompt = common_chat_template_direct_apply_impl(tmpl, inputs);
|
||||
// generation_prompt is the exact assistant-turn prefix the server strips when recovering
|
||||
// the model's output from the full decoded text. Hardcode it to "<|start|>assistant" (the
|
||||
// model emits its own " to=<recipient><|message|>" after it). Re-rendering the template
|
||||
// with add_generation_prompt would also inject a default system message, which must not
|
||||
// become part of this stripped prefix.
|
||||
data.generation_prompt = "<|start|>assistant";
|
||||
data.format = COMMON_CHAT_FORMAT_PEG_NATIVE;
|
||||
data.supports_thinking = true;
|
||||
@@ -2944,27 +2939,16 @@ static common_chat_params common_chat_params_init_onyx(const common_chat_templat
|
||||
auto extract_reasoning = inputs.reasoning_format != COMMON_REASONING_FORMAT_NONE;
|
||||
|
||||
auto has_tools = inputs.tools.is_array() && !inputs.tools.empty();
|
||||
// Constrained grammar whenever tools are offered. For tool_choice=REQUIRED the grammar is
|
||||
// non-lazy (forces a valid ATEM tool call from the first token). For the auto path it is lazy
|
||||
// and only activates once the model commits to a tool recipient (" to=<tool><|message|>"), so
|
||||
// the free-text reasoning (to=self) and final (to=user) turns generate unconstrained. See the
|
||||
// grammar_triggers below.
|
||||
// Constrained grammar whenever tools are offered.
|
||||
auto include_grammar = has_tools && inputs.tool_choice != COMMON_CHAT_TOOL_CHOICE_NONE;
|
||||
|
||||
auto parser = build_chat_peg_parser([&](common_chat_peg_builder & p) {
|
||||
auto start = p.rule("start", p.literal("<|start|>assistant"));
|
||||
|
||||
// No reasoning to split out and no tools: emit the raw stream as content.
|
||||
// Tool parsing does not depend on reasoning_format. include_grammar (above) forces a tool
|
||||
// call whenever tools are offered, so the parser runs the tool rules even under NONE.
|
||||
// reasoning_format only decides whether the to=self block is tagged reasoning or content.
|
||||
if (!extract_reasoning && !include_grammar) {
|
||||
return start + p.content(p.rest());
|
||||
}
|
||||
|
||||
// chain-of-thought message: " to=self<|message|>{body}<|eom|>". The literals stay
|
||||
// structural so a partial " to=" prefix does not emit a content delta before it resolves
|
||||
// to a tool recipient. Under NONE the body is tagged content instead of reasoning.
|
||||
if (extract_reasoning) {
|
||||
p.rule("analysis", p.literal(" to=self<|message|>") + p.reasoning(p.until("<|eom|>")) + p.literal("<|eom|>"));
|
||||
} else {
|
||||
@@ -2972,29 +2956,10 @@ static common_chat_params common_chat_params_init_onyx(const common_chat_templat
|
||||
}
|
||||
auto analysis = p.ref("analysis");
|
||||
|
||||
// final answer: "[ to=user]<|message|>{content}<|eot|>". The recipient is kept fixed
|
||||
// (not a wildcard) so a partial chain-of-thought (" to=self...") never matches this
|
||||
// branch and leaks into content before its <|eom|> terminator arrives.
|
||||
auto recipient = p.optional(p.literal(" to=user"));
|
||||
auto final_msg = p.rule("final", recipient + p.literal("<|message|>") + p.content(p.until("<|eot|>")));
|
||||
|
||||
// Tool-call turn: " to=<tool><|message|><atem:function_calls>...</atem:function_calls>{term}".
|
||||
// Ported from the ATEM tool-call parser. Only added when tools are offered; otherwise the
|
||||
// grammar and top-level rule are exactly the reasoning/final structure above.
|
||||
//
|
||||
// These tool-call rules assume grammar-constrained input, which is what the
|
||||
// server produces during generation: canonical whitespace and in-schema params,
|
||||
// with no inline preamble because onyx puts preamble in the to=self message.
|
||||
// Parsing unconstrained or historical assistant text is strict on purpose and
|
||||
// will reject the whole tool call rather than guess.
|
||||
if (has_tools && inputs.tool_choice != COMMON_CHAT_TOOL_CHOICE_NONE) {
|
||||
// A tool-call message closes with "</atem:function_calls>", terminated by <|eom|>
|
||||
// (more calls this turn) or <|eot|> (end of turn). We do NOT consume the trailing
|
||||
// <|eot|> -- it fires as the server's end-of-turn EOG stop; the <|eom|> between
|
||||
// parallel calls IS consumed (see tool_calls below).
|
||||
|
||||
// Schema-declared string args are captured verbatim up to the closing tag; anything
|
||||
// else is parsed as JSON (via the per-arg schema rule) before its closing tag.
|
||||
auto string_value = p.ac(
|
||||
p.tool_arg_string_value(p.until("</atem:parameter>")) + p.tool_arg_close(p.literal("</atem:parameter>")),
|
||||
"</atem:parameter>");
|
||||
@@ -3030,13 +2995,6 @@ static common_chat_params common_chat_params_init_onyx(const common_chat_templat
|
||||
args = p.zero_or_more(arg_choice + p.space());
|
||||
}
|
||||
|
||||
// The tool identity is keyed on the <atem:invoke name="..."> (the
|
||||
// actual call), NOT the " to=<recipient>". Onyx models sometimes emit
|
||||
// a recipient that disagrees with the invoke name (e.g. " to=
|
||||
// container.exec" with <atem:invoke name="bash.exec">); accept any
|
||||
// recipient here so such calls are parsed rather than dropped. The
|
||||
// "<atem:function_calls>" literal disambiguates this from the to=user
|
||||
// final turn.
|
||||
auto tool_parser = p.tool(
|
||||
p.tool_open(p.literal(" to=") + p.until("<|message|>") +
|
||||
p.literal("<|message|><atem:function_calls>") + p.space() +
|
||||
@@ -3047,12 +3005,6 @@ static common_chat_params common_chat_params_init_onyx(const common_chat_templat
|
||||
tool_choice |= p.rule("tool-" + name, tool_parser);
|
||||
});
|
||||
|
||||
// Parallel tool calls are consecutive assistant messages separated by <|eom|>, the turn
|
||||
// ending with <|eot|> after the last. Gate the loop on parallel_tool_calls like the other
|
||||
// formats: disabled -> single call; enabled -> multiple. NOTE: this model is unreliable at
|
||||
// single-turn parallel (it often ends the turn after the first call), so
|
||||
// parallel_tool_calls=false (sequential, driven by the agent loop) is recommended; true is
|
||||
// honored but may under-emit.
|
||||
auto tool_calls = inputs.parallel_tool_calls
|
||||
? p.trigger_rule("tool-call", tool_choice + p.zero_or_more(p.literal("<|eom|>") + start + tool_choice))
|
||||
: p.trigger_rule("tool-call", tool_choice);
|
||||
@@ -3079,11 +3031,6 @@ static common_chat_params common_chat_params_init_onyx(const common_chat_templat
|
||||
});
|
||||
parser.build_grammar(builder, data.grammar_lazy);
|
||||
});
|
||||
// Trigger only on a TOOL recipient, not the to=self (reasoning) or to=user (final) turns,
|
||||
// which are free text and would empty the grammar stack. The capture group starts at " to="
|
||||
// (matching where the grammar's tool rule begins) so the replayed input lines up with the
|
||||
// grammar; requiring the trailing "<|message|>" (and excluding self/user before it) means
|
||||
// the pattern only fires once the full tool recipient is known.
|
||||
data.grammar_triggers = {
|
||||
{ COMMON_GRAMMAR_TRIGGER_TYPE_PATTERN,
|
||||
"<\\|start\\|>assistant( to=(?!self<\\|message\\|>)(?!user<\\|message\\|>)[^<]*?<\\|message\\|>)" },
|
||||
@@ -3121,12 +3068,10 @@ std::optional<common_chat_params> common_chat_try_specialized_template(
|
||||
return common_chat_params_init_gpt_oss(tmpl, params);
|
||||
}
|
||||
|
||||
// Onyx format using " to=<recipient>" recipients and <|eom|>/<|eot|> message
|
||||
// terminators. The ATEM tool-call markup "<atem:function_calls>" is unique to
|
||||
// this template.
|
||||
// Muse Glimmer format using " to=<recipient>" recipients and <|eom|>/<|eot|> message terminators.
|
||||
if (src.find("<atem:function_calls>") != std::string::npos && src.find("<|eom|>") != std::string::npos) {
|
||||
LOG_DBG("Using specialized template: Onyx\n");
|
||||
return common_chat_params_init_onyx(tmpl, params);
|
||||
LOG_DBG("Using specialized template: Muse Glimmer\n");
|
||||
return common_chat_params_init_muse_glimmer(tmpl, params);
|
||||
}
|
||||
|
||||
// Functionary v3.2 - uses recipient-based format with >>>recipient\n{content}
|
||||
|
||||
@@ -180,8 +180,8 @@ TEXT_MODEL_MAP: dict[str, str] = {
|
||||
"Olmo3ForCausalLM": "olmo",
|
||||
"OlmoForCausalLM": "olmo",
|
||||
"OlmoeForCausalLM": "olmo",
|
||||
"OnyxAssistantModel": "onyx",
|
||||
"OnyxForConditionalGeneration": "onyx",
|
||||
"MuseGlimmerAssistantModel": "muse-glimmer",
|
||||
"MuseGlimmerForConditionalGeneration": "muse-glimmer",
|
||||
"OpenELMForCausalLM": "openelm",
|
||||
"OrionForCausalLM": "orion",
|
||||
"PLMForCausalLM": "plm",
|
||||
@@ -296,7 +296,7 @@ MMPROJ_MODEL_MAP: dict[str, str] = {
|
||||
"MiniCPMV4_6ForConditionalGeneration": "minicpm",
|
||||
"Mistral3ForConditionalGeneration": "llava",
|
||||
"NemotronH_Nano_VL_V2": "nemotron",
|
||||
"OnyxForConditionalGeneration": "onyx",
|
||||
"MuseGlimmerForConditionalGeneration": "muse-glimmer",
|
||||
"PaddleOCRVisionModel": "ernie",
|
||||
"Phi4ForCausalLMV": "phi",
|
||||
"Qwen2AudioForConditionalGeneration": "ultravox",
|
||||
|
||||
@@ -23,9 +23,9 @@ def _unpermute_for_rope(tensor: "Tensor", n_heads: int) -> "Tensor":
|
||||
raise ValueError(f"_unpermute_for_rope: unexpected shape {tuple(tensor.shape)}")
|
||||
|
||||
|
||||
@ModelBase.register("OnyxForConditionalGeneration")
|
||||
class OnyxModel(TextModel):
|
||||
model_arch = gguf.MODEL_ARCH.ONYX
|
||||
@ModelBase.register("MuseGlimmerForConditionalGeneration")
|
||||
class MuseGlimmerModel(TextModel):
|
||||
model_arch = gguf.MODEL_ARCH.MUSE_GLIMMER
|
||||
|
||||
def norm_shift(self, name: str) -> float:
|
||||
# All four layer norms use 1, the final norm uses 0.
|
||||
@@ -61,7 +61,7 @@ class OnyxModel(TextModel):
|
||||
data_torch = _unpermute_for_rope(data_torch, int(self.hparams["num_key_value_heads"]))
|
||||
|
||||
# Synthesize QK-norm weights to absorb qk_scale_factor.
|
||||
# Onyx implementation: scaleless RMSNorm followed by qk_scale_factor..
|
||||
# MuseGlimmer implementation: scaleless RMSNorm followed by qk_scale_factor..
|
||||
if bid is not None and name.endswith(f"model.layers.{bid}.self_attn.q_proj.weight"):
|
||||
head_dim = self.hparams["head_dim"]
|
||||
q_scale = float(self.hparams["qk_scale_factor"])
|
||||
@@ -77,13 +77,13 @@ class OnyxModel(TextModel):
|
||||
yield from super().modify_tensors(data_torch, name, bid)
|
||||
|
||||
|
||||
@ModelBase.register("OnyxForConditionalGeneration")
|
||||
class OnyxVisionModel(MmprojModel):
|
||||
@ModelBase.register("MuseGlimmerForConditionalGeneration")
|
||||
class MuseGlimmerVisionModel(MmprojModel):
|
||||
def get_vision_config(self) -> dict[str, Any] | None:
|
||||
c = self.global_config.get("vision_config")
|
||||
if not c:
|
||||
return None
|
||||
# Onyx actually uses dynamic size, initialize with nominal size
|
||||
# MuseGlimmer actually uses dynamic size, initialize with nominal size
|
||||
image_size = c["pos_emb_height"] * c["patch_size"] * c["merge_size"]
|
||||
return {**c, "image_size": image_size}
|
||||
|
||||
@@ -91,7 +91,7 @@ class OnyxVisionModel(MmprojModel):
|
||||
super().set_gguf_parameters()
|
||||
c = self.hparams_vision # enriched vision_config from get_vision_config()
|
||||
|
||||
self.gguf_writer.add_clip_projector_type(gguf.VisionProjectorType.ONYX)
|
||||
self.gguf_writer.add_clip_projector_type(gguf.VisionProjectorType.MUSE_GLIMMER)
|
||||
self.gguf_writer.add_vision_attention_layernorm_eps(float(c["layer_norm_eps"]))
|
||||
self.gguf_writer.add_vision_spatial_merge_size(int(c["merge_size"]))
|
||||
|
||||
@@ -128,15 +128,15 @@ class OnyxVisionModel(MmprojModel):
|
||||
yield (self.map_tensor_name(name), data_torch)
|
||||
|
||||
|
||||
@ModelBase.register("OnyxAssistantModel")
|
||||
class OnyxAssistantModel(TextModel):
|
||||
@ModelBase.register("MuseGlimmerAssistantModel")
|
||||
class MuseGlimmerAssistantModel(TextModel):
|
||||
model_arch = gguf.MODEL_ARCH.DFLASH
|
||||
|
||||
def set_vocab(self):
|
||||
if self.target_model_dir is None:
|
||||
raise ValueError(
|
||||
"OnyxAssistant (DFlash drafter) requires --target-model-dir pointing to the "
|
||||
"target Onyx HF directory"
|
||||
"MuseGlimmerAssistant (DFlash drafter) requires --target-model-dir pointing to the "
|
||||
"target MuseGlimmer HF directory"
|
||||
)
|
||||
|
||||
original_dir = self.dir_model
|
||||
@@ -488,7 +488,7 @@ class MODEL_ARCH(IntEnum):
|
||||
OLMO = auto()
|
||||
OLMO2 = auto()
|
||||
OLMOE = auto()
|
||||
ONYX = auto()
|
||||
MUSE_GLIMMER = auto()
|
||||
OPENELM = auto()
|
||||
ARCTIC = auto()
|
||||
DEEPSEEK = auto()
|
||||
@@ -1100,7 +1100,7 @@ MODEL_ARCH_NAMES: dict[MODEL_ARCH, str] = {
|
||||
MODEL_ARCH.OLMO: "olmo",
|
||||
MODEL_ARCH.OLMO2: "olmo2",
|
||||
MODEL_ARCH.OLMOE: "olmoe",
|
||||
MODEL_ARCH.ONYX: "onyx",
|
||||
MODEL_ARCH.MUSE_GLIMMER: "muse-glimmer",
|
||||
MODEL_ARCH.OPENELM: "openelm",
|
||||
MODEL_ARCH.ARCTIC: "arctic",
|
||||
MODEL_ARCH.DEEPSEEK: "deepseek",
|
||||
@@ -3129,7 +3129,7 @@ MODEL_TENSORS: dict[MODEL_ARCH, list[MODEL_TENSOR]] = {
|
||||
MODEL_TENSOR.FFN_UP_EXP,
|
||||
MODEL_TENSOR.FFN_DOWN_EXP,
|
||||
],
|
||||
MODEL_ARCH.ONYX: [
|
||||
MODEL_ARCH.MUSE_GLIMMER: [
|
||||
MODEL_TENSOR.TOKEN_EMBD,
|
||||
MODEL_TENSOR.OUTPUT,
|
||||
MODEL_TENSOR.OUTPUT_NORM,
|
||||
@@ -4906,7 +4906,7 @@ class VisionProjectorType:
|
||||
MIMOVL = "mimovl"
|
||||
MIMO_AUDIO = "mimo_audio"
|
||||
GRANITE4_VISION = "granite4_vision"
|
||||
ONYX = "onyx"
|
||||
MUSE_GLIMMER = "muse-glimmer"
|
||||
|
||||
|
||||
# Items here are (block size, type size)
|
||||
|
||||
@@ -382,7 +382,7 @@ class TensorNameMap:
|
||||
),
|
||||
|
||||
MODEL_TENSOR.ATTN_GATE: (
|
||||
"model.layers.{bid}.self_attn.gate_proj", # afmoe onyx
|
||||
"model.layers.{bid}.self_attn.gate_proj", # afmoe muse-glimmer
|
||||
"model.layers.{bid}.linear_attn.in_proj_z", # qwen3.5
|
||||
"model.layers.{bid}.self_attn.g_proj", # step3.5 head-wise attention gate
|
||||
),
|
||||
@@ -1298,12 +1298,12 @@ class TensorNameMap:
|
||||
"encoder.final_layer_norm", # t5
|
||||
"layer_norm", # neobert
|
||||
"model.hidden_norm", # dflash
|
||||
"encoder.output_norm_enc", # dflash (transformers OnyxAssistant)
|
||||
"encoder.output_norm_enc", # dflash (transformers MuseGlimmerAssistant)
|
||||
),
|
||||
|
||||
MODEL_TENSOR.FC: (
|
||||
"model.fc", # dflash
|
||||
"encoder.fc", # dflash (transformers OnyxAssistant)
|
||||
"encoder.fc", # dflash (transformers MuseGlimmerAssistant)
|
||||
),
|
||||
|
||||
MODEL_TENSOR.DSPARK_MARKOV_W1: (
|
||||
@@ -1469,7 +1469,7 @@ class TensorNameMap:
|
||||
"vision_tower.patch_embed.patchifier.proj", # dots.ocr
|
||||
"vision_model.conv1", # Step3-VL
|
||||
"model.vision_embedder.patch_dense", # gemma4 unified
|
||||
"model.vision_tower.patch_embedder.patch_embedding", # onyx
|
||||
"model.vision_tower.patch_embedder.patch_embedding", # muse-glimmer
|
||||
),
|
||||
|
||||
MODEL_TENSOR.V_ENC_EMBD_NORM: (
|
||||
@@ -1538,7 +1538,7 @@ class TensorNameMap:
|
||||
"model.vision_model.transformer.layers.{bid}.self_attn.q_proj", # Deepseek-OCR CLIP, generated
|
||||
"vision_model.model.layers.{bid}.self_attn.q_proj.linear", # gemma4
|
||||
"model.qwen2_model.model.model.layers.{bid}.self_attn.q_proj", # Deepseek-OCR-2 qwen2
|
||||
"model.vision_tower.layers.{bid}.attn.q_proj", # onyx
|
||||
"model.vision_tower.layers.{bid}.attn.q_proj", # muse-glimmer
|
||||
),
|
||||
|
||||
MODEL_TENSOR.V_ENC_ATTN_Q_NORM: (
|
||||
@@ -1565,7 +1565,7 @@ class TensorNameMap:
|
||||
"siglip2.vision_model.encoder.layers.{bid}.self_attn.k_proj",
|
||||
"vision_model.model.layers.{bid}.self_attn.k_proj.linear", # gemma4
|
||||
"model.qwen2_model.model.model.layers.{bid}.self_attn.k_proj", # Deepseek-OCR-2 qwen2
|
||||
"model.vision_tower.layers.{bid}.attn.k_proj", # onyx
|
||||
"model.vision_tower.layers.{bid}.attn.k_proj", # muse-glimmer
|
||||
),
|
||||
|
||||
MODEL_TENSOR.V_ENC_ATTN_K_NORM: (
|
||||
@@ -1592,7 +1592,7 @@ class TensorNameMap:
|
||||
"model.vision_model.transformer.layers.{bid}.self_attn.v_proj", # Deepseek-OCR CLIP, generated
|
||||
"vision_model.model.layers.{bid}.self_attn.v_proj.linear", # gemma4
|
||||
"model.qwen2_model.model.model.layers.{bid}.self_attn.v_proj", # Deepseek-OCR-2 qwen2
|
||||
"model.vision_tower.layers.{bid}.attn.v_proj", # onyx
|
||||
"model.vision_tower.layers.{bid}.attn.v_proj", # muse-glimmer
|
||||
),
|
||||
|
||||
MODEL_TENSOR.V_ENC_INPUT_NORM: (
|
||||
@@ -1616,7 +1616,7 @@ class TensorNameMap:
|
||||
"vision_tower.blocks.{bid}.norm1", # dots.ocr
|
||||
"vision_model.transformer.resblocks.{bid}.ln_1", # Step3-VL
|
||||
"model.qwen2_model.model.model.layers.{bid}.input_layernorm", # Deepseek-OCR-2 qwen2
|
||||
"model.vision_tower.layers.{bid}.norm1", # onyx
|
||||
"model.vision_tower.layers.{bid}.norm1", # muse-glimmer
|
||||
),
|
||||
|
||||
MODEL_TENSOR.V_ENC_ATTN_O: (
|
||||
@@ -1642,7 +1642,7 @@ class TensorNameMap:
|
||||
"vision_model.model.layers.{bid}.self_attn.o_proj.linear", # gemma4
|
||||
"vision_tower.blocks.{bid}.attn.proj", # dots.ocr
|
||||
"vision_model.transformer.resblocks.{bid}.attn.out_proj", # Step3-VL
|
||||
"model.vision_tower.layers.{bid}.attn.proj", # onyx
|
||||
"model.vision_tower.layers.{bid}.attn.proj", # muse-glimmer
|
||||
),
|
||||
|
||||
MODEL_TENSOR.V_ENC_ATTN_SINKS: (
|
||||
@@ -1671,7 +1671,7 @@ class TensorNameMap:
|
||||
"vision_tower.blocks.{bid}.norm2", # dots.ocr
|
||||
"vision_model.transformer.resblocks.{bid}.ln_2", # Step3-VL
|
||||
"model.qwen2_model.model.model.layers.{bid}.post_attention_layernorm", # Deepseek-OCR-2 qwen2
|
||||
"model.vision_tower.layers.{bid}.norm2", # onyx
|
||||
"model.vision_tower.layers.{bid}.norm2", # muse-glimmer
|
||||
),
|
||||
|
||||
MODEL_TENSOR.V_ENC_FFN_UP: (
|
||||
@@ -1696,7 +1696,7 @@ class TensorNameMap:
|
||||
"vision_model.model.layers.{bid}.mlp.up_proj", # gemma4
|
||||
"vision_model.transformer.resblocks.{bid}.mlp.c_fc", # Step3-VL
|
||||
"model.qwen2_model.model.model.layers.{bid}.mlp.up_proj", # Deepseek-OCR-2 qwen2
|
||||
"model.vision_tower.layers.{bid}.mlp.fc1", # onyx
|
||||
"model.vision_tower.layers.{bid}.mlp.fc1", # muse-glimmer
|
||||
),
|
||||
|
||||
MODEL_TENSOR.V_ENC_FFN_GATE: (
|
||||
@@ -1729,7 +1729,7 @@ class TensorNameMap:
|
||||
"model.qwen2_model.model.model.layers.{bid}.mlp.down_proj" , # Deepseek-OCR-2 qwen2
|
||||
"vision_model.model.layers.{bid}.mlp.down_proj", # gemma4
|
||||
"vision_model.transformer.resblocks.{bid}.mlp.c_proj", # Step3-VL
|
||||
"model.vision_tower.layers.{bid}.mlp.fc2", # onyx
|
||||
"model.vision_tower.layers.{bid}.mlp.fc2", # muse-glimmer
|
||||
),
|
||||
|
||||
MODEL_TENSOR.V_ENC_ATTN_POST_NORM: (
|
||||
@@ -1764,7 +1764,7 @@ class TensorNameMap:
|
||||
"model.vision_model.pre_layrnorm", # Deepseek-OCR CLIP
|
||||
"vision_tower.patch_embed.patchifier.norm", # dots.ocr
|
||||
"vision_model.ln_pre", # Step3-VL
|
||||
"model.vision_tower.ln_pre", # onyx
|
||||
"model.vision_tower.ln_pre", # muse-glimmer
|
||||
),
|
||||
|
||||
MODEL_TENSOR.V_POST_NORM: (
|
||||
@@ -1778,7 +1778,7 @@ class TensorNameMap:
|
||||
"visual.post_layernorm", # glm4v
|
||||
"siglip2.vision_model.post_layernorm",
|
||||
"model.qwen2_model.model.model.norm", # Deepseek-OCR-2 qwen2
|
||||
"model.vision_tower.ln_post", # onyx
|
||||
"model.vision_tower.ln_post", # muse-glimmer
|
||||
),
|
||||
|
||||
MODEL_TENSOR.V_MM_POST_NORM: (
|
||||
|
||||
+1
-1
@@ -71,7 +71,7 @@ static const std::map<llm_arch, const char *> LLM_ARCH_NAMES = {
|
||||
{ LLM_ARCH_OLMO, "olmo" },
|
||||
{ LLM_ARCH_OLMO2, "olmo2" },
|
||||
{ LLM_ARCH_OLMOE, "olmoe" },
|
||||
{ LLM_ARCH_ONYX, "onyx" },
|
||||
{ LLM_ARCH_MUSE_GLIMMER, "muse-glimmer" },
|
||||
{ LLM_ARCH_OPENELM, "openelm" },
|
||||
{ LLM_ARCH_ARCTIC, "arctic" },
|
||||
{ LLM_ARCH_DEEPSEEK, "deepseek" },
|
||||
|
||||
+1
-1
@@ -76,7 +76,7 @@ enum llm_arch {
|
||||
LLM_ARCH_OLMO,
|
||||
LLM_ARCH_OLMO2,
|
||||
LLM_ARCH_OLMOE,
|
||||
LLM_ARCH_ONYX,
|
||||
LLM_ARCH_MUSE_GLIMMER,
|
||||
LLM_ARCH_OPENELM,
|
||||
LLM_ARCH_ARCTIC,
|
||||
LLM_ARCH_DEEPSEEK,
|
||||
|
||||
+3
-3
@@ -171,8 +171,8 @@ static llama_model * llama_model_mapping(llm_arch arch, const llama_model_params
|
||||
return new llama_model_olmo2(params);
|
||||
case LLM_ARCH_OLMOE:
|
||||
return new llama_model_olmoe(params);
|
||||
case LLM_ARCH_ONYX:
|
||||
return new llama_model_onyx(params);
|
||||
case LLM_ARCH_MUSE_GLIMMER:
|
||||
return new llama_model_muse_glimmer(params);
|
||||
case LLM_ARCH_OPENELM:
|
||||
return new llama_model_openelm(params);
|
||||
case LLM_ARCH_GPTNEOX:
|
||||
@@ -2530,7 +2530,7 @@ llama_rope_type llama_model_rope_type(const llama_model * model) {
|
||||
case LLM_ARCH_DEEPSEEK2OCR:
|
||||
case LLM_ARCH_DEEPSEEK32:
|
||||
case LLM_ARCH_DEEPSEEK4:
|
||||
case LLM_ARCH_ONYX:
|
||||
case LLM_ARCH_MUSE_GLIMMER:
|
||||
case LLM_ARCH_PLM:
|
||||
case LLM_ARCH_CHATGLM:
|
||||
case LLM_ARCH_GRANITE:
|
||||
|
||||
+2
-2
@@ -1023,8 +1023,8 @@ struct llama_model_olmoe : public llama_model_base {
|
||||
};
|
||||
|
||||
|
||||
struct llama_model_onyx : public llama_model_base {
|
||||
llama_model_onyx(const struct llama_model_params & params) : llama_model_base(params) {}
|
||||
struct llama_model_muse_glimmer : public llama_model_base {
|
||||
llama_model_muse_glimmer(const struct llama_model_params & params) : llama_model_base(params) {}
|
||||
void load_arch_hparams(llama_model_loader & ml) override;
|
||||
void load_arch_tensors(llama_model_loader & ml) override;
|
||||
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
#include "models.h"
|
||||
|
||||
void llama_model_onyx::load_arch_hparams(llama_model_loader & ml) {
|
||||
void llama_model_muse_glimmer::load_arch_hparams(llama_model_loader & ml) {
|
||||
ml.get_key(LLM_KV_ATTENTION_LAYERNORM_RMS_EPS, hparams.f_norm_rms_eps);
|
||||
ml.get_key(LLM_KV_ATTENTION_SLIDING_WINDOW, hparams.n_swa, false);
|
||||
ml.get_key(LLM_KV_FINAL_LOGIT_SOFTCAPPING, hparams.f_final_logit_softcapping, false);
|
||||
@@ -23,7 +23,7 @@ void llama_model_onyx::load_arch_hparams(llama_model_loader & ml) {
|
||||
}
|
||||
}
|
||||
|
||||
void llama_model_onyx::load_arch_tensors(llama_model_loader &) {
|
||||
void llama_model_muse_glimmer::load_arch_tensors(llama_model_loader &) {
|
||||
LLAMA_LOAD_LOCALS;
|
||||
|
||||
tok_embd = create_tensor(tn(LLM_TENSOR_TOKEN_EMBD, "weight"), {n_embd, n_vocab}, 0);
|
||||
@@ -33,7 +33,7 @@ void llama_model_onyx::load_arch_tensors(llama_model_loader &) {
|
||||
for (int i = 0; i < n_layer; ++i) {
|
||||
auto & layer = layers[i];
|
||||
|
||||
// Pre/post-attention norms (Onyx's `weight + 1` applied at conversion time).
|
||||
// Pre/post-attention norms (Muse Glimmer's `weight + 1` applied at conversion time).
|
||||
layer.attn_norm = create_tensor(tn(LLM_TENSOR_ATTN_NORM, "weight", i), {n_embd}, 0);
|
||||
layer.attn_post_norm = create_tensor(tn(LLM_TENSOR_ATTN_POST_NORM, "weight", i), {n_embd}, 0);
|
||||
|
||||
@@ -59,7 +59,7 @@ void llama_model_onyx::load_arch_tensors(llama_model_loader &) {
|
||||
}
|
||||
}
|
||||
|
||||
llama_model_onyx::graph::graph(const llama_model & model, const llm_graph_params & params)
|
||||
llama_model_muse_glimmer::graph::graph(const llama_model & model, const llm_graph_params & params)
|
||||
: llm_graph_context(params) {
|
||||
const int64_t n_embd_head = hparams.n_embd_head_v();
|
||||
GGML_ASSERT(n_embd_head == hparams.n_embd_head_k());
|
||||
@@ -203,6 +203,6 @@ llama_model_onyx::graph::graph(const llama_model & model, const llm_graph_params
|
||||
ggml_build_forward_expand(gf, cur);
|
||||
}
|
||||
|
||||
std::unique_ptr<llm_graph_context> llama_model_onyx::build_arch_graph(const llm_graph_params & params) const {
|
||||
std::unique_ptr<llm_graph_context> llama_model_muse_glimmer::build_arch_graph(const llm_graph_params & params) const {
|
||||
return std::make_unique<graph>(*this, params);
|
||||
}
|
||||
@@ -41,7 +41,7 @@ add_library(mtmd
|
||||
models/kimivl.cpp
|
||||
models/kimik25.cpp
|
||||
models/nemotron-v2-vl.cpp
|
||||
models/onyx.cpp
|
||||
models/muse-glimmer.cpp
|
||||
models/llama4.cpp
|
||||
models/llava.cpp
|
||||
models/minicpmv.cpp
|
||||
|
||||
@@ -408,7 +408,7 @@ enum projector_type {
|
||||
PROJECTOR_TYPE_MINIMAX_M3,
|
||||
PROJECTOR_TYPE_GRANITE4_VISION,
|
||||
PROJECTOR_TYPE_MIMO_AUDIO,
|
||||
PROJECTOR_TYPE_ONYX,
|
||||
PROJECTOR_TYPE_MUSE_GLIMMER,
|
||||
PROJECTOR_TYPE_UNKNOWN,
|
||||
};
|
||||
|
||||
@@ -466,7 +466,7 @@ static std::map<projector_type, std::string> PROJECTOR_TYPE_NAMES = {
|
||||
{ PROJECTOR_TYPE_GRANITE4_VISION, "granite4_vision"},
|
||||
{ PROJECTOR_TYPE_MIMO_AUDIO, "mimo_audio"},
|
||||
{ PROJECTOR_TYPE_PARAKEET, "parakeet"},
|
||||
{ PROJECTOR_TYPE_ONYX, "onyx"},
|
||||
{ PROJECTOR_TYPE_MUSE_GLIMMER, "muse-glimmer"},
|
||||
};
|
||||
|
||||
static projector_type clip_projector_type_from_string(const std::string & str) {
|
||||
|
||||
@@ -109,9 +109,10 @@ struct clip_hparams {
|
||||
int32_t downsample_query_side;
|
||||
int32_t downsample_window_side;
|
||||
|
||||
// Onyx vision (per-block sparse-window pattern, learned pos-emb, patch-temporal)
|
||||
int32_t onyx_patch_temporal = 0;
|
||||
int32_t onyx_sparse_factor = 0;
|
||||
// Muse Glimmer vision (per-block sparse-window pattern, learned pos-emb, patch-temporal)
|
||||
// NOTE: these perhaps shouldn't have the architecture prefix
|
||||
int32_t muse_glimmer_patch_temporal = 0;
|
||||
int32_t muse_glimmer_sparse_factor = 0;
|
||||
|
||||
// audio
|
||||
int32_t n_mel_bins = 0; // whisper preprocessor
|
||||
|
||||
+17
-17
@@ -928,9 +928,9 @@ static std::unique_ptr<clip_graph> clip_get_graph_builder(clip_ctx * ctx, const
|
||||
{
|
||||
builder = std::make_unique<clip_graph_minimax_m3>(ctx, img);
|
||||
} break;
|
||||
case PROJECTOR_TYPE_ONYX:
|
||||
case PROJECTOR_TYPE_MUSE_GLIMMER:
|
||||
{
|
||||
builder = std::make_unique<clip_graph_onyx>(ctx, img);
|
||||
builder = std::make_unique<clip_graph_muse_glimmer>(ctx, img);
|
||||
} break;
|
||||
case PROJECTOR_TYPE_STEP3VL:
|
||||
{
|
||||
@@ -1520,13 +1520,13 @@ struct clip_model_loader {
|
||||
hparams.set_limit_image_tokens(8, 576);
|
||||
hparams.set_warmup_n_tokens(16*16);
|
||||
} break;
|
||||
case PROJECTOR_TYPE_ONYX:
|
||||
case PROJECTOR_TYPE_MUSE_GLIMMER:
|
||||
{
|
||||
hparams.n_merge = 2; // pixel-shuffle downsample after the ViT
|
||||
hparams.image_resize_algo = RESIZE_ALGO_LANCZOS;
|
||||
hparams.rope_theta = 10000.0f;
|
||||
hparams.onyx_patch_temporal = 2; // Onyx-arch invariant
|
||||
hparams.onyx_sparse_factor = 4; // 3 sparse layers + 1 global, repeating
|
||||
hparams.muse_glimmer_patch_temporal = 2;
|
||||
hparams.muse_glimmer_sparse_factor = 4; // 3 sparse layers + 1 global, repeating
|
||||
get_u32(KEY_SPATIAL_MERGE_SIZE, hparams.n_merge, false);
|
||||
hparams.set_limit_image_tokens(1, 4096);
|
||||
hparams.set_warmup_n_tokens(32*32);
|
||||
@@ -2233,7 +2233,7 @@ struct clip_model_loader {
|
||||
model.mm_merger_fc2_w = get_tensor(string_format(TN_MM_MERGER_FC2, "weight"));
|
||||
model.mm_merger_fc2_b = get_tensor(string_format(TN_MM_MERGER_FC2, "bias"));
|
||||
} break;
|
||||
case PROJECTOR_TYPE_ONYX:
|
||||
case PROJECTOR_TYPE_MUSE_GLIMMER:
|
||||
{
|
||||
// 3-linear MLP: fc -> erf-GELU -> proj -> erf-GELU -> vision_proj (into LLM residual dim)
|
||||
model.mm_0_w = get_tensor(string_format(TN_LLAVA_PROJ, 0, "weight"));
|
||||
@@ -3520,7 +3520,7 @@ int clip_n_output_tokens_x(const clip_ctx * ctx, const clip_image_f32 * img) {
|
||||
case PROJECTOR_TYPE_PADDLEOCR:
|
||||
case PROJECTOR_TYPE_HUNYUANVL:
|
||||
case PROJECTOR_TYPE_YOUTUVL:
|
||||
case PROJECTOR_TYPE_ONYX:
|
||||
case PROJECTOR_TYPE_MUSE_GLIMMER:
|
||||
return (img->nx() / params.patch_size) / 2;
|
||||
case PROJECTOR_TYPE_STEP3VL:
|
||||
return img->nx() / (params.patch_size * params.n_merge);
|
||||
@@ -3546,7 +3546,7 @@ int clip_n_output_tokens_y(const clip_ctx * ctx, const clip_image_f32 * img) {
|
||||
case PROJECTOR_TYPE_PADDLEOCR:
|
||||
case PROJECTOR_TYPE_HUNYUANVL:
|
||||
case PROJECTOR_TYPE_YOUTUVL:
|
||||
case PROJECTOR_TYPE_ONYX:
|
||||
case PROJECTOR_TYPE_MUSE_GLIMMER:
|
||||
return (img->ny() / params.patch_size) / 2;
|
||||
case PROJECTOR_TYPE_STEP3VL:
|
||||
return img->ny() / (params.patch_size * params.n_merge);
|
||||
@@ -3626,7 +3626,7 @@ int clip_n_output_tokens(const clip_ctx * ctx, const clip_image_f32 * img) {
|
||||
case PROJECTOR_TYPE_MINIMAX_M3:
|
||||
case PROJECTOR_TYPE_GLM4V:
|
||||
case PROJECTOR_TYPE_YOUTUVL:
|
||||
case PROJECTOR_TYPE_ONYX:
|
||||
case PROJECTOR_TYPE_MUSE_GLIMMER:
|
||||
{
|
||||
// dynamic size (2 conv, so double patch size)
|
||||
int x_patch = img->nx() / (params.patch_size * 2);
|
||||
@@ -3953,7 +3953,7 @@ bool clip_image_batch_encode(clip_ctx * ctx, int n_threads, const clip_image_f32
|
||||
|
||||
// set input per projector
|
||||
switch (ctx->model.proj_type) {
|
||||
case PROJECTOR_TYPE_ONYX:
|
||||
case PROJECTOR_TYPE_MUSE_GLIMMER:
|
||||
{
|
||||
const int grid_w = pos_w; // image_size_width / patch_size
|
||||
const int grid_h = pos_h; // image_size_height / patch_size
|
||||
@@ -3990,10 +3990,10 @@ bool clip_image_batch_encode(clip_ctx * ctx, int n_threads, const clip_image_f32
|
||||
rpos_h[i] = (orig / grid_w) + 1;
|
||||
inv_perm[orig] = i;
|
||||
}
|
||||
set_input_i32("onyx_sp_perm", sp_perm);
|
||||
set_input_i32("onyx_inv_perm", inv_perm);
|
||||
set_input_i32("onyx_pos_w", rpos_w);
|
||||
set_input_i32("onyx_pos_h", rpos_h);
|
||||
set_input_i32("muse_glimmer_sp_perm", sp_perm);
|
||||
set_input_i32("muse_glimmer_inv_perm", inv_perm);
|
||||
set_input_i32("muse_glimmer_pos_w", rpos_w);
|
||||
set_input_i32("muse_glimmer_pos_h", rpos_h);
|
||||
|
||||
// block-diagonal window mask (permuted order)
|
||||
std::vector<float> sp_mask((size_t) n_tok * n_tok, -INFINITY);
|
||||
@@ -4006,7 +4006,7 @@ bool clip_image_batch_encode(clip_ctx * ctx, int n_threads, const clip_image_f32
|
||||
off += s;
|
||||
}
|
||||
}
|
||||
set_input_f32("onyx_sp_mask", sp_mask);
|
||||
set_input_f32("muse_glimmer_sp_mask", sp_mask);
|
||||
|
||||
// pixel-shuffle gather (original order): f*f spatial neighbours grouped
|
||||
std::vector<int32_t> dsp; dsp.reserve(n_tok);
|
||||
@@ -4015,7 +4015,7 @@ bool clip_image_batch_encode(clip_ctx * ctx, int n_threads, const clip_image_f32
|
||||
for (int ry = 0; ry < f; ry++)
|
||||
for (int rx = 0; rx < f; rx++)
|
||||
dsp.push_back((oy * f + ry) * grid_w + (ox * f + rx));
|
||||
set_input_i32("onyx_ds_perm", dsp);
|
||||
set_input_i32("muse_glimmer_ds_perm", dsp);
|
||||
} break;
|
||||
case PROJECTOR_TYPE_MINICPMV:
|
||||
{
|
||||
@@ -5065,7 +5065,7 @@ int clip_n_mmproj_embd(const struct clip_ctx * ctx) {
|
||||
return ctx->model.mm_model_mlp_3_w->ne[1];
|
||||
case PROJECTOR_TYPE_MINIMAX_M3:
|
||||
return ctx->model.mm_merger_fc2_b->ne[0];
|
||||
case PROJECTOR_TYPE_ONYX:
|
||||
case PROJECTOR_TYPE_MUSE_GLIMMER:
|
||||
return ctx->model.mm_2_w->ne[1];
|
||||
case PROJECTOR_TYPE_QWEN2VL:
|
||||
case PROJECTOR_TYPE_QWEN25VL:
|
||||
|
||||
@@ -255,7 +255,7 @@ private:
|
||||
ggml_tensor * append_rowwise_newlines(ggml_context * ctx0, ggml_tensor * tile_output);
|
||||
};
|
||||
|
||||
struct clip_graph_onyx : clip_graph {
|
||||
clip_graph_onyx(clip_ctx * ctx, const clip_image_f32 & img) : clip_graph(ctx, img) {}
|
||||
struct clip_graph_muse_glimmer : clip_graph {
|
||||
clip_graph_muse_glimmer(clip_ctx * ctx, const clip_image_f32 & img) : clip_graph(ctx, img) {}
|
||||
ggml_cgraph * build() override;
|
||||
};
|
||||
|
||||
@@ -1,19 +1,19 @@
|
||||
#include "models.h"
|
||||
|
||||
// Onyx vision encoder: 50-layer ViT with 2D RoPE, sparse block-diagonal
|
||||
// MuseGlimmer vision encoder: 50-layer ViT with 2D RoPE, sparse block-diagonal
|
||||
// window attention (every 4th + last layer global), pixel-shuffle downsample, then
|
||||
// adapter MLP + LLM's vision_projection.
|
||||
//
|
||||
// Several quantities are precomputed on host and fed as named graph inputs (filled in
|
||||
// clip.cpp set_input, PROJECTOR_TYPE_ONYX branch):
|
||||
// onyx_pos_w/_h [n_tok] i32 : 1-indexed RoPE positions (sparse-permuted order)
|
||||
// onyx_sp_perm [n_tok] i32 : window grouping permutation (applied after ln_pre)
|
||||
// onyx_inv_perm [n_tok] i32 : inverse of sp_perm (applied after blocks)
|
||||
// onyx_ds_perm [n_tok] i32 : pixel-shuffle gather (original order)
|
||||
// onyx_sp_mask [n_tok, n_tok] f32 : block-diagonal window mask (sparse layers)
|
||||
ggml_cgraph * clip_graph_onyx::build() {
|
||||
// clip.cpp set_input, PROJECTOR_TYPE_MUSE_GLIMMER branch):
|
||||
// muse_glimmer_pos_w/_h [n_tok] i32 : 1-indexed RoPE positions (sparse-permuted order)
|
||||
// muse_glimmer_sp_perm [n_tok] i32 : window grouping permutation (applied after ln_pre)
|
||||
// muse_glimmer_inv_perm [n_tok] i32 : inverse of sp_perm (applied after blocks)
|
||||
// muse_glimmer_ds_perm [n_tok] i32 : pixel-shuffle gather (original order)
|
||||
// muse_glimmer_sp_mask [n_tok, n_tok] f32 : block-diagonal window mask (sparse layers)
|
||||
ggml_cgraph * clip_graph_muse_glimmer::build() {
|
||||
const int ds = hparams.n_merge; // downsample factor (2)
|
||||
const int sf = hparams.onyx_sparse_factor; // 4
|
||||
const int sf = hparams.muse_glimmer_sparse_factor; // 4
|
||||
const int n_tok = n_patches;
|
||||
const int n_out = (n_patches_x / ds) * (n_patches_y / ds);
|
||||
const float rope_base = hparams.rope_theta; // 10000
|
||||
@@ -25,14 +25,14 @@ ggml_cgraph * clip_graph_onyx::build() {
|
||||
return t;
|
||||
};
|
||||
|
||||
ggml_tensor * pos_w = inp_i32("onyx_pos_w", n_tok);
|
||||
ggml_tensor * pos_h = inp_i32("onyx_pos_h", n_tok);
|
||||
ggml_tensor * sp_perm = inp_i32("onyx_sp_perm", n_tok);
|
||||
ggml_tensor * inv_perm = inp_i32("onyx_inv_perm", n_tok);
|
||||
ggml_tensor * ds_perm = inp_i32("onyx_ds_perm", n_tok);
|
||||
ggml_tensor * pos_w = inp_i32("muse_glimmer_pos_w", n_tok);
|
||||
ggml_tensor * pos_h = inp_i32("muse_glimmer_pos_h", n_tok);
|
||||
ggml_tensor * sp_perm = inp_i32("muse_glimmer_sp_perm", n_tok);
|
||||
ggml_tensor * inv_perm = inp_i32("muse_glimmer_inv_perm", n_tok);
|
||||
ggml_tensor * ds_perm = inp_i32("muse_glimmer_ds_perm", n_tok);
|
||||
|
||||
ggml_tensor * sp_mask = ggml_new_tensor_2d(ctx0, GGML_TYPE_F32, n_tok, n_tok);
|
||||
ggml_set_name(sp_mask, "onyx_sp_mask");
|
||||
ggml_set_name(sp_mask, "muse_glimmer_sp_mask");
|
||||
ggml_set_input(sp_mask);
|
||||
|
||||
// patchify via build_inp (conv2d over raw pixels) + bilinear-resized learned pos-emb
|
||||
@@ -1593,11 +1593,11 @@ mtmd_image_preproc_out mtmd_image_preprocessor_granite::preprocess(const clip_im
|
||||
}
|
||||
|
||||
//
|
||||
// mtmd_image_preprocessor_onyx
|
||||
// mtmd_image_preprocessor_muse_glimmer
|
||||
//
|
||||
|
||||
// Replicates transformers' get_aspect_ratio_preserving_size (image_processing_onyx.py)
|
||||
static clip_image_size onyx_grid_size(int img_w, int img_h, int patch_hw, int max_tokens) {
|
||||
// Replicates transformers' get_aspect_ratio_preserving_size
|
||||
static clip_image_size muse_glimmer_grid_size(int img_w, int img_h, int patch_hw, int max_tokens) {
|
||||
double i_nph = (double) img_h / patch_hw;
|
||||
double i_npw = (double) img_w / patch_hw;
|
||||
const double ratio = i_nph > 0.0 ? i_npw / i_nph : 1.0;
|
||||
@@ -1635,14 +1635,14 @@ static clip_image_size onyx_grid_size(int img_w, int img_h, int patch_hw, int ma
|
||||
return clip_image_size{ best_npw * patch_hw, best_nph * patch_hw };
|
||||
}
|
||||
|
||||
mtmd_image_preproc_out mtmd_image_preprocessor_onyx::preprocess(const clip_image_u8 & img) {
|
||||
mtmd_image_preproc_out mtmd_image_preprocessor_muse_glimmer::preprocess(const clip_image_u8 & img) {
|
||||
const int patch_hw = hparams.patch_size * hparams.n_merge;
|
||||
const int patch_area = hparams.patch_size * hparams.patch_size * hparams.n_merge * hparams.n_merge;
|
||||
GGML_ASSERT(patch_area > 0 && hparams.image_max_pixels > 0);
|
||||
const int max_tokens = hparams.image_max_pixels / patch_area;
|
||||
|
||||
const clip_image_size original_size = img.get_size();
|
||||
const clip_image_size target_size = onyx_grid_size(
|
||||
const clip_image_size target_size = muse_glimmer_grid_size(
|
||||
original_size.width, original_size.height, patch_hw, max_tokens);
|
||||
|
||||
// PIL resizes directly to (target_w, target_h) -- a stretch, no padding.
|
||||
|
||||
@@ -226,7 +226,7 @@ struct mtmd_image_preprocessor_granite : mtmd_image_preprocessor_llava_uhd {
|
||||
};
|
||||
|
||||
// pick the patch grid closest to the input aspect ratio under the per-image token cap, stretch-resize.
|
||||
struct mtmd_image_preprocessor_onyx : mtmd_image_preprocessor {
|
||||
mtmd_image_preprocessor_onyx(const clip_ctx * ctx) : mtmd_image_preprocessor(ctx) {}
|
||||
struct mtmd_image_preprocessor_muse_glimmer : mtmd_image_preprocessor {
|
||||
mtmd_image_preprocessor_muse_glimmer(const clip_ctx * ctx) : mtmd_image_preprocessor(ctx) {}
|
||||
mtmd_image_preproc_out preprocess(const clip_image_u8 & img) override;
|
||||
};
|
||||
|
||||
+2
-2
@@ -470,11 +470,11 @@ struct mtmd_context {
|
||||
img_end = "]<]end of image[>[";
|
||||
image_preproc = std::make_unique<mtmd_image_preprocessor_dyn_size>(ctx_v);
|
||||
} break;
|
||||
case PROJECTOR_TYPE_ONYX:
|
||||
case PROJECTOR_TYPE_MUSE_GLIMMER:
|
||||
{
|
||||
img_beg = "<|image_start|>";
|
||||
img_end = "<|image_end|>";
|
||||
image_preproc = std::make_unique<mtmd_image_preprocessor_onyx>(ctx_v);
|
||||
image_preproc = std::make_unique<mtmd_image_preprocessor_muse_glimmer>(ctx_v);
|
||||
} break;
|
||||
case PROJECTOR_TYPE_YOUTUVL:
|
||||
{
|
||||
|
||||
Reference in New Issue
Block a user