From bb8bf2c42e37189ec94378f5c837f4080a46c1ad Mon Sep 17 00:00:00 2001 From: Xuan Son Nguyen Date: Fri, 2 Oct 2026 19:08:26 +0200 Subject: [PATCH] nits --- common/common.cpp | 2 +- conversion/__init__.py | 1 + conversion/clef.py | 11 +++++++++-- src/llama-batch.cpp | 2 +- src/llama-ext.h | 3 ++- 5 files changed, 14 insertions(+), 5 deletions(-) diff --git a/common/common.cpp b/common/common.cpp index 0eb4878468..0269bc895c 100644 --- a/common/common.cpp +++ b/common/common.cpp @@ -2215,7 +2215,7 @@ llama_batch_ext * common_batch::get_sub_batch(int32_t off, int32_t n) { llama_batch_ext_set_output_logits(res, idx, true); } if (t.decision_order != 0) { - llama_batch_ext_set_decision_order(res, idx, t.decision_order); + llama_batch_ext_set_decision_order(res, idx, (llama_decision_order) t.decision_order); } } diff --git a/conversion/__init__.py b/conversion/__init__.py index 612bd5f189..6d8ae9c1a5 100644 --- a/conversion/__init__.py +++ b/conversion/__init__.py @@ -297,6 +297,7 @@ TEXT_MODEL_MAP: dict[str, str] = { MMPROJ_MODEL_MAP: dict[str, str] = { "AudioFlamingo3ForConditionalGeneration": "ultravox", + "ClefModel": "clef", "CogVLMForCausalLM": "cogvlm", "DeepseekOCR2ForCausalLM": "deepseek", "DeepseekOCRForCausalLM": "deepseek", diff --git a/conversion/clef.py b/conversion/clef.py index 3cc57a4e35..a6f7243945 100644 --- a/conversion/clef.py +++ b/conversion/clef.py @@ -11,7 +11,7 @@ import torch if TYPE_CHECKING: from torch import Tensor -from .base import ModelBase, gguf, logger +from .base import MmprojModel, ModelBase, gguf, logger from .qwen import Qwen3_5TextModel @@ -29,7 +29,6 @@ def _load_clef_hparams(dir_model: Path) -> dict[str, Any]: return hparams -# TODO: image input needs token and embedding entries in the same batch, see https://github.com/ggml-org/llama.cpp/pull/29622 @ModelBase.register("ClefModel") class ClefModel(Qwen3_5TextModel): model_arch = gguf.MODEL_ARCH.CLEF @@ -141,3 +140,11 @@ class ClefModel(Qwen3_5TextModel): return yield self.map_tensor_name(name), data_torch + + +@ModelBase.register("ClefModel") +class ClefVisionModel(MmprojModel): + def __init__(self, *args, **kwargs): + del args, kwargs + raise NotImplementedError( + "multimodal input is not supported yet for Clef, requires https://github.com/ggml-org/llama.cpp/pull/29622 to be merged first") diff --git a/src/llama-batch.cpp b/src/llama-batch.cpp index 6e58af8727..1b3d70627a 100644 --- a/src/llama-batch.cpp +++ b/src/llama-batch.cpp @@ -1268,7 +1268,7 @@ bool llama_batch_ext_set_output_logits(llama_batch_ext * batch, int32_t idx, boo return batch->set_output(idx, value); } -bool llama_batch_ext_set_decision_order(llama_batch_ext * batch, int32_t idx, int32_t order) { +bool llama_batch_ext_set_decision_order(llama_batch_ext * batch, int32_t idx, llama_decision_order order) { return batch->set_decision_order(idx, order); } diff --git a/src/llama-ext.h b/src/llama-ext.h index 3db728645d..fcc78f47e4 100644 --- a/src/llama-ext.h +++ b/src/llama-ext.h @@ -101,6 +101,7 @@ LLAMA_API void llama_set_embeddings_nextn(struct llama_context * ctx, bool value LLAMA_API void llama_set_nextn_layer_offset(struct llama_context * ctx, int32_t offset); // Marks the entries that a joint decision head (clef) reads, the default is 0 +// See https://github.com/ggml-org/llama.cpp/pull/29831 for details // A run of entries with the same value is one span, spans must be separated by entries with value 0 // An option belongs to the last question before it enum llama_decision_order { @@ -111,7 +112,7 @@ enum llama_decision_order { LLAMA_DECISION_ORDER_OPTION = 4, // text of an option }; // The embeddings output has one value per entry: row i is the score of option i -LLAMA_API bool llama_batch_ext_set_decision_order(struct llama_batch_ext * batch, int32_t idx, int32_t order); +LLAMA_API bool llama_batch_ext_set_decision_order(struct llama_batch_ext * batch, int32_t idx, enum llama_decision_order order); // mirrors: // LLAMA_API float * llama_get_embeddings(struct llama_context * ctx);