This commit is contained in:
Xuan Son Nguyen
2026-10-02 19:08:26 +02:00
parent ff15e5724b
commit bb8bf2c42e
5 changed files with 14 additions and 5 deletions
+1 -1
View File
@@ -2215,7 +2215,7 @@ llama_batch_ext * common_batch::get_sub_batch(int32_t off, int32_t n) {
llama_batch_ext_set_output_logits(res, idx, true);
}
if (t.decision_order != 0) {
llama_batch_ext_set_decision_order(res, idx, t.decision_order);
llama_batch_ext_set_decision_order(res, idx, (llama_decision_order) t.decision_order);
}
}
+1
View File
@@ -297,6 +297,7 @@ TEXT_MODEL_MAP: dict[str, str] = {
MMPROJ_MODEL_MAP: dict[str, str] = {
"AudioFlamingo3ForConditionalGeneration": "ultravox",
"ClefModel": "clef",
"CogVLMForCausalLM": "cogvlm",
"DeepseekOCR2ForCausalLM": "deepseek",
"DeepseekOCRForCausalLM": "deepseek",
+9 -2
View File
@@ -11,7 +11,7 @@ import torch
if TYPE_CHECKING:
from torch import Tensor
from .base import ModelBase, gguf, logger
from .base import MmprojModel, ModelBase, gguf, logger
from .qwen import Qwen3_5TextModel
@@ -29,7 +29,6 @@ def _load_clef_hparams(dir_model: Path) -> dict[str, Any]:
return hparams
# TODO: image input needs token and embedding entries in the same batch, see https://github.com/ggml-org/llama.cpp/pull/29622
@ModelBase.register("ClefModel")
class ClefModel(Qwen3_5TextModel):
model_arch = gguf.MODEL_ARCH.CLEF
@@ -141,3 +140,11 @@ class ClefModel(Qwen3_5TextModel):
return
yield self.map_tensor_name(name), data_torch
@ModelBase.register("ClefModel")
class ClefVisionModel(MmprojModel):
def __init__(self, *args, **kwargs):
del args, kwargs
raise NotImplementedError(
"multimodal input is not supported yet for Clef, requires https://github.com/ggml-org/llama.cpp/pull/29622 to be merged first")
+1 -1
View File
@@ -1268,7 +1268,7 @@ bool llama_batch_ext_set_output_logits(llama_batch_ext * batch, int32_t idx, boo
return batch->set_output(idx, value);
}
bool llama_batch_ext_set_decision_order(llama_batch_ext * batch, int32_t idx, int32_t order) {
bool llama_batch_ext_set_decision_order(llama_batch_ext * batch, int32_t idx, llama_decision_order order) {
return batch->set_decision_order(idx, order);
}
+2 -1
View File
@@ -101,6 +101,7 @@ LLAMA_API void llama_set_embeddings_nextn(struct llama_context * ctx, bool value
LLAMA_API void llama_set_nextn_layer_offset(struct llama_context * ctx, int32_t offset);
// Marks the entries that a joint decision head (clef) reads, the default is 0
// See https://github.com/ggml-org/llama.cpp/pull/29831 for details
// A run of entries with the same value is one span, spans must be separated by entries with value 0
// An option belongs to the last question before it
enum llama_decision_order {
@@ -111,7 +112,7 @@ enum llama_decision_order {
LLAMA_DECISION_ORDER_OPTION = 4, // text of an option
};
// The embeddings output has one value per entry: row i is the score of option i
LLAMA_API bool llama_batch_ext_set_decision_order(struct llama_batch_ext * batch, int32_t idx, int32_t order);
LLAMA_API bool llama_batch_ext_set_decision_order(struct llama_batch_ext * batch, int32_t idx, enum llama_decision_order order);
// mirrors:
// LLAMA_API float * llama_get_embeddings(struct llama_context * ctx);