mirror of
https://github.com/ggml-org/llama.cpp.git
synced 2026-10-03 03:17:32 -05:00
nits
This commit is contained in:
+1
-1
@@ -2215,7 +2215,7 @@ llama_batch_ext * common_batch::get_sub_batch(int32_t off, int32_t n) {
|
||||
llama_batch_ext_set_output_logits(res, idx, true);
|
||||
}
|
||||
if (t.decision_order != 0) {
|
||||
llama_batch_ext_set_decision_order(res, idx, t.decision_order);
|
||||
llama_batch_ext_set_decision_order(res, idx, (llama_decision_order) t.decision_order);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -297,6 +297,7 @@ TEXT_MODEL_MAP: dict[str, str] = {
|
||||
|
||||
MMPROJ_MODEL_MAP: dict[str, str] = {
|
||||
"AudioFlamingo3ForConditionalGeneration": "ultravox",
|
||||
"ClefModel": "clef",
|
||||
"CogVLMForCausalLM": "cogvlm",
|
||||
"DeepseekOCR2ForCausalLM": "deepseek",
|
||||
"DeepseekOCRForCausalLM": "deepseek",
|
||||
|
||||
+9
-2
@@ -11,7 +11,7 @@ import torch
|
||||
if TYPE_CHECKING:
|
||||
from torch import Tensor
|
||||
|
||||
from .base import ModelBase, gguf, logger
|
||||
from .base import MmprojModel, ModelBase, gguf, logger
|
||||
from .qwen import Qwen3_5TextModel
|
||||
|
||||
|
||||
@@ -29,7 +29,6 @@ def _load_clef_hparams(dir_model: Path) -> dict[str, Any]:
|
||||
return hparams
|
||||
|
||||
|
||||
# TODO: image input needs token and embedding entries in the same batch, see https://github.com/ggml-org/llama.cpp/pull/29622
|
||||
@ModelBase.register("ClefModel")
|
||||
class ClefModel(Qwen3_5TextModel):
|
||||
model_arch = gguf.MODEL_ARCH.CLEF
|
||||
@@ -141,3 +140,11 @@ class ClefModel(Qwen3_5TextModel):
|
||||
return
|
||||
|
||||
yield self.map_tensor_name(name), data_torch
|
||||
|
||||
|
||||
@ModelBase.register("ClefModel")
|
||||
class ClefVisionModel(MmprojModel):
|
||||
def __init__(self, *args, **kwargs):
|
||||
del args, kwargs
|
||||
raise NotImplementedError(
|
||||
"multimodal input is not supported yet for Clef, requires https://github.com/ggml-org/llama.cpp/pull/29622 to be merged first")
|
||||
|
||||
+1
-1
@@ -1268,7 +1268,7 @@ bool llama_batch_ext_set_output_logits(llama_batch_ext * batch, int32_t idx, boo
|
||||
return batch->set_output(idx, value);
|
||||
}
|
||||
|
||||
bool llama_batch_ext_set_decision_order(llama_batch_ext * batch, int32_t idx, int32_t order) {
|
||||
bool llama_batch_ext_set_decision_order(llama_batch_ext * batch, int32_t idx, llama_decision_order order) {
|
||||
return batch->set_decision_order(idx, order);
|
||||
}
|
||||
|
||||
|
||||
+2
-1
@@ -101,6 +101,7 @@ LLAMA_API void llama_set_embeddings_nextn(struct llama_context * ctx, bool value
|
||||
LLAMA_API void llama_set_nextn_layer_offset(struct llama_context * ctx, int32_t offset);
|
||||
|
||||
// Marks the entries that a joint decision head (clef) reads, the default is 0
|
||||
// See https://github.com/ggml-org/llama.cpp/pull/29831 for details
|
||||
// A run of entries with the same value is one span, spans must be separated by entries with value 0
|
||||
// An option belongs to the last question before it
|
||||
enum llama_decision_order {
|
||||
@@ -111,7 +112,7 @@ enum llama_decision_order {
|
||||
LLAMA_DECISION_ORDER_OPTION = 4, // text of an option
|
||||
};
|
||||
// The embeddings output has one value per entry: row i is the score of option i
|
||||
LLAMA_API bool llama_batch_ext_set_decision_order(struct llama_batch_ext * batch, int32_t idx, int32_t order);
|
||||
LLAMA_API bool llama_batch_ext_set_decision_order(struct llama_batch_ext * batch, int32_t idx, enum llama_decision_order order);
|
||||
|
||||
// mirrors:
|
||||
// LLAMA_API float * llama_get_embeddings(struct llama_context * ctx);
|
||||
|
||||
Reference in New Issue
Block a user