mirror of
https://github.com/ggml-org/llama.cpp.git
synced 2026-10-03 11:27:25 -05:00
Revert rope_type derivation from target
NOTE: this breaks compatibility with Meta's distributed DFlash GGUFs, as the Q/K are stored in "NEOX" (rotated half) format, like in transformers.
This commit is contained in:
+2
-8
@@ -126,12 +126,6 @@ class OnyxAssistantModel(TextModel):
|
||||
self.gguf_writer.add_sliding_window_pattern([t == "sliding_attention" for t in h["layer_types"]])
|
||||
|
||||
def modify_tensors(self, data_torch: Tensor, name: str, bid: int | None) -> Iterable[tuple[str, Tensor]]:
|
||||
# Invert transformers' permute_rope
|
||||
if ".self_attn.q_proj." in name:
|
||||
data_torch = _unpermute_for_rope(data_torch, int(self.hparams["num_attention_heads"]))
|
||||
elif ".self_attn.k_proj." in name:
|
||||
data_torch = _unpermute_for_rope(data_torch, int(self.hparams["num_key_value_heads"]))
|
||||
elif ".self_attn.q_norm." in name or ".self_attn.k_norm." in name:
|
||||
data_torch = _unpermute_for_rope(data_torch, 1)
|
||||
|
||||
# DFlash defaults to NEOX (rotate_half) rope, matching transformers HF layout for Q/K, QK-norms
|
||||
# no permutation needed.
|
||||
yield (self.map_tensor_name(name), data_torch)
|
||||
|
||||
+1
-4
@@ -1350,10 +1350,7 @@ llm_graph_context::llm_graph_context(const llm_graph_params & params) :
|
||||
n_outputs (params.n_outputs),
|
||||
n_ctx_orig (cparams.n_ctx_orig_yarn),
|
||||
pooling_type (cparams.pooling_type),
|
||||
// DFlash: inherit rope type from the linked target
|
||||
rope_type ((arch == LLM_ARCH_DFLASH && cparams.ctx_other != nullptr)
|
||||
? llama_get_model(cparams.ctx_other)->hparams.rope_type
|
||||
: hparams.rope_type),
|
||||
rope_type (hparams.rope_type),
|
||||
sched (params.sched),
|
||||
backend_cpu (params.backend_cpu),
|
||||
cvec (params.cvec),
|
||||
|
||||
Reference in New Issue
Block a user