Revert rope_type derivation from target

NOTE: this breaks compatibility with Meta's distributed DFlash GGUFs, as
the Q/K are stored in "NEOX" (rotated half) format, like in
transformers.
This commit is contained in:
Pedro Cuenca
2026-08-08 13:40:49 +02:00
parent 1a75103a2b
commit 83dd146b75
2 changed files with 3 additions and 12 deletions
+2 -8
View File
@@ -126,12 +126,6 @@ class OnyxAssistantModel(TextModel):
self.gguf_writer.add_sliding_window_pattern([t == "sliding_attention" for t in h["layer_types"]])
def modify_tensors(self, data_torch: Tensor, name: str, bid: int | None) -> Iterable[tuple[str, Tensor]]:
# Invert transformers' permute_rope
if ".self_attn.q_proj." in name:
data_torch = _unpermute_for_rope(data_torch, int(self.hparams["num_attention_heads"]))
elif ".self_attn.k_proj." in name:
data_torch = _unpermute_for_rope(data_torch, int(self.hparams["num_key_value_heads"]))
elif ".self_attn.q_norm." in name or ".self_attn.k_norm." in name:
data_torch = _unpermute_for_rope(data_torch, 1)
# DFlash defaults to NEOX (rotate_half) rope, matching transformers HF layout for Q/K, QK-norms
# no permutation needed.
yield (self.map_tensor_name(name), data_torch)
+1 -4
View File
@@ -1350,10 +1350,7 @@ llm_graph_context::llm_graph_context(const llm_graph_params & params) :
n_outputs (params.n_outputs),
n_ctx_orig (cparams.n_ctx_orig_yarn),
pooling_type (cparams.pooling_type),
// DFlash: inherit rope type from the linked target
rope_type ((arch == LLM_ARCH_DFLASH && cparams.ctx_other != nullptr)
? llama_get_model(cparams.ctx_other)->hparams.rope_type
: hparams.rope_type),
rope_type (hparams.rope_type),
sched (params.sched),
backend_cpu (params.backend_cpu),
cvec (params.cvec),