mirror of
https://github.com/ggml-org/llama.cpp.git
synced 2026-10-02 02:47:26 -05:00
More FP8 conversion fixes
This commit is contained in:
+9
-3
@@ -699,9 +699,15 @@ class ModelBase:
|
||||
cast(list[tuple[int, float]], entries).append((expert_id, float(input_scale[0])))
|
||||
else:
|
||||
for new_name in self._map_fp8_weight_names(weight_name):
|
||||
scale_tensors[new_name.replace(".weight", ".scale")] = scale.numpy()
|
||||
mapped = self.tensor_map.get_type_and_name(new_name, try_suffixes=(".weight",)) if weight.ndim == 3 else None
|
||||
is_packed_expert = mapped is not None and mapped[0] in (
|
||||
gguf.MODEL_TENSOR.FFN_GATE_EXP, gguf.MODEL_TENSOR.FFN_UP_EXP, gguf.MODEL_TENSOR.FFN_DOWN_EXP,
|
||||
)
|
||||
# Packed expert weights need one scale value per expert.
|
||||
n_scales = weight.shape[0] if is_packed_expert else 1
|
||||
scale_tensors[new_name.replace(".weight", ".scale")] = np.repeat(scale.numpy(), n_scales)
|
||||
if input_scale is not None:
|
||||
input_scale_tensors[new_name.replace(".weight", ".input_scale")] = input_scale.numpy()
|
||||
input_scale_tensors[new_name.replace(".weight", ".input_scale")] = np.repeat(input_scale.numpy(), n_scales)
|
||||
|
||||
for name in consumed:
|
||||
self.model_tensors.pop(name, None)
|
||||
@@ -771,7 +777,7 @@ class ModelBase:
|
||||
if mapped is None:
|
||||
continue
|
||||
tensor_type, new_name = mapped
|
||||
if tensor_type not in (gguf.MODEL_TENSOR.FFN_GATE_EXP, gguf.MODEL_TENSOR.FFN_UP_EXP) or not new_name.endswith(".weight"):
|
||||
if tensor_type not in (gguf.MODEL_TENSOR.FFN_GATE_EXP, gguf.MODEL_TENSOR.FFN_UP_EXP, gguf.MODEL_TENSOR.FFN_GATE_UP_EXP) or not new_name.endswith(".weight"):
|
||||
continue
|
||||
if gen().dtype == torch.float8_e4m3fn:
|
||||
bid = next(int(part) for part in new_name.split(".") if part.isdecimal())
|
||||
|
||||
@@ -78,6 +78,15 @@ class Qwen2Model(TextModel):
|
||||
class Qwen2MoeModel(TextModel):
|
||||
model_arch = gguf.MODEL_ARCH.QWEN2MOE
|
||||
|
||||
def _map_fp8_weight_names(self, name: str) -> tuple[str, ...]:
|
||||
if name.removesuffix(".weight").endswith(".mlp.experts.gate_up_proj"):
|
||||
bid = next(int(part) for part in self.map_tensor_name(name).split(".") if part.isdecimal())
|
||||
return (
|
||||
self.format_tensor_name(gguf.MODEL_TENSOR.FFN_GATE_EXP, bid),
|
||||
self.format_tensor_name(gguf.MODEL_TENSOR.FFN_UP_EXP, bid),
|
||||
)
|
||||
return super()._map_fp8_weight_names(name)
|
||||
|
||||
def set_gguf_parameters(self):
|
||||
super().set_gguf_parameters()
|
||||
if (moe_intermediate_size := self.hparams.get("moe_intermediate_size")) is not None:
|
||||
|
||||
Reference in New Issue
Block a user