From db33d3cb89d8b0d954df66047b13e9cc22df8f10 Mon Sep 17 00:00:00 2001 From: Toki Nasin <141258697+tokinasin@users.noreply.github.com> Date: Thu, 1 Oct 2026 14:31:42 +0900 Subject: [PATCH] vocab : honor BOS/EOS settings for PLaMo-2 and PLaMo-3 (#29734) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * vocab : honor BOS/EOS settings for PLaMo-2 and PLaMo-3 The original tokenizer configs for PLaMo-2 and PLaMo-3 have `add_bos_token: true` and `add_eos_token: false` , but _set_vocab_plamo() did not write the BOS/EOS metadata. The PLAMO2 tokenizer path also ignored add_bos/add_eos during tokenization. Write the settings from tokenizer_config.json and honor them in the PLAMO2 tokenization path. GGUFs without these keys keep the previous behavior. * Update conversion/base.py Co-authored-by: Sigbjørn Skjæret --------- Co-authored-by: Sigbjørn Skjæret --- conversion/base.py | 5 +++++ src/llama-vocab.cpp | 16 ++++++++++++++++ 2 files changed, 21 insertions(+) diff --git a/conversion/base.py b/conversion/base.py index 35f564e40c..a39a728f5f 100644 --- a/conversion/base.py +++ b/conversion/base.py @@ -2566,6 +2566,11 @@ class TextModel(ModelBase): self.gguf_writer.add_add_space_prefix(False) + if (add_bos := tokenizer_config.get("add_bos_token")) is not None: + self.gguf_writer.add_add_bos_token(add_bos) + if (add_eos := tokenizer_config.get("add_eos_token")) is not None: + self.gguf_writer.add_add_eos_token(add_eos) + class MmprojModel(ModelBase): model_type = ModelType.MMPROJ diff --git a/src/llama-vocab.cpp b/src/llama-vocab.cpp index 93468b6b4c..e5703f3e81 100644 --- a/src/llama-vocab.cpp +++ b/src/llama-vocab.cpp @@ -3599,6 +3599,10 @@ std::vector llama_vocab::impl::tokenize( } break; case LLAMA_VOCAB_TYPE_PLAMO2: { + if (add_special && add_bos) { + GGML_ASSERT(special_bos_id != LLAMA_TOKEN_NULL); + output.push_back(special_bos_id); + } llm_tokenizer_plamo2_session session(*static_cast(tokenizer.get())); for (const auto & fragment : fragment_buffer) { if (fragment.type == FRAGMENT_BUFFER_VARIANT_TYPE_RAW_TEXT) { @@ -3613,6 +3617,18 @@ std::vector llama_vocab::impl::tokenize( output.push_back(fragment.token); } } + + if (add_special && add_bos && output.size() >= 2 && output[1] == special_bos_id) { + LLAMA_LOG_WARN( + "%s: Added a BOS token to the prompt as specified by the model but the prompt " + "also starts with a BOS token. So now the final prompt starts with 2 BOS tokens. " + "Are you sure this is what you want?\n", __FUNCTION__); + } + + if (add_special && add_eos) { + GGML_ASSERT(special_eos_id != LLAMA_TOKEN_NULL); + output.push_back(special_eos_id); + } } break; case LLAMA_VOCAB_TYPE_TEST: {