fix: use tokenizer-specific pre-tokenization rules (#1975)

This commit is contained in:
leejet
2026-09-15 02:37:52 +08:00
committed by GitHub
parent 07a85c74cb
commit 59c23bce0d
9 changed files with 92 additions and 1021 deletions
+3 -1
View File
@@ -46,7 +46,9 @@ void GemmaTokenizer::load_from_merges(const std::string& merges_utf8_str, const
bpe_len = rank;
}
GemmaTokenizer::GemmaTokenizer(const std::string& merges_utf8_str, const std::string& vocab_utf8_str) {
GemmaTokenizer::GemmaTokenizer(const std::string& merges_utf8_str, const std::string& vocab_utf8_str)
: BPETokenizer("") {
// Gemma replaces spaces with metaspace before its literal-space Split, so no regex boundaries apply.
byte_level_bpe = false;
byte_fallback = true;
add_bos_token = true;