fix: normalize CLIP prompts before special-token splitting (#1670)

This commit is contained in:
leejet
2026-06-17 00:33:00 +08:00
committed by GitHub
parent 92a3b73cdb
commit 7f0e728b7d
4 changed files with 28 additions and 28 deletions
+3 -2
View File
@@ -134,7 +134,8 @@ std::vector<int> BPETokenizer::encode(const std::string& text, on_new_token_cb_t
std::vector<int32_t> bpe_tokens;
std::vector<std::string> token_strs;
auto splited_texts = split_with_special_tokens(text, special_tokens);
std::string normalized_text = normalize_before_split ? normalize(text) : text;
auto splited_texts = split_with_special_tokens(normalized_text, special_tokens);
for (auto& splited_text : splited_texts) {
if (is_special_token(splited_text)) {
@@ -159,7 +160,7 @@ std::vector<int> BPETokenizer::encode(const std::string& text, on_new_token_cb_t
}
}
std::string token_str = normalize(token);
std::string token_str = normalize_before_split ? token : normalize(token);
std::u32string utf32_token;
if (byte_level_bpe) {
for (int i = 0; i < token_str.length(); i++) {