common : stop accepting draft tokens at EOG (#29638)

* common : stop accepting draft tokens at EOG

* cont : remove the test
This commit is contained in:
Ethan Guo
2026-09-29 18:25:41 +02:00
committed by GitHub
parent ba0ba54d93
commit d280808f5d
+5 -1
View File
@@ -681,6 +681,8 @@ std::vector<llama_token> common_sampler_sample_and_accept_n(struct common_sample
std::vector<llama_token> result;
result.reserve(idxs.size());
const llama_vocab * vocab = llama_model_get_vocab(llama_get_model(ctx));
size_t i = 0;
for (; i < draft.size(); i++) {
const llama_token id = common_sampler_sample(gsmpl, ctx, idxs[i], grammar_first);
@@ -689,7 +691,9 @@ std::vector<llama_token> common_sampler_sample_and_accept_n(struct common_sample
result.push_back(id);
if (draft[i] != id) {
// do not accept draft tokens after an EOG - they are not output but would stay in the context
// on replay the last token is from the target and can be EOG, so a trailing EOG is still accepted
if (draft[i] != id || (llama_vocab_is_eog(vocab, id) && i + 1 < draft.size())) {
break;
}
}