From d280808f5d82fcc3142b53f94ea5f594250cd765 Mon Sep 17 00:00:00 2001 From: Ethan Guo Date: Wed, 30 Sep 2026 00:25:41 +0800 Subject: [PATCH] common : stop accepting draft tokens at EOG (#29638) * common : stop accepting draft tokens at EOG * cont : remove the test --- common/sampling.cpp | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/common/sampling.cpp b/common/sampling.cpp index 27a752cc8d..e9e1cb372e 100644 --- a/common/sampling.cpp +++ b/common/sampling.cpp @@ -681,6 +681,8 @@ std::vector common_sampler_sample_and_accept_n(struct common_sample std::vector result; result.reserve(idxs.size()); + const llama_vocab * vocab = llama_model_get_vocab(llama_get_model(ctx)); + size_t i = 0; for (; i < draft.size(); i++) { const llama_token id = common_sampler_sample(gsmpl, ctx, idxs[i], grammar_first); @@ -689,7 +691,9 @@ std::vector common_sampler_sample_and_accept_n(struct common_sample result.push_back(id); - if (draft[i] != id) { + // do not accept draft tokens after an EOG - they are not output but would stay in the context + // on replay the last token is from the target and can be EOG, so a trailing EOG is still accepted + if (draft[i] != id || (llama_vocab_is_eog(vocab, id) && i + 1 < draft.size())) { break; } }