diff --git a/common/sampling.cpp b/common/sampling.cpp index 27a752cc8d..e9e1cb372e 100644 --- a/common/sampling.cpp +++ b/common/sampling.cpp @@ -681,6 +681,8 @@ std::vector common_sampler_sample_and_accept_n(struct common_sample std::vector result; result.reserve(idxs.size()); + const llama_vocab * vocab = llama_model_get_vocab(llama_get_model(ctx)); + size_t i = 0; for (; i < draft.size(); i++) { const llama_token id = common_sampler_sample(gsmpl, ctx, idxs[i], grammar_first); @@ -689,7 +691,9 @@ std::vector common_sampler_sample_and_accept_n(struct common_sample result.push_back(id); - if (draft[i] != id) { + // do not accept draft tokens after an EOG - they are not output but would stay in the context + // on replay the last token is from the target and can be EOG, so a trailing EOG is still accepted + if (draft[i] != id || (llama_vocab_is_eog(vocab, id) && i + 1 < draft.size())) { break; } }