mirror of
https://github.com/ggml-org/llama.cpp.git
synced 2026-09-30 01:47:39 -05:00
common : stop accepting draft tokens at EOG (#29638)
* common : stop accepting draft tokens at EOG * cont : remove the test
This commit is contained in:
+5
-1
@@ -681,6 +681,8 @@ std::vector<llama_token> common_sampler_sample_and_accept_n(struct common_sample
|
||||
std::vector<llama_token> result;
|
||||
result.reserve(idxs.size());
|
||||
|
||||
const llama_vocab * vocab = llama_model_get_vocab(llama_get_model(ctx));
|
||||
|
||||
size_t i = 0;
|
||||
for (; i < draft.size(); i++) {
|
||||
const llama_token id = common_sampler_sample(gsmpl, ctx, idxs[i], grammar_first);
|
||||
@@ -689,7 +691,9 @@ std::vector<llama_token> common_sampler_sample_and_accept_n(struct common_sample
|
||||
|
||||
result.push_back(id);
|
||||
|
||||
if (draft[i] != id) {
|
||||
// do not accept draft tokens after an EOG - they are not output but would stay in the context
|
||||
// on replay the last token is from the target and can be EOG, so a trailing EOG is still accepted
|
||||
if (draft[i] != id || (llama_vocab_is_eog(vocab, id) && i + 1 < draft.size())) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user