Commit d280808f5 for llama.cpp
commit d280808f5d82fcc3142b53f94ea5f594250cd765
Author: Ethan Guo <ethanguo.dev@gmail.com>
Date: Wed Sep 30 00:25:41 2026 +0800
common : stop accepting draft tokens at EOG (#29638)
* common : stop accepting draft tokens at EOG
* cont : remove the test
diff --git a/common/sampling.cpp b/common/sampling.cpp
index 27a752cc8..e9e1cb372 100644
--- a/common/sampling.cpp
+++ b/common/sampling.cpp
@@ -681,6 +681,8 @@ std::vector<llama_token> common_sampler_sample_and_accept_n(struct common_sample
std::vector<llama_token> result;
result.reserve(idxs.size());
+ const llama_vocab * vocab = llama_model_get_vocab(llama_get_model(ctx));
+
size_t i = 0;
for (; i < draft.size(); i++) {
const llama_token id = common_sampler_sample(gsmpl, ctx, idxs[i], grammar_first);
@@ -689,7 +691,9 @@ std::vector<llama_token> common_sampler_sample_and_accept_n(struct common_sample
result.push_back(id);
- if (draft[i] != id) {
+ // do not accept draft tokens after an EOG - they are not output but would stay in the context
+ // on replay the last token is from the target and can be EOG, so a trailing EOG is still accepted
+ if (draft[i] != id || (llama_vocab_is_eog(vocab, id) && i + 1 < draft.size())) {
break;
}
}