Commit 8330e9696 for llama.cpp
commit 8330e9696708fcfd5d465cdbd333e4c5a67e2808
Author: Pranesh Gonegandla <pranesh.iitp@gmail.com>
Date: Sun Oct 4 14:15:22 2026 +0530
spec : fix n-gram drafts rejected at temp > 0 after truncation (#29924)
Co-authored-by: Pranesh Gonegandla <pgonegandla@nvidia.com>
diff --git a/common/speculative.cpp b/common/speculative.cpp
index 328ed241a..33dda825d 100644
--- a/common/speculative.cpp
+++ b/common/speculative.cpp
@@ -2915,8 +2915,8 @@ void common_speculative_draft(common_speculative * spec) {
SPC_DBG("truncating draft to %d tokens\n", dp.n_max);
result.resize(dp.n_max);
- // the candidates are one per drafted token and must be cut with them
- if (dp.result_q) {
+ // trim the candidates only if the drafter produced them (n-gram drafters do not)
+ if (dp.result_q && !dp.result_q->empty()) {
dp.result_q->resize(dp.n_max);
}
}