Commit 8330e9696 for llama.cpp

commit 8330e9696708fcfd5d465cdbd333e4c5a67e2808
Author: Pranesh Gonegandla <pranesh.iitp@gmail.com>
Date:   Sun Oct 4 14:15:22 2026 +0530

    spec : fix n-gram drafts rejected at temp > 0 after truncation (#29924)

    Co-authored-by: Pranesh Gonegandla <pgonegandla@nvidia.com>

diff --git a/common/speculative.cpp b/common/speculative.cpp
index 328ed241a..33dda825d 100644
--- a/common/speculative.cpp
+++ b/common/speculative.cpp
@@ -2915,8 +2915,8 @@ void common_speculative_draft(common_speculative * spec) {
                         SPC_DBG("truncating draft to %d tokens\n", dp.n_max);
                         result.resize(dp.n_max);

-                        // the candidates are one per drafted token and must be cut with them
-                        if (dp.result_q) {
+                        // trim the candidates only if the drafter produced them (n-gram drafters do not)
+                        if (dp.result_q && !dp.result_q->empty()) {
                             dp.result_q->resize(dp.n_max);
                         }
                     }