Commit db33d3cb8 for llama.cpp

commit db33d3cb89d8b0d954df66047b13e9cc22df8f10
Author: Toki Nasin <141258697+tokinasin@users.noreply.github.com>
Date:   Thu Oct 1 14:31:42 2026 +0900

    vocab : honor BOS/EOS settings for PLaMo-2 and PLaMo-3 (#29734)

    * vocab : honor BOS/EOS settings for PLaMo-2 and PLaMo-3

    The original tokenizer configs for PLaMo-2 and PLaMo-3 have
    `add_bos_token: true` and `add_eos_token: false` , but
    _set_vocab_plamo() did not write the BOS/EOS metadata. The
    PLAMO2 tokenizer path also ignored add_bos/add_eos during
    tokenization.

    Write the settings from tokenizer_config.json and honor them in
    the PLAMO2 tokenization path. GGUFs without these keys keep the
    previous behavior.

    * Update conversion/base.py

    Co-authored-by: Sigbjørn Skjæret <sigbjorn.skjaeret@huggingface.co>

    ---------

    Co-authored-by: Sigbjørn Skjæret <sigbjorn.skjaeret@huggingface.co>

diff --git a/conversion/base.py b/conversion/base.py
index 35f564e40..a39a728f5 100644
--- a/conversion/base.py
+++ b/conversion/base.py
@@ -2566,6 +2566,11 @@ class TextModel(ModelBase):

         self.gguf_writer.add_add_space_prefix(False)

+        if (add_bos := tokenizer_config.get("add_bos_token")) is not None:
+            self.gguf_writer.add_add_bos_token(add_bos)
+        if (add_eos := tokenizer_config.get("add_eos_token")) is not None:
+            self.gguf_writer.add_add_eos_token(add_eos)
+

 class MmprojModel(ModelBase):
     model_type = ModelType.MMPROJ
diff --git a/src/llama-vocab.cpp b/src/llama-vocab.cpp
index 93468b6b4..e5703f3e8 100644
--- a/src/llama-vocab.cpp
+++ b/src/llama-vocab.cpp
@@ -3599,6 +3599,10 @@ std::vector<llama_token> llama_vocab::impl::tokenize(
             } break;
         case LLAMA_VOCAB_TYPE_PLAMO2:
             {
+                if (add_special && add_bos) {
+                    GGML_ASSERT(special_bos_id != LLAMA_TOKEN_NULL);
+                    output.push_back(special_bos_id);
+                }
                 llm_tokenizer_plamo2_session session(*static_cast<const llm_tokenizer_plamo2 *>(tokenizer.get()));
                 for (const auto & fragment : fragment_buffer) {
                     if (fragment.type == FRAGMENT_BUFFER_VARIANT_TYPE_RAW_TEXT) {
@@ -3613,6 +3617,18 @@ std::vector<llama_token> llama_vocab::impl::tokenize(
                         output.push_back(fragment.token);
                     }
                 }
+
+                if (add_special && add_bos && output.size() >= 2 && output[1] == special_bos_id) {
+                    LLAMA_LOG_WARN(
+                        "%s: Added a BOS token to the prompt as specified by the model but the prompt "
+                        "also starts with a BOS token. So now the final prompt starts with 2 BOS tokens. "
+                        "Are you sure this is what you want?\n", __FUNCTION__);
+                }
+
+                if (add_special && add_eos) {
+                    GGML_ASSERT(special_eos_id != LLAMA_TOKEN_NULL);
+                    output.push_back(special_eos_id);
+                }
             } break;
         case LLAMA_VOCAB_TYPE_TEST:
             {