Commit 8df332de1 for llama.cpp
commit 8df332de1b7d036952631ef9d0a25d8aa60aeea3
Author: Daniel Bevenius <daniel.bevenius@gmail.com>
Date: Wed Sep 30 12:52:02 2026 +0200
model-conversion : add --add-bos to run org model script (#29558)
This commit adds an optional --add-bos token command line option to the
run-org-model.py script.
The motivation for this is that there are models, for example Gemma4,
that explicitely set the add_bos value to true in llama-vocab.cpp even
if the original model does not set this value to True.
It would be nice to be able to force the models to agree on the bos
token so that logit verification can proceed.
Refs: https://github.com/ggml-org/llama.cpp/pull/21500
diff --git a/examples/model-conversion/scripts/causal/run-org-model.py b/examples/model-conversion/scripts/causal/run-org-model.py
index 6f85ee448..b0f6e886c 100755
--- a/examples/model-conversion/scripts/causal/run-org-model.py
+++ b/examples/model-conversion/scripts/causal/run-org-model.py
@@ -19,6 +19,8 @@ def parse_arguments():
parser.add_argument("--prompt-file", "-f", help="Optional prompt file", required=False)
parser.add_argument("--verbose", "-v", action="store_true", help="Enable verbose debug output")
parser.add_argument("--device", "-d", help="Device to use (cpu, cuda, mps, auto)", default="auto")
+ parser.add_argument("--add-bos", action=argparse.BooleanOptionalAction, default=None,
+ help="Override BOS token setting (default: use model's own setting)")
return parser.parse_args()
def load_model_and_tokenizer(model_path, device="auto"):
@@ -119,6 +121,9 @@ def main():
model, tokenizer, config = load_model_and_tokenizer(model_path, args.device)
+ if args.add_bos is not None and hasattr(tokenizer, "add_bos_token"):
+ tokenizer.add_bos_token = args.add_bos
+
if args.verbose:
enable_torch_debugging(model)