Commit 2b36825cb for llama.cpp
commit 2b36825cbc39b06ed9512b487a67127206e474af
Author: Yu Chengye <60293095+kabu1204@users.noreply.github.com>
Date: Thu Oct 1 22:34:32 2026 +0800
convert : write Gemma embedding scale for DFlash drafts (#29802)
* convert : write Gemma embedding scale for DFlash drafts
A DFlash draft shares the target's token embeddings. Gemma scales them by sqrt(hidden_size) in the forward pass, and the draft config does not state that scale, so the converted draft read unscaled embeddings.
Take the scale from the target config when the draft config has none.
Assisted-by: Claude
* convert : check with get_model_architecture for gemma models
diff --git a/conversion/qwen.py b/conversion/qwen.py
index 95a41fb3a..4946456e9 100644
--- a/conversion/qwen.py
+++ b/conversion/qwen.py
@@ -729,6 +729,12 @@ class DFlashModel(Qwen3Model):
embedding_scale = dflash_config.get(
"input_embedding_scale", self.hparams.get("input_embedding_scale")
)
+ if embedding_scale is None and self.target_model_dir is not None:
+ # the draft shares the target's token embeddings, and Gemma scales them by sqrt(hidden_size) in the forward pass
+ target_hparams = ModelBase.load_hparams(self.target_model_dir, False)
+ if get_model_architecture(target_hparams, ModelType.TEXT).startswith("Gemma"):
+ target_hparams = {**target_hparams, **target_hparams.get("text_config", {})}
+ embedding_scale = target_hparams["hidden_size"] ** 0.5
if embedding_scale is not None:
self.gguf_writer.add_embedding_scale(float(embedding_scale))