Commit a11f57ba9 for llama.cpp
commit a11f57ba93797579a5d1855ee216a31f10242676
Author: Hrishith Thadicherla <99313418+hthadicherla@users.noreply.github.com>
Date: Thu Oct 8 21:12:28 2026 +0530
model : fix DFlash output head sharing (#30111)
* llama : fix DFlash output head sharing
Assisted-by: Codex
* dflash : read tied output weights from GGUF metadata
Assisted-by: Codex
* llama : share tied word embedding metadata
Assisted-by: Codex
* llama : remove DFlash embedding head fallback
Assisted-by: Codex
diff --git a/conversion/gemma.py b/conversion/gemma.py
index 2a1f6931f..924d8e784 100644
--- a/conversion/gemma.py
+++ b/conversion/gemma.py
@@ -849,8 +849,8 @@ class Gemma4DSparkModel(DFlashModel):
raise ValueError("Gemma4 DSpark attention bias and MoE are not supported")
if (self.hparams.get("draft_vocab_size") or self.hparams["vocab_size"]) != self.hparams["vocab_size"]:
raise ValueError("Gemma4 DSpark currently requires a full draft vocabulary")
- if "model.lm_head.weight" not in self.model_tensors and self.hparams.get("tie_word_embeddings") is not True:
- raise ValueError("Gemma4 DSpark requires lm_head.weight unless tie_word_embeddings is true")
+ if "model.lm_head.weight" not in self.model_tensors:
+ raise ValueError("Gemma4 DSpark requires lm_head.weight")
self.dflash_config = self.hparams.get("dflash_config", {})
markov_type = self.dflash_config.get("markov_head_type", self.hparams.get("markov_head_type", "vanilla"))
diff --git a/src/models/dflash.cpp b/src/models/dflash.cpp
index c8c2895b6..b6cb6bfb9 100644
--- a/src/models/dflash.cpp
+++ b/src/models/dflash.cpp
@@ -167,9 +167,6 @@ void llama_model_dflash::load_arch_tensors(llama_model_loader &) {
// optional: reduced-vocab drafts ship their own lm head, full-vocab drafts can share the target's via ctx_other
// a draft with its own embeddings + head references no target tensors and can run on devices the target does not use (e.g. -devd with a tensor-split target)
output = create_tensor(tn(LLM_TENSOR_OUTPUT, "weight"), { n_embd, n_vocab_draft }, TENSOR_NOT_REQUIRED);
- if (output == nullptr && tok_embd != nullptr) {
- output = create_tensor(tn(LLM_TENSOR_TOKEN_EMBD, "weight"), { n_embd, n_vocab_draft }, TENSOR_DUPLICATED);
- }
if (hparams.dsv4_hc_mult > 0) {
const int64_t q_lora_rank = hparams.n_lora_q;