Commit bd4f514db for llama.cpp
commit bd4f514db14d87fded667787a7a963bfbaa98e89
Author: Tarek Dakhran <tarek@liquid.ai>
Date: Thu Sep 24 01:16:43 2026 +0200
convert : allow vision target for DFlash/Dspark (#29339)
Resolve the target arch with get_model_architecture so vision targets
(e.g. Lfm2VlForConditionalGeneration) map to their text model for the vocab.
Fix double rope reorder for LFM2/LFM2.5 DSpark drafters
diff --git a/conversion/qwen.py b/conversion/qwen.py
index ef5504f3d..64d606176 100644
--- a/conversion/qwen.py
+++ b/conversion/qwen.py
@@ -10,7 +10,7 @@ import torch
if TYPE_CHECKING:
from torch import Tensor
-from .base import LazyTorchTensor, ModelBase, TextModel, gguf, logger
+from .base import LazyTorchTensor, ModelBase, ModelType, TextModel, get_model_architecture, gguf, logger
@ModelBase.register("QWenLMHeadModel")
@@ -666,7 +666,7 @@ class DFlashModel(Qwen3Model):
from . import get_model_class
with open(self.target_model_dir / "config.json", "r", encoding="utf-8") as f:
target_hparams = json.load(f)
- target_arch = target_hparams["architectures"][0]
+ target_arch = get_model_architecture(target_hparams, ModelType.TEXT)
target_cls = get_model_class(target_arch)
if target_cls is not type(self):
@@ -841,13 +841,6 @@ class DSparkModel(DFlashModel):
return None
return super().filter_tensors(item)
- _ROPE_PERMUTE_SUFFIXES = (
- "self_attn.q_proj.weight",
- "self_attn.k_proj.weight",
- "self_attn.q_norm.weight",
- "self_attn.k_norm.weight",
- )
-
def modify_tensors(self, data_torch: Tensor, name: str, bid: int | None) -> Iterable[tuple[str, Tensor]]:
if name == "model.d2t":
self._d2t = data_torch
@@ -856,12 +849,6 @@ class DSparkModel(DFlashModel):
if self._n_vocab_draft == self.hparams["vocab_size"] and name.endswith("lm_head.weight"):
return
- # interleaved-rope checkpoints (rope_is_neox_style = false) -> NeoX layout: per head, even dims first then odd
- if not self.hparams.get("rope_is_neox_style", True) and name.endswith(self._ROPE_PERMUTE_SUFFIXES):
- head_dim = self.hparams["head_dim"]
- shape = data_torch.shape
- data_torch = data_torch.reshape(-1, head_dim // 2, 2, *shape[1:]).transpose(1, 2).reshape(shape)
-
yield from super().modify_tensors(data_torch, name, bid)
def prepare_tensors(self):