Commit a657f7e98 for llama.cpp
commit a657f7e981ff8764d2ccce1e74ec3c7e9bf2cfa2
Author: Tarek Dakhran <tarek@liquid.ai>
Date: Thu Oct 8 02:07:49 2026 +0200
model : add LiquidAI/d1-omni-600M decision model (#30114)
* model : add LiquidAI/d1-omni-600M decision model
Assisted-by: Claude Opus 5.5
* mtmd : keep conformer GLU sigmoid on CUDA
Assisted-by: Claude Opus 5.5
* server : take d1omni audio through images and input_audio, scope memory-less lfm2 to non-causal
Assisted-by: Claude Opus 5.5
* common : rename decision type d1omni to lfm2-d1-omni, server : make images an alias of files
Assisted-by: Claude Opus 5.5
diff --git a/common/common.cpp b/common/common.cpp
index 0e891cea6..28ea8680e 100644
--- a/common/common.cpp
+++ b/common/common.cpp
@@ -1171,6 +1171,7 @@ static const std::map<common_decision_type, std::string> COMMON_DECISION_TYPE_NA
{ COMMON_DECISION_TYPE_CLEF, "clef" },
{ COMMON_DECISION_TYPE_PPLX_DECIDER, "pplx-decider" },
{ COMMON_DECISION_TYPE_LFM2_D1, "lfm2-d1" },
+ { COMMON_DECISION_TYPE_LFM2_D1_OMNI, "lfm2-d1-omni" },
};
static common_decision_type common_decision_type_from_string(const std::string & str) {
@@ -1284,7 +1285,8 @@ common_init_result::common_init_result(common_params & params, bool model_only)
// these decision models return a score for each token via the embeddings output
// TODO: maybe improve this in the future
const auto decision_type = common_get_decision_type(model);
- if (decision_type == COMMON_DECISION_TYPE_LAYA || decision_type == COMMON_DECISION_TYPE_KEV || decision_type == COMMON_DECISION_TYPE_CLEF) {
+ if (decision_type == COMMON_DECISION_TYPE_LAYA || decision_type == COMMON_DECISION_TYPE_KEV || decision_type == COMMON_DECISION_TYPE_CLEF ||
+ decision_type == COMMON_DECISION_TYPE_LFM2_D1_OMNI) {
params.embedding = true;
params.pooling_type = LLAMA_POOLING_TYPE_NONE;
diff --git a/common/common.h b/common/common.h
index 6f8acf31d..0a85f11f9 100644
--- a/common/common.h
+++ b/common/common.h
@@ -964,6 +964,7 @@ enum common_decision_type {
COMMON_DECISION_TYPE_CLEF, // all questions in one prompt, score of option i read from the embeddings output at row i
COMMON_DECISION_TYPE_PPLX_DECIDER, // same as openjev, label codes of 1 or 2 letters
COMMON_DECISION_TYPE_LFM2_D1, // same as openjev, the labels depend on the question type
+ COMMON_DECISION_TYPE_LFM2_D1_OMNI, // same as laya, other prompt layout
COMMON_DECISION_TYPE_UNKNOWN, // a decision model of a type that is not supported
};
diff --git a/conversion/__init__.py b/conversion/__init__.py
index 34c56bbee..72960be1a 100644
--- a/conversion/__init__.py
+++ b/conversion/__init__.py
@@ -162,6 +162,7 @@ TEXT_MODEL_MAP: dict[str, str] = {
"Lfm2BidirectionalModel": "lfm2",
"Lfm2ForCausalLM": "lfm2",
"D1Model": "lfm2",
+ "D1OmniModel": "lfm2",
"Lfm2Model": "lfm2",
"Lfm2MoeForCausalLM": "lfm2",
"Llama4ForCausalLM": "llama",
@@ -335,6 +336,7 @@ MMPROJ_MODEL_MAP: dict[str, str] = {
"KimiK25ForConditionalGeneration": "kimivl",
"KimiVLForConditionalGeneration": "kimivl",
"Lfm2AudioForConditionalGeneration": "lfm2",
+ "D1OmniModel": "lfm2",
"Lfm2VlForConditionalGeneration": "lfm2",
"LightOnOCRForConditionalGeneration": "lighton_ocr",
"Llama4ForConditionalGeneration": "llama4",
diff --git a/conversion/lfm2.py b/conversion/lfm2.py
index d4343675e..113c91116 100644
--- a/conversion/lfm2.py
+++ b/conversion/lfm2.py
@@ -161,6 +161,121 @@ class LFM2ColBertModel(LFM2Model):
yield f"{self.dense_tensor_name}.weight", tensor.clone()
+def _is_d1_omni_checkpoint(dir_model: Path) -> bool:
+ if not (dir_model / "config.json").is_file():
+ return False
+ with open(dir_model / "config.json", encoding="utf-8") as f:
+ return json.load(f).get("model_type") == "d1_omni"
+
+
+@ModelBase.register_hparams_loader(_is_d1_omni_checkpoint)
+def _load_d1_omni_hparams(dir_model: Path) -> dict[str, Any]:
+ logger.info("gguf: detected d1-omni checkpoint")
+ hparams = ModelBase.load_hparams(dir_model, False, guess=False)
+ text = hparams["text_config"]
+ n_layer, n_layer_head = text["num_hidden_layers"], hparams["head_layers"]
+ # the trunk uses the LFM2 FFN sizing, the head blocks are appended with a plain 4x MLP
+ n_ff = int(text["block_ffn_dim_multiplier"] * int(2 * text["intermediate_size"] / 3))
+ n_ff = text["block_multiple_of"] * ((n_ff + text["block_multiple_of"] - 1) // text["block_multiple_of"])
+ text["num_hidden_layers"] = n_layer + n_layer_head
+ text["intermediate_size"] = [n_ff] * n_layer + [4 * text["hidden_size"]] * n_layer_head
+ text["block_auto_adjust_ff_dim"] = False
+ return hparams
+
+
+@ModelBase.register("D1OmniModel")
+@ModelBase.example("LiquidAI/d1-omni-600M")
+class D1OmniModel(LFM2Model):
+ model_arch = gguf.MODEL_ARCH.LFM2
+
+ # the server cuts the text to these lengths, see server-decision.cpp
+ _MAX_LENGTH = 16384
+ _IMAGE_TEXT_LENGTH = 896
+ _AUDIO_TEXT_LENGTH = 15360
+
+ def set_vocab(self):
+ super().set_vocab()
+ # the systemone template writes the BOS, after the media
+ self.gguf_writer.remove_key(gguf.Keys.Tokenizer.ADD_BOS)
+ self.gguf_writer.add_add_bos_token(False)
+ self.gguf_writer.add_token_type_count(3) # choice, score, noul
+ self.gguf_writer.add_chat_template([{"name": "systemone", "template": self._systemone_template()}])
+
+ @staticmethod
+ def _systemone_template() -> str:
+ # follows prompt.py of the model repo, the server cuts each marked piece to its token budget
+ # the media (images, or an audio clip if audio is true) come first
+ description = jinja_str_or_json("o.description")
+ has_description = "o.description is not none and o.description != ''"
+ yes_no = "{{ 'yes' if o.key == 'true' else 'no' }}"
+ option_code = "{% if loop.index0 < 10 %}00{% elif loop.index0 < 100 %}0{% endif %}{{ loop.index0 }}"
+ option = (
+ "{% if type == 'choice' and audio %}option_" + option_code + ": "
+ "{% if " + has_description + " %}" + description + "{% else %}{{ o.key }}{% endif %}"
+ "{% elif type == 'choice' %}{{ o.key }}{% if " + has_description + " %}: " + description + "{% endif %}"
+ "{% elif type == 'score' %}level {{ o.key }}: " + description
+ + "{% elif audio %}{{ o.key }}: " + yes_no
+ + "{% else %}{{ o.key }}: {% if " + has_description + " %}" + description
+ + "{% elif images and not ns.criteria %}" + yes_no
+ + "{% elif o.key == 'true' %}yes, the statement holds"
+ "{% else %}no, the statement does not hold{% endif %}{% endif %}"
+ )
+ state = "{% if state is string %}{{ state }}{% elif state is not none %}{{ state | tojson }}{% elif audio %}{}{% endif %}"
+ return (
+ "{% set ns = namespace(criteria=false) %}"
+ "{% for o in options %}{% if o.description is not none %}{% set ns.criteria = true %}{% endif %}{% endfor %}"
+ "{% for image in images %}{{ image }}{% endfor %}{{ sep }}"
+ "<|startoftext|><|reserved_7|>{{ sep }}{{ mark_state }}" + state
+ + "{{ sep }}{{ mark_question }}<|reserved_8|>" + jinja_str_or_json("instructions")
+ + "{% for o in options %}{{ sep }}<|reserved_9|><|mask|>{{ sep }}{{ mark_option }} " + option
+ + "{{ sep }}<|reserved_10|>{% endfor %}{{ sep }}<|reserved_11|>"
+ )
+
+ def set_gguf_parameters(self):
+ lengths = (self.hparams["max_length"], self.hparams["image_text_length"], self.hparams["audio_text_length"])
+ if lengths != (self._MAX_LENGTH, self._IMAGE_TEXT_LENGTH, self._AUDIO_TEXT_LENGTH):
+ raise ValueError(f"unexpected text lengths: {lengths}")
+ n_head, n_layer_head = self.hparams["num_attention_heads"], self.hparams["head_layers"]
+ self.hparams["num_key_value_heads"] = [
+ self.hparams["num_key_value_heads"] if t != "conv" else 0 for t in self.hparams["layer_types"]
+ ] + [n_head] * n_layer_head
+
+ # the head needs per-layer sizes, LFM2Model writes a single feed forward length
+ TextModel.set_gguf_parameters(self)
+ self.gguf_writer.add_vocab_size(self.hparams["vocab_size"])
+ self.gguf_writer.add_shortconv_l_cache(self.hparams["conv_L_cache"])
+ self.gguf_writer.add_layer_norm_eps(1e-5) # nn.LayerNorm of the head
+ self.gguf_writer.add_causal_attention(False)
+
+ self.gguf_writer.add_decision_type(gguf.DecisionType.LFM2_D1_OMNI)
+ self.gguf_writer.add_decision_block_count(n_layer_head)
+ # "choice:3-5" -> "choice.3_5", "choice:11+" -> "choice.11"
+ for name, value in self.hparams["temperatures"].items():
+ self.gguf_writer.add_decision_temperature(name.replace(":", ".").replace("-", "_").rstrip("+"), value)
+
+ @classmethod
+ def filter_tensors(cls, item: tuple[str, Callable[[], Tensor]]) -> tuple[str, Callable[[], Tensor]] | None:
+ name, gen = item
+
+ if name.startswith(("vision.", "audio.")):
+ return None
+
+ name = name.replace("encoder.", "model.", 1) if name.startswith("encoder.") else name
+ name = name.replace("head.head.layers.", "head.layers.").replace("in_proj_", "in_proj.")
+ name = name.removeprefix("head.") if name.startswith(("head.type_emb", "head.scorer")) else name
+
+ return super().filter_tensors((name, gen))
+
+ def modify_tensors(self, data_torch: Tensor, name: str, bid: int | None) -> Iterable[tuple[str, Tensor]]:
+ if name.startswith("head.layers.") and bid is not None:
+ # the head blocks come after the trunk blocks
+ suffix = name.split(".", 3)[3]
+ bid += self.block_count - self.hparams["head_layers"]
+ name = f"head.layers.{bid}.{suffix}"
+
+ yield from super().modify_tensors(data_torch, name, bid)
+
+
@ModelBase.register("Lfm2MoeForCausalLM")
@ModelBase.example("LiquidAI/LFM2-8B-A1B")
class LFM2MoeModel(TextModel):
@@ -276,6 +391,58 @@ class LFM2VLModel(MmprojModel):
yield from super().modify_tensors(data_torch, name, bid)
+@ModelBase.register("D1OmniModel")
+@ModelBase.example("LiquidAI/d1-omni-600M")
+class D1OmniMmprojModel(ConformerAudioModel):
+ has_vision_encoder = True
+ has_audio_encoder = True
+
+ def __init__(self, *args, **kwargs):
+ super().__init__(*args, **kwargs)
+ assert self.hparams_vision is not None and self.hparams_audio is not None
+ # dynamic resolution, as LFM2VLModel
+ self.hparams_vision["image_size"] = 256
+ # the images are normalized to [-1, 1] (vision.py of the model repo)
+ self.preprocessor_config = {**self.preprocessor_config, "image_mean": [0.5] * 3, "image_std": [0.5] * 3}
+ self.hparams_audio["hidden_size"] = self.hparams_audio["d_model"]
+ self.hparams_audio["intermediate_size"] = self.hparams_audio["d_model"] * self.hparams_audio["ff_expansion_factor"]
+ self.hparams_audio["num_attention_heads"] = self.hparams_audio["n_heads"]
+
+ def set_gguf_parameters(self):
+ super().set_gguf_parameters()
+ self.gguf_writer.add_clip_vision_projector_type(gguf.VisionProjectorType.D1OMNI_V)
+ self.gguf_writer.add_vision_attention_layernorm_eps(self.find_vparam(["layer_norm_eps"]))
+ self.gguf_writer.add_vision_projector_scale_factor(self.global_config.get("downsample_factor", 2))
+ self.gguf_writer.add_vision_use_gelu(True)
+
+ assert self.hparams_audio is not None
+ self.gguf_writer.add_clip_audio_projector_type(gguf.VisionProjectorType.D1OMNI_A)
+ self.gguf_writer.add_audio_num_mel_bins(self.hparams_audio["feat_in"])
+ self.gguf_writer.add_audio_attention_layernorm_eps(1e-5)
+
+ @classmethod
+ def filter_tensors(cls, item: tuple[str, Callable[[], Tensor]]) -> tuple[str, Callable[[], Tensor]] | None:
+ name, gen = item
+
+ if name.startswith(("encoder.", "head.")):
+ return None
+
+ name = name.replace("vision.tower.", "vision_tower.").replace("vision.projector.", "multi_modal_projector.")
+ name = name.replace("audio.encoder.", "conformer.")
+ # the residual block continues the adapter: norm, linear, gelu, linear, then norm, down, up
+ for old, new in (("adapter.norm", 0), ("adapter.linear_1", 1), ("adapter.linear_2", 3),
+ ("residual.ln", 4), ("residual.down", 5), ("residual.up", 6)):
+ name = name.replace(f"audio.{old}.", f"audio_adapter.model.{new}.")
+
+ return super().filter_tensors((name, gen))
+
+ def modify_tensors(self, data_torch: Tensor, name: str, bid: int | None) -> Iterable[tuple[str, Tensor]]:
+ if "patch_embedding.weight" in name:
+ data_torch = data_torch.view(data_torch.shape[0], 16, 16, 3).permute(0, 3, 1, 2)
+
+ yield from super().modify_tensors(data_torch, name, bid)
+
+
@ModelBase.register("Lfm2AudioForConditionalGeneration")
@ModelBase.example("LiquidAI/LFM2.5-Audio-1.5B", "LiquidAI/LFM2-Audio-1.5B")
class LFM2AudioModel(ConformerAudioModel):
diff --git a/gguf-py/gguf/constants.py b/gguf-py/gguf/constants.py
index c181cb44a..fcf325fe9 100644
--- a/gguf-py/gguf/constants.py
+++ b/gguf-py/gguf/constants.py
@@ -5172,6 +5172,10 @@ MODEL_TENSORS: dict[MODEL_ARCH, list[MODEL_TENSOR]] = {
MODEL_TENSOR.ATTN_OUT,
MODEL_TENSOR.OUTPUT,
MODEL_TENSOR.DENSE_2_OUT, # LFM2-ColBert-350M
+ MODEL_TENSOR.TOKEN_TYPES, # decision head
+ MODEL_TENSOR.CLS,
+ MODEL_TENSOR.CLS_NORM,
+ MODEL_TENSOR.CLS_OUT,
],
MODEL_ARCH.LFM2MOE: [
MODEL_TENSOR.TOKEN_EMBD,
@@ -6072,6 +6076,7 @@ class DecisionType:
CLEF = "clef" # joint head over all questions, one score per option
PPLX_DECIDER = "pplx-decider" # same as openjev, label codes of 1 or 2 letters
LFM2_D1 = "lfm2-d1" # same as openjev, the labels depend on the question type
+ LFM2_D1_OMNI = "lfm2-d1-omni" # same head as laya on a bidirectional LFM2 trunk, other prompt layout
class VisionProjectorType:
@@ -6133,6 +6138,8 @@ class VisionProjectorType:
GRANITE4_VISION = "granite4_vision"
MUSE_GLIMMER = "muse-glimmer"
COHERE2V = "cohere2v"
+ D1OMNI_V = "d1omni_v" # lfm2 vision, without separator tokens
+ D1OMNI_A = "d1omni_a" # lfm2a audio, with a residual block after the projector
# Items here are (block size, type size)
diff --git a/src/llama-model.cpp b/src/llama-model.cpp
index 9f05a66e4..bf04945bf 100644
--- a/src/llama-model.cpp
+++ b/src/llama-model.cpp
@@ -2368,6 +2368,11 @@ ggml_tensor * llama_model::get_rope_factors(const llama_cparams & cparams, int i
llama_memory_i * llama_model::create_memory(const llama_memory_params & params, const llama_cparams & cparams) const {
llama_memory_i * res;
+ // the non-causal LFM2 decision graph reads the whole prompt in one batch, nothing is kept
+ if (arch == LLM_ARCH_LFM2 && !hparams.causal_attn && hparams.n_layer_decision > 0) {
+ return nullptr;
+ }
+
switch (arch) {
// Models that need specific instantiation should be handled in the
// switch statement
diff --git a/src/models/lfm2.cpp b/src/models/lfm2.cpp
index 07b71ccd3..2488c0746 100644
--- a/src/models/lfm2.cpp
+++ b/src/models/lfm2.cpp
@@ -4,6 +4,9 @@
#include <algorithm>
+// question types of a decision model: choice, score, noul
+static const uint32_t N_DECISION_TYPES = 3;
+
void llama_model_lfm2::load_arch_hparams(llama_model_loader & ml) {
ml.get_key(LLM_KV_SHORTCONV_L_CACHE, hparams.n_shortconv_l_cache);
ml.get_key(LLM_KV_ATTENTION_LAYERNORM_RMS_EPS, hparams.f_norm_rms_eps);
@@ -23,6 +26,15 @@ void llama_model_lfm2::load_arch_hparams(llama_model_loader & ml) {
default: type = LLM_TYPE_UNKNOWN;
}
+ ml.get_key(LLM_KV_DECISION_BLOCK_COUNT, hparams.n_layer_decision, false);
+ if (hparams.n_layer_decision > 0) {
+ if (hparams.n_layer_decision >= hparams.n_layer() || hparams.causal_attn) {
+ throw std::runtime_error("invalid decision head");
+ }
+ ml.get_key(LLM_KV_ATTENTION_LAYERNORM_EPS, hparams.f_norm_eps);
+ hparams.n_embd_out_impl = N_DECISION_TYPES;
+ }
+
if (const auto is_swa = ml.get_key(LLM_KV_ATTENTION_SLIDING_WINDOW, hparams.n_swa, false); is_swa && hparams.n_swa > 0) {
hparams.swa_type = LLAMA_SWA_TYPE_STANDARD;
for (uint32_t il = 0; il < hparams.n_layer(); ++il) {
@@ -37,13 +49,49 @@ void llama_model_lfm2::load_arch_tensors(llama_model_loader &) {
tok_embd = create_tensor(tn(LLM_TENSOR_TOKEN_EMBD, "weight"), {n_embd, n_vocab}, 0);
output_norm = create_tensor(tn(LLM_TENSOR_OUTPUT_NORM_LFM2, "weight"), {n_embd}, 0);
- output = create_tensor(tn(LLM_TENSOR_OUTPUT, "weight"), {n_embd, n_vocab}, TENSOR_NOT_REQUIRED);
- if (output == NULL) {
- output = create_tensor(tn(LLM_TENSOR_TOKEN_EMBD, "weight"), {n_embd, n_vocab}, TENSOR_DUPLICATED);
+ if (hparams.n_layer_decision > 0) {
+ // decision head: plain pre-norm blocks with biases
+ for (int i = n_layer - (int) hparams.n_layer_decision; i < n_layer; ++i) {
+ auto & layer = layers[i];
+ const int64_t n_ff_head = hparams.n_ff(i);
+
+ layer.attn_norm = create_tensor(tn(LLM_TENSOR_ATTN_NORM, "weight", i), {n_embd}, 0);
+ layer.attn_norm_b = create_tensor(tn(LLM_TENSOR_ATTN_NORM, "bias", i), {n_embd}, 0);
+
+ layer.wqkv = create_tensor(tn(LLM_TENSOR_ATTN_QKV, "weight", i), {n_embd, 3 * n_embd}, 0);
+ layer.wqkv_b = create_tensor(tn(LLM_TENSOR_ATTN_QKV, "bias", i), {3 * n_embd}, 0);
+ layer.wo = create_tensor(tn(LLM_TENSOR_ATTN_OUT, "weight", i), {n_embd, n_embd}, 0);
+ layer.wo_b = create_tensor(tn(LLM_TENSOR_ATTN_OUT, "bias", i), {n_embd}, 0);
+
+ layer.ffn_norm = create_tensor(tn(LLM_TENSOR_FFN_NORM, "weight", i), {n_embd}, 0);
+ layer.ffn_norm_b = create_tensor(tn(LLM_TENSOR_FFN_NORM, "bias", i), {n_embd}, 0);
+ layer.ffn_up = create_tensor(tn(LLM_TENSOR_FFN_UP, "weight", i), {n_embd, n_ff_head}, 0);
+ layer.ffn_up_b = create_tensor(tn(LLM_TENSOR_FFN_UP, "bias", i), {n_ff_head}, 0);
+ layer.ffn_down = create_tensor(tn(LLM_TENSOR_FFN_DOWN, "weight", i), {n_ff_head, n_embd}, 0);
+ layer.ffn_down_b = create_tensor(tn(LLM_TENSOR_FFN_DOWN, "bias", i), {n_embd}, 0);
+ }
+
+ if (n_token_types != N_DECISION_TYPES) {
+ throw std::runtime_error("decision model must have one token type per question type");
+ }
+ type_embd = create_tensor(tn(LLM_TENSOR_TOKEN_TYPES, "weight"), {n_embd, n_token_types}, 0);
+
+ cls_norm = create_tensor(tn(LLM_TENSOR_CLS_NORM, "weight"), {n_embd}, 0);
+ cls_norm_b = create_tensor(tn(LLM_TENSOR_CLS_NORM, "bias"), {n_embd}, 0);
+ cls = create_tensor(tn(LLM_TENSOR_CLS, "weight"), {n_embd, n_embd}, 0);
+ cls_b = create_tensor(tn(LLM_TENSOR_CLS, "bias"), {n_embd}, 0);
+ cls_out = create_tensor(tn(LLM_TENSOR_CLS_OUT, "weight"), {n_embd, 1}, 0);
+ cls_out_b = create_tensor(tn(LLM_TENSOR_CLS_OUT, "bias"), {1}, 0);
+ } else {
+ output = create_tensor(tn(LLM_TENSOR_OUTPUT, "weight"), {n_embd, n_vocab}, TENSOR_NOT_REQUIRED);
+
+ if (output == NULL) {
+ output = create_tensor(tn(LLM_TENSOR_TOKEN_EMBD, "weight"), {n_embd, n_vocab}, TENSOR_DUPLICATED);
+ }
}
- for (int i = 0; i < n_layer; ++i) {
+ for (int i = 0; i < n_layer - (int) hparams.n_layer_decision; ++i) {
auto & layer = layers[i];
const bool is_moe_layer = i >= static_cast<int>(hparams.n_layer_dense_lead);
@@ -87,6 +135,9 @@ void llama_model_lfm2::load_arch_tensors(llama_model_loader &) {
}
std::unique_ptr<llm_graph_context> llama_model_lfm2::build_arch_graph(const llm_graph_params & params) const {
+ if (hparams.n_layer_decision > 0) {
+ return std::make_unique<graph_decision>(*this, params);
+ }
if (hparams.swa_type == LLAMA_SWA_TYPE_STANDARD) {
return std::make_unique<graph<true>>(*this, params);
} else {
@@ -294,6 +345,242 @@ llama_model_lfm2::graph<iswa>::graph(const llama_model & model, const llm_graph_
ggml_build_forward_expand(gf, cur);
}
+// media entries (an image or audio prefix) are embeddings, text entries are tokens
+static bool lfm2_is_media(const llama_ubatch & ubatch, int64_t i) {
+ return ubatch.is_mixed() ? ubatch.type[i] != 0 : ubatch.token == nullptr;
+}
+
+// non-causal within a sequence, the media never reads the text, so it is a function of the media alone
+// in the head, the text and the media only read their own kind
+class llm_graph_input_attn_media : public llm_graph_input_attn_no_cache {
+public:
+ llm_graph_input_attn_media(const llama_hparams & hparams, const llama_cparams & cparams, bool is_head) :
+ llm_graph_input_attn_no_cache(hparams, cparams), is_head(is_head) {}
+
+ void set_input(const llama_ubatch * ubatch) override {
+ const int64_t n_tokens = ubatch->n_tokens;
+
+ std::vector<bool> is_media(n_tokens);
+ for (int64_t i = 0; i < n_tokens; ++i) {
+ is_media[i] = lfm2_is_media(*ubatch, i);
+ }
+
+ const auto fill_mask = [&](auto * data, auto zero, auto ninf) {
+ for (int64_t i1 = 0; i1 < n_tokens; ++i1) {
+ for (int64_t i0 = 0; i0 < n_tokens; ++i0) {
+ bool visible = ubatch->seq_id[i0][0] == ubatch->seq_id[i1][0];
+ if (is_head) {
+ visible = visible && is_media[i0] == is_media[i1];
+ } else {
+ visible = visible && !(is_media[i1] && !is_media[i0]);
+ }
+ data[i1 * n_tokens + i0] = visible ? zero : ninf;
+ }
+ }
+ };
+
+ GGML_ASSERT(ggml_backend_buffer_is_host(self_kq_mask->buffer));
+ if (self_kq_mask->type == GGML_TYPE_F16) {
+ fill_mask((ggml_fp16_t *) self_kq_mask->data, ggml_fp32_to_fp16(0.0f), ggml_fp32_to_fp16(-INFINITY));
+ } else {
+ fill_mask((float *) self_kq_mask->data, 0.0f, -INFINITY);
+ }
+ }
+
+ const bool is_head;
+};
+
+// 1 where the previous (next) token is the left (right) neighbor in the same sequence
+// the last media entry does not read the text on its right
+class llm_graph_input_conv_mask : public llm_graph_input_i {
+public:
+ void set_input(const llama_ubatch * ubatch) override {
+ const int64_t n_tokens = ubatch->n_tokens;
+
+ std::vector<float> data_left(n_tokens, 0.0f);
+ std::vector<float> data_right(n_tokens, 0.0f);
+ for (int64_t i = 0; i + 1 < n_tokens; ++i) {
+ const bool is_next = ubatch->seq_id[i][0] == ubatch->seq_id[i + 1][0] && ubatch->pos[i] + 1 == ubatch->pos[i + 1];
+ data_right[i] = is_next && !(lfm2_is_media(*ubatch, i) && !lfm2_is_media(*ubatch, i + 1));
+ data_left[i + 1] = is_next;
+ }
+ ggml_backend_tensor_set(left, data_left.data(), 0, ggml_nbytes(left));
+ ggml_backend_tensor_set(right, data_right.data(), 0, ggml_nbytes(right));
+ }
+
+ ggml_tensor * left = nullptr; // F32 [1, n_tokens]
+ ggml_tensor * right = nullptr; // F32 [1, n_tokens]
+};
+
+llama_model_lfm2::graph_decision::graph_decision(const llama_model & model, const llm_graph_params & params) :
+ llm_graph_context(params) {
+ const int64_t n_embd_head = hparams.n_embd_head_v();
+ const int n_layer_enc = n_layer - hparams.n_layer_decision;
+
+ ggml_tensor * cur = build_inp_embd(model.tok_embd);
+ cb(cur, "model.embed_tokens", -1);
+
+ ggml_tensor * inp_pos = build_inp_pos();
+ ggml_tensor * inp_out_ids = build_inp_out_ids();
+
+ const auto type_mask = cparams.flash_attn ? GGML_TYPE_F16 : GGML_TYPE_F32;
+
+ llm_graph_input_attn_no_cache * inp_attn[2];
+ for (bool is_head : {false, true}) {
+ auto inp = std::make_unique<llm_graph_input_attn_media>(hparams, cparams, is_head);
+ inp->self_kq_mask = ggml_new_tensor_4d(ctx0, type_mask, n_tokens, n_tokens, 1, 1);
+ ggml_set_input(inp->self_kq_mask);
+ inp->self_kq_mask_cnv = inp->self_kq_mask;
+ inp_attn[is_head] = (llm_graph_input_attn_no_cache *) res->add_input(std::move(inp));
+ }
+
+ auto inp_conv = std::make_unique<llm_graph_input_conv_mask>();
+ inp_conv->left = ggml_new_tensor_2d(ctx0, GGML_TYPE_F32, 1, n_tokens);
+ inp_conv->right = ggml_new_tensor_2d(ctx0, GGML_TYPE_F32, 1, n_tokens);
+ ggml_set_input(inp_conv->left);
+ ggml_set_input(inp_conv->right);
+ ggml_tensor * conv_left = inp_conv->left;
+ ggml_tensor * conv_right = inp_conv->right;
+ res->add_input(std::move(inp_conv));
+
+ for (int il = 0; il < n_layer_enc; ++il) {
+ const auto & layer = model.layers[il];
+
+ ggml_tensor * inpL = cur;
+ cur = build_norm(cur, layer.attn_norm, NULL, LLM_NORM_RMS, il);
+ cb(cur, "model.layers.{}.operator_norm", il);
+
+ if (hparams.is_recr(il)) {
+ ggml_tensor * bcx = build_lora_mm(layer.shortconv.in_proj, cur);
+ cb(bcx, "model.layers.{}.conv.in_proj", il);
+
+ ggml_tensor * b = ggml_view_2d(ctx0, bcx, n_embd, n_tokens, bcx->nb[1], 0 * n_embd * ggml_element_size(bcx));
+ ggml_tensor * c = ggml_view_2d(ctx0, bcx, n_embd, n_tokens, bcx->nb[1], 1 * n_embd * ggml_element_size(bcx));
+ ggml_tensor * x = ggml_view_2d(ctx0, bcx, n_embd, n_tokens, bcx->nb[1], 2 * n_embd * ggml_element_size(bcx));
+
+ // centred 3-tap conv, a tap outside the sequence reads 0
+ ggml_tensor * bx = ggml_mul(ctx0, b, x);
+ ggml_tensor * bxp = ggml_pad_ext(ctx0, bx, 0, 0, 1, 1, 0, 0, 0, 0);
+ ggml_tensor * prv = ggml_view_2d(ctx0, bxp, n_embd, n_tokens, bxp->nb[1], 0);
+ ggml_tensor * nxt = ggml_view_2d(ctx0, bxp, n_embd, n_tokens, bxp->nb[1], 2 * bxp->nb[1]);
+
+ GGML_ASSERT(hparams.n_shortconv_l_cache == 3);
+ ggml_tensor * taps = ggml_cont(ctx0, ggml_transpose(ctx0, layer.shortconv.conv));
+ ggml_tensor * tap0 = ggml_view_1d(ctx0, taps, n_embd, 0 * taps->nb[1]);
+ ggml_tensor * tap1 = ggml_view_1d(ctx0, taps, n_embd, 1 * taps->nb[1]);
+ ggml_tensor * tap2 = ggml_view_1d(ctx0, taps, n_embd, 2 * taps->nb[1]);
+
+ ggml_tensor * y = ggml_mul(ctx0, bx, tap1);
+ y = ggml_add(ctx0, y, ggml_mul(ctx0, ggml_mul(ctx0, prv, tap0), conv_left));
+ y = ggml_add(ctx0, y, ggml_mul(ctx0, ggml_mul(ctx0, nxt, tap2), conv_right));
+ cb(y, "model.layers.{}.conv.conv", il);
+
+ cur = build_lora_mm(layer.shortconv.out_proj, ggml_mul(ctx0, c, y));
+ cb(cur, "model.layers.{}.conv.out_proj", il);
+ } else {
+ auto [q, k, v] = build_qkv(layer, cur, n_embd_head, n_head, hparams.n_head_kv(il), il);
+
+ q = build_norm(q, layer.attn_q_norm, NULL, LLM_NORM_RMS, il);
+ k = build_norm(k, layer.attn_k_norm, NULL, LLM_NORM_RMS, il);
+
+ q = ggml_rope_ext(ctx0, q, inp_pos, nullptr, n_rot, rope_type, n_ctx_orig, freq_base, freq_scale, ext_factor,
+ attn_factor, beta_fast, beta_slow);
+ k = ggml_rope_ext(ctx0, k, inp_pos, nullptr, n_rot, rope_type, n_ctx_orig, freq_base, freq_scale, ext_factor,
+ attn_factor, beta_fast, beta_slow);
+
+ cur = build_attn(inp_attn[0],
+ layer.wo, NULL, layer.wo_s,
+ q, k, v, nullptr, nullptr, nullptr, 1.0f / sqrtf(float(n_embd_head)), il);
+ cb(cur, "model.layers.{}.self_attn.out_proj", il);
+ }
+
+ cur = ggml_add(ctx0, cur, inpL);
+
+ ggml_tensor * ffn_out = build_norm(cur, layer.ffn_norm, NULL, LLM_NORM_RMS, il);
+ ffn_out = build_ffn(ffn_out,
+ layer.ffn_up, NULL, NULL,
+ layer.ffn_gate, NULL, NULL,
+ layer.ffn_down, NULL, NULL,
+ NULL, LLM_FFN_SILU, LLM_FFN_PAR, il);
+
+ cur = ggml_add(ctx0, cur, ffn_out);
+ cb(cur, "l_out", il);
+ }
+
+ cur = build_norm(cur, model.output_norm, NULL, LLM_NORM_RMS, -1);
+ cb(cur, "result_norm", -1);
+
+ cur = build_decision_head(model, cur, inp_attn[1], inp_out_ids);
+
+ res->t_embd = cur;
+ ggml_build_forward_expand(gf, cur);
+}
+
+// same as llama_model_modern_bert::graph::build_decision_head(), with the head counts of the head layers
+ggml_tensor * llama_model_lfm2::graph_decision::build_decision_head(
+ const llama_model & model,
+ ggml_tensor * inp,
+ llm_graph_input_attn_no_cache * inp_attn,
+ ggml_tensor * inp_out_ids) {
+ const int64_t n_embd_head = hparams.n_embd_head_v();
+ const int n_layer_enc = n_layer - hparams.n_layer_decision;
+
+ ggml_tensor * scores = nullptr;
+
+ // the question type is not a graph input, so the head is evaluated for each of them
+ for (uint32_t it = 0; it < N_DECISION_TYPES; ++it) {
+ ggml_tensor * type_row = ggml_view_1d(ctx0, model.type_embd, n_embd, it * model.type_embd->nb[1]);
+ ggml_tensor * inpL = ggml_add(ctx0, inp, type_row);
+
+ for (int il = n_layer_enc; il < n_layer; ++il) {
+ const auto & layer = model.layers[il];
+
+ ggml_tensor * cur = build_norm(inpL, layer.attn_norm, layer.attn_norm_b, LLM_NORM, il);
+ cb(cur, "attn_norm", il);
+
+ // no positional encoding in the head
+ auto [Qcur, Kcur, Vcur] = build_qkv(layer, cur, n_embd_head, hparams.n_head(il), hparams.n_head_kv(il), il);
+
+ cur = build_attn(inp_attn,
+ layer.wo, layer.wo_b, layer.wo_s,
+ Qcur, Kcur, Vcur, nullptr, nullptr, nullptr, 1.0f/sqrtf(float(n_embd_head)), il);
+ cb(cur, "kqv_out", il);
+
+ if (il == n_layer - 1 && inp_out_ids) {
+ cur = ggml_get_rows(ctx0, cur, inp_out_ids);
+ inpL = ggml_get_rows(ctx0, inpL, inp_out_ids);
+ }
+
+ ggml_tensor * ffn_inp = ggml_add(ctx0, cur, inpL);
+ cb(ffn_inp, "ffn_inp", il);
+
+ cur = build_norm(ffn_inp, layer.ffn_norm, layer.ffn_norm_b, LLM_NORM, il);
+ cb(cur, "ffn_norm", il);
+
+ cur = build_ffn(cur,
+ layer.ffn_up, layer.ffn_up_b, NULL,
+ NULL, NULL, NULL,
+ layer.ffn_down, layer.ffn_down_b, NULL,
+ NULL,
+ LLM_FFN_RELU,
+ LLM_FFN_SEQ, il);
+
+ inpL = ggml_add(ctx0, cur, ffn_inp);
+ }
+
+ // scorer
+ ggml_tensor * cur = build_norm(inpL, model.cls_norm, model.cls_norm_b, LLM_NORM, -1);
+ cur = ggml_add(ctx0, build_lora_mm(model.cls, cur), model.cls_b);
+ cur = ggml_gelu_erf(ctx0, cur);
+ cur = ggml_add(ctx0, build_lora_mm(model.cls_out, cur), model.cls_out_b);
+
+ scores = scores ? ggml_concat(ctx0, scores, cur, 0) : cur;
+ }
+ cb(scores, "decision_scores", -1);
+
+ return scores;
+}
+
// Explicit template instantiations
template struct llama_model_lfm2::graph<true>;
template struct llama_model_lfm2::graph<false>;
diff --git a/src/models/models.h b/src/models/models.h
index 1ef0c5156..d6ddf9d16 100644
--- a/src/models/models.h
+++ b/src/models/models.h
@@ -2186,6 +2186,17 @@ struct llama_model_lfm2 : public llama_model_base {
graph(const llama_model & model, const llm_graph_params & params);
};
+ // non-causal trunk without memory, then the decision head
+ struct graph_decision : public llm_graph_context {
+ graph_decision(const llama_model & model, const llm_graph_params & params);
+
+ ggml_tensor * build_decision_head(
+ const llama_model & model,
+ ggml_tensor * inp,
+ llm_graph_input_attn_no_cache * inp_attn,
+ ggml_tensor * inp_out_ids);
+ };
+
std::unique_ptr<llm_graph_context> build_arch_graph(const llm_graph_params & params) const override;
};
diff --git a/tools/mtmd/clip-impl.h b/tools/mtmd/clip-impl.h
index abf75d998..de171b7a5 100644
--- a/tools/mtmd/clip-impl.h
+++ b/tools/mtmd/clip-impl.h
@@ -476,6 +476,7 @@ enum projector_type {
PROJECTOR_TYPE_MERALION,
PROJECTOR_TYPE_MUSIC_FLAMINGO,
PROJECTOR_TYPE_LFM2,
+ PROJECTOR_TYPE_D1OMNI_V,
PROJECTOR_TYPE_KIMIVL,
PROJECTOR_TYPE_PADDLEOCR,
PROJECTOR_TYPE_LIGHTONOCR,
@@ -488,6 +489,7 @@ enum projector_type {
PROJECTOR_TYPE_DEEPSEEKOCR2,
PROJECTOR_TYPE_DEEPSEEK4V,
PROJECTOR_TYPE_LFM2A,
+ PROJECTOR_TYPE_D1OMNI_A,
PROJECTOR_TYPE_GLM4V,
PROJECTOR_TYPE_GLM5V,
PROJECTOR_TYPE_YOUTUVL,
@@ -544,6 +546,7 @@ static std::map<projector_type, std::string> PROJECTOR_TYPE_NAMES = {
{ PROJECTOR_TYPE_MERALION, "meralion"},
{ PROJECTOR_TYPE_MUSIC_FLAMINGO, "musicflamingo"},
{ PROJECTOR_TYPE_LFM2, "lfm2"},
+ { PROJECTOR_TYPE_D1OMNI_V, "d1omni_v"},
{ PROJECTOR_TYPE_KIMIVL, "kimivl"},
{ PROJECTOR_TYPE_PADDLEOCR, "paddleocr"},
{ PROJECTOR_TYPE_LIGHTONOCR, "lightonocr"},
@@ -556,6 +559,7 @@ static std::map<projector_type, std::string> PROJECTOR_TYPE_NAMES = {
{ PROJECTOR_TYPE_DEEPSEEKOCR2, "deepseekocr2"},
{ PROJECTOR_TYPE_DEEPSEEK4V, "deepseek4v"},
{ PROJECTOR_TYPE_LFM2A, "lfm2a"},
+ { PROJECTOR_TYPE_D1OMNI_A, "d1omni_a"},
{ PROJECTOR_TYPE_GLM4V, "glm4v"},
{ PROJECTOR_TYPE_GLM5V, "glm5v"},
{ PROJECTOR_TYPE_YOUTUVL, "youtuvl"},
diff --git a/tools/mtmd/clip-model.h b/tools/mtmd/clip-model.h
index 33f679fc0..a5ff38822 100644
--- a/tools/mtmd/clip-model.h
+++ b/tools/mtmd/clip-model.h
@@ -623,6 +623,10 @@ struct clip_model {
ggml_tensor * mm_3_b = nullptr;
ggml_tensor * mm_4_w = nullptr;
ggml_tensor * mm_4_b = nullptr;
+ ggml_tensor * mm_5_w = nullptr;
+ ggml_tensor * mm_5_b = nullptr;
+ ggml_tensor * mm_6_w = nullptr;
+ ggml_tensor * mm_6_b = nullptr;
// GLMV-Edge projection
ggml_tensor * mm_model_adapter_conv_w = nullptr;
diff --git a/tools/mtmd/clip.cpp b/tools/mtmd/clip.cpp
index b7302e4bf..ca38d76df 100644
--- a/tools/mtmd/clip.cpp
+++ b/tools/mtmd/clip.cpp
@@ -939,6 +939,7 @@ static std::unique_ptr<clip_graph> clip_get_graph_builder(clip_ctx * ctx, const
case PROJECTOR_TYPE_IDEFICS3:
case PROJECTOR_TYPE_COHERE2V:
case PROJECTOR_TYPE_LFM2:
+ case PROJECTOR_TYPE_D1OMNI_V:
case PROJECTOR_TYPE_JANUS_PRO:
case PROJECTOR_TYPE_PHI4:
{
@@ -1073,6 +1074,7 @@ static std::unique_ptr<clip_graph> clip_get_graph_builder(clip_ctx * ctx, const
builder = std::make_unique<clip_graph_deepseekocr2>(ctx, img);
} break;
case PROJECTOR_TYPE_LFM2A:
+ case PROJECTOR_TYPE_D1OMNI_A:
{
builder = std::make_unique<clip_graph_conformer>(ctx, img);
} break;
@@ -1525,6 +1527,7 @@ struct clip_model_loader {
}
} break;
case PROJECTOR_TYPE_LFM2:
+ case PROJECTOR_TYPE_D1OMNI_V:
{
// default for older GGUFs
std::string resize_algo = "bilinear";
@@ -1540,6 +1543,10 @@ struct clip_model_loader {
}
hparams.image_resize_algo_rf = hparams.image_resize_algo;
hparams.image_resize_algo_ov = hparams.image_resize_algo;
+ // the tiles stretch the image to the grid (d1-omni vision.py)
+ if (model.proj_type == PROJECTOR_TYPE_D1OMNI_V) {
+ hparams.image_pad_rf = PAD_NONE;
+ }
get_u32(KEY_PROJ_SCALE_FACTOR, hparams.n_merge, false);
// ref: https://huggingface.co/LiquidAI/LFM2.5-VL-1.6B/blob/main/processor_config.json
hparams.set_limit_image_tokens(64, 256);
@@ -1980,6 +1987,7 @@ struct clip_model_loader {
hparams.set_warmup_n_tokens(32*32);
} break;
case PROJECTOR_TYPE_LFM2A:
+ case PROJECTOR_TYPE_D1OMNI_A:
{
// audio preprocessing params
hparams.audio_chunk_len = 1; // in seconds
@@ -2799,6 +2807,7 @@ struct clip_model_loader {
model.mm_2_b = get_tensor(string_format(TN_LLAVA_PROJ, 2, "bias"));
} break;
case PROJECTOR_TYPE_LFM2:
+ case PROJECTOR_TYPE_D1OMNI_V:
{
model.mm_input_norm_w = get_tensor(TN_MM_INP_NORM, false);
model.mm_input_norm_b = get_tensor(TN_MM_INP_NORM_B, false);
@@ -3411,6 +3420,7 @@ struct clip_model_loader {
model.mm_input_proj_w = get_tensor(string_format(TN_A_MM_INP_PROJ, "weight"));
} break;
case PROJECTOR_TYPE_LFM2A:
+ case PROJECTOR_TYPE_D1OMNI_A:
{
for (int i : {0, 2, 3, 5, 6}) {
model.pre_encode_conv_X_w[i] = get_tensor(string_format(TN_CONV1D, i, "weight"));
@@ -3426,6 +3436,16 @@ struct clip_model_loader {
model.mm_3_w = get_tensor(string_format(TN_MM_AUDIO_MLP, 3, "weight"));
model.mm_3_b = get_tensor(string_format(TN_MM_AUDIO_MLP, 3, "bias"));
+ // residual block after the projector: norm, down, up
+ if (model.proj_type == PROJECTOR_TYPE_D1OMNI_A) {
+ model.mm_4_w = get_tensor(string_format(TN_MM_AUDIO_MLP, 4, "weight"));
+ model.mm_4_b = get_tensor(string_format(TN_MM_AUDIO_MLP, 4, "bias"));
+ model.mm_5_w = get_tensor(string_format(TN_MM_AUDIO_MLP, 5, "weight"));
+ model.mm_5_b = get_tensor(string_format(TN_MM_AUDIO_MLP, 5, "bias"));
+ model.mm_6_w = get_tensor(string_format(TN_MM_AUDIO_MLP, 6, "weight"));
+ model.mm_6_b = get_tensor(string_format(TN_MM_AUDIO_MLP, 6, "bias"));
+ }
+
for (int il = 0; il < hparams.n_layer; ++il) {
auto & layer = model.layers[il];
@@ -4277,6 +4297,7 @@ int clip_n_output_tokens(const clip_ctx * ctx, const clip_image_f32 * img) {
n_patches = ctx->model.hparams.image_size / ctx->model.hparams.patch_size;
} break;
case PROJECTOR_TYPE_LFM2:
+ case PROJECTOR_TYPE_D1OMNI_V:
case PROJECTOR_TYPE_KIMIVL:
case PROJECTOR_TYPE_KIMIK25:
{
@@ -4400,6 +4421,7 @@ int clip_n_output_tokens(const clip_ctx * ctx, const clip_image_f32 * img) {
}
} break;
case PROJECTOR_TYPE_LFM2A:
+ case PROJECTOR_TYPE_D1OMNI_A:
{
n_patches = ((((img->nx() + 1) / 2) + 1) / 2 + 1) / 2;
} break;
@@ -5358,6 +5380,7 @@ bool clip_encode(struct clip_ctx * ctx, struct clip_encode_params * params) {
case PROJECTOR_TYPE_GLMA:
case PROJECTOR_TYPE_ULTRAVOX:
case PROJECTOR_TYPE_LFM2:
+ case PROJECTOR_TYPE_D1OMNI_V:
case PROJECTOR_TYPE_VOXTRAL:
case PROJECTOR_TYPE_MERALION:
case PROJECTOR_TYPE_MUSIC_FLAMINGO:
@@ -5617,6 +5640,7 @@ bool clip_encode(struct clip_ctx * ctx, struct clip_encode_params * params) {
}
} break;
case PROJECTOR_TYPE_LFM2A:
+ case PROJECTOR_TYPE_D1OMNI_A:
{
GGML_ASSERT(imgs.entries.size() == 1);
const auto n_frames = clip_n_output_tokens(ctx, &imgs.entries.front());
@@ -6080,6 +6104,7 @@ int clip_n_mmproj_embd(const struct clip_ctx * ctx) {
return ctx->model.mm_2_w->ne[1];
case PROJECTOR_TYPE_GLMA:
case PROJECTOR_TYPE_LFM2:
+ case PROJECTOR_TYPE_D1OMNI_V:
case PROJECTOR_TYPE_KIMIVL:
case PROJECTOR_TYPE_PADDLEOCR:
case PROJECTOR_TYPE_KIMIK25:
@@ -6096,6 +6121,8 @@ int clip_n_mmproj_embd(const struct clip_ctx * ctx) {
return ctx->model.mm_fc_w->ne[1];
case PROJECTOR_TYPE_LFM2A:
return ctx->model.position_embeddings->ne[0];
+ case PROJECTOR_TYPE_D1OMNI_A:
+ return ctx->model.mm_3_w->ne[1];
case PROJECTOR_TYPE_GRANITE_SPEECH:
return ctx->model.qf_proj_blocks[0].qf_proj_linear_w->ne[1];
case PROJECTOR_TYPE_GRANITE4_VISION:
diff --git a/tools/mtmd/models/conformer.cpp b/tools/mtmd/models/conformer.cpp
index 18c3d27bc..9463f9f1c 100644
--- a/tools/mtmd/models/conformer.cpp
+++ b/tools/mtmd/models/conformer.cpp
@@ -4,7 +4,7 @@ ggml_cgraph * clip_graph_conformer::build() {
const int n_frames = img.nx();
const int n_pos = n_frames / 2;
const int n_pos_embd = (((((n_frames + 1) / 2) + 1) / 2 + 1) / 2) * 2 - 1;
- GGML_ASSERT(model.position_embeddings->ne[1] >= n_pos);
+ GGML_ASSERT(!model.position_embeddings || model.position_embeddings->ne[1] >= n_pos);
ggml_tensor * pos_emb = ggml_new_tensor_2d(ctx0, GGML_TYPE_F32, 512, n_pos_embd);
ggml_set_name(pos_emb, "pos_emb");
@@ -164,7 +164,7 @@ ggml_cgraph * clip_graph_conformer::build() {
// TODO @ngxson : support this ops in ggml
{
int64_t d = x->ne[0] / 2;
- ggml_tensor * gate = ggml_sigmoid(ctx0, ggml_view_2d(ctx0, x, d, x->ne[1], x->nb[1], d * x->nb[0]));
+ ggml_tensor * gate = ggml_sigmoid(ctx0, ggml_cont(ctx0, ggml_view_2d(ctx0, x, d, x->ne[1], x->nb[1], d * x->nb[0])));
x = ggml_mul(ctx0, ggml_view_2d(ctx0, x, d, x->ne[1], x->nb[1], 0), gate);
x = ggml_cont(ctx0, ggml_transpose(ctx0, x));
}
@@ -207,6 +207,13 @@ ggml_cgraph * clip_graph_conformer::build() {
cb(cur, "audio_adapter.model.{}", 0);
cur = build_ffn(cur, model.mm_1_w, model.mm_1_b, nullptr, nullptr, model.mm_3_w, model.mm_3_b, FFN_GELU_ERF, -1);
+ // d1omni_a: residual block after the projector
+ if (model.mm_4_w) {
+ ggml_tensor * x = build_norm(cur, model.mm_4_w, model.mm_4_b, NORM_TYPE_NORMAL, 1e-5, -1);
+ x = build_ffn(x, model.mm_5_w, model.mm_5_b, nullptr, nullptr, model.mm_6_w, model.mm_6_b, FFN_GELU_ERF, -1);
+ cur = ggml_add(ctx0, cur, x);
+ }
+
cb(cur, "projected", -1);
ggml_build_forward_expand(gf, cur);
diff --git a/tools/mtmd/models/siglip.cpp b/tools/mtmd/models/siglip.cpp
index 96d21ceee..8cc901a28 100644
--- a/tools/mtmd/models/siglip.cpp
+++ b/tools/mtmd/models/siglip.cpp
@@ -4,7 +4,7 @@ ggml_cgraph * clip_graph_siglip::build() {
ggml_tensor * inp = build_inp();
ggml_tensor * learned_pos_embd = model.position_embeddings;
- if (proj_type == PROJECTOR_TYPE_LFM2 || proj_type == PROJECTOR_TYPE_PHI4) {
+ if (proj_type == PROJECTOR_TYPE_LFM2 || proj_type == PROJECTOR_TYPE_D1OMNI_V || proj_type == PROJECTOR_TYPE_PHI4) {
learned_pos_embd = resize_position_embeddings();
}
@@ -55,7 +55,7 @@ ggml_cgraph * clip_graph_siglip::build() {
cur = build_mm(model.mm_2_w, cur);
cur = ggml_add(ctx0, cur, model.mm_2_b);
- } else if (proj_type == PROJECTOR_TYPE_LFM2) {
+ } else if (proj_type == PROJECTOR_TYPE_LFM2 || proj_type == PROJECTOR_TYPE_D1OMNI_V) {
// pixel unshuffle block
const int scale_factor = model.hparams.n_merge;
cur = build_patch_merge_permute(cur, scale_factor);
@@ -70,11 +70,12 @@ ggml_cgraph * clip_graph_siglip::build() {
cur = ggml_add(ctx0, cur, model.mm_input_norm_b);
}
+ // d1-omni uses the exact gelu in the projector
cur = build_ffn(cur,
model.mm_1_w, model.mm_1_b,
nullptr, nullptr,
model.mm_2_w, model.mm_2_b,
- FFN_GELU,
+ proj_type == PROJECTOR_TYPE_D1OMNI_V ? FFN_GELU_ERF : FFN_GELU,
-1);
} else if (proj_type == PROJECTOR_TYPE_JANUS_PRO) {
diff --git a/tools/mtmd/mtmd-audio.cpp b/tools/mtmd/mtmd-audio.cpp
index 6c8b19441..1b4ea60bf 100644
--- a/tools/mtmd/mtmd-audio.cpp
+++ b/tools/mtmd/mtmd-audio.cpp
@@ -994,6 +994,42 @@ bool mtmd_audio_preprocessor_conformer::preprocess(const float *
return true;
}
+//
+// mtmd_audio_preprocessor_d1omni
+//
+
+bool mtmd_audio_preprocessor_d1omni::preprocess(const float * samples,
+ size_t n_samples,
+ std::vector<mtmd_audio_mel> & output) const {
+ if (n_samples == 0) {
+ return false;
+ }
+ const size_t n_max = 30 * hparams.audio_sample_rate;
+ const size_t n_min = hparams.audio_sample_rate / 2;
+
+ std::vector<float> buf(samples, samples + std::min(n_samples, n_max));
+ buf.resize(std::max(buf.size(), n_min), 0.0f);
+
+ if (!mtmd_audio_preprocessor_conformer::preprocess(buf.data(), buf.size(), output)) {
+ return false;
+ }
+
+ // the encoder reads one frame per hop, not the extra frame of the centre padding (NeMo: seq_len)
+ const int64_t n_frames = buf.size() / hparams.audio_hop_len;
+ for (auto & mel : output) {
+ if (mel.n_len <= n_frames) {
+ continue;
+ }
+ std::vector<float> data((size_t) mel.n_mel * n_frames);
+ for (int64_t j = 0; j < mel.n_mel; ++j) {
+ std::copy_n(mel.data.begin() + (size_t) j * mel.n_len, n_frames, data.begin() + (size_t) j * n_frames);
+ }
+ mel.n_len = n_frames;
+ mel.data = std::move(data);
+ }
+ return true;
+}
+
//
// mtmd_audio_preprocessor_granite_speech
//
diff --git a/tools/mtmd/mtmd-audio.h b/tools/mtmd/mtmd-audio.h
index 36db50526..4ca6cbd55 100644
--- a/tools/mtmd/mtmd-audio.h
+++ b/tools/mtmd/mtmd-audio.h
@@ -80,6 +80,12 @@ struct mtmd_audio_preprocessor_conformer : mtmd_audio_preprocessor {
mtmd_audio_cache cache;
};
+// same as conformer, the audio is cut to 30 s and padded to 0.5 s (d1-omni audio.py)
+struct mtmd_audio_preprocessor_d1omni : mtmd_audio_preprocessor_conformer {
+ using mtmd_audio_preprocessor_conformer::mtmd_audio_preprocessor_conformer;
+ bool preprocess(const float * samples, size_t n_samples, std::vector<mtmd_audio_mel> & output) const override;
+};
+
struct mtmd_audio_preprocessor_granite_speech : mtmd_audio_preprocessor {
mtmd_audio_preprocessor_granite_speech(const clip_ctx * ctx) : mtmd_audio_preprocessor(ctx) {}
void initialize() override;
diff --git a/tools/mtmd/mtmd.cpp b/tools/mtmd/mtmd.cpp
index b2319b901..b6e331bc7 100644
--- a/tools/mtmd/mtmd.cpp
+++ b/tools/mtmd/mtmd.cpp
@@ -871,6 +871,11 @@ struct mtmd_context {
ov_img_first = false;
image_preproc = std::make_unique<mtmd_image_preprocessor_lfm2>(ctx_v);
} break;
+ case PROJECTOR_TYPE_D1OMNI_V:
+ {
+ // same tiles and thumbnail as lfm2, without separator tokens
+ image_preproc = std::make_unique<mtmd_image_preprocessor_lfm2>(ctx_v);
+ } break;
case PROJECTOR_TYPE_GLM4V:
{
// <|begin_of_image|> ... (image embeddings) ... <|end_of_image|>
@@ -982,6 +987,10 @@ struct mtmd_context {
{
audio_preproc = std::make_unique<mtmd_audio_preprocessor_conformer>(ctx_a);
} break;
+ case PROJECTOR_TYPE_D1OMNI_A:
+ {
+ audio_preproc = std::make_unique<mtmd_audio_preprocessor_d1omni>(ctx_a);
+ } break;
case PROJECTOR_TYPE_GRANITE_SPEECH:
{
audio_preproc = std::make_unique<mtmd_audio_preprocessor_granite_speech>(ctx_a);
diff --git a/tools/server/README.md b/tools/server/README.md
index 8b271395b..9231f06e8 100644
--- a/tools/server/README.md
+++ b/tools/server/README.md
@@ -1703,9 +1703,11 @@ Follows the [TypeSafe API](https://docs.typesafe.ai/api), streaming is not suppo
*Options:*
-`state`: The content to evaluate. Can be a string, an object or an array. A value that is not a string is given to the model as JSON text. For lfm2-d1, it can be `null`, for example to ask about images only.
+`state`: The content to evaluate. Can be a string, an object or an array. A value that is not a string is given to the model as JSON text. For lfm2-d1 and lfm2-d1-omni, it can be `null`, for example to ask about images only.
-`images`: Optional. An array of images, the maximum number may be limited depending on the model. Each one is a data URL (`data:image/...;base64,...`). See the image input section below.
+`files`: Optional. An array of input files, the maximum number may be limited depending on the model. Each one is a data URL (`data:image/...;base64,...`). For audio-capable models, it can be audio clips (`data:audio/...;base64,...`). See the image input section below.
+
+`images`: Optional. An alias of `files`.
`questions`: An object that maps a question id to a question. Each question has these fields:
@@ -1718,20 +1720,20 @@ Follows the [TypeSafe API](https://docs.typesafe.ai/api), streaming is not suppo
The questions of a request are answered independently, an answer does not depend on the other questions. The exception is clef: it reads all the questions in one prompt and decides them jointly.
-The number of options of a `choice` question is limited by the model, for example: 52 for openjev, 255 for laya, clef, pplx-decider and lfm2-d1. For laya, long questions and options are truncated to the token budget the model was trained with.
+The number of options of a `choice` question is limited by the model, for example: 52 for openjev, 255 for laya, clef, pplx-decider, lfm2-d1 and lfm2-d1-omni. For laya, long questions and options are truncated to the token budget the model was trained with.
-For laya and clef, the whole prompt is evaluated in one batch: it must fit in `--ubatch-size`. A server that runs clef only serves this endpoint, text generation is not available.
+For laya, clef and lfm2-d1-omni, the whole prompt is evaluated in one batch: it must fit in `--ubatch-size`. An lfm2-d1-omni prompt is cut to 16384 tokens. A server that runs clef only serves this endpoint, text generation is not available.
*Image input:*
-Image input needs a model that supports it (for example: openjev, clef, pplx-decider, lfm2-d1) and its multimodal projector, see `--mmproj`.
+Image input needs a model that supports it (for example: openjev, clef, pplx-decider, lfm2-d1, lfm2-d1-omni) and its multimodal projector, see `--mmproj`.
Images can be given in two ways, and both can be used in the same request:
-- The `images` field.
-- A `state` made of chat messages, either an array of messages or an object with a `messages` array. An `image_url` part in the `content` of a message is taken as an image, in the same format as chat completions. Only data URLs are accepted.
+- The `files` field, or its alias `images`.
+- A `state` made of chat messages, either an array of messages or an object with a `messages` array. An `image_url` part in the `content` of a message is taken as an image, in the same format as chat completions. Only data URLs are accepted. For lfm2-d1-omni, an `input_audio` part is taken as an audio clip, as base64 data.
-All the images are placed before the state in the prompt, the ones from `images` first. The image parts are removed from the state.
+All the images are placed before the state in the prompt, the ones from `files` and `images` first. The image parts are removed from the state.
*Response:*
diff --git a/tools/server/server-common.cpp b/tools/server/server-common.cpp
index f56595231..d364ba898 100644
--- a/tools/server/server-common.cpp
+++ b/tools/server/server-common.cpp
@@ -157,6 +157,7 @@ std::vector<std::string> server_model_output_modalities(common_decision_type dec
case COMMON_DECISION_TYPE_CLEF:
case COMMON_DECISION_TYPE_PPLX_DECIDER:
case COMMON_DECISION_TYPE_LFM2_D1:
+ case COMMON_DECISION_TYPE_LFM2_D1_OMNI:
return {"decisions"};
default:
// fallback when there is no decision type or the metadata is bad
diff --git a/tools/server/server-context.cpp b/tools/server/server-context.cpp
index 6f3519215..e3270747b 100644
--- a/tools/server/server-context.cpp
+++ b/tools/server/server-context.cpp
@@ -5660,7 +5660,7 @@ void server_routes::init_routes() {
scores.push_back(result->scores);
n_tokens += result->n_tokens;
}
- answers[question.id] = decision.format_answer(question, scores);
+ answers[question.id] = decision.format_answer(question, scores, !files.empty());
}
res->ok(json{
diff --git a/tools/server/server-decision.cpp b/tools/server/server-decision.cpp
index 71657a359..11ad66a19 100644
--- a/tools/server/server-decision.cpp
+++ b/tools/server/server-decision.cpp
@@ -126,6 +126,12 @@ void server_decision_context::init(const llama_model * model) {
} else if (model_type == COMMON_DECISION_TYPE_LFM2_D1) {
n_options_max = 255;
noul_true_first = true;
+ } else if (model_type == COMMON_DECISION_TYPE_LFM2_D1_OMNI) {
+ token_marker = llama_vocab_mask(vocab);
+ if (token_marker == LLAMA_TOKEN_NULL) {
+ throw std::runtime_error("decision model has no mask token");
+ }
+ n_options_max = 255;
} else {
throw std::runtime_error("unsupported decision model type: " + type_name);
}
@@ -139,8 +145,8 @@ void server_decision_context::init(const llama_model * model) {
//
std::vector<server_decision_question> server_decision_context::parse_questions(const json & body) const {
- // d1 accepts a null state (images only)
- if (!body.contains("state") || (body.at("state").is_null() && type != COMMON_DECISION_TYPE_LFM2_D1)) {
+ // lfm2-d1 and lfm2-d1-omni accept a null state, for example to ask about images only
+ if (!body.contains("state") || (body.at("state").is_null() && type != COMMON_DECISION_TYPE_LFM2_D1 && type != COMMON_DECISION_TYPE_LFM2_D1_OMNI)) {
throw std::invalid_argument("\"state\" must be provided");
}
if (!body.contains("questions") || !body.at("questions").is_object() || body.at("questions").empty()) {
@@ -193,7 +199,15 @@ std::vector<server_decision_question> server_decision_context::parse_questions(c
throw err("\"criteria\" must be an object");
}
for (const char * key : {"false", "true"}) {
- question.options.push_back({key, criteria.is_object() && criteria.contains(key) ? criteria.at(key) : json()});
+ json description;
+ if (criteria.is_object() && criteria.contains(key)) {
+ description = criteria.at(key);
+ } else if (criteria.is_object() && type == COMMON_DECISION_TYPE_LFM2_D1_OMNI) {
+ // lfm2-d1-omni also reads the descriptions under "no" and "yes"
+ const char * alias = std::string(key) == "true" ? "yes" : "no";
+ description = criteria.contains(alias) ? criteria.at(alias) : json();
+ }
+ question.options.push_back({key, description});
}
if (noul_true_first) {
std::swap(question.options[0], question.options[1]);
@@ -217,9 +231,10 @@ std::vector<server_decision_question> server_decision_context::parse_questions(c
static const size_t DECISION_MAX_IMAGES = 8;
+// any media, mtmd tells an audio clip from an image by its content
static void decision_load_image(const json & url, std::vector<raw_buffer> & files) {
- if (!url.is_string() || !string_starts_with(url.get<std::string>(), "data:image/")) {
- throw std::invalid_argument("images must be data URLs (data:image/...;base64,...)");
+ if (!url.is_string() || !string_starts_with(url.get<std::string>(), "data:")) {
+ throw std::invalid_argument("images must be data URLs (data:image/...;base64,... or data:audio/...;base64,...)");
}
if (files.size() >= DECISION_MAX_IMAGES) {
throw std::invalid_argument(string_format("too many images, the maximum is %zu", DECISION_MAX_IMAGES));
@@ -227,15 +242,30 @@ static void decision_load_image(const json & url, std::vector<raw_buffer> & file
handle_media(files, url.get<std::string>(), "");
}
+// audio is base64 data, as a data URL or not (OpenAI input_audio)
+static void decision_load_audio(const json & data, std::vector<raw_buffer> & files) {
+ if (!data.is_string() || string_starts_with(data.get<std::string>(), "http") || string_starts_with(data.get<std::string>(), "file://")) {
+ throw std::invalid_argument("audio must be base64 data");
+ }
+ if (files.size() >= DECISION_MAX_IMAGES) {
+ throw std::invalid_argument(string_format("too many media files, the maximum is %zu", DECISION_MAX_IMAGES));
+ }
+ handle_media(files, data.get<std::string>(), "");
+}
+
json server_decision_context::parse_state(const json & body, std::vector<raw_buffer> & files) const {
if (body.contains("videos") && !body.at("videos").is_null() && !body.at("videos").empty()) {
throw std::invalid_argument("\"videos\" is not supported");
}
- if (body.contains("images") && !body.at("images").is_null()) {
- if (!body.at("images").is_array()) {
- throw std::invalid_argument("\"images\" must be an array");
+ // "images" is an alias of "files"
+ for (const char * key : {"files", "images"}) {
+ if (!body.contains(key) || body.at(key).is_null()) {
+ continue;
+ }
+ if (!body.at(key).is_array()) {
+ throw std::invalid_argument(string_format("\"%s\" must be an array", key));
}
- for (const auto & url : body.at("images")) {
+ for (const auto & url : body.at(key)) {
decision_load_image(url, files);
}
}
@@ -247,7 +277,7 @@ json server_decision_context::parse_state(const json & body, std::vector<raw_buf
return state;
}
- // chat messages: take the image parts out of the content
+ // chat messages: take the image and audio parts out of the content
json messages_out = json::array();
for (const auto & msg : messages) {
if (!msg.is_object() || !msg.contains("content") || !msg.at("content").is_array()) {
@@ -259,6 +289,9 @@ json server_decision_context::parse_state(const json & body, std::vector<raw_buf
if (part.is_object() && json_value(part, "type", std::string()) == "image_url" && part.contains("image_url")) {
const json & image_url = part.at("image_url");
decision_load_image(image_url.is_object() && image_url.contains("url") ? image_url.at("url") : image_url, files);
+ } else if (part.is_object() && json_value(part, "type", std::string()) == "input_audio" && part.contains("input_audio")) {
+ const json input_audio = json_value(part, "input_audio", json::object());
+ decision_load_audio(input_audio.contains("data") ? input_audio.at("data") : json_value(input_audio, "url", json()), files);
} else {
content.push_back(part);
}
@@ -364,6 +397,41 @@ static std::string decision_kev_text(const json & val) {
return std::regex_replace(decision_kev_render(val), re_special, "<\xC2\xA6$1\xC2\xA6>");
}
+// lfm2-d1-omni: special tokens written in the input must not be parsed as such, in keys too (d1-omni prompt.py: escape)
+static json decision_d1omni_escape(const json & val) {
+ static const std::regex re_special("<\\|([A-Za-z0-9_]+)\\|>");
+ if (val.is_string()) {
+ return std::regex_replace(val.get<std::string>(), re_special, "<\xC2\xA6$1\xC2\xA6>");
+ }
+ if (val.is_array()) {
+ json out = json::array();
+ for (const auto & item : val) {
+ out.push_back(decision_d1omni_escape(item));
+ }
+ return out;
+ }
+ if (val.is_object()) {
+ json out = json::object();
+ for (const auto & [key, item] : val.items()) {
+ out[std::regex_replace(key, re_special, "<\xC2\xA6$1\xC2\xA6>")] = decision_d1omni_escape(item);
+ }
+ return out;
+ }
+ return val;
+}
+
+// given to the lfm2-d1-omni template: text between the pieces of the prompt, and at the start of the pieces that are cut to a token budget
+static const std::string D1OMNI_MARKER = "<<d1omni:";
+static const std::string D1OMNI_SEP = "<<d1omni:sep>>";
+static const std::string D1OMNI_MARK_STATE = "<<d1omni:state>>";
+static const std::string D1OMNI_MARK_QUESTION = "<<d1omni:question>>";
+static const std::string D1OMNI_MARK_OPTION = "<<d1omni:option>>";
+
+// max_length, image_text_length and audio_text_length of the model config, the converter checks them
+static const size_t D1OMNI_MAX_TOKENS = 16384;
+static const size_t D1OMNI_MAX_TOKENS_IMAGE = 896;
+static const size_t D1OMNI_MAX_TOKENS_AUDIO = 15360;
+
size_t server_decision_context::n_variants(const server_decision_question & question) const {
// lev shows the options of a choice in 2 orders, to cancel the preference for the first label
if (type == COMMON_DECISION_TYPE_LEV && question.type == SERVER_DECISION_QUESTION_CHOICE && question.options.size() > 1) {
@@ -515,7 +583,8 @@ std::string server_decision_context::render(
const std::vector<server_decision_question> & questions,
const server_decision_question & question,
size_t variant,
- size_t n_images) const {
+ size_t n_images,
+ bool is_audio) const {
// the template is given raw JSON values, it serializes the ones that are not strings
json inp = json{
{"id", question.id},
@@ -554,6 +623,15 @@ std::string server_decision_context::render(
inp = decision_replace_text(inp, text_marker, " ");
}
+ if (type == COMMON_DECISION_TYPE_LFM2_D1_OMNI) {
+ inp = decision_replace_text(decision_d1omni_escape(inp), D1OMNI_MARKER, "<<d1omni ");
+ inp["audio"] = is_audio;
+ inp["sep"] = D1OMNI_SEP;
+ inp["mark_state"] = D1OMNI_MARK_STATE;
+ inp["mark_question"] = D1OMNI_MARK_QUESTION;
+ inp["mark_option"] = D1OMNI_MARK_OPTION;
+ }
+
// the template puts one media marker per image
json images = json::array();
if (n_images > 0) {
@@ -580,6 +658,11 @@ void server_decision_context::fill_task(
mtmd_context * mctx,
const mtmd_helper_init_opt & init_opt,
server_task & task) const {
+ if (type == COMMON_DECISION_TYPE_LFM2_D1_OMNI) {
+ fill_task_d1omni(state, questions, question, variant, files, mctx, init_opt, task);
+ return;
+ }
+
const std::string prompt = render(state, questions, question, variant, files.size());
if (type == COMMON_DECISION_TYPE_OPENJEV || type == COMMON_DECISION_TYPE_LEV || type == COMMON_DECISION_TYPE_NIMBLE || type == COMMON_DECISION_TYPE_PPLX_DECIDER) {
@@ -678,6 +761,129 @@ void server_decision_context::fill_task_laya(llama_tokens & tokens, const server
task.decision.column = question.type;
}
+// each piece is cut to its budget as the model was trained (d1-omni prompt.py: encode), the state gets the room that is left
+// the media come first, the text after them has a budget of its own
+void server_decision_context::fill_task_d1omni(
+ const json & state,
+ const std::vector<server_decision_question> & questions,
+ const server_decision_question & question,
+ size_t variant,
+ const std::vector<raw_buffer> & files,
+ mtmd_context * mctx,
+ const mtmd_helper_init_opt & init_opt,
+ server_task & task) const {
+ const auto invalid = std::runtime_error("unexpected layout of the decision prompt");
+
+ // the prompt depends on the kind of media, mtmd tells an audio clip from an image by its content
+ std::string markers;
+ server_tokens media(llama_tokens(), false);
+ bool is_audio = false;
+ if (!files.empty()) {
+ for (size_t i = 0; i < files.size(); i++) {
+ markers += get_media_marker();
+ }
+ media = process_mtmd_prompt(mctx, markers, files, init_opt);
+ for (size_t i = 0; i < media.size(); i++) {
+ if (media[i] == LLAMA_TOKEN_NULL) {
+ const auto & chunk = media.find_chunk(i);
+ is_audio = is_audio || mtmd_input_chunk_get_type(chunk.get()) == MTMD_INPUT_CHUNK_TYPE_AUDIO;
+ i += mtmd_input_chunk_get_n_tokens(chunk.get()) - 1;
+ }
+ }
+ }
+ if (is_audio && files.size() > 1) {
+ throw std::invalid_argument("a request has images or one audio clip, not both");
+ }
+ const size_t n_media = media.size();
+
+ const std::string prompt = render(state, questions, question, variant, files.size(), is_audio);
+ std::vector<std::string> pieces = string_split(prompt, D1OMNI_SEP);
+ if (pieces.empty() || pieces[0] != markers) {
+ throw invalid;
+ }
+
+ size_t n_max = D1OMNI_MAX_TOKENS;
+ if (!files.empty()) {
+ n_max = std::min(is_audio ? D1OMNI_MAX_TOKENS_AUDIO : D1OMNI_MAX_TOKENS_IMAGE, D1OMNI_MAX_TOKENS - std::min(D1OMNI_MAX_TOKENS, n_media));
+ if (n_max < 64) {
+ throw std::invalid_argument(string_format("the media take %zu of the %zu positions, send fewer images", n_media, D1OMNI_MAX_TOKENS));
+ }
+ }
+
+ // the options get max(96, min(24 n + 32, max / 2)) tokens, shared evenly
+ const int64_t n_options = question.options.size();
+ const int64_t n_budget = std::max<int64_t>(96, std::min<int64_t>(24 * n_options + 32, n_max / 2));
+ const int64_t n_option_max = std::max<int64_t>(2, (n_budget - 3 * n_options) / n_options);
+ const int64_t n_question_max = std::max<int64_t>(16, n_budget);
+
+ llama_tokens head; // before the state
+ llama_tokens body; // the state
+ llama_tokens tail; // after the state
+ bool has_state = false;
+ for (size_t i_piece = 1; i_piece < pieces.size(); i_piece++) {
+ std::string piece = pieces[i_piece];
+ int64_t n_piece_max = -1;
+ bool is_state = false;
+ if (string_starts_with(piece, D1OMNI_MARK_STATE)) {
+ piece = piece.substr(D1OMNI_MARK_STATE.size());
+ is_state = true;
+ } else if (string_starts_with(piece, D1OMNI_MARK_QUESTION)) {
+ piece = piece.substr(D1OMNI_MARK_QUESTION.size());
+ n_piece_max = n_question_max;
+ } else if (string_starts_with(piece, D1OMNI_MARK_OPTION)) {
+ piece = piece.substr(D1OMNI_MARK_OPTION.size());
+ n_piece_max = n_option_max;
+ }
+
+ llama_tokens tokens = common_tokenize(vocab, piece, false, true);
+ if (n_piece_max >= 0 && (int64_t) tokens.size() > n_piece_max) {
+ tokens.resize(n_piece_max);
+ }
+
+ if (is_state) {
+ if (has_state) {
+ throw invalid;
+ }
+ body = std::move(tokens);
+ has_state = true;
+ } else {
+ llama_tokens & dst = has_state ? tail : head;
+ dst.insert(dst.end(), tokens.begin(), tokens.end());
+ }
+ }
+ if (!has_state) {
+ throw invalid;
+ }
+
+ const size_t n_room = n_max - std::min(n_max, head.size() + tail.size());
+ body.resize(std::min(body.size(), n_room));
+
+ llama_tokens tokens = std::move(head);
+ tokens.insert(tokens.end(), body.begin(), body.end());
+ tokens.insert(tokens.end(), tail.begin(), tail.end());
+ tokens.resize(std::min(tokens.size(), n_max));
+
+ for (size_t i = 0; i < tokens.size(); i++) {
+ if (tokens[i] == token_marker) {
+ task.decision.markers.push_back(n_media + i);
+ }
+ }
+ if ((int64_t) task.decision.markers.size() != n_options) {
+ throw std::invalid_argument("the options do not fit in the context");
+ }
+
+ // the output has one score per question type
+ task.decision.column = question.type;
+ if (files.empty()) {
+ task.tokens = server_tokens(tokens, false);
+ } else {
+ task.tokens = std::move(media);
+ for (const llama_token token : tokens) {
+ task.tokens.push_back(token);
+ }
+ }
+}
+
//
// joint prompt (clef)
//
@@ -850,14 +1056,15 @@ static double decision_confidence_score(const std::vector<double> & probs) {
return std::max(0.0, 1.0 - dist / dist_uniform);
}
-json server_decision_context::format_answer(const server_decision_question & question, const std::vector<std::vector<float>> & scores) const {
+json server_decision_context::format_answer(const server_decision_question & question, const std::vector<std::vector<float>> & scores, bool has_media) const {
const size_t n = n_outputs(question);
if (scores.size() != n_variants(question)) {
throw std::runtime_error("decision result does not match the number of variants");
}
// softmax over the outputs of each variant, then the average of the variants
- const float temperature = get_temperature(question);
+ // lfm2-d1-omni: image and audio answers are not calibrated
+ const float temperature = has_media && type == COMMON_DECISION_TYPE_LFM2_D1_OMNI ? 1.0f : get_temperature(question);
std::vector<double> probs(n, 0.0);
for (size_t v = 0; v < scores.size(); v++) {
const auto & s = scores[v];
diff --git a/tools/server/server-decision.h b/tools/server/server-decision.h
index 8f357f374..5c7e44f37 100644
--- a/tools/server/server-decision.h
+++ b/tools/server/server-decision.h
@@ -62,6 +62,7 @@ struct server_decision_context {
case COMMON_DECISION_TYPE_CLEF:
case COMMON_DECISION_TYPE_PPLX_DECIDER:
case COMMON_DECISION_TYPE_LFM2_D1:
+ case COMMON_DECISION_TYPE_LFM2_D1_OMNI:
return true;
default:
return false;
@@ -72,7 +73,8 @@ struct server_decision_context {
std::vector<server_decision_question> parse_questions(const json & body) const;
// returns the state without its images, they are appended to files in order
- // images come from "images" and from the image_url parts of a state made of chat messages
+ // images come from "files" (alias "images") and from the image_url and input_audio parts of a state made of chat messages
+ // an image can be an audio clip if the model supports it
json parse_state(const json & body, std::vector<raw_buffer> & files) const;
// number of prompts that are evaluated to answer this question, each one shows the options in a different order
@@ -101,7 +103,7 @@ struct server_decision_context {
server_task & task) const;
// scores: the raw model outputs of each variant
- json format_answer(const server_decision_question & question, const std::vector<std::vector<float>> & scores) const;
+ json format_answer(const server_decision_question & question, const std::vector<std::vector<float>> & scores, bool has_media = false) const;
private:
const llama_vocab * vocab = nullptr;
@@ -128,12 +130,22 @@ private:
const std::vector<server_decision_question> & questions,
const server_decision_question & question,
size_t variant,
- size_t n_images) const;
+ size_t n_images,
+ bool is_audio = false) const;
json render_options(const server_decision_question & question, size_t variant) const;
size_t n_outputs(const server_decision_question & question) const;
// LFM2_D1: label text and tokens of each option
void d1_labels(const server_decision_question & question, std::vector<std::string> & texts, std::vector<llama_tokens> & groups) const;
void fill_task_laya(llama_tokens & tokens, const server_decision_question & question, server_task & task) const;
+ void fill_task_d1omni(
+ const json & state,
+ const std::vector<server_decision_question> & questions,
+ const server_decision_question & question,
+ size_t variant,
+ const std::vector<raw_buffer> & files,
+ mtmd_context * mctx,
+ const mtmd_helper_init_opt & init_opt,
+ server_task & task) const;
float get_temperature(const server_decision_question & question) const;
};