Commit 88dcc460d for llama.cpp
commit 88dcc460d628698bb8305b98c200c34f1edfdc04
Author: Tarek Dakhran <tarek@liquid.ai>
Date: Wed Oct 7 22:04:39 2026 +0200
model : add LiquidAI/d1-3B decision model (#30110)
* model : add LiquidAI/d1-3b decision model
mtmd : read LFM2 image resize algo from GGUF
Assisted-by: Claude Opus 5.5
* common : rename decision type d1 to lfm2-d1
Assisted-by: Claude Opus 5.5
diff --git a/common/common.cpp b/common/common.cpp
index 768b2e9a5..0e891cea6 100644
--- a/common/common.cpp
+++ b/common/common.cpp
@@ -1170,6 +1170,7 @@ static const std::map<common_decision_type, std::string> COMMON_DECISION_TYPE_NA
{ COMMON_DECISION_TYPE_LAYA, "laya" },
{ COMMON_DECISION_TYPE_CLEF, "clef" },
{ COMMON_DECISION_TYPE_PPLX_DECIDER, "pplx-decider" },
+ { COMMON_DECISION_TYPE_LFM2_D1, "lfm2-d1" },
};
static common_decision_type common_decision_type_from_string(const std::string & str) {
diff --git a/common/common.h b/common/common.h
index 0f912d901..6f8acf31d 100644
--- a/common/common.h
+++ b/common/common.h
@@ -963,6 +963,7 @@ enum common_decision_type {
COMMON_DECISION_TYPE_LAYA, // score of one marker token per option, read from the embeddings output
COMMON_DECISION_TYPE_CLEF, // all questions in one prompt, score of option i read from the embeddings output at row i
COMMON_DECISION_TYPE_PPLX_DECIDER, // same as openjev, label codes of 1 or 2 letters
+ COMMON_DECISION_TYPE_LFM2_D1, // same as openjev, the labels depend on the question type
COMMON_DECISION_TYPE_UNKNOWN, // a decision model of a type that is not supported
};
diff --git a/conversion/__init__.py b/conversion/__init__.py
index 9ee55d5a6..34c56bbee 100644
--- a/conversion/__init__.py
+++ b/conversion/__init__.py
@@ -161,6 +161,7 @@ TEXT_MODEL_MAP: dict[str, str] = {
"Lfm2BidirectionalForMaskedLM": "lfm2",
"Lfm2BidirectionalModel": "lfm2",
"Lfm2ForCausalLM": "lfm2",
+ "D1Model": "lfm2",
"Lfm2Model": "lfm2",
"Lfm2MoeForCausalLM": "lfm2",
"Llama4ForCausalLM": "llama",
diff --git a/conversion/lfm2.py b/conversion/lfm2.py
index e50bbdd0f..d4343675e 100644
--- a/conversion/lfm2.py
+++ b/conversion/lfm2.py
@@ -1,5 +1,8 @@
from __future__ import annotations
+import json
+
+from pathlib import Path
from typing import Any, Callable, Iterable, TYPE_CHECKING
import torch
@@ -7,7 +10,7 @@ import torch
if TYPE_CHECKING:
from torch import Tensor
-from .base import MmprojModel, ModelBase, TextModel, gguf
+from .base import MmprojModel, ModelBase, TextModel, gguf, jinja_str_or_json, logger
from .gemma import ConformerAudioModel
@@ -65,6 +68,68 @@ class LFM2Model(TextModel):
yield from super().modify_tensors(data_torch, name, bid)
+def _is_d1_checkpoint(dir_model: Path) -> bool:
+ if not (dir_model / "config.json").is_file():
+ return False
+ with open(dir_model / "config.json", encoding="utf-8") as f:
+ return json.load(f).get("auto_map", {}).get("AutoModel", "").endswith(".D1Model")
+
+
+@ModelBase.register_hparams_loader(_is_d1_checkpoint)
+def _load_d1_hparams(dir_model: Path) -> dict[str, Any]:
+ logger.info("gguf: detected d1 checkpoint")
+ hparams = ModelBase.load_hparams(dir_model, False, guess=False)
+ # the mmproj stays LFM2-VL
+ hparams["text_config"]["architectures"] = ["D1Model"]
+ return hparams
+
+
+@ModelBase.register("D1Model")
+@ModelBase.example("LiquidAI/d1-3b")
+class D1Model(LFM2Model):
+ model_arch = gguf.MODEL_ARCH.LFM2
+
+ def set_vocab(self):
+ super().set_vocab()
+ self.gguf_writer.add_chat_template([{"name": "systemone", "template": self._systemone_template()}])
+
+ @staticmethod
+ def _systemone_template() -> str:
+ # follows prompt.py of the model repo
+ description = jinja_str_or_json("o.description")
+ choice = (
+ "{{ '\\n\\nOptions:\\n' }}"
+ "{% for o in options %}{{ o.label }} {% if o.description %}" + description + "{% else %}{{ o.key | replace('_', ' ') }}{% endif %}"
+ "{% if not loop.last %}{{ '\\n' }}{% endif %}{% endfor %}"
+ "{{ '\\n\\nReply with the option code only.' }}"
+ )
+ # with criteria, a missing description is written as None
+ noul = (
+ "{% set ns = namespace(criteria=false) %}{% for o in options %}{% if o.description is not none %}{% set ns.criteria = true %}{% endif %}{% endfor %}"
+ "{% if ns.criteria %}"
+ "{% for o in options %}{{ '\\nYes: ' if o.key == 'true' else '\\nNo: ' }}"
+ "{% if o.description is none %}None{% else %}" + description + "{% endif %}{% endfor %}{% endif %}"
+ "{{ '\\n\\nReply with yes or no only.' }}"
+ )
+ score = (
+ "{{ '\\n\\n' }}{% for o in options %}{{ o.key }} " + description + "{{ '\\n' }}{% endfor %}"
+ "{{ '\\nReply with a single digit 0-' }}{{ options | length - 1 }}{{ ' only.' }}"
+ )
+ return (
+ "<|startoftext|><|im_start|>user\n"
+ "{% for image in images %}{{ image }}{% endfor %}"
+ "{% if state is not none %}{% if state is string %}{{ state }}{% else %}{{ state | tojson(indent=2) }}{% endif %}"
+ "{{ '\\n\\n\\nQUESTION:\\n' }}{% endif %}"
+ + jinja_str_or_json("instructions")
+ + "{% if type == 'choice' %}" + choice + "{% elif type == 'noul' %}" + noul + "{% else %}" + score + "{% endif %}"
+ "{{ '<|im_end|>\\n<|im_start|>assistant\\n' }}"
+ )
+
+ def set_gguf_parameters(self):
+ super().set_gguf_parameters()
+ self.gguf_writer.add_decision_type(gguf.DecisionType.LFM2_D1)
+
+
@ModelBase.register("Lfm2Model", "Lfm2BidirectionalModel", "Lfm2BidirectionalForMaskedLM")
@ModelBase.example("LiquidAI/LFM2.5-ColBERT-350M", "LiquidAI/LFM2.5-Embedding-350M", "LiquidAI/LFM2.5-Encoder-350M", "LiquidAI/LFM2.5-Encoder-230M")
class LFM2ColBertModel(LFM2Model):
@@ -188,6 +253,12 @@ class LFM2VLModel(MmprojModel):
# python notation, e.g. for vision_feature_layer == -1, we pick last layer -> vision_feature_layers_to_drop = 0
vision_feature_layers_to_drop = -(self.global_config.get("vision_feature_layer", -1) + 1)
self.gguf_writer.add_vision_block_count(self.find_vparam(self.n_block_keys) - vision_feature_layers_to_drop)
+ # PIL resample enum
+ if (resample := self.preprocessor_config.get("resample")) is not None:
+ resize_algo = {1: "lanczos", 2: "bilinear", 3: "bicubic"}.get(resample)
+ if resize_algo is None:
+ raise ValueError(f"unsupported resample: {resample}")
+ self.gguf_writer.add_vision_image_resize_algo(resize_algo)
@classmethod
def filter_tensors(cls, item: tuple[str, Callable[[], Tensor]]) -> tuple[str, Callable[[], Tensor]] | None:
diff --git a/gguf-py/gguf/constants.py b/gguf-py/gguf/constants.py
index 842cd7891..c181cb44a 100644
--- a/gguf-py/gguf/constants.py
+++ b/gguf-py/gguf/constants.py
@@ -408,6 +408,7 @@ class Keys:
BLOCK_COUNT = "clip.vision.block_count"
IMAGE_MEAN = "clip.vision.image_mean"
IMAGE_STD = "clip.vision.image_std"
+ IMAGE_RESIZE_ALGO = "clip.vision.image_resize_algo"
SPATIAL_MERGE_SIZE = "clip.vision.spatial_merge_size"
SWIGLU_CLAMP = "clip.vision.swiglu_clamp"
EXPERT_COUNT_PER_LAYER = "clip.vision.expert_count_per_layer" # dots3note pyramid MoE, 0 = dense layer
@@ -6070,6 +6071,7 @@ class DecisionType:
NIMBLE = "nimble" # same as openjev, the prompt lists all the questions of the request
CLEF = "clef" # joint head over all questions, one score per option
PPLX_DECIDER = "pplx-decider" # same as openjev, label codes of 1 or 2 letters
+ LFM2_D1 = "lfm2-d1" # same as openjev, the labels depend on the question type
class VisionProjectorType:
diff --git a/gguf-py/gguf/gguf_writer.py b/gguf-py/gguf/gguf_writer.py
index e814691d7..f33a8a551 100644
--- a/gguf-py/gguf/gguf_writer.py
+++ b/gguf-py/gguf/gguf_writer.py
@@ -1432,6 +1432,9 @@ class GGUFWriter:
def add_vision_image_mean(self, values: Sequence[float]) -> None:
self.add_array(Keys.ClipVision.IMAGE_MEAN, values)
+ def add_vision_image_resize_algo(self, value: str) -> None:
+ self.add_string(Keys.ClipVision.IMAGE_RESIZE_ALGO, value)
+
def add_vision_image_std(self, values: Sequence[float]) -> None:
self.add_array(Keys.ClipVision.IMAGE_STD, values)
diff --git a/tools/mtmd/clip-impl.h b/tools/mtmd/clip-impl.h
index e73b74aff..abf75d998 100644
--- a/tools/mtmd/clip-impl.h
+++ b/tools/mtmd/clip-impl.h
@@ -58,6 +58,7 @@
#define KEY_PATCH_SIZE "clip.vision.patch_size"
#define KEY_IMAGE_MEAN "clip.vision.image_mean"
#define KEY_IMAGE_STD "clip.vision.image_std"
+#define KEY_IMAGE_RESIZE_ALGO "clip.vision.image_resize_algo"
#define KEY_PROJ_SCALE_FACTOR "clip.vision.projector.scale_factor"
#define KEY_PROJ_SAMPLE_QUERY_SIDE "clip.vision.projector.query_side"
#define KEY_PROJ_SAMPLE_WINDOW_SIDE "clip.vision.projector.window_side"
diff --git a/tools/mtmd/clip.cpp b/tools/mtmd/clip.cpp
index 06c562897..b7302e4bf 100644
--- a/tools/mtmd/clip.cpp
+++ b/tools/mtmd/clip.cpp
@@ -1526,9 +1526,20 @@ struct clip_model_loader {
} break;
case PROJECTOR_TYPE_LFM2:
{
- hparams.image_resize_algo = RESIZE_ALGO_BILINEAR;
- hparams.image_resize_algo_rf = RESIZE_ALGO_BILINEAR;
- hparams.image_resize_algo_ov = RESIZE_ALGO_BILINEAR;
+ // default for older GGUFs
+ std::string resize_algo = "bilinear";
+ get_string(KEY_IMAGE_RESIZE_ALGO, resize_algo, false);
+ if (resize_algo == "bilinear") {
+ hparams.image_resize_algo = RESIZE_ALGO_BILINEAR;
+ } else if (resize_algo == "bicubic") {
+ hparams.image_resize_algo = RESIZE_ALGO_BICUBIC;
+ } else if (resize_algo == "lanczos") {
+ hparams.image_resize_algo = RESIZE_ALGO_LANCZOS;
+ } else {
+ throw std::runtime_error(string_format("%s: unsupported image resize algo: %s\n", __func__, resize_algo.c_str()));
+ }
+ hparams.image_resize_algo_rf = hparams.image_resize_algo;
+ hparams.image_resize_algo_ov = hparams.image_resize_algo;
get_u32(KEY_PROJ_SCALE_FACTOR, hparams.n_merge, false);
// ref: https://huggingface.co/LiquidAI/LFM2.5-VL-1.6B/blob/main/processor_config.json
hparams.set_limit_image_tokens(64, 256);
diff --git a/tools/server/README.md b/tools/server/README.md
index 79def4079..8b271395b 100644
--- a/tools/server/README.md
+++ b/tools/server/README.md
@@ -1703,7 +1703,7 @@ Follows the [TypeSafe API](https://docs.typesafe.ai/api), streaming is not suppo
*Options:*
-`state`: The content to evaluate. Can be a string, an object or an array. A value that is not a string is given to the model as JSON text.
+`state`: The content to evaluate. Can be a string, an object or an array. A value that is not a string is given to the model as JSON text. For lfm2-d1, it can be `null`, for example to ask about images only.
`images`: Optional. An array of images, the maximum number may be limited depending on the model. Each one is a data URL (`data:image/...;base64,...`). See the image input section below.
@@ -1718,13 +1718,13 @@ Follows the [TypeSafe API](https://docs.typesafe.ai/api), streaming is not suppo
The questions of a request are answered independently, an answer does not depend on the other questions. The exception is clef: it reads all the questions in one prompt and decides them jointly.
-The number of options of a `choice` question is limited by the model, for example: 52 for openjev, 255 for laya, clef and pplx-decider. For laya, long questions and options are truncated to the token budget the model was trained with.
+The number of options of a `choice` question is limited by the model, for example: 52 for openjev, 255 for laya, clef, pplx-decider and lfm2-d1. For laya, long questions and options are truncated to the token budget the model was trained with.
For laya and clef, the whole prompt is evaluated in one batch: it must fit in `--ubatch-size`. A server that runs clef only serves this endpoint, text generation is not available.
*Image input:*
-Image input needs a model that supports it (for example: openjev, clef, pplx-decider) and its multimodal projector, see `--mmproj`.
+Image input needs a model that supports it (for example: openjev, clef, pplx-decider, lfm2-d1) and its multimodal projector, see `--mmproj`.
Images can be given in two ways, and both can be used in the same request:
diff --git a/tools/server/server-common.cpp b/tools/server/server-common.cpp
index 77dbec5e1..f56595231 100644
--- a/tools/server/server-common.cpp
+++ b/tools/server/server-common.cpp
@@ -156,6 +156,7 @@ std::vector<std::string> server_model_output_modalities(common_decision_type dec
case COMMON_DECISION_TYPE_LAYA:
case COMMON_DECISION_TYPE_CLEF:
case COMMON_DECISION_TYPE_PPLX_DECIDER:
+ case COMMON_DECISION_TYPE_LFM2_D1:
return {"decisions"};
default:
// fallback when there is no decision type or the metadata is bad
diff --git a/tools/server/server-context.cpp b/tools/server/server-context.cpp
index 23835e185..6f3519215 100644
--- a/tools/server/server-context.cpp
+++ b/tools/server/server-context.cpp
@@ -2344,6 +2344,16 @@ private:
GGML_ASSERT(label >= 0 && label < n_vocab);
res->scores.push_back(logits[label]);
}
+ if (!decision.label_groups.empty()) {
+ std::vector<float> scores;
+ size_t i = 0;
+ for (const int32_t n : decision.label_groups) {
+ GGML_ASSERT(n > 0 && i + n <= res->scores.size());
+ scores.push_back(*std::max_element(res->scores.begin() + i, res->scores.begin() + i + n));
+ i += n;
+ }
+ res->scores = std::move(scores);
+ }
} else {
// the outputs of this slot in this batch are the last tokens of the prompt
std::vector<int32_t> idx;
diff --git a/tools/server/server-decision.cpp b/tools/server/server-decision.cpp
index 9ed68009f..71657a359 100644
--- a/tools/server/server-decision.cpp
+++ b/tools/server/server-decision.cpp
@@ -3,6 +3,7 @@
#include "../../src/llama-ext.h" // staging API: llama_decision_order
#include <algorithm>
+#include <cctype>
#include <cmath>
#include <regex>
#include <stdexcept>
@@ -122,6 +123,9 @@ void server_decision_context::init(const llama_model * model) {
n_options_max = 255;
noul_true_first = true;
choice_sorted = true;
+ } else if (model_type == COMMON_DECISION_TYPE_LFM2_D1) {
+ n_options_max = 255;
+ noul_true_first = true;
} else {
throw std::runtime_error("unsupported decision model type: " + type_name);
}
@@ -135,7 +139,8 @@ void server_decision_context::init(const llama_model * model) {
//
std::vector<server_decision_question> server_decision_context::parse_questions(const json & body) const {
- if (!body.contains("state") || body.at("state").is_null()) {
+ // d1 accepts a null state (images only)
+ if (!body.contains("state") || (body.at("state").is_null() && type != COMMON_DECISION_TYPE_LFM2_D1)) {
throw std::invalid_argument("\"state\" must be provided");
}
if (!body.contains("questions") || !body.at("questions").is_object() || body.at("questions").empty()) {
@@ -374,9 +379,112 @@ size_t server_decision_context::n_outputs(const server_decision_question & quest
return question.options.size();
}
+// label codes follow prompt.py of the model repo
+void server_decision_context::d1_labels(const server_decision_question & question, std::vector<std::string> & texts, std::vector<llama_tokens> & groups) const {
+ const size_t n_options = question.options.size();
+
+ auto get_single_tokens = [&](const std::vector<std::string> & forms) {
+ llama_tokens out;
+ for (const auto & form : forms) {
+ const auto toks = common_tokenize(vocab, form, false, false);
+ if (toks.size() == 1 && std::find(out.begin(), out.end(), toks[0]) == out.end()) {
+ out.push_back(toks[0]);
+ }
+ }
+ return out;
+ };
+
+ if (question.type != SERVER_DECISION_QUESTION_CHOICE) {
+ for (const auto & opt : question.options) {
+ llama_tokens group;
+ if (question.type == SERVER_DECISION_QUESTION_SCORE) {
+ group = get_single_tokens({opt.key});
+ } else if (opt.key == "true") {
+ group = get_single_tokens({"yes", "Yes", "YES"});
+ } else {
+ group = get_single_tokens({"no", "No", "NO"});
+ }
+ if (group.empty()) {
+ throw std::runtime_error("decision label is not a single token: " + opt.key);
+ }
+ texts.push_back(opt.key);
+ groups.push_back(group);
+ }
+ return;
+ }
+
+ bool is_letters = true;
+ for (const auto & opt : question.options) {
+ is_letters = is_letters && opt.key.size() == 1 && std::isalpha((unsigned char) opt.key[0]);
+ }
+
+ std::vector<std::string> codes;
+ for (size_t i = 0; i < n_options; i++) {
+ if (is_letters) {
+ codes.push_back(question.options[i].key);
+ } else if (n_options <= 26) {
+ codes.push_back(std::string(1, 'A' + i));
+ } else {
+ codes.push_back(string_format("%02zu", i));
+ }
+ }
+
+ std::vector<std::string> pool;
+ for (char c = 'A'; c <= 'Z'; c++) {
+ pool.push_back(std::string(1, c));
+ }
+ for (int i = 0; i < 100; i++) {
+ pool.push_back(string_format("%02d", i));
+ }
+ for (char c = 'a'; c <= 'z'; c++) {
+ pool.push_back(std::string(1, c));
+ }
+ for (int i = 0; i < 200; i++) {
+ pool.push_back(string_format("#%d", i));
+ }
+ for (char a = 'A'; a <= 'Z'; a++) {
+ for (char b = 'A'; b <= 'Z'; b++) {
+ pool.push_back(std::string{a, b});
+ }
+ }
+
+ llama_tokens used;
+ auto take = [&](const std::string & code) {
+ const auto toks = common_tokenize(vocab, code, false, false);
+ if (toks.size() != 1 || std::find(used.begin(), used.end(), toks[0]) != used.end()) {
+ return false;
+ }
+ used.push_back(toks[0]);
+ llama_tokens group = {toks[0]};
+ for (const llama_token tok : get_single_tokens({" " + code})) {
+ if (tok != toks[0]) {
+ group.push_back(tok);
+ }
+ }
+ texts.push_back(code);
+ groups.push_back(group);
+ return true;
+ };
+ for (const auto & code : codes) {
+ bool is_taken = take(code);
+ for (size_t i = 0; !is_taken && i < pool.size(); i++) {
+ is_taken = take(pool[i]);
+ }
+ if (!is_taken) {
+ throw std::invalid_argument(string_format("no single-token label left for %zu options", n_options));
+ }
+ }
+}
+
json server_decision_context::render_options(const server_decision_question & question, size_t variant) const {
const size_t n_options = question.options.size();
+ std::vector<std::string> d1_texts;
+ std::vector<llama_tokens> d1_groups;
+ if (type == COMMON_DECISION_TYPE_LFM2_D1) {
+ d1_labels(question, d1_texts, d1_groups);
+ }
+
// the second variant shows the options in the reverse order
json options = json::array();
for (size_t i = 0; i < n_options; i++) {
@@ -394,6 +502,9 @@ json server_decision_context::render_options(const server_decision_question & qu
if (!label_texts.empty()) {
option["label"] = label_texts[i];
}
+ if (!d1_texts.empty()) {
+ option["label"] = d1_texts[i];
+ }
options.push_back(option);
}
return options;
@@ -474,6 +585,17 @@ void server_decision_context::fill_task(
if (type == COMMON_DECISION_TYPE_OPENJEV || type == COMMON_DECISION_TYPE_LEV || type == COMMON_DECISION_TYPE_NIMBLE || type == COMMON_DECISION_TYPE_PPLX_DECIDER) {
// lev reads the ratings of a noul question at its first labels, not at the digits
task.decision.labels.assign(labels.begin(), labels.begin() + n_outputs(question));
+ }
+ if (type == COMMON_DECISION_TYPE_LFM2_D1) {
+ std::vector<std::string> texts;
+ std::vector<llama_tokens> groups;
+ d1_labels(question, texts, groups);
+ for (const auto & group : groups) {
+ task.decision.labels.insert(task.decision.labels.end(), group.begin(), group.end());
+ task.decision.label_groups.push_back(group.size());
+ }
+ }
+ if (type == COMMON_DECISION_TYPE_OPENJEV || type == COMMON_DECISION_TYPE_LEV || type == COMMON_DECISION_TYPE_NIMBLE || type == COMMON_DECISION_TYPE_PPLX_DECIDER || type == COMMON_DECISION_TYPE_LFM2_D1) {
if (!files.empty()) {
task.tokens = process_mtmd_prompt(mctx, prompt, files, init_opt);
return;
diff --git a/tools/server/server-decision.h b/tools/server/server-decision.h
index f29211941..8f357f374 100644
--- a/tools/server/server-decision.h
+++ b/tools/server/server-decision.h
@@ -43,6 +43,7 @@ struct server_decision_context {
case COMMON_DECISION_TYPE_KEV:
case COMMON_DECISION_TYPE_NIMBLE:
case COMMON_DECISION_TYPE_PPLX_DECIDER:
+ case COMMON_DECISION_TYPE_LFM2_D1:
return true;
default:
return false;
@@ -60,6 +61,7 @@ struct server_decision_context {
case COMMON_DECISION_TYPE_OPENJEV:
case COMMON_DECISION_TYPE_CLEF:
case COMMON_DECISION_TYPE_PPLX_DECIDER:
+ case COMMON_DECISION_TYPE_LFM2_D1:
return true;
default:
return false;
@@ -129,6 +131,8 @@ private:
size_t n_images) const;
json render_options(const server_decision_question & question, size_t variant) const;
size_t n_outputs(const server_decision_question & question) const;
+ // LFM2_D1: label text and tokens of each option
+ void d1_labels(const server_decision_question & question, std::vector<std::string> & texts, std::vector<llama_tokens> & groups) const;
void fill_task_laya(llama_tokens & tokens, const server_decision_question & question, server_task & task) const;
float get_temperature(const server_decision_question & question) const;
diff --git a/tools/server/server-task.h b/tools/server/server-task.h
index 552c03b07..0cbc5f353 100644
--- a/tools/server/server-task.h
+++ b/tools/server/server-task.h
@@ -178,8 +178,9 @@ struct server_task {
// used by SERVER_TASK_TYPE_DECISION
// where to read the model output of each option, exactly one of the two lists is used
struct decision {
- std::vector<llama_token> labels; // logits of these tokens, at the last prompt token
- std::vector<int32_t> markers; // embeddings[column] at these prompt positions
+ std::vector<llama_token> labels; // logits of these tokens, at the last prompt token
+ std::vector<int32_t> label_groups; // if set, number of labels per output, the output is their max
+ std::vector<int32_t> markers; // embeddings[column] at these prompt positions
int32_t column = 0;
// if set, embeddings is [q | k], and the output is instead the scaled dot product of q[pointer] and k[marker]
int32_t pointer = -1;