Commit a4cb4c61f for llama.cpp
commit a4cb4c61fd9d9c2066c7c1747821d3d65b8943bd
Author: Xuan-Son Nguyen <son@huggingface.co>
Date: Fri Oct 2 11:56:04 2026 +0200
llama, server: add /v1/systemone API (models: laya, julia-1, lev, openjev, kev) (#29818)
* init conversion
* convert: ok
* model loaded
* add server code
* improve conversion script
* support shared prompt prefix
* add docs, imorove UX a bit
* add vision support
* add openjev tiny model for testing
* add dev docs
* support lev & kev
* clean up
* fix lev noul
* fix py lint
* nits docs
* clarify about not supporting date_facts
diff --git a/common/common.cpp b/common/common.cpp
index 9c07dd84b..aca194983 100644
--- a/common/common.cpp
+++ b/common/common.cpp
@@ -1180,6 +1180,34 @@ struct common_init_result::impl {
std::vector<llama_sampler_seq_config> samplers_seq_config;
};
+static const std::map<common_decision_type, std::string> COMMON_DECISION_TYPE_NAMES = {
+ { COMMON_DECISION_TYPE_OPENJEV, "openjev" },
+ { COMMON_DECISION_TYPE_LEV, "lev" },
+ { COMMON_DECISION_TYPE_KEV, "kev" },
+ { COMMON_DECISION_TYPE_LAYA, "laya" },
+};
+
+static common_decision_type common_decision_type_from_string(const std::string & str) {
+ for (const auto & pair : COMMON_DECISION_TYPE_NAMES) {
+ if (pair.second == str) {
+ return pair.first;
+ }
+ }
+ return COMMON_DECISION_TYPE_UNKNOWN;
+}
+
+common_decision_type common_get_decision_type(const struct llama_model * model) {
+ char buf[64];
+ if (llama_model_meta_val_str(model, "general.architecture", buf, sizeof(buf)) < 0) {
+ return COMMON_DECISION_TYPE_NONE;
+ }
+ const std::string key = std::string(buf) + ".decision.type";
+ if (llama_model_meta_val_str(model, key.c_str(), buf, sizeof(buf)) < 0) {
+ return COMMON_DECISION_TYPE_NONE;
+ }
+ return common_decision_type_from_string(buf);
+}
+
common_init_result::common_init_result(common_params & params, bool model_only) :
pimpl(new impl{}) {
auto mparams = common_model_params_to_llama(params);
@@ -1232,6 +1260,21 @@ common_init_result::common_init_result(common_params & params, bool model_only)
const llama_vocab * vocab = llama_model_get_vocab(model);
+ // this decision model returns a score for each token via the embeddings output
+ // TODO: maybe improve this in the future
+ const auto decision_type = common_get_decision_type(model);
+ if (decision_type == COMMON_DECISION_TYPE_LAYA || decision_type == COMMON_DECISION_TYPE_KEV) {
+ params.embedding = true;
+ params.pooling_type = LLAMA_POOLING_TYPE_NONE;
+
+ cparams.embeddings = true;
+ cparams.pooling_type = LLAMA_POOLING_TYPE_NONE;
+ cparams.n_outputs_max = cparams.n_batch;
+ cparams.n_outputs_max_per_seq = 1;
+
+ LOG_INF("%s", "decision model reads the embeddings output, enabling embedding mode\n");
+ }
+
// load and optionally apply lora adapters
for (auto & la : params.lora_adapters) {
llama_adapter_lora_ptr lora;
diff --git a/common/common.h b/common/common.h
index 49322c81b..04ffbcd1c 100644
--- a/common/common.h
+++ b/common/common.h
@@ -950,6 +950,18 @@ bool tty_can_use_colors();
struct common_sampler;
+// typed decision models, see "<arch>.decision.type" in the model metadata
+enum common_decision_type {
+ COMMON_DECISION_TYPE_NONE, // not a decision model
+ COMMON_DECISION_TYPE_OPENJEV, // logits of one label token per option, read at the last prompt token
+ COMMON_DECISION_TYPE_LEV, // same as openjev, noul is read from a rating scale
+ COMMON_DECISION_TYPE_KEV, // dot product of the hidden states of the last token and of one end token per option
+ COMMON_DECISION_TYPE_LAYA, // score of one marker token per option, read from the embeddings output
+ COMMON_DECISION_TYPE_UNKNOWN, // a decision model of a type that is not supported
+};
+
+common_decision_type common_get_decision_type(const struct llama_model * model);
+
// note: defines the model, context, samplers, ets. lifetimes
struct common_init_result {
common_init_result(common_params & params, bool model_only = false);
diff --git a/conversion/__init__.py b/conversion/__init__.py
index 777b2f982..cd8c4d3ba 100644
--- a/conversion/__init__.py
+++ b/conversion/__init__.py
@@ -150,6 +150,8 @@ TEXT_MODEL_MAP: dict[str, str] = {
"LLaDAMoEModelLM": "llada",
"LLaDAModelLM": "llada",
"LLaMAForCausalLM": "llama",
+ "KevModel": "lev",
+ "LevModel": "lev",
"Lfm25AudioTokenizer": "lfm2",
"Lfm2BidirectionalModel": "lfm2",
"Lfm2ForCausalLM": "lfm2",
@@ -188,6 +190,7 @@ TEXT_MODEL_MAP: dict[str, str] = {
"Mistral3ForConditionalGeneration": "mistral3",
"MistralForCausalLM": "llama",
"MixtralForCausalLM": "llama",
+ "ModernBertDecisionModel": "bert",
"ModernBertForMaskedLM": "bert",
"ModernBertForSequenceClassification": "bert",
"ModernBertModel": "bert",
@@ -207,6 +210,7 @@ TEXT_MODEL_MAP: dict[str, str] = {
"MuseGlimmerAssistantModel": "muse_glimmer",
"MuseGlimmerForConditionalGeneration": "muse_glimmer",
"OpenELMForCausalLM": "openelm",
+ "OpenJevModel": "qwen",
"OrionForCausalLM": "orion",
"PLMForCausalLM": "plm",
"PLaMo2ForCausalLM": "plamo",
@@ -344,6 +348,7 @@ MMPROJ_MODEL_MAP: dict[str, str] = {
"Qwen3TTSForConditionalGeneration": "qwen3tts",
"Qwen3VLForConditionalGeneration": "qwen3vl",
"Qwen3VLMoeForConditionalGeneration": "qwen3vl",
+ "OpenJevModel": "qwen3vl",
"Qwen3_5ForConditionalGeneration": "qwen3vl",
"Qwen3_5MoeForConditionalGeneration": "qwen3vl",
"Qwen4ExpForConditionalGeneration": "qwen4exp",
diff --git a/conversion/base.py b/conversion/base.py
index a39a728f5..0f3bd9a7b 100644
--- a/conversion/base.py
+++ b/conversion/base.py
@@ -1268,22 +1268,24 @@ class ModelBase:
return inner
@staticmethod
- def load_hparams(dir_model: Path, is_mistral_format: bool):
+ def load_hparams(dir_model: Path, is_mistral_format: bool, guess: bool = True):
if is_mistral_format:
with open(dir_model / "params.json", "r", encoding="utf-8") as f:
config = json.load(f)
return config
+ # checkpoints with a non-HF layout are matched by their own loader
+ # models with a HF layout can also register a hparams loader to switch to a custom class
+ config = ModelBase.load_hparams_guess(dir_model) if guess and dir_model.is_dir() else None
+ if config is not None:
+ return config
+
try:
# for security reason, we don't allow loading remote code by default
# if a model need remote code, we will fallback to config.json
config = AutoConfig.from_pretrained(dir_model, trust_remote_code=False).to_dict()
except Exception as e:
logger.warning(f"Failed to load model config from {dir_model}: {e}")
- if not (dir_model / "config.json").is_file():
- config = ModelBase.load_hparams_guess(dir_model)
- if config is not None:
- return config
logger.warning("Trying to load config.json instead")
with open(dir_model / "config.json", "r", encoding="utf-8") as f:
config = json.load(f)
@@ -1936,6 +1938,9 @@ class TextModel(ModelBase):
if chkhsh == "653660222fb704f61cbf2b618a8ae6502b7f8b20c980f9a5de07ed78e13319cd":
# ref: https://huggingface.co/ufakai/ufakzeka-1
res = "ufakzeka"
+ if chkhsh == "4b05e02dad1c5ae07d266fd3342ddb644c6f6be058d728bc0a33af31a1d6ee66":
+ # ref: https://huggingface.co/jhu-clsp/mmBERT-base
+ res = "mmbert"
if res is None:
logger.warning("\n")
@@ -2880,6 +2885,11 @@ else:
LazyTorchTensor._dtype_str_map["F8_E8M0"] = torch.uint8
+def jinja_str_or_json(name: str) -> str:
+ # jinja expression that renders a variable as-is if it is a string, as JSON otherwise
+ return "{{ " + name + " if " + name + " is string else " + name + " | tojson }}"
+
+
def get_model_architecture(hparams: dict[str, Any], model_type: ModelType) -> str:
# TODO @ngxson : this won't work correctly if the model has both audio & vision encoders
# maybe we should fallback to text model's arch in that case, since not many models have both
diff --git a/conversion/bert.py b/conversion/bert.py
index 23d8b9333..f24715426 100644
--- a/conversion/bert.py
+++ b/conversion/bert.py
@@ -11,7 +11,7 @@ import torch
if TYPE_CHECKING:
from torch import Tensor
-from .base import ModelBase, SentencePieceTokenTypes, TextModel, gguf, logger
+from .base import ModelBase, SentencePieceTokenTypes, TextModel, gguf, jinja_str_or_json, logger
@ModelBase.register("BertModel", "BertForMaskedLM", "CamembertModel", "BertForSequenceClassification")
@@ -606,6 +606,17 @@ class ModernBertModel(BertModel):
self.gguf_writer.add_add_sep_token(True)
self._set_vocab_gpt2()
+ def get_vocab_base(self) -> tuple[list[str], list[int], str]:
+ tokens, toktypes, tokpre = super().get_vocab_base()
+ if tokpre == "mmbert":
+ # the added tokens for runs of spaces are never matched by the reference tokenizer
+ space = b"\xe2\x96\x81".decode("utf-8")
+ for i, token in enumerate(tokens):
+ if toktypes[i] == gguf.TokenType.USER_DEFINED and token and not token.strip(" "):
+ tokens[i] = space * len(token)
+ toktypes[i] = gguf.TokenType.NORMAL
+ return tokens, toktypes, tokpre
+
def set_gguf_parameters(self):
super().set_gguf_parameters()
self.gguf_writer.add_sliding_window(self.hparams["local_attention"])
@@ -639,3 +650,98 @@ class ModernBertModel(BertModel):
name = "classifier.out_proj.bias"
yield from super().modify_tensors(data_torch, name, bid)
+
+
+def _is_decision_checkpoint(dir_model: Path) -> bool:
+ if not (dir_model / "encoder" / "config.json").is_file():
+ return False
+ return (dir_model / "rl_agent_config.json").is_file() or (dir_model / "julia_config.json").is_file()
+
+
+@ModelBase.register_hparams_loader(_is_decision_checkpoint)
+def _load_decision_hparams(dir_model: Path) -> dict[str, Any]:
+ logger.info("gguf: detected ModernBert decision checkpoint")
+ hparams = ModelBase.load_hparams(dir_model / "encoder", False, guess=False)
+ is_julia = (dir_model / "julia_config.json").is_file()
+ with open(dir_model / ("julia_config.json" if is_julia else "rl_agent_config.json"), encoding="utf-8") as f:
+ decision = json.load(f)
+ n_layer = hparams["num_hidden_layers"]
+ n_layer_head = decision["head_layers"]
+ hparams["architectures"] = ["ModernBertDecisionModel"]
+ hparams["decision"] = decision
+ # the head blocks are appended to the encoder blocks, they use a plain 4x MLP
+ hparams["num_hidden_layers"] = n_layer + n_layer_head
+ hparams["intermediate_size"] = [hparams["intermediate_size"]] * n_layer + [4 * hparams["hidden_size"]] * n_layer_head
+ return hparams
+
+
+@ModelBase.register("ModernBertDecisionModel")
+@ModelBase.example("convaiinnovations/laya", "SupersonicLabs/Julia-1")
+class ModernBertDecisionModel(ModernBertModel):
+ model_arch = gguf.MODEL_ARCH.MODERN_BERT
+
+ def set_vocab(self):
+ # vocab loaders read self.dir_model, point it to the tokenizer sub-directory
+ dir_model = self.dir_model
+ self.dir_model = dir_model / "tokenizer"
+ try:
+ super().set_vocab()
+ finally:
+ self.dir_model = dir_model
+ self.gguf_writer.add_token_type_count(3) # choice, score, noul
+ self.gguf_writer.add_chat_template([{"name": "systemone", "template": self._systemone_template()}])
+
+ def _systemone_template(self) -> str:
+ with open(self.dir_model / "tokenizer" / "tokenizer_config.json", encoding="utf-8") as f:
+ tokenizer_config = json.load(f)
+ tok_cls, tok_sep, tok_mask = (tokenizer_config[k] for k in ("cls_token", "sep_token", "mask_token"))
+ description = jinja_str_or_json("o.description")
+ if self.hparams["decision"].get("architecture") == "JuliaDecisionModel":
+ option = "{% if o.description %}" + description + "{% else %}{{ o.key }}{% endif %}"
+ else:
+ option = (
+ "{% if type == 'choice' %}{{ o.key }}{% if o.description %}: " + description + "{% endif %}"
+ "{% elif type == 'score' %}level {{ o.key }}: " + description
+ + "{% else %}{{ o.key }}: {% if o.description %}" + description
+ + "{% elif o.key == 'true' %}yes, the statement holds"
+ "{% else %}no, the statement does not hold{% endif %}{% endif %}"
+ )
+ return (
+ tok_cls + "{{ type }} question: " + jinja_str_or_json("instructions") + tok_sep
+ + "{% for o in options %}" + tok_mask + " " + option + "{% endfor %}"
+ + tok_sep + jinja_str_or_json("state") + tok_sep
+ )
+
+ def set_gguf_parameters(self):
+ super().set_gguf_parameters()
+ decision = self.hparams["decision"]
+ self.gguf_writer.add_decision_type(gguf.DecisionType.LAYA)
+ self.gguf_writer.add_decision_block_count(decision["head_layers"])
+ self.gguf_writer.add_decision_max_head_tokens(decision.get("head_max_len", 256))
+ for name, value in zip(("choice", "score", "noul"), decision.get("temperature", [])):
+ self.gguf_writer.add_decision_temperature(name, value)
+ # "choice:3-5" -> "choice.3_5", "choice:11+" -> "choice.11"
+ for name, value in decision.get("temperature_by_options", {}).items():
+ self.gguf_writer.add_decision_temperature(name.replace(":", ".").replace("-", "_").rstrip("+"), value)
+
+ @classmethod
+ def filter_tensors(cls, item: tuple[str, Callable[[], Tensor]]) -> tuple[str, Callable[[], Tensor]] | None:
+ name, gen = item
+
+ # act_head is not used for the answer, the fitted temperatures come from the config
+ if name.startswith("act_head.") or name == "temperature":
+ return None
+
+ if name.startswith("encoder."):
+ name = name[8:]
+
+ return super().filter_tensors((name, gen))
+
+ def modify_tensors(self, data_torch: Tensor, name: str, bid: int | None) -> Iterable[tuple[str, Tensor]]:
+ if name.startswith("head.layers.") and bid is not None:
+ # the head blocks come after the encoder blocks
+ suffix = name.split(".", 3)[3].replace("in_proj_", "in_proj.")
+ bid += self.block_count - self.hparams["decision"]["head_layers"]
+ name = f"head.layers.{bid}.{suffix}"
+
+ yield from super().modify_tensors(data_torch, name, bid)
diff --git a/conversion/lev.py b/conversion/lev.py
new file mode 100644
index 000000000..b10421cb4
--- /dev/null
+++ b/conversion/lev.py
@@ -0,0 +1,205 @@
+from __future__ import annotations
+
+import json
+
+from pathlib import Path
+from typing import Any, Iterable, TYPE_CHECKING
+
+import torch
+
+if TYPE_CHECKING:
+ from torch import Tensor
+
+from .base import LazyTorchTensor, ModelBase, gguf, jinja_str_or_json, logger
+from .qwen import Qwen3_5TextModel
+
+
+def _decision_lora_base(dir_model: Path) -> tuple[str, str | None]:
+ # the base model of a LoRA adapter: (repo id, revision)
+ with open(dir_model / "adapter_config.json", encoding="utf-8") as f:
+ lora_config = json.load(f)
+ revision = lora_config.get("revision")
+ if revision is None and (dir_model / "training_config.json").is_file():
+ with open(dir_model / "training_config.json", encoding="utf-8") as f:
+ revision = json.load(f).get("base_revision")
+ return lora_config["base_model_name_or_path"], revision
+
+
+def _load_decision_lora_hparams(dir_model: Path, arch: str) -> dict[str, Any]:
+ from huggingface_hub import hf_hub_download
+ repo_id, revision = _decision_lora_base(dir_model)
+ with open(hf_hub_download(repo_id, "config.json", revision=revision), encoding="utf-8") as f:
+ hparams = json.load(f)
+ hparams["architectures"] = [arch]
+ return hparams
+
+
+class _DecisionLoraMixin:
+ # decision model released as a LoRA adapter: the base model is downloaded and the adapter is merged into it
+ no_mtp = True
+
+ def __init__(self, dir_model: Path, *args, **kwargs):
+ from huggingface_hub import snapshot_download
+ from safetensors.torch import load_file
+
+ repo_id, revision = _decision_lora_base(dir_model)
+ logger.info(f"gguf: downloading the base model {repo_id}")
+ dir_base = Path(snapshot_download(repo_id, revision=revision, allow_patterns=["*.json", "*.jinja", "*.safetensors"]))
+ super().__init__(dir_base, *args, **kwargs) # ty: ignore[too-many-positional-arguments]
+ self.dir_adapter = dir_model
+ self.dir_model_card = dir_model
+
+ with open(dir_model / "adapter_config.json", encoding="utf-8") as f:
+ lora_config = json.load(f)
+ # only a plain LoRA can be merged as scale * B @ A
+ assert lora_config["peft_type"] == "LORA"
+ assert lora_config.get("bias", "none") == "none"
+ assert not lora_config.get("use_dora") and not lora_config.get("use_rslora") and not lora_config.get("lora_bias")
+ assert not lora_config.get("rank_pattern") and not lora_config.get("alpha_pattern")
+ assert not lora_config.get("modules_to_save")
+ self.lora_scale = lora_config["lora_alpha"] / lora_config["r"]
+
+ # "layers.0.mlp.up_proj.weight" -> {"A": tensor, "B": tensor}
+ self.lora: dict[str, dict[str, Tensor]] = {}
+ for name, tensor in load_file(dir_model / "adapter_model.safetensors").items():
+ base_name, _, part = name[name.index("layers."):].partition(".lora_")
+ assert part in ("A.weight", "B.weight"), f"unexpected LoRA tensor: {name}"
+ self.lora.setdefault(base_name + ".weight", {})[part[0]] = tensor.float()
+ self.lora_merged: set[str] = set()
+
+ def modify_tensors(self, data_torch: Tensor, name: str, bid: int | None) -> Iterable[tuple[str, Tensor]]:
+ lora = self.lora.get(name[name.index("layers."):]) if "layers." in name else None
+ if lora is not None:
+ assert set(lora) == {"A", "B"} and data_torch.shape == (lora["B"].shape[0], lora["A"].shape[1])
+ delta = self.lora_scale * (lora["B"] @ lora["A"])
+ data_torch = data_torch.float() + LazyTorchTensor.from_eager(delta)
+ self.lora_merged.add(name[name.index("layers."):])
+ yield from super().modify_tensors(data_torch, name, bid) # ty: ignore[unresolved-attribute]
+
+ def prepare_tensors(self):
+ super().prepare_tensors() # ty: ignore[unresolved-attribute]
+ if len(self.lora_merged) != len(self.lora):
+ raise ValueError(f"only {len(self.lora_merged)} of {len(self.lora)} LoRA tensors were merged into the base model")
+
+
+@ModelBase.register_hparams_loader(lambda dir_model: (dir_model / "lev_release.json").is_file())
+def _load_lev_hparams(dir_model: Path) -> dict[str, Any]:
+ logger.info("gguf: detected Lev checkpoint")
+ return _load_decision_lora_hparams(dir_model, "LevModel")
+
+
+@ModelBase.register("LevModel")
+@ModelBase.example("interfaze-ai/lev")
+class LevModel(_DecisionLoraMixin, Qwen3_5TextModel):
+ model_arch = gguf.MODEL_ARCH.QWEN35
+
+ # TODO: the head for large option sets (mode B, mode_b_head.pt) is not converted, only the label readout is supported
+ # TODO: a description that is not text is given as JSON without the escaping of non-ASCII characters used in training
+
+ # prompt follows packages/lev/src/lev/prompt.py of https://github.com/Abhinavexists/lev (chat style, state first)
+ _SYSTEM_PROMPT = (
+ "You are a System One decision model. You read the Evidence and answer each "
+ "Criterion by choosing exactly one of the listed options. You never explain. "
+ "You answer with the single option label only."
+ )
+
+ def set_vocab(self):
+ super().set_vocab()
+ self.gguf_writer.add_chat_template([{"name": "systemone", "template": self._systemone_template()}])
+
+ def _systemone_template(self) -> str:
+ description = jinja_str_or_json("o.description")
+ options = (
+ "{{ '# Options\\n' }}{% for o in options %}{{ o.label }}. "
+ "{% if type == 'score' %}(level {{ o.key }} of {{ options | length - 1 }}) " + description
+ + "{% else %}{{ o.key }}{% if o.description %}: " + description + "{% endif %}{% endif %}"
+ "{{ '\\n' }}{% endfor %}"
+ "{{ '\\nRespond with only the letter of ' }}"
+ "{% if type == 'score' %}the level that best matches.{% else %}the best option.{% endif %}"
+ )
+ # noul is answered on a rating scale, its 2 options are only used for their description
+ scale = "{{ '# Scale\\n0 = certainly no ... 8 = certainly yes\\n' }}"
+ for key, name in (("true", "yes"), ("false", "no")):
+ scale += (
+ "{% for o in options %}{% if o.key == '" + key + "' and o.description %}"
+ + name + ": " + description + "{{ '\\n' }}{% endif %}{% endfor %}"
+ )
+ scale += "{{ '\\nRespond with only a digit from 0 to 8.' }}"
+ return (
+ "<|im_start|>system\n" + self._SYSTEM_PROMPT + "<|im_end|>\n"
+ "<|im_start|>user\n# Evidence\n" + jinja_str_or_json("state") + "\n\n# Criterion\n"
+ "{% if instructions %}" + jinja_str_or_json("instructions") + "{% else %}{{ id }}{% endif %}"
+ "{{ '\\n\\n' }}{% if type == 'noul' %}" + scale + "{% else %}" + options + "{% endif %}"
+ "{{ '\\n<|im_end|>\\n<|im_start|>assistant\\n<think>\\n\\n</think>\\n\\n' }}"
+ )
+
+ def set_gguf_parameters(self):
+ super().set_gguf_parameters()
+ self.gguf_writer.add_decision_type(gguf.DecisionType.LEV)
+ with open(self.dir_adapter / "calibration.json", encoding="utf-8") as f:
+ temperatures = json.load(f)["temperatures"]
+ # "choice:A:small" -> "choice.small", only the label readout (mode A) is supported
+ for name, value in temperatures.items():
+ qtype, mode, *band = name.split(":")
+ if mode == "A":
+ self.gguf_writer.add_decision_temperature(".".join([qtype] + band), value)
+
+
+def _is_kev_checkpoint(dir_model: Path) -> bool:
+ # a LoRA adapter with the pointer head and the config of the kev training code
+ if not all((dir_model / name).is_file() for name in ("adapter_config.json", "head.pt", "training_config.json")):
+ return False
+ with open(dir_model / "training_config.json", encoding="utf-8") as f:
+ return "head_dim" in json.load(f).get("args", {})
+
+
+@ModelBase.register_hparams_loader(_is_kev_checkpoint)
+def _load_kev_hparams(dir_model: Path) -> dict[str, Any]:
+ logger.info("gguf: detected Kev checkpoint")
+ return _load_decision_lora_hparams(dir_model, "KevModel")
+
+
+@ModelBase.register("KevModel")
+@ModelBase.example("jaredpalmer/kev-4b")
+class KevModel(_DecisionLoraMixin, Qwen3_5TextModel):
+ model_arch = gguf.MODEL_ARCH.QWEN35
+
+ # TODO: the server needs a question and its options in one batch, the state can be in previous batches
+ # note: no plan to support date_facts (kev/api.py), its regex matching is fragile, a more generic impl is needed
+
+ def __init__(self, *args, **kwargs):
+ super().__init__(*args, **kwargs)
+ self.head = torch.load(self.dir_adapter / "head.pt", map_location="cpu", weights_only=True)
+ assert set(self.head["head"]) == {"q.weight", "q.bias", "k.weight", "k.bias"}
+ assert self.head["head"]["q.weight"].shape[0] == self.head["head_dim"]
+
+ def set_vocab(self):
+ super().set_vocab()
+ self.gguf_writer.add_chat_template([{"name": "systemone", "template": self._systemone_template()}])
+
+ def _systemone_template(self) -> str:
+ # prompt follows kev/model.py and kev/api.py of https://github.com/jaredpalmer/kev
+ # state, instructions and descriptions are given as text
+ name = "{% if type != 'noul' %}{{ o.key }}{% elif o.key == 'true' %}yes{% else %}no{% endif %}"
+ option = (
+ "{% if type == 'score' %}{% if o.description %}{{ o.description }}{% endif %}"
+ "{% else %}" + name + "{% if o.description %}: {{ o.description }}{% endif %}{% endif %}"
+ )
+ return (
+ "<|fim_prefix|>{{ state }}<|fim_middle|>{{ instructions }}"
+ "{% for o in options %}<|box_start|>" + option + "<|box_end|>{% endfor %}<|fim_suffix|>"
+ )
+
+ def set_gguf_parameters(self):
+ super().set_gguf_parameters()
+ self.gguf_writer.add_decision_type(gguf.DecisionType.KEV)
+ self.gguf_writer.add_embedding_length_out(2 * self.head["head_dim"])
+ for name in ("choice", "score", "noul"):
+ self.gguf_writer.add_decision_temperature(name, self.head["temperature"])
+
+ def generate_extra_tensors(self) -> Iterable[tuple[str, Tensor]]:
+ yield from super().generate_extra_tensors()
+ # pointer head: the output of a token is [q | k]
+ head = self.head["head"]
+ yield "classifier.out_proj.weight", torch.cat([head["q.weight"], head["k.weight"]], dim=0)
+ yield "classifier.out_proj.bias", torch.cat([head["q.bias"], head["k.bias"]], dim=0)
diff --git a/conversion/qwen.py b/conversion/qwen.py
index 4946456e9..0b3918733 100644
--- a/conversion/qwen.py
+++ b/conversion/qwen.py
@@ -2,6 +2,7 @@ from __future__ import annotations
import json
+from pathlib import Path
from typing import Any, Callable, Iterable, TYPE_CHECKING
import numpy as np
@@ -10,7 +11,7 @@ import torch
if TYPE_CHECKING:
from torch import Tensor
-from .base import LazyTorchTensor, ModelBase, ModelType, TextModel, get_model_architecture, gguf, logger
+from .base import LazyTorchTensor, ModelBase, ModelType, TextModel, get_model_architecture, gguf, jinja_str_or_json, logger
@ModelBase.register("QWenLMHeadModel")
@@ -655,6 +656,62 @@ class Qwen3_5TextModel(_Qwen35MRopeMixin, _LinearAttentionVReorderBase):
model_arch = gguf.MODEL_ARCH.QWEN35
+def _is_openjev_checkpoint(dir_model: Path) -> bool:
+ return (dir_model / "helper" / "shim.py").is_file() and (dir_model / "config.json").is_file()
+
+
+@ModelBase.register_hparams_loader(_is_openjev_checkpoint)
+def _load_openjev_hparams(dir_model: Path) -> dict[str, Any]:
+ logger.info("gguf: detected OpenJev checkpoint")
+ hparams = ModelBase.load_hparams(dir_model, False, guess=False)
+ hparams["architectures"] = ["OpenJevModel"]
+ return hparams
+
+
+@ModelBase.register("OpenJevModel")
+@ModelBase.example("openjev/openjev")
+class OpenJevModel(Qwen3_5TextModel):
+ model_arch = gguf.MODEL_ARCH.QWEN35
+ no_mtp = True # the checkpoint has no MTP head
+
+ # prompt and calibration follow helper/shim.py of the model repo (text lane)
+ _LETTERS = "ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz"
+ _TEMPERATURE = 0.85
+ _TEMPERATURE_NOUL = 1.829074 # applied on top of _TEMPERATURE
+
+ def set_vocab(self):
+ super().set_vocab()
+ self.gguf_writer.add_chat_template([{"name": "systemone", "template": self._systemone_template()}])
+
+ def _systemone_template(self) -> str:
+ description = jinja_str_or_json("o.description")
+ option = (
+ "{% if type != 'noul' %}{{ o.key }}: {% if o.description %}" + description + "{% endif %}"
+ "{% elif o.key == 'true' %}yes: {% if o.description %}" + description + "{% else %}The statement is true.{% endif %}"
+ "{% else %}no: {% if o.description %}" + description + "{% else %}The statement is false.{% endif %}{% endif %}"
+ )
+ # TODO: only the layout with one image is known (image first), the one with several images is not verified
+ images = (
+ "{% for image in images %}{{ image }}{% endfor %}"
+ "{% if images %}{{ 'The screenshot shows the current screen.\\n' }}{% endif %}"
+ )
+ return (
+ "{% set letters = '" + self._LETTERS + "' %}"
+ "<|im_start|>user\n" + images + "State:\n" + jinja_str_or_json("state") + "\n\nQuestion: " + jinja_str_or_json("instructions")
+ + "{% if type == 'score' %} Rate along the ordered levels below (lowest first).{% endif %}"
+ "{{ '\\nOptions:\\n' }}"
+ "{% for o in options %}[{{ letters[loop.index0] }}] " + option + "{{ '\\n' }}{% endfor %}"
+ "{{ '\\nAnswer with the letter of the best option only.<|im_end|>\\n<|im_start|>assistant\\n<think>\\n\\n</think>\\n\\n' }}"
+ )
+
+ def set_gguf_parameters(self):
+ super().set_gguf_parameters()
+ self.gguf_writer.add_decision_type(gguf.DecisionType.OPENJEV)
+ self.gguf_writer.add_decision_temperature("choice", self._TEMPERATURE)
+ self.gguf_writer.add_decision_temperature("score", self._TEMPERATURE)
+ self.gguf_writer.add_decision_temperature("noul", self._TEMPERATURE * self._TEMPERATURE_NOUL)
+
+
@ModelBase.register("Qwen3_5MoeForConditionalGeneration", "Qwen3_5MoeForCausalLM")
@ModelBase.example("Qwen/Qwen3.5-35B-A3B")
class Qwen3_5MoeTextModel(_Qwen35MRopeMixin, _LinearAttentionVReorderBase):
diff --git a/conversion/qwen3vl.py b/conversion/qwen3vl.py
index 11ce68515..638bcbb24 100644
--- a/conversion/qwen3vl.py
+++ b/conversion/qwen3vl.py
@@ -13,7 +13,7 @@ from .qwen import Qwen3Model, Qwen3MoeModel
from .qwenvl import Qwen25AudioModel
-@ModelBase.register("Qwen3VLForConditionalGeneration", "Qwen3VLMoeForConditionalGeneration", "Qwen3_5ForConditionalGeneration", "Qwen3_5MoeForConditionalGeneration")
+@ModelBase.register("Qwen3VLForConditionalGeneration", "Qwen3VLMoeForConditionalGeneration", "Qwen3_5ForConditionalGeneration", "Qwen3_5MoeForConditionalGeneration", "OpenJevModel")
@ModelBase.example("Qwen/Qwen3-VL-4B-Instruct", "Qwen/Qwen3-VL-30B-A3B-Instruct", "Qwen/Qwen3.5-9B", "Qwen/Qwen3.5-35B-A3B")
class Qwen3VLVisionModel(MmprojModel):
def __init__(self, *args, **kwargs):
diff --git a/convert_hf_to_gguf_update.py b/convert_hf_to_gguf_update.py
index 24b9bc075..d39d2f6fe 100755
--- a/convert_hf_to_gguf_update.py
+++ b/convert_hf_to_gguf_update.py
@@ -164,6 +164,7 @@ models = [
{"name": "mellum2", "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/JetBrains/Mellum2-12B-A2.5B-Base"},
{"name": "laguna", "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/poolside/Laguna-XS.2", },
{"name": "ufakzeka", "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/ufakai/ufakzeka-1", },
+ {"name": "mmbert", "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/jhu-clsp/mmBERT-base", },
]
# some models are known to be broken upstream, so we will skip them as exceptions
diff --git a/docs/development/HOWTO-add-model.md b/docs/development/HOWTO-add-model.md
index 31b3f2686..abf6503f2 100644
--- a/docs/development/HOWTO-add-model.md
+++ b/docs/development/HOWTO-add-model.md
@@ -139,6 +139,23 @@ Note:
- In most cases, `llama-mtmd-cli` should not be modified. If a model requires a specific prompt, either let the user provide it or bake it into the Jinja chat template.
- For audio generation models, see `tools/mtmd/README-dev.md`
+## Add a decision model
+
+A decision model answers typed questions about a state in one forward pass. It is served by `POST /v1/systemone` in `llama-server`, see [the server docs](../../tools/server/README.md).
+
+The conversion is the same as above, but a new model needs its own `DecisionType` in `gguf-py/gguf/constants.py`. See the existing models and follow the pattern.
+
+> [!IMPORTANT]
+>
+> Most of the logic is handled in `tools/server/server-decision.cpp`, to avoid too many changes to `libllama`.
+
+Note:
+- If a new public API is needed in `libllama`, add it to `llama-ext.h`.
+- Metadata with a single use case must be hard-coded in `server-decision.cpp` instead of being saved to the GGUF. This avoids bloating the conversion code.
+- Most importantly, keep your change as small and as self-contained as possible. Reuse the existing infrastructure whenever you can.
+
+For more information, see [PR #29818](https://github.com/ggml-org/llama.cpp/pull/29818).
+
## Tips and tricks
### Prefer conversion-time tensor modifications over graph-time ones
diff --git a/gguf-py/gguf/constants.py b/gguf-py/gguf/constants.py
index 8075e82a3..a8fd9f0dd 100644
--- a/gguf-py/gguf/constants.py
+++ b/gguf-py/gguf/constants.py
@@ -320,6 +320,13 @@ class Keys:
class ShortConv:
L_CACHE = "{arch}.shortconv.l_cache"
+ class Decision:
+ TYPE = "{arch}.decision.type"
+ # note: single-use-case keys can be hard-coded in cpp code
+ BLOCK_COUNT = "{arch}.decision.block_count"
+ MAX_HEAD_TOKENS = "{arch}.decision.max_head_tokens"
+ TEMPERATURE = "{arch}.decision.temperature.{name}" # name: "<type>" or "<type>.<n_opt bucket>"
+
class Tokenizer:
MODEL = "tokenizer.ggml.model"
PRE = "tokenizer.ggml.pre"
@@ -2487,6 +2494,7 @@ MODEL_TENSORS: dict[MODEL_ARCH, list[MODEL_TENSOR]] = {
MODEL_ARCH.MODERN_BERT: [
MODEL_TENSOR.TOKEN_EMBD,
MODEL_TENSOR.TOKEN_EMBD_NORM,
+ MODEL_TENSOR.TOKEN_TYPES,
MODEL_TENSOR.OUTPUT_NORM,
MODEL_TENSOR.ATTN_NORM,
MODEL_TENSOR.ATTN_OUT,
@@ -2881,6 +2889,7 @@ MODEL_TENSORS: dict[MODEL_ARCH, list[MODEL_TENSOR]] = {
MODEL_TENSOR.TOKEN_EMBD,
MODEL_TENSOR.OUTPUT_NORM,
MODEL_TENSOR.OUTPUT,
+ MODEL_TENSOR.CLS_OUT,
MODEL_TENSOR.ATTN_NORM,
MODEL_TENSOR.ATTN_Q,
MODEL_TENSOR.ATTN_Q_NORM,
@@ -5906,6 +5915,13 @@ class GGUFValueType(IntEnum):
raise ValueError(f"Unknown type: {type(val)}")
+class DecisionType:
+ LAYA = "laya" # head blocks + scorer on the hidden state of one marker token per option
+ OPENJEV = "openjev" # logits of one label token per option
+ LEV = "lev" # same as openjev, noul is read from a rating scale
+ KEV = "kev" # dot product of the hidden states of the last token and of one end token per option
+
+
class VisionProjectorType:
GEMMA3 = "gemma3"
GEMMA3NV = "gemma3nv"
diff --git a/gguf-py/gguf/gguf_writer.py b/gguf-py/gguf/gguf_writer.py
index 342f6eb4d..1dee3fe11 100644
--- a/gguf-py/gguf/gguf_writer.py
+++ b/gguf-py/gguf/gguf_writer.py
@@ -1340,6 +1340,18 @@ class GGUFWriter:
def add_classifier_pooling_type(self, value: PoolingType) -> None:
self.add_uint32(Keys.Classifier.POOLING_TYPE.format(arch=self.arch), value.value)
+ def add_decision_type(self, value: str) -> None:
+ self.add_string(Keys.Decision.TYPE.format(arch=self.arch), value)
+
+ def add_decision_block_count(self, value: int) -> None:
+ self.add_uint32(Keys.Decision.BLOCK_COUNT.format(arch=self.arch), value)
+
+ def add_decision_max_head_tokens(self, value: int) -> None:
+ self.add_uint32(Keys.Decision.MAX_HEAD_TOKENS.format(arch=self.arch), value)
+
+ def add_decision_temperature(self, name: str, value: float) -> None:
+ self.add_float32(Keys.Decision.TEMPERATURE.format(arch=self.arch, name=name), value)
+
# for vision models
def add_clip_has_vision_encoder(self, value: bool) -> None:
diff --git a/gguf-py/gguf/tensor_mapping.py b/gguf-py/gguf/tensor_mapping.py
index ed7f1c2be..c8d52f2fe 100644
--- a/gguf-py/gguf/tensor_mapping.py
+++ b/gguf-py/gguf/tensor_mapping.py
@@ -48,6 +48,7 @@ class TensorNameMap:
# Token type embeddings
MODEL_TENSOR.TOKEN_TYPES: (
"embeddings.token_type_embeddings", # bert nomic-bert
+ "type_emb", # laya
),
# Normalization of token embeddings
@@ -216,6 +217,7 @@ class TensorNameMap:
"layers.{bid}.input_layernorm", # qwen3-embedding
"model.layers.{bid}.attention_layernorm", # apertus
"model.layers.{bid}.pre_attention_layernorm", # kormo
+ "head.layers.{bid}.norm1", # laya
),
# Attention norm 2
@@ -250,6 +252,7 @@ class TensorNameMap:
"layers.{bid}.attn.Wqkv", # modern-bert
"model.layers.{bid}.self_attn.language_expert_query_key_value", # cogvlm
"model.layers.{bid}.linear_attn.in_proj_qkv", # qwen3.5
+ "head.layers.{bid}.self_attn.in_proj", # laya
),
# Attention query
@@ -355,6 +358,7 @@ class TensorNameMap:
"backbone.layers.{bid}.mixer.o_proj", # nemotron-h
"model.layers.{bid}.self_attn.language_expert_dense", # cogvlm
"model.blocks.{bid}.attn.attn_resid", # talkie
+ "head.layers.{bid}.self_attn.out_proj", # laya
),
# Attention output norm
@@ -420,7 +424,8 @@ class TensorNameMap:
"layers.{bid}.post_attention_layernorm", # qwen3-embedding
"model.layers.{bid}.feedforward_layernorm", # apertus
"model.layers.{bid}.pre_mlp_layernorm", # kormo
- "layers.{bid}.mlp_norm" # modern-bert
+ "layers.{bid}.mlp_norm", # modern-bert
+ "head.layers.{bid}.norm2", # laya
),
# Pre feed-forward norm
@@ -533,6 +538,7 @@ class TensorNameMap:
"backbone.layers.{bid}.mixer.up_proj", # nemotron-h
"model.layers.{bid}.mlp.language_mlp.up_proj", # cogvlm
"model.blocks.{bid}.mlp.mlp_linear", # talkie
+ "head.layers.{bid}.linear1", # laya
),
MODEL_TENSOR.FFN_UP_EXP: (
@@ -663,6 +669,7 @@ class TensorNameMap:
"backbone.layers.{bid}.mixer.down_proj", # nemotron-h
"model.layers.{bid}.mlp.language_mlp.down_proj", # cogvlm
"model.blocks.{bid}.mlp.mlp_resid", # talkie
+ "head.layers.{bid}.linear2", # laya
),
MODEL_TENSOR.FFN_DOWN_EXP: (
@@ -1441,14 +1448,17 @@ class TensorNameMap:
"pre_classifier", # distillbert
"dense", # neobert
"head.dense", # modern-bert
+ "scorer.1", # laya
),
MODEL_TENSOR.CLS_OUT: (
"classifier.out_proj", # roberta
+ "scorer.3", # laya
),
MODEL_TENSOR.CLS_NORM: (
"head.norm", # modern-bert
+ "scorer.0", # laya
),
#############################################################################
diff --git a/src/llama-arch.cpp b/src/llama-arch.cpp
index cc361c7f2..9f205b942 100644
--- a/src/llama-arch.cpp
+++ b/src/llama-arch.cpp
@@ -366,6 +366,8 @@ static const std::map<llm_kv, const char *> LLM_KV_NAMES = {
{ LLM_KV_CLASSIFIER_OUTPUT_LABELS, "%s.classifier.output_labels" },
{ LLM_KV_CLASSIFIER_POOLING_TYPE, "%s.classifier.pooling_type" },
+ { LLM_KV_DECISION_BLOCK_COUNT, "%s.decision.block_count" },
+
{ LLM_KV_TARGET_LAYERS, "%s.target_layers" },
{ LLM_KV_TARGET_HIDDEN_SIZE, "%s.target_hidden_size" },
{ LLM_KV_NORM_BEFORE_RESIDUAL, "%s.norm_before_residual" },
diff --git a/src/llama-arch.h b/src/llama-arch.h
index 612b6797a..ca390fcb3 100644
--- a/src/llama-arch.h
+++ b/src/llama-arch.h
@@ -412,6 +412,8 @@ enum llm_kv {
LLM_KV_CLASSIFIER_OUTPUT_LABELS,
LLM_KV_CLASSIFIER_POOLING_TYPE,
+ LLM_KV_DECISION_BLOCK_COUNT,
+
LLM_KV_TARGET_LAYERS,
LLM_KV_TARGET_HIDDEN_SIZE,
LLM_KV_DFLASH_BLOCK_SIZE,
diff --git a/src/llama-hparams.h b/src/llama-hparams.h
index 756007e1f..6c504c5dd 100644
--- a/src/llama-hparams.h
+++ b/src/llama-hparams.h
@@ -65,6 +65,7 @@ struct llama_hparams {
uint32_t n_embd;
uint32_t n_layer_all;
uint32_t n_layer_nextn = 0;
+ uint32_t n_layer_decision = 0; // trailing blocks that form the decision head
// granite-switch: index of the single-head "router" KV layer that encodes
// per-token adapter selection. -1 when the model has no such layer.
diff --git a/src/llama-model.h b/src/llama-model.h
index 25e514869..2243377cc 100644
--- a/src/llama-model.h
+++ b/src/llama-model.h
@@ -684,6 +684,7 @@ struct llama_model {
struct ggml_tensor * cls_out = nullptr;
struct ggml_tensor * cls_out_b = nullptr;
struct ggml_tensor * cls_norm = nullptr;
+ struct ggml_tensor * cls_norm_b = nullptr;
struct ggml_tensor * conv1d = nullptr;
struct ggml_tensor * conv1d_b = nullptr;
diff --git a/src/llama-vocab.cpp b/src/llama-vocab.cpp
index e5703f3e8..015e98374 100644
--- a/src/llama-vocab.cpp
+++ b/src/llama-vocab.cpp
@@ -543,6 +543,13 @@ struct llm_tokenizer_bpe : llm_tokenizer {
};
byte_encode = false;
break;
+ case LLAMA_VOCAB_PRE_TYPE_MMBERT:
+ // same as Gemma4, the words are split in tokenize()
+ regex_exprs = {
+ "[^\\n]+|[\\n]+",
+ };
+ byte_encode = false;
+ break;
case LLAMA_VOCAB_PRE_TYPE_MINICPM5:
regex_exprs = {
// original regex from tokenizer.json (openbmb/MiniCPM5-1B)
@@ -618,11 +625,34 @@ struct llm_tokenizer_bpe_session {
virtual void tokenize(const std::string & text, std::vector<llama_token> & output) {
int final_prev_index = -1;
- const auto word_collection = unicode_regex_split(text, tokenizer.regex_exprs, tokenizer.byte_encode);
+ auto word_collection = unicode_regex_split(text, tokenizer.regex_exprs, tokenizer.byte_encode);
symbols_final.clear();
auto tok_pre = vocab.get_pre_type();
+ if (tok_pre == LLAMA_VOCAB_PRE_TYPE_MMBERT) {
+ // Metaspace pre-tokenizer: a text starts with an escaped space, and each escaped space starts a new word
+ static const std::string space = "\xe2\x96\x81";
+ std::vector<std::string> words;
+ for (auto & word : word_collection) {
+ if (word.find_first_not_of('\n') == std::string::npos) {
+ words.push_back(word);
+ continue;
+ }
+ if (word.compare(0, space.size(), space) != 0) {
+ word = space + word;
+ }
+ size_t start = 0;
+ while (start < word.size()) {
+ size_t end = word.find(space, start + space.size());
+ end = end == std::string::npos ? word.size() : end;
+ words.push_back(word.substr(start, end - start));
+ start = end;
+ }
+ }
+ word_collection = std::move(words);
+ }
+
for (const auto & word : word_collection) {
work_queue = llm_bigram_bpe::queue();
symbols.clear();
@@ -634,7 +664,7 @@ struct llm_tokenizer_bpe_session {
if (vocab.get_ignore_merges() && vocab.text_to_token(word) != LLAMA_TOKEN_NULL) {
symbols.emplace_back(llm_symbol{-1, -1, word.c_str(), word.size()});
offset = word.size();
- } else if (tok_pre == LLAMA_VOCAB_PRE_TYPE_GEMMA4 && word.find_first_not_of('\n') == std::string::npos) {
+ } else if ((tok_pre == LLAMA_VOCAB_PRE_TYPE_GEMMA4 || tok_pre == LLAMA_VOCAB_PRE_TYPE_MMBERT) && word.find_first_not_of('\n') == std::string::npos) {
// fix for gemma 4, ref: https://github.com/ggml-org/llama.cpp/pull/21343
auto tok = vocab.text_to_token(word);
if (tok != LLAMA_TOKEN_NULL) {
@@ -2233,6 +2263,10 @@ void llama_vocab::impl::load(llama_model_loader & ml, const LLM_KV & kv) {
tokenizer_pre == "granite-embed-multi-311m") {
pre_type = LLAMA_VOCAB_PRE_TYPE_GEMMA4;
escape_whitespaces = true;
+ } else if (
+ tokenizer_pre == "mmbert") {
+ pre_type = LLAMA_VOCAB_PRE_TYPE_MMBERT;
+ escape_whitespaces = true;
} else if (
tokenizer_pre == "sarvam-moe") {
pre_type = LLAMA_VOCAB_PRE_TYPE_SARVAM_MOE;
@@ -3107,7 +3141,7 @@ void llama_vocab::impl::load(llama_model_loader & ml, const LLM_KV & kv) {
// set attributes by model/tokenizer/architecture name
if (false
- || _contains_any(tokenizer_pre, {"jina-v2-de", "jina-v2-es", "jina-v2-code"})
+ || _contains_any(tokenizer_pre, {"jina-v2-de", "jina-v2-es", "jina-v2-code", "mmbert"})
|| _contains_any(general_arch, {"nomic-bert-moe", "jina-bert-v3"})
) {
if (token_to_id.count("<mask>") == 0) {
diff --git a/src/llama-vocab.h b/src/llama-vocab.h
index 3fb061f0e..033fa6208 100644
--- a/src/llama-vocab.h
+++ b/src/llama-vocab.h
@@ -68,6 +68,7 @@ enum llama_vocab_pre_type {
LLAMA_VOCAB_PRE_TYPE_HY_V4 = 57,
LLAMA_VOCAB_PRE_TYPE_SPARK2_5 = 58,
LLAMA_VOCAB_PRE_TYPE_UFAKZEKA = 59,
+ LLAMA_VOCAB_PRE_TYPE_MMBERT = 60,
};
struct LLM_KV;
diff --git a/src/models/models.h b/src/models/models.h
index a800d3fc0..8f04933a6 100644
--- a/src/models/models.h
+++ b/src/models/models.h
@@ -351,6 +351,12 @@ struct llama_model_modern_bert : public llama_model_base {
struct graph : public llm_graph_context {
graph(const llama_model & model, const llm_graph_params & params);
+
+ ggml_tensor * build_decision_head(
+ const llama_model & model,
+ ggml_tensor * inp,
+ llm_graph_input_attn_no_cache * inp_attn,
+ ggml_tensor * inp_out_ids);
};
std::unique_ptr<llm_graph_context> build_arch_graph(const llm_graph_params & params) const override;
diff --git a/src/models/modern-bert.cpp b/src/models/modern-bert.cpp
index 158e3160c..45455acbb 100644
--- a/src/models/modern-bert.cpp
+++ b/src/models/modern-bert.cpp
@@ -1,5 +1,8 @@
#include "models.h"
+// question types of a decision model: choice, score, noul
+static const uint32_t N_DECISION_TYPES = 3;
+
void llama_model_modern_bert::load_arch_hparams(llama_model_loader & ml) {
const bool found_swa = ml.get_key(LLM_KV_ATTENTION_SLIDING_WINDOW, hparams.n_swa, false);
if (found_swa && hparams.n_swa > 0) {
@@ -25,6 +28,19 @@ void llama_model_modern_bert::load_arch_hparams(llama_model_loader & ml) {
hparams.pooling_type_cls = LLAMA_POOLING_TYPE_MEAN;
}
+ ml.get_key(LLM_KV_DECISION_BLOCK_COUNT, hparams.n_layer_decision, false);
+ if (hparams.n_layer_decision > 0) {
+ if (hparams.n_layer_decision >= hparams.n_layer()) {
+ throw std::runtime_error("invalid number of decision blocks");
+ }
+ // the head blocks always use full attention
+ for (uint32_t il = hparams.n_layer() - hparams.n_layer_decision; il < hparams.n_layer(); il++) {
+ hparams.is_swa_impl[il] = false;
+ }
+ // the output is one score per question type
+ hparams.n_embd_out_impl = N_DECISION_TYPES;
+ }
+
switch (hparams.n_layer()) {
case 12:
type = LLM_TYPE_47M; break; // granite-embedding-small
@@ -44,7 +60,9 @@ void llama_model_modern_bert::load_arch_tensors(llama_model_loader &) {
output_norm = create_tensor(tn(LLM_TENSOR_OUTPUT_NORM, "weight"), {n_embd}, 0);
- for(int i = 0; i < n_layer; ++i) {
+ const int n_layer_enc = n_layer - hparams.n_layer_decision;
+
+ for(int i = 0; i < n_layer_enc; ++i) {
auto& layer = layers[i];
if ( i != 0 ) {
@@ -68,6 +86,43 @@ void llama_model_modern_bert::load_arch_tensors(llama_model_loader &) {
cls = create_tensor(tn(LLM_TENSOR_CLS, "weight"), {n_embd, n_embd}, TENSOR_NOT_REQUIRED);
cls_norm = create_tensor(tn(LLM_TENSOR_CLS_NORM, "weight"), {n_embd}, TENSOR_NOT_REQUIRED);
+ if (hparams.n_layer_decision == 0) {
+ return;
+ }
+
+ // decision head: plain pre-norm blocks with biases
+ for (int i = n_layer_enc; i < n_layer; ++i) {
+ auto & layer = layers[i];
+ const int64_t n_ff_head = hparams.n_ff(i);
+
+ layer.attn_norm = create_tensor(tn(LLM_TENSOR_ATTN_NORM, "weight", i), {n_embd}, 0);
+ layer.attn_norm_b = create_tensor(tn(LLM_TENSOR_ATTN_NORM, "bias", i), {n_embd}, 0);
+
+ layer.wqkv = create_tensor(tn(LLM_TENSOR_ATTN_QKV, "weight", i), {n_embd, 3 * n_embd}, 0);
+ layer.wqkv_b = create_tensor(tn(LLM_TENSOR_ATTN_QKV, "bias", i), {3 * n_embd}, 0);
+ layer.wo = create_tensor(tn(LLM_TENSOR_ATTN_OUT, "weight", i), {n_embd, n_embd}, 0);
+ layer.wo_b = create_tensor(tn(LLM_TENSOR_ATTN_OUT, "bias", i), {n_embd}, 0);
+
+ layer.ffn_norm = create_tensor(tn(LLM_TENSOR_FFN_NORM, "weight", i), {n_embd}, 0);
+ layer.ffn_norm_b = create_tensor(tn(LLM_TENSOR_FFN_NORM, "bias", i), {n_embd}, 0);
+ layer.ffn_up = create_tensor(tn(LLM_TENSOR_FFN_UP, "weight", i), {n_embd, n_ff_head}, 0);
+ layer.ffn_up_b = create_tensor(tn(LLM_TENSOR_FFN_UP, "bias", i), {n_ff_head}, 0);
+ layer.ffn_down = create_tensor(tn(LLM_TENSOR_FFN_DOWN, "weight", i), {n_ff_head, n_embd}, 0);
+ layer.ffn_down_b = create_tensor(tn(LLM_TENSOR_FFN_DOWN, "bias", i), {n_embd}, 0);
+ }
+
+ if (n_token_types != N_DECISION_TYPES) {
+ throw std::runtime_error("decision model must have one token type per question type");
+ }
+ type_embd = create_tensor(tn(LLM_TENSOR_TOKEN_TYPES, "weight"), {n_embd, n_token_types}, 0);
+
+ cls_b = create_tensor(tn(LLM_TENSOR_CLS, "bias"), {n_embd}, 0);
+ cls_norm_b = create_tensor(tn(LLM_TENSOR_CLS_NORM, "bias"), {n_embd}, 0);
+
+ if (!cls || !cls_norm || !cls_out || !cls_out_b) {
+ throw std::runtime_error("decision model is missing the scorer tensors");
+ }
+
}
std::unique_ptr<llm_graph_context> llama_model_modern_bert::build_arch_graph(const llm_graph_params & params) const {
@@ -95,7 +150,10 @@ llama_model_modern_bert::graph::graph(const llama_model & model, const llm_graph
auto * inp_attn = build_attn_inp_no_cache();
- for (int il = 0; il < n_layer; ++il) {
+ const int n_layer_dec = hparams.n_layer_decision;
+ const int n_layer_enc = n_layer - n_layer_dec;
+
+ for (int il = 0; il < n_layer_enc; ++il) {
const float freq_base_l = model.get_rope_freq_base(cparams, il);
const float freq_scale_l = model.get_rope_freq_scale(cparams, il);
@@ -172,6 +230,75 @@ llama_model_modern_bert::graph::graph(const llama_model & model, const llm_graph
LLM_NORM, -1);
cb(cur, "final_norm_out", -1);
+ if (n_layer_dec > 0) {
+ cur = build_decision_head(model, cur, inp_attn, inp_out_ids);
+ }
+
res->t_embd = cur;
ggml_build_forward_expand(gf, cur);
}
+
+// returns one score per question type for each output token, the caller reads them at the option markers
+ggml_tensor * llama_model_modern_bert::graph::build_decision_head(
+ const llama_model & model,
+ ggml_tensor * inp,
+ llm_graph_input_attn_no_cache * inp_attn,
+ ggml_tensor * inp_out_ids) {
+ const int64_t n_embd_head = hparams.n_embd_head_v();
+ const int n_layer_enc = n_layer - hparams.n_layer_decision;
+
+ ggml_tensor * scores = nullptr;
+
+ // the question type is not a graph input, so the head is evaluated for each of them
+ for (uint32_t it = 0; it < N_DECISION_TYPES; ++it) {
+ ggml_tensor * type_row = ggml_view_1d(ctx0, model.type_embd, n_embd, it * model.type_embd->nb[1]);
+ ggml_tensor * inpL = ggml_add(ctx0, inp, type_row);
+
+ for (int il = n_layer_enc; il < n_layer; ++il) {
+ const auto & layer = model.layers[il];
+
+ ggml_tensor * cur = build_norm(inpL, layer.attn_norm, layer.attn_norm_b, LLM_NORM, il);
+ cb(cur, "attn_norm", il);
+
+ // no positional encoding in the head
+ auto [Qcur, Kcur, Vcur] = build_qkv(layer, cur, n_embd_head, n_head, n_head_kv, il);
+
+ cur = build_attn(inp_attn,
+ layer.wo, layer.wo_b, layer.wo_s,
+ Qcur, Kcur, Vcur, nullptr, nullptr, nullptr, 1.0f/sqrtf(float(n_embd_head)), il);
+ cb(cur, "kqv_out", il);
+
+ if (il == n_layer - 1 && inp_out_ids) {
+ cur = ggml_get_rows(ctx0, cur, inp_out_ids);
+ inpL = ggml_get_rows(ctx0, inpL, inp_out_ids);
+ }
+
+ ggml_tensor * ffn_inp = ggml_add(ctx0, cur, inpL);
+ cb(ffn_inp, "ffn_inp", il);
+
+ cur = build_norm(ffn_inp, layer.ffn_norm, layer.ffn_norm_b, LLM_NORM, il);
+ cb(cur, "ffn_norm", il);
+
+ cur = build_ffn(cur,
+ layer.ffn_up, layer.ffn_up_b, NULL,
+ NULL, NULL, NULL,
+ layer.ffn_down, layer.ffn_down_b, NULL,
+ NULL,
+ LLM_FFN_RELU,
+ LLM_FFN_SEQ, il);
+
+ inpL = ggml_add(ctx0, cur, ffn_inp);
+ }
+
+ // scorer
+ ggml_tensor * cur = build_norm(inpL, model.cls_norm, model.cls_norm_b, LLM_NORM, -1);
+ cur = ggml_add(ctx0, build_lora_mm(model.cls, cur), model.cls_b);
+ cur = ggml_gelu_erf(ctx0, cur);
+ cur = ggml_add(ctx0, build_lora_mm(model.cls_out, cur), model.cls_out_b);
+
+ scores = scores ? ggml_concat(ctx0, scores, cur, 0) : cur;
+ }
+ cb(scores, "decision_scores", -1);
+
+ return scores;
+}
diff --git a/src/models/qwen35.cpp b/src/models/qwen35.cpp
index a1e263500..d50f067a5 100644
--- a/src/models/qwen35.cpp
+++ b/src/models/qwen35.cpp
@@ -43,6 +43,10 @@ void llama_model_qwen35::load_arch_tensors(llama_model_loader & ml) {
output_norm = create_tensor(tn(LLM_TENSOR_OUTPUT_NORM, "weight"), { n_embd }, 0);
output = create_tensor(tn(LLM_TENSOR_OUTPUT, "weight"), { n_embd, n_vocab }, TENSOR_NOT_REQUIRED);
+ // optional projection of the embeddings output
+ cls_out = create_tensor(tn(LLM_TENSOR_CLS_OUT, "weight"), { n_embd, hparams.n_embd_out() }, TENSOR_NOT_REQUIRED);
+ cls_out_b = create_tensor(tn(LLM_TENSOR_CLS_OUT, "bias"), { hparams.n_embd_out() }, TENSOR_NOT_REQUIRED);
+
// if output is NULL, init from the input tok embed
if (output == NULL) {
output = create_tensor(tn(LLM_TENSOR_TOKEN_EMBD, "weight"), { n_embd, n_vocab }, TENSOR_DUPLICATED);
@@ -215,6 +219,16 @@ llama_model_qwen35::graph::graph(const llama_model & model, const llm_graph_para
cb(cur, "result_norm", -1);
res->t_embd = cur;
+ if (model.cls_out) {
+ ggml_tensor * embd = build_lora_mm(model.cls_out, cur);
+ if (model.cls_out_b) {
+ embd = ggml_add(ctx0, embd, model.cls_out_b);
+ }
+ cb(embd, "result_embd_proj", -1);
+ res->t_embd = embd;
+ ggml_build_forward_expand(gf, embd);
+ }
+
// LM head
cur = build_lora_mm(model.output, cur, model.output_s);
diff --git a/tools/server/CMakeLists.txt b/tools/server/CMakeLists.txt
index 4adaaceef..70b8d7da8 100644
--- a/tools/server/CMakeLists.txt
+++ b/tools/server/CMakeLists.txt
@@ -13,6 +13,8 @@ add_library(${TARGET} STATIC
server-queue.h
server-common.cpp
server-common.h
+ server-decision.cpp
+ server-decision.h
server-context.cpp
server-context.h
server-stream.cpp
diff --git a/tools/server/README.md b/tools/server/README.md
index 09fdb87e5..47953cf11 100644
--- a/tools/server/README.md
+++ b/tools/server/README.md
@@ -1665,6 +1665,140 @@ curl http://localhost:8080/v1/messages/count_tokens \
{"input_tokens": 10}
```
+## TypeSafe-compatible API Endpoints
+
+### POST `/v1/systemone`: TypeSafe-compatible System One API
+
+Answers typed questions about a `state` with a decision model.
+
+Follows the [TypeSafe API](https://docs.typesafe.ai/api), streaming is not supported. Multimodal input is an extension to this API, see the [OpenJev multimodal API](https://jev-skills.github.io/openjev-multimodal/api) for reference.
+
+*Options:*
+
+`state`: The content to evaluate. Can be a string, an object or an array. A value that is not a string is given to the model as JSON text.
+
+`images`: Optional. An array of images, the maximum number may be limited depending on the model. Each one is a data URL (`data:image/...;base64,...`). See the image input section below.
+
+`questions`: An object that maps a question id to a question. Each question has these fields:
+
+- `type`: One of `choice`, `score`, `noul`.
+- `instructions`: The question. Can be a string, an object or an array.
+- `criteria`: The possible answers, the shape depends on `type`:
+ - `choice`: An object that maps each option to its description. The description can be `null`.
+ - `score`: An array of 2 to 10 level descriptions, lowest level first.
+ - `noul`: Optional. An object with the descriptions of `true` and `false`.
+
+The questions of a request are answered independently, an answer does not depend on the other questions.
+
+The number of options of a `choice` question is limited by the model, for example: 52 for openjev, 255 for laya. For laya, long questions and options are truncated to the token budget the model was trained with.
+
+*Image input:*
+
+Image input needs a model that supports it (for example: openjev) and its multimodal projector, see `--mmproj`.
+
+Images can be given in two ways, and both can be used in the same request:
+
+- The `images` field.
+- A `state` made of chat messages, either an array of messages or an object with a `messages` array. An `image_url` part in the `content` of a message is taken as an image, in the same format as chat completions. Only data URLs are accepted.
+
+All the images are placed before the state in the prompt, the ones from `images` first. The image parts are removed from the state.
+
+*Response:*
+
+`answers`: An object that maps each question id to its answer. The fields depend on the question type:
+
+- `choice`:
+ - `choice`: The option with the highest probability.
+ - `probabilities`: The probability of each option, they sum to 1.
+ - `confidence`: A value from 0 to 1, where 0 means all options are equally likely.
+- `score`:
+ - `score`: The expected level index, weighted by probability. It can be between two levels.
+ - `legend`: The description of each level index.
+ - `probabilities`: The probability of each level index, they sum to 1.
+ - `confidence`: A value from 0 to 1.
+- `noul`:
+ - `noul`: The probability that the answer is true.
+
+`usage`: `input_tokens` is the number of prompt tokens of all questions. `output_tokens` is always 0.
+
+The probabilities are scaled with the temperatures stored in the model file. They are not guaranteed to be calibrated for your data.
+
+*Examples:*
+
+```shell
+curl http://127.0.0.1:8080/v1/systemone \
+ -H "Content-Type: application/json" \
+ -d '{
+ "state": "Customer message: I was charged twice for my order last week and nobody has replied.",
+ "questions": {
+ "route": {
+ "type": "choice",
+ "instructions": "Which team should handle this?",
+ "criteria": {"billing": null, "shipping": null, "technical": null}
+ },
+ "angry": {
+ "type": "noul",
+ "instructions": "Is the customer angry?"
+ },
+ "urgency": {
+ "type": "score",
+ "instructions": "How urgent is this?",
+ "criteria": ["can wait", "this week", "today", "right now"]
+ }
+ }
+ }' | jq
+```
+
+Response (values are shortened):
+
+```json
+{
+ "model": "openjev",
+ "answers": {
+ "route": {
+ "type": "choice",
+ "choice": "billing",
+ "probabilities": {"billing": 0.9998, "shipping": 0.0001, "technical": 0.0001},
+ "confidence": 0.9997
+ },
+ "angry": {
+ "type": "noul",
+ "noul": 0.6328
+ },
+ "urgency": {
+ "type": "score",
+ "score": 2.0858,
+ "legend": {"0": "can wait", "1": "this week", "2": "today", "3": "right now"},
+ "probabilities": {"0": 0.0023, "1": 0.116, "2": 0.6753, "3": 0.2064},
+ "confidence": 0.673
+ }
+ },
+ "usage": {
+ "input_tokens": 239,
+ "output_tokens": 0
+ }
+}
+```
+
+Example with an image:
+
+```shell
+curl http://127.0.0.1:8080/v1/systemone \
+ -H "Content-Type: application/json" \
+ -d '{
+ "state": "The document was received by the accounting team this morning.",
+ "images": ["data:image/jpeg;base64,/9j/4AAQSkZJRg..."],
+ "questions": {
+ "has_table": {
+ "type": "noul",
+ "instructions": "Does the image contain a table?"
+ }
+ }
+ }' | jq
+```
+
+An invalid request returns the error `400`. A model that is not a decision model returns the error `501`. A request with images returns the error `501` if the model does not support image input, or if no multimodal projector is loaded.
+
## Server tools
The server exposes a REST API under `/tools` that allows the Web UI to call server tools. This endpoint is intended to be used internally by the Web UI and subject to change or to be removed in the future.
diff --git a/tools/server/server-common.cpp b/tools/server/server-common.cpp
index a37826abe..2d90fcad2 100644
--- a/tools/server/server-common.cpp
+++ b/tools/server/server-common.cpp
@@ -1076,7 +1076,7 @@ json oaicompat_completion_params_parse(const json & body) {
// - file:// for local files (only allowed if media_path is set)
// - data: for base64 encoded data with uri scheme (e.g. data:image/png;base64,...)
// - raw base64 encoded data
-static void handle_media(
+void handle_media(
std::vector<raw_buffer> & out_files,
const std::string & url,
const std::string & media_path) {
diff --git a/tools/server/server-common.h b/tools/server/server-common.h
index 8cb6b90da..7e4ebed2a 100644
--- a/tools/server/server-common.h
+++ b/tools/server/server-common.h
@@ -270,6 +270,12 @@ llama_tokens tokenize_mixed(const llama_vocab * vocab, const json & json_prompt,
// if validate_utf8(text) == text.size(), then the whole text is valid utf8
size_t validate_utf8(const std::string& text);
+// load a media file from an URL (http, file, data) or from raw base64 data
+void handle_media(
+ std::vector<raw_buffer> & out_files,
+ const std::string & url,
+ const std::string & media_path);
+
// process mtmd prompt, return the server_tokens containing both text tokens and media chunks
// if is_placeholder is true, the media chunk will be treated as placeholder for counting tokens; the output tokens are not usable for actual inference (e.g. for submitting a task to server_queue)
server_tokens process_mtmd_prompt(
diff --git a/tools/server/server-context.cpp b/tools/server/server-context.cpp
index 470fbd977..edb8e2d86 100644
--- a/tools/server/server-context.cpp
+++ b/tools/server/server-context.cpp
@@ -1,6 +1,7 @@
#include "server-context.h"
#include "server-chat.h"
#include "server-common.h"
+#include "server-decision.h"
#include "server-http.h"
#include "server-task.h"
#include "server-queue.h"
@@ -413,6 +414,10 @@ struct server_slot {
if (pooling == LLAMA_POOLING_TYPE_RANK && llama_get_causal_attn(ctx_tgt)) {
return true;
}
+ // a decision task reads its outputs from the last batch
+ if (task->type == SERVER_TASK_TYPE_DECISION) {
+ return true;
+ }
return false;
}
@@ -689,6 +694,14 @@ struct server_slot {
return res;
}
+ // the other slot continues from the tokens processed so far
+ void copy_prompt_to(server_slot & other) const {
+ mem.seq_rm(other.id, -1, -1);
+ mem.seq_cp(id, other.id, -1, -1);
+
+ other.prompt = prompt.clone();
+ }
+
void copy_state_to(server_slot & other) const {
GGML_ASSERT(state == SLOT_STATE_DONE_PROMPT);
@@ -825,6 +838,8 @@ public:
mtmd_helper_init_opt init_opt = mtmd_helper_init_opt_default();
const llama_vocab * vocab = nullptr;
+ server_decision_context decision;
+
server_queue queue_tasks;
server_response queue_results;
@@ -1102,6 +1117,13 @@ private:
vocab = llama_model_get_vocab(model_tgt);
+ try {
+ decision.init(model_tgt);
+ } catch (const std::exception & e) {
+ SRV_ERR("failed to init decision model: %s\n", e.what());
+ return false;
+ }
+
n_ctx = llama_n_ctx(ctx_tgt);
add_bos_token = llama_vocab_get_add_bos(vocab);
@@ -2194,6 +2216,67 @@ private:
queue_results.send(std::move(res));
}
+ void send_decision(const server_slot & slot, const common_batch & batch, int32_t i_batch) {
+ auto res = std::make_unique<server_task_result_decision>();
+ res->id = slot.task->id;
+ res->index = slot.task->index;
+ res->n_tokens = slot.task->n_tokens();
+
+ const auto & decision = slot.task->decision;
+
+ if (!decision.labels.empty()) {
+ const float * logits = llama_get_logits_ith(slot.ctx_tgt, i_batch);
+ if (logits == nullptr) {
+ send_error(slot, "failed to get logits", ERROR_TYPE_SERVER);
+ return;
+ }
+ const int32_t n_vocab = llama_vocab_n_tokens(vocab);
+ for (const llama_token label : decision.labels) {
+ GGML_ASSERT(label >= 0 && label < n_vocab);
+ res->scores.push_back(logits[label]);
+ }
+ } else {
+ // the outputs of this slot in this batch are the last tokens of the prompt
+ std::vector<int32_t> idx;
+ for (int i = 0; i < batch.size(); ++i) {
+ if (batch.tokens[i].output && batch.tokens[i].seq_id == slot.id) {
+ idx.push_back(i);
+ }
+ }
+ const int32_t pos_first = slot.prompt.n_tokens() - (int32_t) idx.size();
+ auto get_embd = [&](int32_t pos) -> const float * {
+ const int32_t i = pos - pos_first;
+ return i >= 0 && i < (int32_t) idx.size() ? llama_get_embeddings_ith(slot.ctx_tgt, idx[i]) : nullptr;
+ };
+
+ const int32_t n_embd_out = llama_model_n_embd_out(model_tgt);
+ const int32_t n_pointer = n_embd_out / 2;
+ const float * embd_q = decision.pointer >= 0 ? get_embd(decision.pointer) : nullptr;
+ GGML_ASSERT(decision.column >= 0 && decision.column < n_embd_out);
+
+ for (const int32_t marker : decision.markers) {
+ const float * embd = get_embd(marker);
+ if (embd == nullptr || (decision.pointer >= 0 && embd_q == nullptr)) {
+ send_error(slot, "failed to get embeddings, the question and its options must fit in one batch", ERROR_TYPE_SERVER);
+ return;
+ }
+ if (decision.pointer < 0) {
+ res->scores.push_back(embd[decision.column]);
+ continue;
+ }
+ float dot = 0.0f;
+ for (int32_t i = 0; i < n_pointer; i++) {
+ dot += embd_q[i] * embd[n_pointer + i];
+ }
+ res->scores.push_back(dot / sqrtf((float) n_pointer));
+ }
+ }
+
+ SLT_DBG(slot, "%s", "sending decision result\n");
+
+ queue_results.send(std::move(res));
+ }
+
void send_rerank(const server_slot & slot, const common_batch & batch) {
auto res = std::make_unique<server_task_result_rerank>();
res->id = slot.task->id;
@@ -2384,6 +2467,7 @@ private:
case SERVER_TASK_TYPE_INFILL:
case SERVER_TASK_TYPE_EMBEDDING:
case SERVER_TASK_TYPE_RERANK:
+ case SERVER_TASK_TYPE_DECISION:
{
// special case: if input is provided via CLI, tokenize it first
// otherwise, no need to tokenize as it's already done inside the HTTP thread
@@ -3212,12 +3296,29 @@ private:
return;
}
+ // the outputs of a decision are read from one batch
+ const int32_t n_decision_first = slot.task->type == SERVER_TASK_TYPE_DECISION ? slot.task->decision.pos_first() : -1;
+ if (n_decision_first >= 0 && slot.task->n_tokens() - n_decision_first > n_batch) {
+ send_error(slot,
+ string_format("the question and its options (%d tokens) are too large to process. "
+ "increase the batch size (current batch size: %d)",
+ slot.task->n_tokens() - n_decision_first, n_batch),
+ ERROR_TYPE_INVALID_REQUEST);
+ slot.release();
+ return;
+ }
+
const bool is_stateless_task = slot.task->type == SERVER_TASK_TYPE_EMBEDDING || slot.task->type == SERVER_TASK_TYPE_RERANK;
if (slot.task->params.cache_prompt && !is_stateless_task) {
// reuse any previously computed tokens that are common with the new prompt
n_past = slot.prompt.tokens.get_common_prefix(input_tokens);
+ // the children start from the shared prefix, do not go past it
+ if (slot.task->n_tokens_shared > 0) {
+ n_past = std::min(n_past, slot.task->n_tokens_shared);
+ }
+
// if there is an alora invoked, don't cache after the invocation start
if (slot.alora_invocation_start > 0) {
SLT_DBG(slot, "only caching to alora invocation start (n_past = %d, alora_invocation_start = %d)\n", n_past, slot.alora_invocation_start);
@@ -3443,6 +3544,24 @@ private:
slot.mem.seq_rm(slot.id, p0, -1);
+ // shared prompt prefix: once it is processed, the children continue from it with their own prompt
+ bool wait_shared = false;
+ if (slot.task->n_tokens_shared > 0) {
+ const bool is_shared_done = slot.prompt.n_tokens() == slot.task->n_tokens_shared;
+ for (auto & other : slots) {
+ if (other.state != SLOT_STATE_WAIT_OTHER || other.task->id_parent != slot.task->id) {
+ continue;
+ }
+ if (is_shared_done) {
+ SLT_TRC(slot, " - copying shared prompt (%d tokens) to child %d\n", slot.prompt.n_tokens(), other.id);
+ slot.copy_prompt_to(other);
+ other.state = SLOT_STATE_STARTED;
+ } else {
+ wait_shared = true;
+ }
+ }
+ }
+
// If using an alora, there may be uncached tokens that come
// before the invocation sequence. When this happens, the
// tokens before the invocation sequence need to be
@@ -3520,6 +3639,8 @@ private:
const auto & spans = slot.task->params.message_spans;
const auto last_user_pos = spans.last_user_message_pos();
+ const int32_t n_decision_first = slot.task->type == SERVER_TASK_TYPE_DECISION ? slot.task->decision.pos_first() : -1;
+
// add prompt tokens for processing in the current batch
while (slot.prompt.n_tokens() < slot.task->n_tokens() && batch.size() < n_batch) {
// get next token to process
@@ -3528,6 +3649,16 @@ private:
break; // end of text chunk
}
+ // stop at the end of the shared prefix, the children are started from this state
+ if (wait_shared && slot.prompt.n_tokens() == slot.task->n_tokens_shared) {
+ break;
+ }
+
+ // the outputs of a decision are read from one batch, do not split them
+ if (slot.prompt.n_tokens() == n_decision_first && batch.size() + slot.task->n_tokens() - n_decision_first > n_batch) {
+ break;
+ }
+
// if this is an alora request with pre-invocation
// tokens that are not cached, we need to stop filling
// this batch at those pre-invocation tokens.
@@ -3771,6 +3902,8 @@ private:
SLT_TRC(slot, " - copying state to child %d\n", child->id);
GGML_ASSERT(child->state == SLOT_STATE_WAIT_OTHER);
+ // children with their own prompt are started at the end of the shared prefix
+ GGML_ASSERT(slot.task->n_tokens_shared == 0);
slot.copy_state_to(*child);
child->state = SLOT_STATE_DONE_PROMPT;
@@ -3831,6 +3964,13 @@ private:
return;
}
+ if (slot.task->type == SERVER_TASK_TYPE_DECISION) {
+ send_decision(slot, batch.view, slot.i_batch - off);
+ slot.release();
+ slot.i_batch = -1;
+ return;
+ }
+
GGML_ASSERT(slot.task->need_sampling());
// prompt evaluated for next-token prediction
@@ -5224,6 +5364,76 @@ void server_routes::init_routes() {
return res;
};
+ this->post_systemone = [this](const server_http_req & req) {
+ auto res = create_response();
+ const auto & decision = ctx_server.decision;
+ if (decision.type == COMMON_DECISION_TYPE_NONE) {
+ res->error(format_error_response("This model is not a decision model", ERROR_TYPE_NOT_SUPPORTED));
+ return res;
+ }
+
+ const json body = json::parse(req.body);
+ const auto questions = decision.parse_questions(body);
+
+ std::vector<raw_buffer> files;
+ const json state = decision.parse_state(body, files);
+ if (!files.empty() && (!decision.can_use_images() || !meta->has_inp_image)) {
+ res->error(format_error_response("This server does not support image input for decisions. For a model that supports it, start it with `--mmproj`", ERROR_TYPE_NOT_SUPPORTED));
+ return res;
+ }
+
+ // one task per variant of each question
+ auto & rd = res->rd;
+ {
+ std::vector<server_task> tasks;
+ for (const auto & question : questions) {
+ for (size_t variant = 0; variant < decision.n_variants(question); variant++) {
+ server_task task = server_task(SERVER_TASK_TYPE_DECISION);
+ task.id = rd.get_new_id();
+ decision.fill_task(state, question, variant, files, ctx_server.mctx, ctx_server.init_opt, task);
+ tasks.push_back(std::move(task));
+ }
+ }
+ if (decision.can_share_prompt()) {
+ tasks = server_decision_group_tasks(std::move(tasks), params.n_parallel);
+ }
+ rd.post_tasks(std::move(tasks));
+ }
+
+ auto all_results = rd.wait_for_all(req.should_stop);
+
+ if (all_results.is_terminated) {
+ return res; // connection is closed
+ } else if (all_results.error) {
+ res->error(all_results.error->to_json());
+ return res;
+ }
+
+ json answers = json::object();
+ int32_t n_tokens = 0;
+ size_t i_result = 0;
+ for (const auto & question : questions) {
+ std::vector<std::vector<float>> scores;
+ for (size_t variant = 0; variant < decision.n_variants(question); variant++) {
+ auto * result = dynamic_cast<server_task_result_decision *>(all_results.results[i_result++].get());
+ GGML_ASSERT(result != nullptr);
+ scores.push_back(result->scores);
+ n_tokens += result->n_tokens;
+ }
+ answers[question.id] = decision.format_answer(question, scores);
+ }
+
+ res->ok(json{
+ {"model", meta->model_name},
+ {"answers", answers},
+ {"usage", {
+ {"input_tokens", n_tokens},
+ {"output_tokens", 0},
+ }},
+ });
+ return res;
+ };
+
this->get_lora_adapters = [this](const server_http_req & req) {
auto res = create_response();
diff --git a/tools/server/server-context.h b/tools/server/server-context.h
index 7265ccad1..c554bb95b 100644
--- a/tools/server/server-context.h
+++ b/tools/server/server-context.h
@@ -152,6 +152,7 @@ struct server_routes {
server_http_context::handler_t post_embeddings;
server_http_context::handler_t post_embeddings_oai;
server_http_context::handler_t post_rerank;
+ server_http_context::handler_t post_systemone;
server_http_context::handler_t get_lora_adapters;
server_http_context::handler_t post_lora_adapters;
diff --git a/tools/server/server-decision.cpp b/tools/server/server-decision.cpp
new file mode 100644
index 000000000..fa654c978
--- /dev/null
+++ b/tools/server/server-decision.cpp
@@ -0,0 +1,681 @@
+#include "server-decision.h"
+
+#include <algorithm>
+#include <cmath>
+#include <regex>
+#include <stdexcept>
+
+static const char * decision_question_type_name(server_decision_question_type type) {
+ switch (type) {
+ case SERVER_DECISION_QUESTION_CHOICE: return "choice";
+ case SERVER_DECISION_QUESTION_SCORE: return "score";
+ case SERVER_DECISION_QUESTION_NOUL: return "noul";
+ }
+ return "";
+}
+
+// lev reads noul from a rating scale: 0 = certainly no, 8 = certainly yes
+static const size_t DECISION_LEV_N_RATINGS = 9;
+
+static std::string decision_meta_str(const llama_model * model, const std::string & key) {
+ char buf[256];
+ const int32_t n = llama_model_meta_val_str(model, key.c_str(), buf, sizeof(buf));
+ return n < 0 ? "" : std::string(buf);
+}
+
+//
+// model-specific setup
+//
+
+void server_decision_context::init(const llama_model * model) {
+ *this = server_decision_context(); // the model can be reloaded
+
+ const common_decision_type model_type = common_get_decision_type(model);
+ if (model_type == COMMON_DECISION_TYPE_NONE) {
+ return;
+ }
+
+ const std::string prefix = decision_meta_str(model, "general.architecture") + ".decision.";
+ const std::string type_name = decision_meta_str(model, prefix + "type");
+
+ vocab = llama_model_get_vocab(model);
+
+ const char * tmpl_src = llama_model_chat_template(model, "systemone");
+ if (tmpl_src == nullptr) {
+ throw std::runtime_error("decision model has no \"systemone\" template");
+ }
+ tmpl = std::make_shared<const common_chat_template>(tmpl_src, "", "");
+
+ const std::string prefix_temp = prefix + "temperature.";
+ for (int32_t i = 0; i < llama_model_meta_count(model); i++) {
+ char key[256];
+ char val[64];
+ if (llama_model_meta_key_by_index(model, i, key, sizeof(key)) < 0 || !string_starts_with(key, prefix_temp)) {
+ continue;
+ }
+ if (llama_model_meta_val_str_by_index(model, i, val, sizeof(val)) < 0) {
+ continue;
+ }
+ const float temp = std::strtof(val, nullptr);
+ if (temp <= 0.0f) {
+ throw std::runtime_error(string_format("invalid decision temperature: %s = %s", key, val));
+ }
+ temperatures[key + prefix_temp.size()] = temp;
+ }
+
+ if (model_type == COMMON_DECISION_TYPE_OPENJEV) {
+ // one letter per option, each must be a single token
+ const std::string letters = "ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz";
+ for (const char c : letters) {
+ const auto toks = common_tokenize(vocab, std::string(1, c), false, false);
+ if (toks.size() != 1) {
+ throw std::runtime_error(string_format("decision label '%c' is not a single token", c));
+ }
+ labels.push_back(toks[0]);
+ }
+ n_options_max = labels.size();
+ noul_true_first = true;
+ } else if (model_type == COMMON_DECISION_TYPE_LEV) {
+ // label codes are A..Z then AA..ZZ, only the ones that are a single token are used
+ std::vector<std::string> codes;
+ for (char a = 'A'; a <= 'Z'; a++) {
+ codes.push_back(std::string(1, a));
+ }
+ for (char a = 'A'; a <= 'Z'; a++) {
+ for (char b = 'A'; b <= 'Z'; b++) {
+ codes.push_back(std::string{a, b});
+ }
+ }
+ for (const auto & code : codes) {
+ const auto toks = common_tokenize(vocab, code, false, false);
+ if (toks.size() == 1 && labels.size() < 255) {
+ labels.push_back(toks[0]);
+ label_texts.push_back(code);
+ }
+ }
+ n_options_max = labels.size();
+ } else if (model_type == COMMON_DECISION_TYPE_KEV) {
+ // the hidden state of an option is read at the token that ends it
+ const auto toks = common_tokenize(vocab, "<|box_end|>", false, true);
+ if (toks.size() != 1) {
+ throw std::runtime_error("decision model has no <|box_end|> token");
+ }
+ token_marker = toks[0];
+ n_options_max = 255;
+ } else if (model_type == COMMON_DECISION_TYPE_LAYA) {
+ token_marker = llama_vocab_mask(vocab);
+ token_sep = llama_vocab_sep(vocab);
+ if (token_marker == LLAMA_TOKEN_NULL || token_sep == LLAMA_TOKEN_NULL) {
+ throw std::runtime_error("decision model has no mask or sep token");
+ }
+ text_marker = common_token_to_piece(vocab, token_marker, true);
+
+ const std::string val = decision_meta_str(model, prefix + "max_head_tokens");
+ max_head_tokens = std::strtoul(val.c_str(), nullptr, 10);
+ if (max_head_tokens == 0) {
+ throw std::runtime_error("decision model has no valid max_head_tokens");
+ }
+ n_options_max = 255;
+ } else {
+ throw std::runtime_error("unsupported decision model type: " + type_name);
+ }
+ type = model_type;
+
+ SRV_INF("decision model type: %s\n", type_name.c_str());
+}
+
+//
+// request parsing
+//
+
+std::vector<server_decision_question> server_decision_context::parse_questions(const json & body) const {
+ if (!body.contains("state") || body.at("state").is_null()) {
+ throw std::invalid_argument("\"state\" must be provided");
+ }
+ if (!body.contains("questions") || !body.at("questions").is_object() || body.at("questions").empty()) {
+ throw std::invalid_argument("\"questions\" must be a non-empty object");
+ }
+
+ std::vector<server_decision_question> questions;
+ for (const auto & [id, q] : body.at("questions").items()) {
+ auto err = [&id = id](const std::string & msg) {
+ return std::invalid_argument("questions." + id + ": " + msg);
+ };
+ if (!q.is_object()) {
+ throw err("must be an object");
+ }
+ if (!q.contains("instructions") || q.at("instructions").is_null()) {
+ throw err("\"instructions\" must be provided");
+ }
+
+ server_decision_question question;
+ question.id = id;
+ question.instructions = q.at("instructions");
+
+ const std::string type_name = json_value(q, "type", std::string());
+ const json criteria = q.contains("criteria") ? q.at("criteria") : json();
+
+ if (type_name == "choice") {
+ question.type = SERVER_DECISION_QUESTION_CHOICE;
+ if (!criteria.is_object() || criteria.empty()) {
+ throw err("\"criteria\" must be a non-empty object");
+ }
+ for (const auto & [key, description] : criteria.items()) {
+ question.options.push_back({key, description});
+ }
+ } else if (type_name == "score") {
+ question.type = SERVER_DECISION_QUESTION_SCORE;
+ if (!criteria.is_array() || criteria.size() < 2 || criteria.size() > 10) {
+ throw err("\"criteria\" must be an array of 2 to 10 levels");
+ }
+ for (size_t i = 0; i < criteria.size(); i++) {
+ question.options.push_back({std::to_string(i), criteria.at(i)});
+ }
+ } else if (type_name == "noul") {
+ question.type = SERVER_DECISION_QUESTION_NOUL;
+ if (!criteria.is_null() && !criteria.is_object()) {
+ throw err("\"criteria\" must be an object");
+ }
+ for (const char * key : {"false", "true"}) {
+ question.options.push_back({key, criteria.is_object() && criteria.contains(key) ? criteria.at(key) : json()});
+ }
+ if (noul_true_first) {
+ std::swap(question.options[0], question.options[1]);
+ }
+ } else {
+ throw err("\"type\" must be one of: choice, score, noul");
+ }
+
+ if (question.options.size() > n_options_max) {
+ throw err(string_format("too many options (%zu), this model supports at most %zu", question.options.size(), n_options_max));
+ }
+
+ questions.push_back(std::move(question));
+ }
+ return questions;
+}
+
+//
+// images
+//
+
+static const size_t DECISION_MAX_IMAGES = 8;
+
+static void decision_load_image(const json & url, std::vector<raw_buffer> & files) {
+ if (!url.is_string() || !string_starts_with(url.get<std::string>(), "data:image/")) {
+ throw std::invalid_argument("images must be data URLs (data:image/...;base64,...)");
+ }
+ if (files.size() >= DECISION_MAX_IMAGES) {
+ throw std::invalid_argument(string_format("too many images, the maximum is %zu", DECISION_MAX_IMAGES));
+ }
+ handle_media(files, url.get<std::string>(), "");
+}
+
+json server_decision_context::parse_state(const json & body, std::vector<raw_buffer> & files) const {
+ if (body.contains("images") && !body.at("images").is_null()) {
+ if (!body.at("images").is_array()) {
+ throw std::invalid_argument("\"images\" must be an array");
+ }
+ for (const auto & url : body.at("images")) {
+ decision_load_image(url, files);
+ }
+ }
+
+ const json & state = body.at("state");
+ const bool is_wrapped = state.is_object() && state.contains("messages");
+ const json & messages = is_wrapped ? state.at("messages") : state;
+ if (!messages.is_array()) {
+ return state;
+ }
+
+ // chat messages: take the image parts out of the content
+ json messages_out = json::array();
+ for (const auto & msg : messages) {
+ if (!msg.is_object() || !msg.contains("content") || !msg.at("content").is_array()) {
+ messages_out.push_back(msg);
+ continue;
+ }
+ json content = json::array();
+ for (const auto & part : msg.at("content")) {
+ if (part.is_object() && json_value(part, "type", std::string()) == "image_url" && part.contains("image_url")) {
+ const json & image_url = part.at("image_url");
+ decision_load_image(image_url.is_object() && image_url.contains("url") ? image_url.at("url") : image_url, files);
+ } else {
+ content.push_back(part);
+ }
+ }
+ json msg_out = msg;
+ msg_out["content"] = content;
+ messages_out.push_back(msg_out);
+ }
+
+ if (!is_wrapped) {
+ return messages_out;
+ }
+ json state_out = state;
+ state_out["messages"] = messages_out;
+ return state_out;
+}
+
+//
+// prompt
+//
+
+// replace text in all strings of a JSON value
+static json decision_replace_text(const json & val, const std::string & search, const std::string & replace) {
+ if (val.is_string()) {
+ std::string str = val.get<std::string>();
+ string_replace_all(str, search, replace);
+ return str;
+ }
+ if (val.is_array()) {
+ json out = json::array();
+ for (const auto & item : val) {
+ out.push_back(decision_replace_text(item, search, replace));
+ }
+ return out;
+ }
+ if (val.is_object()) {
+ json out = json::object();
+ for (const auto & [key, item] : val.items()) {
+ out[key] = decision_replace_text(item, search, replace);
+ }
+ return out;
+ }
+ return val;
+}
+
+// sort the keys of all objects of a JSON value
+static json decision_sort_keys(const json & val) {
+ if (val.is_array()) {
+ json out = json::array();
+ for (const auto & item : val) {
+ out.push_back(decision_sort_keys(item));
+ }
+ return out;
+ }
+ if (val.is_object()) {
+ std::map<std::string, json> sorted;
+ for (const auto & [key, item] : val.items()) {
+ sorted[key] = decision_sort_keys(item);
+ }
+ json out = json::object();
+ for (const auto & [key, item] : sorted) {
+ out[key] = item;
+ }
+ return out;
+ }
+ return val;
+}
+
+// kev flattens a JSON value into text, the keys of an object are kept as labels (kev/api.py: render)
+static std::string decision_kev_render(const json & val, int indent = 0) {
+ const std::string pad(2 * indent, ' ');
+ if (val.is_null()) {
+ return "";
+ }
+ if (val.is_string()) {
+ return val.get<std::string>();
+ }
+ if (val.is_boolean()) {
+ return val.get<bool>() ? "True" : "False";
+ }
+ if (val.is_array()) {
+ std::string out;
+ for (const auto & item : val) {
+ const std::string text = decision_kev_render(item, indent + 1);
+ out += (out.empty() ? "" : "\n") + pad + "- " + text.substr(std::min(text.size(), text.find_first_not_of(" \t\n\r")));
+ }
+ return out;
+ }
+ if (val.is_object()) {
+ std::string out;
+ for (const auto & [key, item] : val.items()) {
+ const bool is_nested = item.is_object() || item.is_array();
+ out += (out.empty() ? "" : "\n") + pad + key + (is_nested ? ":\n" : ": ") + decision_kev_render(item, is_nested ? indent + 1 : 0);
+ }
+ return out;
+ }
+ return val.dump();
+}
+
+// kev text input: special tokens written in the text must not be parsed as such
+static std::string decision_kev_text(const json & val) {
+ static const std::regex re_special("<\\|([A-Za-z0-9_]+)\\|>");
+ return std::regex_replace(decision_kev_render(val), re_special, "<\xC2\xA6$1\xC2\xA6>");
+}
+
+size_t server_decision_context::n_variants(const server_decision_question & question) const {
+ // lev shows the options of a choice in 2 orders, to cancel the preference for the first label
+ if (type == COMMON_DECISION_TYPE_LEV && question.type == SERVER_DECISION_QUESTION_CHOICE && question.options.size() > 1) {
+ return 2;
+ }
+ return 1;
+}
+
+size_t server_decision_context::n_outputs(const server_decision_question & question) const {
+ if (type == COMMON_DECISION_TYPE_LEV && question.type == SERVER_DECISION_QUESTION_NOUL) {
+ return DECISION_LEV_N_RATINGS;
+ }
+ return question.options.size();
+}
+
+std::string server_decision_context::render(const json & state, const server_decision_question & question, size_t variant, size_t n_images) const {
+ const size_t n_options = question.options.size();
+
+ // the second variant shows the options in the reverse order
+ json options = json::array();
+ for (size_t i = 0; i < n_options; i++) {
+ const auto & opt = question.options[variant == 0 ? i : n_options - 1 - i];
+ json option = json{
+ {"key", opt.key},
+ {"description", opt.description},
+ };
+ if (type == COMMON_DECISION_TYPE_KEV) {
+ option["key"] = decision_kev_text(opt.key);
+ if (!opt.description.is_null()) {
+ option["description"] = decision_kev_text(opt.description);
+ }
+ }
+ if (!label_texts.empty()) {
+ option["label"] = label_texts[i];
+ }
+ options.push_back(option);
+ }
+
+ // the template is given raw JSON values, it serializes the ones that are not strings
+ json inp = json{
+ {"id", question.id},
+ {"type", decision_question_type_name(question.type)},
+ {"instructions", question.instructions},
+ {"state", state},
+ {"options", options},
+ };
+
+ // lev was trained with sorted keys
+ if (type == COMMON_DECISION_TYPE_LEV) {
+ inp = decision_sort_keys(inp);
+ }
+
+ // the kev template only takes text
+ if (type == COMMON_DECISION_TYPE_KEV) {
+ inp["state"] = decision_kev_text(state);
+ inp["instructions"] = decision_kev_text(question.instructions);
+ }
+
+ // the input must not contain the marker of the options
+ if (!text_marker.empty()) {
+ inp = decision_replace_text(inp, text_marker, " ");
+ }
+
+ // the template puts one media marker per image
+ json images = json::array();
+ if (n_images > 0) {
+ inp = decision_replace_text(inp, get_media_marker(), " ");
+ for (size_t i = 0; i < n_images; i++) {
+ images.push_back(get_media_marker());
+ }
+ }
+ inp["images"] = images;
+
+ jinja::context ctx(tmpl->source());
+ jinja::global_from_json(ctx, inp, false);
+ jinja::runtime runtime(ctx);
+ const jinja::value results = runtime.execute(tmpl->prog);
+ return jinja::runtime::gather_string_parts(results)->as_string().str();
+}
+
+void server_decision_context::fill_task(
+ const json & state,
+ const server_decision_question & question,
+ size_t variant,
+ const std::vector<raw_buffer> & files,
+ mtmd_context * mctx,
+ const mtmd_helper_init_opt & init_opt,
+ server_task & task) const {
+ const std::string prompt = render(state, question, variant, files.size());
+
+ if (type == COMMON_DECISION_TYPE_OPENJEV || type == COMMON_DECISION_TYPE_LEV) {
+ // lev reads the ratings of a noul question at its first labels, not at the digits
+ task.decision.labels.assign(labels.begin(), labels.begin() + n_outputs(question));
+ if (!files.empty()) {
+ task.tokens = process_mtmd_prompt(mctx, prompt, files, init_opt);
+ return;
+ }
+ }
+
+ llama_tokens tokens = common_tokenize(vocab, prompt, false, true);
+ if (type == COMMON_DECISION_TYPE_LAYA) {
+ fill_task_laya(tokens, question, task);
+ }
+ if (type == COMMON_DECISION_TYPE_KEV) {
+ // an option is read at its end token, the question at the last token
+ for (size_t i = 0; i < tokens.size(); i++) {
+ if (tokens[i] == token_marker) {
+ task.decision.markers.push_back(i);
+ }
+ }
+ if (task.decision.markers.size() != question.options.size()) {
+ throw std::runtime_error("unexpected layout of the decision prompt");
+ }
+ task.decision.pointer = tokens.size() - 1;
+ }
+ task.tokens = server_tokens(tokens, false);
+}
+
+// the prompt is: [cls] question [sep] ([marker] option)* [sep] state [sep]
+// options and question are cut to fit max_head_tokens, the same way the model was trained
+void server_decision_context::fill_task_laya(llama_tokens & tokens, const server_decision_question & question, server_task & task) const {
+ const size_t n_options = question.options.size();
+
+ std::vector<size_t> markers;
+ for (size_t i = 0; i < tokens.size(); i++) {
+ if (tokens[i] == token_marker) {
+ markers.push_back(i);
+ }
+ }
+ const auto invalid = std::runtime_error("unexpected layout of the decision prompt");
+ if (markers.size() != n_options || markers[0] < 2 || tokens[markers[0] - 1] != token_sep || tokens.back() != token_sep) {
+ throw invalid;
+ }
+ const size_t head_end = markers[0] - 1;
+ const size_t opts_end = std::find(tokens.begin() + markers.back(), tokens.end(), token_sep) - tokens.begin();
+ if (opts_end + 1 >= tokens.size()) {
+ throw invalid;
+ }
+
+ // marker + text of each option
+ std::vector<llama_tokens> options;
+ size_t n_options_tokens = 0;
+ auto set_max = [&](size_t n_max) {
+ n_options_tokens = 0;
+ for (auto & opt : options) {
+ opt.resize(std::min(opt.size(), n_max));
+ n_options_tokens += opt.size();
+ }
+ };
+ for (size_t i = 0; i < n_options; i++) {
+ const size_t end = i + 1 < n_options ? markers[i + 1] : opts_end;
+ options.emplace_back(tokens.begin() + markers[i], tokens.begin() + end);
+ }
+ set_max(max_option_tokens + 1);
+ if (n_options_tokens + 16 > max_head_tokens) {
+ // too many or too long options, shrink them evenly
+ set_max(std::max((size_t) 4, (max_head_tokens - std::min(max_head_tokens, (size_t) 16)) / n_options));
+ }
+ const size_t n_question_max = std::max((size_t) 8, max_head_tokens - std::min(max_head_tokens, n_options_tokens));
+
+ llama_tokens out;
+ out.push_back(tokens[0]);
+ out.insert(out.end(), tokens.begin() + 1, tokens.begin() + std::min(head_end, 1 + n_question_max));
+ out.push_back(token_sep);
+ for (const auto & opt : options) {
+ task.decision.markers.push_back(out.size());
+ out.insert(out.end(), opt.begin(), opt.end());
+ }
+ out.insert(out.end(), tokens.begin() + opts_end, tokens.end());
+ tokens = std::move(out);
+
+ // the output has one score per question type
+ task.decision.column = question.type;
+}
+
+//
+// answer
+//
+
+float server_decision_context::get_temperature(const server_decision_question & question) const {
+ const size_t n = question.options.size();
+ const std::string type_name = decision_question_type_name(question.type);
+
+ // the temperature can depend on the number of options, the buckets are the ones used to fit it
+ std::string bucket;
+ if (type == COMMON_DECISION_TYPE_LEV) {
+ bucket = n <= 8 ? "small" : n <= 26 ? "mid" : "large";
+ } else {
+ bucket = n <= 2 ? "2" : n <= 5 ? "3_5" : n <= 10 ? "6_10" : "11";
+ }
+
+ for (const auto & name : {type_name + "." + bucket, type_name}) {
+ const auto it = temperatures.find(name);
+ if (it != temperatures.end()) {
+ return it->second;
+ }
+ }
+ return 1.0f;
+}
+
+// confidence formulas are the ones published by TypeSafe
+
+static double decision_confidence_choice(const std::vector<double> & probs) {
+ if (probs.size() < 2) {
+ return 1.0;
+ }
+ const double uniform = 1.0 / probs.size();
+ const double p_max = *std::max_element(probs.begin(), probs.end());
+ return std::max(0.0, (p_max - uniform) / (1.0 - uniform));
+}
+
+static double decision_confidence_score(const std::vector<double> & probs) {
+ if (probs.size() < 2) {
+ return 1.0;
+ }
+ const size_t n = probs.size();
+ const size_t mode = std::max_element(probs.begin(), probs.end()) - probs.begin();
+
+ // mean distance to the mode, relative to the one of a uniform distribution around its center
+ double dist = 0.0;
+ double dist_uniform = 0.0;
+ for (size_t i = 0; i < n; i++) {
+ dist += probs[i] * std::fabs((double) i - (double) mode);
+ dist_uniform += std::fabs((double) i - (n - 1) / 2.0) / n;
+ }
+ return std::max(0.0, 1.0 - dist / dist_uniform);
+}
+
+json server_decision_context::format_answer(const server_decision_question & question, const std::vector<std::vector<float>> & scores) const {
+ const size_t n = n_outputs(question);
+ if (scores.size() != n_variants(question)) {
+ throw std::runtime_error("decision result does not match the number of variants");
+ }
+
+ // softmax over the outputs of each variant, then the average of the variants
+ const float temperature = get_temperature(question);
+ std::vector<double> probs(n, 0.0);
+ for (size_t v = 0; v < scores.size(); v++) {
+ const auto & s = scores[v];
+ if (s.size() != n) {
+ throw std::runtime_error("decision result does not match the number of options");
+ }
+ const float score_max = *std::max_element(s.begin(), s.end());
+ std::vector<double> p(n);
+ double sum = 0.0;
+ for (size_t i = 0; i < n; i++) {
+ p[i] = std::exp((double) (s[i] - score_max) / temperature);
+ sum += p[i];
+ }
+ for (size_t i = 0; i < n; i++) {
+ // the second variant is in the reverse order
+ probs[v == 0 ? i : n - 1 - i] += p[i] / sum / scores.size();
+ }
+ }
+
+ json answer = json{{"type", decision_question_type_name(question.type)}};
+
+ if (question.type == SERVER_DECISION_QUESTION_NOUL) {
+ if (type == COMMON_DECISION_TYPE_LEV) {
+ double expected = 0.0;
+ for (size_t i = 0; i < n; i++) {
+ expected += probs[i] * i / (n - 1);
+ }
+ answer["noul"] = expected;
+ return answer;
+ }
+ for (size_t i = 0; i < n; i++) {
+ if (question.options[i].key == "true") {
+ answer["noul"] = probs[i];
+ }
+ }
+ return answer;
+ }
+
+ json probabilities = json::object();
+ for (size_t i = 0; i < n; i++) {
+ probabilities[question.options[i].key] = probs[i];
+ }
+
+ if (question.type == SERVER_DECISION_QUESTION_CHOICE) {
+ const size_t best = std::max_element(probs.begin(), probs.end()) - probs.begin();
+ answer["choice"] = question.options[best].key;
+ answer["probabilities"] = probabilities;
+ answer["confidence"] = decision_confidence_choice(probs);
+ } else {
+ double expected = 0.0;
+ json legend = json::object();
+ for (size_t i = 0; i < n; i++) {
+ expected += i * probs[i];
+ legend[question.options[i].key] = question.options[i].description;
+ }
+ answer["score"] = expected;
+ answer["legend"] = legend;
+ answer["probabilities"] = probabilities;
+ answer["confidence"] = decision_confidence_score(probs);
+ }
+ return answer;
+}
+
+//
+// shared prompt prefix
+//
+
+std::vector<server_task> server_decision_group_tasks(std::vector<server_task> && tasks, size_t n_slots) {
+ n_slots = std::max(n_slots, (size_t) 1);
+
+ std::vector<server_task> groups;
+ for (size_t i = 0; i < tasks.size(); i += n_slots) {
+ const size_t end = std::min(tasks.size(), i + n_slots);
+ server_task & parent = tasks[i];
+
+ // every task must have at least one token of its own to evaluate
+ size_t n_shared = parent.tokens.size() - 1;
+ for (size_t j = i + 1; j < end; j++) {
+ n_shared = std::min(n_shared, parent.tokens.get_common_prefix(tasks[j].tokens));
+ n_shared = std::min(n_shared, tasks[j].tokens.size() - 1);
+ }
+
+ if (end - i < 2 || n_shared == 0) {
+ for (size_t j = i; j < end; j++) {
+ groups.push_back(std::move(tasks[j]));
+ }
+ continue;
+ }
+
+ parent.n_tokens_shared = n_shared;
+ for (size_t j = i + 1; j < end; j++) {
+ tasks[j].id_parent = parent.id;
+ parent.child_tasks.push_back(std::move(tasks[j]));
+ }
+ groups.push_back(std::move(parent));
+ }
+ return groups;
+}
diff --git a/tools/server/server-decision.h b/tools/server/server-decision.h
new file mode 100644
index 000000000..5e38feb44
--- /dev/null
+++ b/tools/server/server-decision.h
@@ -0,0 +1,113 @@
+#pragma once
+
+#include "server-common.h"
+#include "server-task.h"
+
+#include <map>
+#include <memory>
+#include <string>
+#include <vector>
+
+// typed decision models (TypeSafe /v1/systemone API)
+// the model answers each question in one forward pass, no token is generated
+
+enum server_decision_question_type {
+ SERVER_DECISION_QUESTION_CHOICE,
+ SERVER_DECISION_QUESTION_SCORE,
+ SERVER_DECISION_QUESTION_NOUL,
+};
+
+struct server_decision_option {
+ std::string key;
+ json description; // null if not provided
+};
+
+struct server_decision_question {
+ std::string id;
+ server_decision_question_type type;
+ json instructions;
+ std::vector<server_decision_option> options; // in the order of the model outputs
+};
+
+struct server_decision_context {
+ common_decision_type type = COMMON_DECISION_TYPE_NONE;
+
+ // read the "<arch>.decision.*" metadata, type stays NONE if the model has none
+ void init(const llama_model * model);
+
+ // true if the questions of a request start with the same tokens, and the model can continue from them
+ bool can_share_prompt() const {
+ switch (type) {
+ case COMMON_DECISION_TYPE_OPENJEV:
+ case COMMON_DECISION_TYPE_LEV:
+ case COMMON_DECISION_TYPE_KEV:
+ return true;
+ default:
+ return false;
+ }
+ }
+
+ // true if the prompt of the model has a place for images
+ bool can_use_images() const {
+ switch (type) {
+ case COMMON_DECISION_TYPE_OPENJEV:
+ return true;
+ default:
+ return false;
+ }
+ }
+
+ // throw std::invalid_argument on bad input
+ std::vector<server_decision_question> parse_questions(const json & body) const;
+
+ // returns the state without its images, they are appended to files in order
+ // images come from "images" and from the image_url parts of a state made of chat messages
+ json parse_state(const json & body, std::vector<raw_buffer> & files) const;
+
+ // number of prompts that are evaluated to answer this question, each one shows the options in a different order
+ size_t n_variants(const server_decision_question & question) const;
+
+ // set the prompt of one variant of this question, and where to read its result
+ // mctx is only used if there are files
+ void fill_task(
+ const json & state,
+ const server_decision_question & question,
+ size_t variant,
+ const std::vector<raw_buffer> & files,
+ mtmd_context * mctx,
+ const mtmd_helper_init_opt & init_opt,
+ server_task & task) const;
+
+ // scores: the raw model outputs of each variant
+ json format_answer(const server_decision_question & question, const std::vector<std::vector<float>> & scores) const;
+
+private:
+ const llama_vocab * vocab = nullptr;
+ std::shared_ptr<const common_chat_template> tmpl; // the "systemone" template
+
+ std::map<std::string, float> temperatures; // "<type>" or "<type>.<n_options bucket>"
+ size_t n_options_max = 0;
+ bool noul_true_first = false; // noul options are [true, false] instead of [false, true]
+
+ // OPENJEV, LEV
+ std::vector<llama_token> labels;
+ std::vector<std::string> label_texts; // only if the label of an option is given to the template
+
+ // LAYA, KEV
+ llama_token token_marker = LLAMA_TOKEN_NULL;
+ llama_token token_sep = LLAMA_TOKEN_NULL;
+ std::string text_marker;
+ size_t max_head_tokens = 0; // question + options
+ size_t max_option_tokens = 48;
+
+ std::string render(const json & state, const server_decision_question & question, size_t variant, size_t n_images) const;
+ size_t n_outputs(const server_decision_question & question) const;
+ void fill_task_laya(llama_tokens & tokens, const server_decision_question & question, server_task & task) const;
+
+ float get_temperature(const server_decision_question & question) const;
+};
+
+// group the tasks so that the common prefix of their prompts is evaluated only once
+// each group is one parent and its children, it takes at most n_slots slots
+// note: the order of the tasks is preserved
+std::vector<server_task> server_decision_group_tasks(std::vector<server_task> && tasks, size_t n_slots);
diff --git a/tools/server/server-task.cpp b/tools/server/server-task.cpp
index 0d3beb313..a5c33c056 100644
--- a/tools/server/server-task.cpp
+++ b/tools/server/server-task.cpp
@@ -1495,6 +1495,17 @@ json server_task_result_rerank::to_json() {
};
}
+//
+// server_task_result_decision
+//
+json server_task_result_decision::to_json() {
+ return json {
+ {"index", index},
+ {"scores", scores},
+ {"tokens_evaluated", n_tokens},
+ };
+}
+
//
// server_task_result_error
//
diff --git a/tools/server/server-task.h b/tools/server/server-task.h
index 9c99143f8..8c5fa9a68 100644
--- a/tools/server/server-task.h
+++ b/tools/server/server-task.h
@@ -16,6 +16,7 @@ enum server_task_type {
SERVER_TASK_TYPE_COMPLETION,
SERVER_TASK_TYPE_EMBEDDING,
SERVER_TASK_TYPE_RERANK,
+ SERVER_TASK_TYPE_DECISION,
SERVER_TASK_TYPE_INFILL,
SERVER_TASK_TYPE_CANCEL,
SERVER_TASK_TYPE_CONTROL,
@@ -148,6 +149,8 @@ struct server_task {
// temporary store of child tasks for scheduling
// note: accessing to elements is invalid after the task is moved to server_slot
std::vector<server_task> child_tasks;
+ // if set on a parent, the children have their own prompt and only share its first n_tokens_shared tokens
+ int32_t n_tokens_shared = 0;
// used by SERVER_TASK_TYPE_INFERENCE
task_params params;
@@ -172,6 +175,26 @@ struct server_task {
// used by SERVER_TASK_TYPE_METRICS
bool metrics_reset_bucket = false;
+ // used by SERVER_TASK_TYPE_DECISION
+ // where to read the model output of each option, exactly one of the two lists is used
+ struct decision {
+ std::vector<llama_token> labels; // logits of these tokens, at the last prompt token
+ std::vector<int32_t> markers; // embeddings[column] at these prompt positions
+ int32_t column = 0;
+ // if set, embeddings is [q | k], and the output is instead the scaled dot product of q[pointer] and k[marker]
+ int32_t pointer = -1;
+
+ // first prompt position that is read, -1 if none
+ int32_t pos_first() const {
+ int32_t pos = pointer;
+ for (const int32_t marker : markers) {
+ pos = pos < 0 ? marker : std::min(pos, marker);
+ }
+ return pos;
+ }
+ };
+ decision decision;
+
// used by SERVER_TASK_TYPE_SET_LORA
std::map<int, float> set_lora; // mapping adapter ID -> scale
@@ -188,6 +211,8 @@ struct server_task {
case SERVER_TASK_TYPE_EMBEDDING:
case SERVER_TASK_TYPE_RERANK:
return true;
+ case SERVER_TASK_TYPE_DECISION:
+ return !decision.markers.empty();
default:
return false;
}
@@ -198,6 +223,8 @@ struct server_task {
case SERVER_TASK_TYPE_COMPLETION:
case SERVER_TASK_TYPE_INFILL:
return true;
+ case SERVER_TASK_TYPE_DECISION:
+ return !decision.labels.empty();
default:
return false;
}
@@ -474,6 +501,14 @@ struct server_task_result_rerank : server_task_result {
virtual json to_json() override;
};
+struct server_task_result_decision : server_task_result {
+ std::vector<float> scores; // one raw model output per option
+
+ int32_t n_tokens;
+
+ virtual json to_json() override;
+};
+
struct server_task_result_error : server_task_result {
error_type err_type = ERROR_TYPE_SERVER;
std::string err_msg;
diff --git a/tools/server/server.cpp b/tools/server/server.cpp
index bcf84e1ae..ad538a6d6 100644
--- a/tools/server/server.cpp
+++ b/tools/server/server.cpp
@@ -231,6 +231,7 @@ int llama_server(common_params & params, int argc, char ** argv) {
routes.post_embeddings = models_routes->proxy_post;
routes.post_embeddings_oai = models_routes->proxy_post;
routes.post_rerank = models_routes->proxy_post;
+ routes.post_systemone = models_routes->proxy_post;
routes.post_tokenize = models_routes->proxy_post;
routes.post_detokenize = models_routes->proxy_post;
routes.post_apply_template = models_routes->proxy_post;
@@ -278,6 +279,7 @@ int llama_server(common_params & params, int argc, char ** argv) {
ctx_http.post("/reranking", ex_wrapper(routes.post_rerank));
ctx_http.post("/v1/rerank", ex_wrapper(routes.post_rerank));
ctx_http.post("/v1/reranking", ex_wrapper(routes.post_rerank));
+ ctx_http.post("/v1/systemone", ex_wrapper(routes.post_systemone));
ctx_http.post("/tokenize", ex_wrapper(routes.post_tokenize));
ctx_http.post("/detokenize", ex_wrapper(routes.post_detokenize));
ctx_http.post("/apply-template", ex_wrapper(routes.post_apply_template));
diff --git a/tools/server/tests/unit/test_systemone.py b/tools/server/tests/unit/test_systemone.py
new file mode 100644
index 000000000..d9d988410
--- /dev/null
+++ b/tools/server/tests/unit/test_systemone.py
@@ -0,0 +1,216 @@
+import pytest
+from utils import *
+from test_vision_api import get_img_url
+
+server = ServerPreset.tinylaya()
+
+
+@pytest.fixture(autouse=True)
+def create_server():
+ global server
+ server = ServerPreset.tinylaya()
+
+
+TEST_STATE = "I was charged twice for my order last week and nobody has replied."
+
+TEST_QUESTIONS = {
+ "route": {
+ "type": "choice",
+ "instructions": "Which team should handle this?",
+ "criteria": {"billing": "payments and refunds", "shipping": None, "technical": None},
+ },
+ "urgency": {
+ "type": "score",
+ "instructions": "How urgent is this?",
+ "criteria": ["can wait", "this week", "today", "right now"],
+ },
+ "angry": {
+ "type": "noul",
+ "instructions": "Is the customer angry?",
+ },
+}
+
+
+def get_prompt_metrics(server: ServerProcess) -> tuple[int, int]:
+ """returns the number of prompt tokens (processed, cached) since the server started"""
+ res = server.make_request("GET", "/metrics")
+ assert res.status_code == 200
+ values = {}
+ for line in res.body.splitlines():
+ if line.startswith("llamacpp:"):
+ name, value = line.split(" ")
+ values[name] = int(float(value))
+ return values["llamacpp:prompt_tokens_total"], values["llamacpp:prompt_tokens_cached_total"]
+
+
+@pytest.mark.parametrize("preset", ["tinylaya", "tinyopenjev"])
+def test_systemone(preset: str):
+ global server
+ server = getattr(ServerPreset, preset)()
+ server.start()
+ res = server.make_request("POST", "/v1/systemone", data={
+ "state": TEST_STATE,
+ "questions": TEST_QUESTIONS,
+ })
+ assert res.status_code == 200
+ assert res.body["usage"]["input_tokens"] > 0
+ assert res.body["usage"]["output_tokens"] == 0
+
+ answers = res.body["answers"]
+ assert list(answers.keys()) == ["route", "urgency", "angry"]
+
+ route = answers["route"]
+ assert route["type"] == "choice"
+ assert list(route["probabilities"].keys()) == ["billing", "shipping", "technical"]
+ assert abs(sum(route["probabilities"].values()) - 1.0) < 1e-4
+ assert route["choice"] == max(route["probabilities"], key=route["probabilities"].get)
+ assert 0.0 <= route["confidence"] <= 1.0
+
+ urgency = answers["urgency"]
+ assert urgency["type"] == "score"
+ assert urgency["legend"] == {"0": "can wait", "1": "this week", "2": "today", "3": "right now"}
+ assert list(urgency["probabilities"].keys()) == ["0", "1", "2", "3"]
+ assert abs(sum(urgency["probabilities"].values()) - 1.0) < 1e-4
+ assert abs(urgency["score"] - sum(i * p for i, p in enumerate(urgency["probabilities"].values()))) < 1e-4
+ assert 0.0 <= urgency["confidence"] <= 1.0
+
+ angry = answers["angry"]
+ assert angry["type"] == "noul"
+ assert 0.0 <= angry["noul"] <= 1.0
+
+
+def test_systemone_json_state():
+ global server
+ server.start()
+ questions = {
+ "refund": {
+ "type": "noul",
+ "instructions": "Is a refund requested?",
+ "criteria": {"false": "no refund is asked", "true": "a refund is asked"},
+ },
+ }
+ res_obj = server.make_request("POST", "/v1/systemone", data={
+ "state": {"ticket": TEST_STATE, "plan": "pro"},
+ "questions": questions,
+ })
+ assert res_obj.status_code == 200
+ # an object is given to the model as JSON text
+ res_str = server.make_request("POST", "/v1/systemone", data={
+ "state": '{"ticket": "' + TEST_STATE + '", "plan": "pro"}',
+ "questions": questions,
+ })
+ assert res_str.status_code == 200
+ assert res_obj.body["usage"] == res_str.body["usage"]
+ assert abs(res_obj.body["answers"]["refund"]["noul"] - res_str.body["answers"]["refund"]["noul"]) < 1e-4
+
+
+@pytest.mark.parametrize("data", [
+ {"questions": TEST_QUESTIONS},
+ {"state": TEST_STATE},
+ {"state": TEST_STATE, "questions": {}},
+ {"state": TEST_STATE, "questions": {"q": {"type": "unknown", "instructions": "x"}}},
+ {"state": TEST_STATE, "questions": {"q": {"type": "noul"}}},
+ {"state": TEST_STATE, "questions": {"q": {"type": "choice", "instructions": "x"}}},
+ {"state": TEST_STATE, "questions": {"q": {"type": "choice", "instructions": "x", "criteria": {}}}},
+ {"state": TEST_STATE, "questions": {"q": {"type": "score", "instructions": "x", "criteria": ["only one"]}}},
+])
+def test_systemone_invalid_request(data: dict):
+ global server
+ server.start()
+ res = server.make_request("POST", "/v1/systemone", data=data)
+ assert res.status_code == 400
+ assert "error" in res.body
+
+
+def test_systemone_shared_prompt():
+ global server
+ server = ServerPreset.tinyopenjev()
+ server.server_metrics = True
+ server.start()
+ res = server.make_request("POST", "/v1/systemone", data={
+ "state": TEST_STATE,
+ "questions": TEST_QUESTIONS,
+ })
+ assert res.status_code == 200
+
+ # the first question evaluates the shared prefix, the 2 others start from it
+ n_processed, n_cached = get_prompt_metrics(server)
+ assert n_cached > 0
+ assert n_cached % 2 == 0
+ assert n_processed + n_cached == res.body["usage"]["input_tokens"]
+
+ # with one slot the prompt cannot be shared, the answers must be the same
+ server.stop()
+ server = ServerPreset.tinyopenjev()
+ server.n_slots = 1
+ server.start()
+ res_single = server.make_request("POST", "/v1/systemone", data={
+ "state": TEST_STATE,
+ "questions": TEST_QUESTIONS,
+ })
+ assert res_single.status_code == 200
+ assert res_single.body["usage"] == res.body["usage"]
+ for qid in ["route", "urgency"]:
+ probs_shared = res.body["answers"][qid]["probabilities"]
+ probs_single = res_single.body["answers"][qid]["probabilities"]
+ for key in probs_shared:
+ assert abs(probs_shared[key] - probs_single[key]) < 0.01
+ assert abs(res.body["answers"]["angry"]["noul"] - res_single.body["answers"]["angry"]["noul"]) < 0.01
+
+
+def test_systemone_images():
+ global server
+ server = ServerPreset.tinyopenjev()
+ server.start()
+ image = get_img_url("IMG_BASE64_URI_0")
+
+ res_text = server.make_request("POST", "/v1/systemone", data={
+ "state": TEST_STATE,
+ "questions": TEST_QUESTIONS,
+ })
+ assert res_text.status_code == 200
+
+ res = server.make_request("POST", "/v1/systemone", data={
+ "state": TEST_STATE,
+ "questions": TEST_QUESTIONS,
+ "images": [image],
+ })
+ assert res.status_code == 200
+ assert list(res.body["answers"].keys()) == ["route", "urgency", "angry"]
+ assert res.body["usage"]["input_tokens"] > res_text.body["usage"]["input_tokens"]
+
+ # same image, given as a part of a chat message
+ res_part = server.make_request("POST", "/v1/systemone", data={
+ "state": [{"role": "user", "content": [
+ {"type": "image_url", "image_url": {"url": image}},
+ {"type": "text", "text": TEST_STATE},
+ ]}],
+ "questions": TEST_QUESTIONS,
+ })
+ assert res_part.status_code == 200
+ assert res_part.body["usage"]["input_tokens"] > res_text.body["usage"]["input_tokens"]
+
+ res = server.make_request("POST", "/v1/systemone", data={
+ "state": TEST_STATE,
+ "questions": TEST_QUESTIONS,
+ "images": [image] * 9,
+ })
+ assert res.status_code == 400
+
+ res = server.make_request("POST", "/v1/systemone", data={
+ "state": TEST_STATE,
+ "questions": TEST_QUESTIONS,
+ "images": ["https://example.com/image.png"],
+ })
+ assert res.status_code == 400
+
+
+def test_systemone_images_not_supported():
+ global server
+ server.start()
+ res = server.make_request("POST", "/v1/systemone", data={
+ "state": TEST_STATE,
+ "questions": TEST_QUESTIONS,
+ "images": [get_img_url("IMG_BASE64_URI_0")],
+ })
+ assert res.status_code == 501
diff --git a/tools/server/tests/utils.py b/tools/server/tests/utils.py
index 3a50ae5c3..90c4ebac8 100644
--- a/tools/server/tests/utils.py
+++ b/tools/server/tests/utils.py
@@ -628,6 +628,37 @@ class ServerPreset:
server.server_reranking = True
return server
+ @staticmethod
+ def tinylaya() -> ServerProcess:
+ server = ServerProcess()
+ server.offline = True # will be downloaded by load_all()
+ local_model = os.environ.get("TINYLAYA_LOCAL_MODEL")
+ server.model_hf_file = None
+ if local_model:
+ server.model_file = local_model
+ server.model_hf_repo = None
+ else:
+ server.model_hf_repo = "ggml-org/tinylaya-for-testing-gguf"
+ server.n_ctx = 1024
+ server.n_batch = 512
+ server.n_ubatch = 512
+ server.n_slots = 2
+ server.seed = 42
+ return server
+
+ @staticmethod
+ def tinyopenjev() -> ServerProcess:
+ server = ServerProcess()
+ server.offline = True # will be downloaded by load_all()
+ # mmproj is already provided by HF registry API
+ server.model_hf_file = None
+ server.model_hf_repo = "ggml-org/tinyopenjev-for-testing-gguf:Q8_0"
+ server.n_ctx = 4096
+ server.n_batch = 512
+ server.n_slots = 4
+ server.seed = 42
+ return server
+
@staticmethod
def tinygemma3() -> ServerProcess:
server = ServerProcess()