Commit 4d60b4d08 for llama.cpp
commit 4d60b4d087103cd96fe8feeb557709cd8998cd56
Author: Adrien Gallouët <angt@huggingface.co>
Date: Mon Oct 5 17:08:29 2026 +0200
common, server : report model input/output modalities in GET /models (#29987)
Signed-off-by: Adrien Gallouët <angt@huggingface.co>
diff --git a/common/arg.cpp b/common/arg.cpp
index 9ed1455b4..14d82f695 100644
--- a/common/arg.cpp
+++ b/common/arg.cpp
@@ -682,7 +682,10 @@ void common_models_handler_apply(common_models_handler & handler, common_params
// if HF repo is a preset repo, we simply run server in router mode with the preset.ini file
params.models_preset_hf = params.model.hf_repo; // only for showing a warning
params.models_preset = hf_cache::finalize_file(plan.preset);
- params.model = common_params_model{}; // make sure to clear model, so server starts in router mode
+ // clear the model so the server starts in router mode
+ params.model.path.clear();
+ params.model.hf_repo.clear();
+ params.model.docker_repo.clear();
});
}
diff --git a/common/common.cpp b/common/common.cpp
index 463137e5a..c301c22ca 100644
--- a/common/common.cpp
+++ b/common/common.cpp
@@ -1,4 +1,5 @@
#include "ggml.h"
+#include "ggml-cpp.h"
#include "gguf.h"
#include "build-info.h"
@@ -1191,6 +1192,41 @@ common_decision_type common_get_decision_type(const struct llama_model * model)
return common_decision_type_from_string(buf);
}
+common_decision_type common_get_decision_type(const std::string & fname) {
+ struct gguf_init_params gguf_params = {
+ /* .no_alloc = */ true,
+ /* .ctx = */ nullptr,
+ };
+
+ gguf_context_ptr gguf_ctx(gguf_init_from_file(fname.c_str(), gguf_params));
+ if (!gguf_ctx) {
+ return COMMON_DECISION_TYPE_UNKNOWN; // missing or unreadable file
+ }
+
+ std::string arch;
+ const int64_t arch_id = gguf_find_key(gguf_ctx.get(), "general.architecture");
+ if (arch_id < 0) {
+ return COMMON_DECISION_TYPE_UNKNOWN; // no architecture in the metadata
+ }
+ if (gguf_get_kv_type(gguf_ctx.get(), arch_id) != GGUF_TYPE_STRING) {
+ return COMMON_DECISION_TYPE_UNKNOWN; // malformed metadata
+ }
+ arch = gguf_get_val_str(gguf_ctx.get(), arch_id);
+ if (arch.empty()) {
+ return COMMON_DECISION_TYPE_UNKNOWN;
+ }
+
+ const std::string key = arch + ".decision.type";
+ const int64_t type_id = gguf_find_key(gguf_ctx.get(), key.c_str());
+ if (type_id < 0) {
+ return COMMON_DECISION_TYPE_NONE;
+ }
+ if (gguf_get_kv_type(gguf_ctx.get(), type_id) != GGUF_TYPE_STRING) {
+ return COMMON_DECISION_TYPE_UNKNOWN; // malformed metadata
+ }
+ return common_decision_type_from_string(gguf_get_val_str(gguf_ctx.get(), type_id));
+}
+
common_init_result::common_init_result(common_params & params, bool model_only) :
pimpl(new impl{}) {
auto mparams = common_model_params_to_llama(params);
diff --git a/common/common.h b/common/common.h
index e1ef70a9e..2f50d90c6 100644
--- a/common/common.h
+++ b/common/common.h
@@ -965,6 +965,10 @@ enum common_decision_type {
common_decision_type common_get_decision_type(const struct llama_model * model);
+// same as above, but reads a GGUF file; it does not load the model
+// returns COMMON_DECISION_TYPE_UNKNOWN if the file is missing, unreadable, or invalid
+common_decision_type common_get_decision_type(const std::string & fname);
+
// note: defines the model, context, samplers, ets. lifetimes
struct common_init_result {
common_init_result(common_params & params, bool model_only = false);
diff --git a/tools/server/README.md b/tools/server/README.md
index d2d6ab2be..4ab9238b2 100644
--- a/tools/server/README.md
+++ b/tools/server/README.md
@@ -1248,6 +1248,30 @@ Returns information about the loaded model. See [OpenAI Models API documentation
The returned list always has one single element. The `meta` field can be `null` (for example, while the model is still loading).
+Each object in `data` has an `architecture` object. It has two string arrays:
+
+- `input_modalities` lists what the model can read. It always has `text`, plus each media type that the model supports.
+- `output_modalities` lists what the model can produce.
+
+One output value is special:
+
+| Value | Meaning |
+|---|---|
+| `decisions` | The model is a native decision model. Serve it with [`/v1/systemone`](#post-v1systemone-typesafe-compatible-system-one-api). |
+
+A language model that classifies with prompts does not get `decisions`. Only native decision models do.
+
+Check for membership. Tolerate values that you do not know:
+
+```js
+const useSystemOne =
+ model.architecture?.output_modalities?.includes("decisions") === true;
+```
+
+Without decision metadata, `output_modalities` is `["text"]`. This default is for compatibility only. It does not mean that the model can generate text. Values can change. New combinations such as `["text", "decisions"]` use the same shape.
+
+The router returns the same `architecture` object in [`GET /models`](#get-models-list-available-models). You can find a native decision model without a probe or a model load. This works for unloaded and sleeping models too. Older servers can omit `architecture`. If it is absent, use the legacy behavior of your client.
+
By default, model `id` field is the path to model file, specified via `-m`. You can set a custom value for model `id` field via `--alias` argument. For example, `--alias gpt-4o-mini`.
Example:
@@ -1259,6 +1283,10 @@ Example:
{
"id": "../models/Meta-Llama-3.1-8B-Instruct-Q4_K_M.gguf",
"object": "model",
+ "architecture": {
+ "input_modalities": ["text"],
+ "output_modalities": ["text"]
+ },
"created": 1735142223,
"owned_by": "llamacpp",
"meta": {
@@ -1990,6 +2018,37 @@ Note:
- If a model is not running, it will be added or updated according to the source
2. When the model is loaded, the info from `/v1/models` is forwarded to router's `/v1/models`. This includes metadata about the model and the runtime instance.
+Each object in `data` has the same `architecture` object as [`GET /v1/models`](#get-v1models-openai-compatible-model-info-api) of a direct server. The server computes both arrays offline. It does not load the model, download files, or run inference. `output_modalities` comes from the GGUF metadata. `input_modalities` comes from the projector file. A native decision model shows `decisions` before its first load, after unload, and while it sleeps:
+
+```json
+{
+ "object": "list",
+ "data": [
+ {
+ "id": "my-decision-model",
+ "object": "model",
+ "tags": ["local"],
+ "architecture": {
+ "input_modalities": ["text"],
+ "output_modalities": ["decisions"]
+ },
+ "status": {
+ "value": "unloaded"
+ }
+ }
+ ]
+}
+```
+
+The values work like this:
+
+- A loaded model reports both arrays. Its values replace the cached values in full.
+- The cache keeps the values across sleep and unload. A known decision model stays advertised.
+- Before the first report, the values come from the offline computation.
+- Offline computation cannot see video. Only a loaded model reports `video` in `input_modalities`.
+- If the metadata or the model file is not available, both arrays are `["text"]`.
+- A source or preset refresh computes both arrays again. A replaced model does not keep old values.
+
The `status` object can be:
```json
diff --git a/tools/server/server-common.cpp b/tools/server/server-common.cpp
index 076a85741..4b73aa908 100644
--- a/tools/server/server-common.cpp
+++ b/tools/server/server-common.cpp
@@ -143,6 +143,47 @@ const char * get_media_marker() {
return marker.c_str();
}
+//
+// model output modalities
+//
+
+std::vector<std::string> server_model_output_modalities(common_decision_type decision_type) {
+ switch (decision_type) {
+ case COMMON_DECISION_TYPE_OPENJEV:
+ case COMMON_DECISION_TYPE_LEV:
+ case COMMON_DECISION_TYPE_KEV:
+ case COMMON_DECISION_TYPE_NIMBLE:
+ case COMMON_DECISION_TYPE_LAYA:
+ case COMMON_DECISION_TYPE_CLEF:
+ return {"decisions"};
+ default:
+ // fallback when there is no decision type or the metadata is bad
+ return {"text"};
+ }
+}
+
+json server_model_architecture_json(
+ bool inp_image,
+ bool inp_audio,
+ bool inp_video,
+ const std::vector<std::string> & output_modalities) {
+ std::vector<std::string> input_modalities = {"text"};
+ if (inp_image) {
+ input_modalities.push_back("image");
+ }
+ if (inp_audio) {
+ input_modalities.push_back("audio");
+ }
+ if (inp_video) {
+ input_modalities.push_back("video");
+ }
+
+ return {
+ {"input_modalities", input_modalities},
+ {"output_modalities", output_modalities},
+ };
+}
+
//
// lora utils
//
diff --git a/tools/server/server-common.h b/tools/server/server-common.h
index 20cdeacf2..6165d871c 100644
--- a/tools/server/server-common.h
+++ b/tools/server/server-common.h
@@ -105,6 +105,20 @@ std::string gen_tool_call_id();
// get a random marker; note: each time the server restarts, the marker will be different
const char * get_media_marker();
+//
+// model output modalities
+//
+
+// output modalities for architecture.output_modalities in GET /models
+std::vector<std::string> server_model_output_modalities(common_decision_type decision_type);
+
+// architecture object of GET /models; shared by the direct server and the router
+json server_model_architecture_json(
+ bool inp_image,
+ bool inp_audio,
+ bool inp_video,
+ const std::vector<std::string> & output_modalities);
+
//
// lora utils
//
diff --git a/tools/server/server-context.cpp b/tools/server/server-context.cpp
index 13be63acd..d1af984e6 100644
--- a/tools/server/server-context.cpp
+++ b/tools/server/server-context.cpp
@@ -4537,40 +4537,41 @@ server_context_meta server_context::get_meta() const {
const char * ftype_name = llama_ftype_name(llama_model_ftype(impl->model_tgt));
return server_context_meta {
- /* build_info */ std::string(llama_build_info()),
- /* model_name */ impl->model_name,
- /* model_aliases */ impl->model_aliases,
- /* model_tags */ impl->model_tags,
- /* model_path */ impl->params_base.model.path,
- /* has_mtmd */ impl->mctx != nullptr,
- /* has_inp_image */ impl->chat_params.allow_image,
- /* has_inp_audio */ impl->chat_params.allow_audio,
- /* has_inp_video */ impl->chat_params.allow_video,
- /* json_ui_settings */ impl->json_ui_settings,
- /* slot_n_ctx */ impl->n_ctx_slot(),
- /* pooling_type */ llama_pooling_type(impl->ctx_tgt),
-
- /* chat_params */ impl->chat_params,
- /* chat_template_caps */ common_chat_templates_get_caps(impl->chat_params.tmpls.get()),
-
- /* bos_token_str */ bos_token_str,
- /* eos_token_str */ eos_token_str,
- /* fim_pre_token */ llama_vocab_fim_pre(impl->vocab),
- /* fim_sub_token */ llama_vocab_fim_suf(impl->vocab),
- /* fim_mid_token */ llama_vocab_fim_mid(impl->vocab),
- /* fim_pad_token */ llama_vocab_fim_pad(impl->vocab),
- /* fim_rep_token */ llama_vocab_fim_rep(impl->vocab),
- /* fim_sep_token */ llama_vocab_fim_sep(impl->vocab),
-
- /* logit_bias_eog */ impl->params_base.sampling.logit_bias_eog,
-
- /* model_vocab_type */ llama_vocab_type(impl->vocab),
- /* model_vocab_n_tokens */ llama_vocab_n_tokens(impl->vocab),
- /* model_n_ctx_train */ llama_model_n_ctx_train(impl->model_tgt),
- /* model_n_embd_inp */ llama_model_n_embd(impl->model_tgt),
- /* model_n_params */ llama_model_n_params(impl->model_tgt),
- /* model_size */ llama_model_size(impl->model_tgt),
- /* model_ftype */ ftype_name,
+ /* build_info */ std::string(llama_build_info()),
+ /* model_name */ impl->model_name,
+ /* model_aliases */ impl->model_aliases,
+ /* model_tags */ impl->model_tags,
+ /* model_path */ impl->params_base.model.path,
+ /* model_output_modalities */ server_model_output_modalities(common_get_decision_type(impl->model_tgt)),
+ /* has_mtmd */ impl->mctx != nullptr,
+ /* has_inp_image */ impl->chat_params.allow_image,
+ /* has_inp_audio */ impl->chat_params.allow_audio,
+ /* has_inp_video */ impl->chat_params.allow_video,
+ /* json_ui_settings */ impl->json_ui_settings,
+ /* slot_n_ctx */ impl->n_ctx_slot(),
+ /* pooling_type */ llama_pooling_type(impl->ctx_tgt),
+
+ /* chat_params */ impl->chat_params,
+ /* chat_template_caps */ common_chat_templates_get_caps(impl->chat_params.tmpls.get()),
+
+ /* bos_token_str */ bos_token_str,
+ /* eos_token_str */ eos_token_str,
+ /* fim_pre_token */ llama_vocab_fim_pre(impl->vocab),
+ /* fim_sub_token */ llama_vocab_fim_suf(impl->vocab),
+ /* fim_mid_token */ llama_vocab_fim_mid(impl->vocab),
+ /* fim_pad_token */ llama_vocab_fim_pad(impl->vocab),
+ /* fim_rep_token */ llama_vocab_fim_rep(impl->vocab),
+ /* fim_sep_token */ llama_vocab_fim_sep(impl->vocab),
+
+ /* logit_bias_eog */ impl->params_base.sampling.logit_bias_eog,
+
+ /* model_vocab_type */ llama_vocab_type(impl->vocab),
+ /* model_vocab_n_tokens */ llama_vocab_n_tokens(impl->vocab),
+ /* model_n_ctx_train */ llama_model_n_ctx_train(impl->model_tgt),
+ /* model_n_embd_inp */ llama_model_n_embd(impl->model_tgt),
+ /* model_n_params */ llama_model_n_params(impl->model_tgt),
+ /* model_size */ llama_model_size(impl->model_tgt),
+ /* model_ftype */ ftype_name,
};
}
@@ -4898,6 +4899,11 @@ static json get_res_model_info(const server_context_meta & meta) {
{"aliases", meta.model_aliases},
{"tags", meta.model_tags},
{"object", "model"},
+ {"architecture", server_model_architecture_json(
+ meta.has_inp_image,
+ meta.has_inp_audio,
+ meta.has_inp_video,
+ meta.model_output_modalities)},
{"created", std::time(0)},
{"owned_by", "llamacpp"},
{"meta", {
diff --git a/tools/server/server-context.h b/tools/server/server-context.h
index c554bb95b..7fe003531 100644
--- a/tools/server/server-context.h
+++ b/tools/server/server-context.h
@@ -19,6 +19,7 @@ struct server_context_meta {
std::set<std::string> model_aliases;
std::set<std::string> model_tags;
std::string model_path;
+ std::vector<std::string> model_output_modalities; // output modalities for GET /models
bool has_mtmd;
bool has_inp_image;
bool has_inp_audio;
diff --git a/tools/server/server-models.cpp b/tools/server/server-models.cpp
index d42b523b2..35d21c876 100644
--- a/tools/server/server-models.cpp
+++ b/tools/server/server-models.cpp
@@ -537,9 +537,16 @@ void server_model_meta::update_args(common_preset_context & ctx_preset, std::str
}
}
-void server_model_meta::update_caps() {
+void server_model_meta::update_caps(const common_params & base) {
+ // reset to the default so a failed refresh cannot keep old values
+ architecture = server_model_architecture_json(false, false, false, {"text"});
+
+ // resolve the model file offline; do not download
+ common_params params;
+ params.model = base.model;
+ // --no-mmproj applies to child models and blocks auto-attached projectors
+ params.no_mmproj = base.no_mmproj;
try {
- common_params params;
preset.apply_to_params(params, {
"LLAMA_ARG_MODEL",
"LLAMA_ARG_MODEL_URL",
@@ -551,16 +558,33 @@ void server_model_meta::update_caps() {
});
params.offline = true;
common_models_handler handler = common_models_handler_init(params, LLAMA_EXAMPLE_SERVER);
- common_models_handler_apply(handler, params); // note: this won't download the model because offline=true
- if (params.no_mmproj || params.mmproj.path.empty()) {
- multimodal = { false, false };
- } else {
- multimodal = mtmd_get_cap_from_file(params.mmproj.path.c_str());
+ common_models_handler_apply(handler, params);
+ } catch (const std::exception & e) {
+ LOG_WRN("failed to resolve the model of '%s': %s\n", name.c_str(), e.what());
+ return;
+ }
+
+ // read the output modalities from the GGUF metadata
+ std::vector<std::string> output_modalities = {"text"};
+ if (!params.model.path.empty()) {
+ output_modalities = server_model_output_modalities(common_get_decision_type(params.model.path));
+ }
+
+ bool inp_image = false;
+ bool inp_audio = false;
+ try {
+ if (!params.no_mmproj && !params.mmproj.path.empty()) {
+ mtmd_caps caps = mtmd_get_cap_from_file(params.mmproj.path.c_str());
+ inp_image = caps.inp_vision;
+ inp_audio = caps.inp_audio;
}
} catch (const std::exception & e) {
- LOG_WRN("failed to initialize common_params for multimodal capability detection: %s\n", e.what());
- multimodal = { false, false };
+ LOG_WRN("failed to read the multimodal capabilities of '%s': %s\n", name.c_str(), e.what());
+ // keep the output modalities from the GGUF metadata
}
+
+ // offline discovery cannot see video; a loaded model reports it
+ architecture = server_model_architecture_json(inp_image, inp_audio, false, output_modalities);
}
//
@@ -651,7 +675,7 @@ void server_models::add_model(server_model_meta && meta) {
}
meta.update_args(ctx_preset, bin_path); // render args
- meta.update_caps();
+ meta.update_caps(base_params);
std::string name = meta.name;
mapping[name] = instance_t{
/* subproc */ std::make_shared<server_subproc>(),
@@ -841,7 +865,6 @@ void server_models::load_models() {
/* progress */ {},
/* exit_code */ 0,
/* stop_timeout */ DEFAULT_STOP_TIMEOUT,
- /* multimodal */ mtmd_caps{false, false},
// /* need_download */ false,
};
add_model(std::move(meta));
@@ -962,7 +985,7 @@ void server_models::load_models() {
inst.meta.exit_code = 0; // clear failed state so the model can be reloaded
inst.meta.update_args(ctx_preset, bin_path);
- inst.meta.update_caps();
+ inst.meta.update_caps(base_params);
}
// add models that are new in this reload, load-on-startup is not honored here since a
@@ -983,7 +1006,6 @@ void server_models::load_models() {
/* progress */ {},
/* exit_code */ 0,
/* stop_timeout */ DEFAULT_STOP_TIMEOUT,
- /* multimodal */ mtmd_caps{false, false},
// /* need_download */ false,
};
add_model(std::move(meta));
@@ -1307,6 +1329,27 @@ void server_models::update_status(const std::string & name, const update_status_
}
if (!args.loaded_info.is_null()) {
meta.loaded_info = args.loaded_info;
+ // the child replaces both arrays in full; a bad or missing value changes nothing
+ if (args.loaded_info.contains("architecture") && args.loaded_info.at("architecture").is_object()) {
+ const json & child_arch = args.loaded_info.at("architecture");
+ for (const char * key : { "input_modalities", "output_modalities" }) {
+ if (!child_arch.contains(key) || !child_arch.at(key).is_array()) {
+ continue;
+ }
+ std::vector<std::string> modalities;
+ bool valid = true;
+ for (const auto & m : child_arch.at(key)) {
+ if (!m.is_string()) {
+ valid = false;
+ break;
+ }
+ modalities.push_back(m.get<std::string>());
+ }
+ if (valid) {
+ meta.architecture[key] = std::move(modalities);
+ }
+ }
+ }
}
if (!args.progress.is_null()) {
meta.progress = args.progress;
@@ -2067,19 +2110,6 @@ void server_models_routes::init_routes() {
status["failed"] = true;
}
- // pi coding agent multimodal compatibility
- json input_modalities = json::array({"text"});
- if (meta.multimodal.inp_vision) {
- input_modalities.push_back("image");
- }
- if (meta.multimodal.inp_audio) {
- input_modalities.push_back("audio");
- }
- json architecture {
- {"input_modalities", input_modalities},
- {"output_modalities", json::array({"text"})},
- };
-
json model_info = json {
{"id", meta.name},
{"aliases", meta.aliases},
@@ -2088,7 +2118,7 @@ void server_models_routes::init_routes() {
{"owned_by", "llamacpp"}, // for OAI-compat
{"created", t}, // for OAI-compat
{"status", status},
- {"architecture", architecture},
+ {"architecture", meta.architecture},
{"source", server_model_source_to_string(meta.source)},
{"can_remove", meta.source == SERVER_MODEL_SOURCE_CACHE},
// {"need_download", meta.need_download},
diff --git a/tools/server/server-models.h b/tools/server/server-models.h
index 0a6999ee3..238955e7c 100644
--- a/tools/server/server-models.h
+++ b/tools/server/server-models.h
@@ -84,8 +84,8 @@ struct server_model_meta {
json progress; // reflect load or download progress info, if any
int exit_code = 0; // exit code of the model instance process (only valid if status == FAILED)
int stop_timeout = 0; // seconds to wait before force-killing the model instance during shutdown
- mtmd_caps multimodal; // multimodal capabilities
bool hidden = false; // hidden from GET /models, but still accept if requested
+ json architecture = server_model_architecture_json(false, false, false, {"text"});
bool is_ready() const {
return status == SERVER_MODEL_STATUS_LOADED;
@@ -104,7 +104,7 @@ struct server_model_meta {
}
void update_args(common_preset_context & ctx_presets, std::string bin_path);
- void update_caps();
+ void update_caps(const common_params & base);
};
struct server_models_routes;