Commit 680a03628 for llama.cpp
commit 680a036285273a3ff56032ec5d7f3352609eba4f
Author: Tim Wang <149349643+timothywang21@users.noreply.github.com>
Date: Mon Sep 28 15:40:38 2026 -0400
server : support typed content (vision/audio/video) input for /v1/embeddings endpoint (#29556)
* server : support multimodal input for /v1/embeddings (Qwen3-VL-Embedding)
Accept the OpenAI-style wrapped content array format for multimodal
embedding requests. Each {"content": [...]} object is one input that
produces one embedding; text parts are concatenated and image_url parts
are decoded via handle_media then spliced with process_mtmd_prompt.
The legacy formats (plain string, token arrays, mixed arrays, and the
{prompt_string, multimodal_data} object) continue to work unchanged via
tokenize_input_prompts. Bare content arrays (the unwrapped shape) are
rejected with a migration message.
Also disables KV prefix reuse for stateless embedding/rerank tasks so
that repeated inputs do not incorrectly share cached KV across requests.
Assisted-by: Opencode Qwen3.8 27B
* clean up comments and docs
* refactor
* add tests
* support video and audio inp
---------
Co-authored-by: timothywang21 <timothywang21@users.noreply.github.com>
Co-authored-by: Xuan Son Nguyen <son@huggingface.co>
diff --git a/tools/server/README.md b/tools/server/README.md
index e665904ba..09fdb87e5 100644
--- a/tools/server/README.md
+++ b/tools/server/README.md
@@ -1509,6 +1509,13 @@ This endpoint requires that the model uses a pooling different than type `none`.
See [OpenAI Embeddings API documentation](https://platform.openai.com/docs/api-reference/embeddings).
+For multimodal models (loaded with `--mmproj`), each element of `input` can also be an object with a `content` array, using the same parts as `/v1/chat/completions`:
+- `{ "type": "text", "text": "..." }`: text is added to the prompt as-is
+- `{ "type": "image_url", "image_url": { "url": "..." } }`: remote URL, base64 data URI, or local file (`file://`, requires `--media-path`)
+- `{ "type": "input_audio", "input_audio": { "data": "..." } }` and `{ "type": "input_video", "input_video": { "url": "..." } }`: same as `/v1/chat/completions`, requires a model with audio or video support
+
+Each object gives one embedding. This input shape is not part of the OpenAI Embeddings API; it follows the shape used by providers like OpenRouter for vision embedding models.
+
*Examples:*
- input as string
@@ -1537,6 +1544,26 @@ See [OpenAI Embeddings API documentation](https://platform.openai.com/docs/api-r
}'
```
+- `input` as multimodal content
+
+ ```shell
+ curl http://localhost:8080/v1/embeddings \
+ -H "Content-Type: application/json" \
+ -H "Authorization: Bearer no-key" \
+ -d '{
+ "input": [
+ { "content": [
+ { "type": "image_url", "image_url": { "url": "data:image/jpeg;base64,/9j/4AAQSkZJRg..." } },
+ { "type": "text", "text": "Describe this image" }
+ ] },
+ { "content": [
+ { "type": "text", "text": "hello" }
+ ] }
+ ],
+ "encoding_format": "float"
+ }'
+ ```
+
### POST `/v1/responses/input_tokens`: Token Counting
Similar to [Response input token counts API](https://developers.openai.com/api/reference/python/resources/responses/subresources/input_tokens/methods/count).
diff --git a/tools/server/server-common.cpp b/tools/server/server-common.cpp
index 2ff790a5d..8cf794f1e 100644
--- a/tools/server/server-common.cpp
+++ b/tools/server/server-common.cpp
@@ -970,7 +970,7 @@ server_tokens process_mtmd_prompt(
}
/**
- * break the input "prompt" object into multiple prompt if needed, then tokenize them
+ * tokenize a single input "prompt" object
* use tokenize_input_prompts() if the input could be an array.
* this supports these cases:
* - "prompt": "string"
@@ -978,7 +978,7 @@ server_tokens process_mtmd_prompt(
* - "prompt": [12, 34, "string", 56, 78]
* - "prompt": { "prompt_string": "string", "multimodal_data": [ "base64" ] }
*/
-static server_tokens tokenize_input_subprompt(const llama_vocab * vocab, mtmd_context * mctx, const json & json_prompt, bool add_special, bool parse_special, const mtmd_helper_init_opt & init_opt) {
+server_tokens tokenize_input_subprompt(const llama_vocab * vocab, mtmd_context * mctx, const json & json_prompt, bool add_special, bool parse_special, const mtmd_helper_init_opt & init_opt) {
constexpr char JSON_STRING_PROMPT_KEY[] = "prompt_string";
constexpr char JSON_MTMD_DATA_KEY[] = "multimodal_data";
const bool has_mtmd = mctx != nullptr;
@@ -1147,6 +1147,79 @@ static void handle_media(
}
}
+// load media files from an OAI content array, then replace each media part with a media marker text part
+static void oaicompat_content_load_media(json & content, const server_chat_params & opt, std::vector<raw_buffer> & out_files) {
+ for (auto & p : content) {
+ std::string type = json_value(p, "type", std::string());
+ if (type == "image_url") {
+ if (!opt.allow_image) {
+ throw std::runtime_error("image input is not supported - hint: if this is unexpected, you may need to provide the mmproj");
+ }
+
+ json image_url = json_value(p, "image_url", json::object());
+ std::string url = json_value(image_url, "url", std::string());
+ handle_media(out_files, url, opt.media_path);
+
+ p["type"] = "media_marker";
+ p["text"] = get_media_marker();
+ p.erase("image_url");
+
+ } else if (type == "input_audio") {
+ if (!opt.allow_audio) {
+ throw std::runtime_error("audio input is not supported - hint: if this is unexpected, you may need to provide the mmproj");
+ }
+
+ // note: don't need to validate "format", it's redundant
+ json input_audio = json_value(p, "input_audio", json::object());
+ std::string url = json_value(input_audio, "data",
+ json_value(input_audio, "url", std::string()));
+ handle_media(out_files, url, opt.media_path);
+
+ p["type"] = "media_marker";
+ p["text"] = get_media_marker();
+ p.erase("input_audio");
+
+ } else if (type == "input_video" || type == "video_url") {
+ if (!opt.allow_video) {
+ throw std::runtime_error("video input is not supported - hint: if this is unexpected, you may need to provide the mmproj");
+ }
+
+ // accept the OpenAI-style "video_url" key as an alias of "input_video"
+ json input_video = json_value(p, type, json::object());
+ std::string url = json_value(input_video, "data",
+ json_value(input_video, "url", std::string()));
+ handle_media(out_files, url, opt.media_path);
+
+ p["type"] = "media_marker";
+ p["text"] = get_media_marker();
+ p.erase("input_video");
+ p.erase("video_url");
+
+ } else if (type != "text") {
+ throw std::invalid_argument("unsupported content[].type");
+ }
+ }
+}
+
+server_tokens tokenize_oai_content_array(const llama_vocab * vocab, mtmd_context * mctx, const server_chat_params & opt, json content, bool add_special, bool parse_special, const mtmd_helper_init_opt & init_opt) {
+ if (!content.is_array()) {
+ throw std::invalid_argument("\"content\" must be an array");
+ }
+
+ std::vector<raw_buffer> files;
+ oaicompat_content_load_media(content, opt, files);
+
+ std::string prompt;
+ for (const auto & p : content) {
+ prompt += json_value(p, "text", std::string());
+ }
+
+ if (files.empty()) {
+ return server_tokens(common_tokenize(vocab, prompt, add_special, parse_special), false);
+ }
+ return process_mtmd_prompt(mctx, prompt, files, init_opt);
+}
+
// used by /chat/completions endpoint
json oaicompat_chat_params_parse(
json & body, /* openai api json semantics */
@@ -1233,56 +1306,7 @@ json oaicompat_chat_params_parse(
throw std::invalid_argument("Expected 'content' to be a string or an array");
}
- for (auto & p : content) {
- std::string type = json_value(p, "type", std::string());
- if (type == "image_url") {
- if (!opt.allow_image) {
- throw std::runtime_error("image input is not supported - hint: if this is unexpected, you may need to provide the mmproj");
- }
-
- json image_url = json_value(p, "image_url", json::object());
- std::string url = json_value(image_url, "url", std::string());
- handle_media(out_files, url, opt.media_path);
-
- p["type"] = "media_marker";
- p["text"] = get_media_marker();
- p.erase("image_url");
-
- } else if (type == "input_audio") {
- if (!opt.allow_audio) {
- throw std::runtime_error("audio input is not supported - hint: if this is unexpected, you may need to provide the mmproj");
- }
-
- // note: don't need to validate "format", it's redundant
- json input_audio = json_value(p, "input_audio", json::object());
- std::string url = json_value(input_audio, "data",
- json_value(input_audio, "url", std::string()));
- handle_media(out_files, url, opt.media_path);
-
- p["type"] = "media_marker";
- p["text"] = get_media_marker();
- p.erase("input_audio");
-
- } else if (type == "input_video" || type == "video_url") {
- if (!opt.allow_video) {
- throw std::runtime_error("video input is not supported - hint: if this is unexpected, you may need to provide the mmproj");
- }
-
- // accept the OpenAI-style "video_url" key as an alias of "input_video"
- json input_video = json_value(p, type, json::object());
- std::string url = json_value(input_video, "data",
- json_value(input_video, "url", std::string()));
- handle_media(out_files, url, opt.media_path);
-
- p["type"] = "media_marker";
- p["text"] = get_media_marker();
- p.erase("input_video");
- p.erase("video_url");
-
- } else if (type != "text") {
- throw std::invalid_argument("unsupported content[].type");
- }
- }
+ oaicompat_content_load_media(content, opt, out_files);
}
auto caps = common_chat_templates_get_caps(opt.tmpls.get());
diff --git a/tools/server/server-common.h b/tools/server/server-common.h
index e00408cd4..8cb6b90da 100644
--- a/tools/server/server-common.h
+++ b/tools/server/server-common.h
@@ -300,6 +300,15 @@ std::vector<server_tokens> tokenize_input_prompts(
bool parse_special,
const mtmd_helper_init_opt & init_opt);
+// tokenize a single prompt, see tokenize_input_prompts() for the supported shapes
+server_tokens tokenize_input_subprompt(
+ const llama_vocab * vocab,
+ mtmd_context * mctx,
+ const json & json_prompt,
+ bool add_special,
+ bool parse_special,
+ const mtmd_helper_init_opt & init_opt);
+
//
// OAI utils
//
@@ -330,6 +339,16 @@ json oaicompat_chat_params_parse(
const server_chat_params & opt,
std::vector<raw_buffer> & out_files);
+// used by /embeddings endpoint, content has the same format as a chat message content array
+server_tokens tokenize_oai_content_array(
+ const llama_vocab * vocab,
+ mtmd_context * mctx,
+ const server_chat_params & opt,
+ json content,
+ bool add_special,
+ bool parse_special,
+ const mtmd_helper_init_opt & init_opt);
+
// TODO: move it to server-task.cpp
json format_embeddings_response_oaicompat(
const json & request,
diff --git a/tools/server/server-context.cpp b/tools/server/server-context.cpp
index efff54997..2036319c1 100644
--- a/tools/server/server-context.cpp
+++ b/tools/server/server-context.cpp
@@ -3193,7 +3193,9 @@ private:
return;
}
- if (slot.task->params.cache_prompt) {
+ const bool is_stateless_task = slot.task->type == SERVER_TASK_TYPE_EMBEDDING || slot.task->type == SERVER_TASK_TYPE_RERANK;
+
+ if (slot.task->params.cache_prompt && !is_stateless_task) {
// reuse any previously computed tokens that are common with the new prompt
n_past = slot.prompt.tokens.get_common_prefix(input_tokens);
@@ -5403,7 +5405,27 @@ std::unique_ptr<server_res_generator> server_routes::handle_embeddings_impl(cons
}
}
- auto tokenized_prompts = tokenize_input_prompts(ctx_server.vocab, ctx_server.mctx, prompt, true, true, ctx_server.init_opt);
+ // same shapes as tokenize_input_prompts(), plus OAI content: { "content": [ { "type": "text"|"image_url"|"input_audio"|"input_video", ... } ] }
+ auto tokenize_entry = [&](const json & p) {
+ if (p.is_object() && p.contains("content")) {
+ return tokenize_oai_content_array(ctx_server.vocab, ctx_server.mctx, meta->chat_params, p.at("content"), true, true, ctx_server.init_opt);
+ }
+ return tokenize_input_subprompt(ctx_server.vocab, ctx_server.mctx, p, true, true, ctx_server.init_opt);
+ };
+
+ std::vector<server_tokens> tokenized_prompts;
+ if (prompt.is_array() && !json_is_array_and_contains_numbers(prompt)) {
+ for (const auto & p : prompt) {
+ tokenized_prompts.push_back(tokenize_entry(p));
+ }
+ } else {
+ tokenized_prompts.push_back(tokenize_entry(prompt));
+ }
+ if (tokenized_prompts.empty()) {
+ res->error(format_error_response("\"input\" must not be empty", ERROR_TYPE_INVALID_REQUEST));
+ return res;
+ }
+
for (const auto & tokens : tokenized_prompts) {
// this check is necessary for models that do not add BOS token to the input
if (tokens.empty()) {
diff --git a/tools/server/tests/unit/test_embedding.py b/tools/server/tests/unit/test_embedding.py
index 17ba09554..c4a7d35fe 100644
--- a/tools/server/tests/unit/test_embedding.py
+++ b/tools/server/tests/unit/test_embedding.py
@@ -83,6 +83,10 @@ def test_embedding_multiple_with_fa():
(["string1", [12, 34, 56]], True),
([[12, 34, 56], [12, 34, 56]], True),
([[12, 34, 56], [12, "string", 34, 56]], True),
+ # object entries
+ ({"prompt_string": "string"}, False),
+ ({"content": [{"type": "text", "text": "string"}]}, False),
+ (["string1", {"prompt_string": "string2"}, {"content": [{"type": "text", "text": "string3"}]}], True),
]
)
def test_embedding_mixed_input(input, is_multi_prompt: bool):
@@ -101,6 +105,40 @@ def test_embedding_mixed_input(input, is_multi_prompt: bool):
assert len(data[0]['embedding']) > 1
+def test_embedding_content_text_same_as_string():
+ global server
+ server.pooling = 'last'
+ server.start()
+ res = server.make_request("POST", "/v1/embeddings", data={
+ "input": [
+ "hello world",
+ {"content": [{"type": "text", "text": "hello "}, {"type": "text", "text": "world"}]},
+ ],
+ })
+ assert res.status_code == 200
+ data = res.body['data']
+ assert data[0]['embedding'] == data[1]['embedding']
+
+
+@pytest.mark.parametrize(
+ "input",
+ [
+ [],
+ {"content": "string"},
+ {"content": [{"type": "unknown"}]},
+ # model is not multimodal
+ {"content": [{"type": "image_url", "image_url": {"url": "data:image/png;base64,AAAA"}}]},
+ {"content": [{"type": "input_audio", "input_audio": {"data": "AAAA", "format": "wav"}}]},
+ {"content": [{"type": "input_video", "input_video": {"url": "data:video/mp4;base64,AAAA"}}]},
+ ]
+)
+def test_embedding_invalid_input(input):
+ global server
+ server.start()
+ res = server.make_request("POST", "/v1/embeddings", data={"input": input})
+ assert res.status_code != 200
+
+
def test_embedding_pooling_mean():
global server
server.pooling = 'mean'
diff --git a/tools/server/tests/unit/test_vision_api.py b/tools/server/tests/unit/test_vision_api.py
index 3bf868e66..dd04e2e2b 100644
--- a/tools/server/tests/unit/test_vision_api.py
+++ b/tools/server/tests/unit/test_vision_api.py
@@ -179,3 +179,28 @@ def test_vision_embeddings(prompt, image_data, success):
assert content[0]['embedding'] != content[2]['embedding']
else:
assert res.status_code != 200
+
+
+def test_vision_embeddings_oai_content():
+ global server
+ server.server_embeddings = True
+ server.pooling = 'mean'
+ server.n_batch = 512
+ server.start()
+ res = server.make_request("POST", "/v1/embeddings", data={
+ "input": [
+ {"content": [
+ {"type": "text", "text": "What is this: "},
+ {"type": "image_url", "image_url": {"url": get_img_url("IMG_BASE64_URI_0")}},
+ {"type": "text", "text": "\n"},
+ ]},
+ {JSON_PROMPT_STRING_KEY: "What is this: <__media__>\n", JSON_MULTIMODAL_KEY: [get_img_url("IMG_BASE64_0")]},
+ "What is this: \n",
+ ],
+ })
+ assert res.status_code == 200
+ data = res.body["data"]
+ assert len(data) == 3
+ # same prompt and image in both formats
+ assert data[0]["embedding"] == data[1]["embedding"]
+ assert data[0]["embedding"] != data[2]["embedding"]