Commit 9871df591 for llama.cpp

commit 9871df5911a03a813518bcd623ee2eba91bf32aa
Author: Xuan-Son Nguyen <son@huggingface.co>
Date:   Mon Oct 5 14:02:40 2026 +0200

    server: support vision input for Clef (#29969)

    * server: support vision input for Clef

    * move input_attn_causal to private

    * extend old server_batch::embd

    * server_batch::token::pos to multi dim

    * nits

    * fix abort

    * fix img tokens cap

    * fix yield_to_queue mutate data

diff --git a/conversion/clef.py b/conversion/clef.py
index a6f724394..b793d92c8 100644
--- a/conversion/clef.py
+++ b/conversion/clef.py
@@ -11,8 +11,9 @@ import torch
 if TYPE_CHECKING:
     from torch import Tensor

-from .base import MmprojModel, ModelBase, gguf, logger
+from .base import ModelBase, gguf, logger
 from .qwen import Qwen3_5TextModel
+from .qwen3vl import Qwen3VLVisionModel


 def _is_clef_checkpoint(dir_model: Path) -> bool:
@@ -65,6 +66,7 @@ class ClefModel(Qwen3_5TextModel):

         # the pieces of the prompt are tokenized one by one, the server gives the text that separates them (sep)
         # and the text that starts the span of a question or of an option (mark_question, mark_option)
+        # images is one media marker per image, the vision start and end tokens are added by the server
         # the keys of JSON objects are given in sorted order
         option = (
             "{% set d = o.description %}"
@@ -75,6 +77,7 @@ class ClefModel(Qwen3_5TextModel):
         )
         return (
             text(f"<|im_start|>system\n{cls._SYSTEM_PROMPT}<|im_end|>\n<|im_start|>user\nSTATE:\n")
+            + "{% if images %}{{ sep }}{% for image in images %}{{ image }}{% endfor %}" + text("\n") + "{% endif %}"
             + "{{ sep }}" + render("state")
             + "{{ sep }}" + text("\n\nSCHEMA FIELDS:\n")
             + "{% for q in questions %}"
@@ -143,8 +146,5 @@ class ClefModel(Qwen3_5TextModel):


 @ModelBase.register("ClefModel")
-class ClefVisionModel(MmprojModel):
-    def __init__(self, *args, **kwargs):
-        del args, kwargs
-        raise NotImplementedError(
-            "multimodal input is not supported yet for Clef, requires https://github.com/ggml-org/llama.cpp/pull/29622 to be merged first")
+class ClefVisionModel(Qwen3VLVisionModel):
+    pass
diff --git a/src/models/clef.cpp b/src/models/clef.cpp
index 1911efa7e..81e4ad65e 100644
--- a/src/models/clef.cpp
+++ b/src/models/clef.cpp
@@ -125,7 +125,7 @@ static clef_spans clef_get_spans(const llama_ubatch & ubatch) {
     std::vector<bool> has_option;

     // TODO: support multiple sequences
-    bool ok = ubatch.decision_order != nullptr && ubatch.n_seqs_unq == 1;
+    bool ok = ubatch.decision_order != nullptr && ubatch.token != nullptr && ubatch.n_seqs_unq == 1;

     const int32_t n_tokens = ubatch.n_tokens;
     for (int32_t i = 0; ok && i < n_tokens;) {
@@ -156,6 +156,11 @@ static clef_spans clef_get_spans(const llama_ubatch & ubatch) {
         i = end;
     }

+    // the head reads the token ids of the spans, they cannot be embeddings
+    for (int32_t i = 0; ok && ubatch.is_mixed() && i < n_tokens; i++) {
+        ok = !ubatch.type[i] || ubatch.decision_order[i] == LLAMA_DECISION_ORDER_NONE;
+    }
+
     // each question needs an option
     res.valid = ok && !res.questions.empty() && std::find(has_option.begin(), has_option.end(), false) == has_option.end();
     if (!res.valid) {
@@ -177,8 +182,9 @@ public:
     }

     void set_input(const llama_ubatch * ubatch) override {
-        GGML_ASSERT(ubatch->token);
-        ggml_backend_tensor_set(tokens, ubatch->token, 0, n_tokens * sizeof(llama_token));
+        // a batch of embeddings has no token ids, and no usable spans
+        const std::vector<llama_token> no_tokens(ubatch->token ? 0 : n_tokens, 0);
+        ggml_backend_tensor_set(tokens, ubatch->token ? ubatch->token : no_tokens.data(), 0, n_tokens * sizeof(llama_token));

         const auto spans = clef_get_spans(*ubatch);
         GGML_ASSERT(spans.questions.size() == n_questions && spans.options.size() == n_options);
@@ -308,12 +314,35 @@ llama_model_clef::graph::graph(const llama_model & model_base, const llm_graph_p
     ggml_build_forward_expand(gf, cur);
 }

-// same as build_attn_inp_no_cache(), but the mask is causal even if the batch is processed by the encoder path
-llm_graph_input_attn_no_cache * llama_model_clef::graph::build_attn_inp_causal() {
-    llama_cparams cparams_causal = cparams;
-    cparams_causal.causal_attn = true;
+// causal mask by batch order: the tokens of an image share the same position
+class llm_graph_input_attn_clef : public llm_graph_input_attn_no_cache {
+public:
+    using llm_graph_input_attn_no_cache::llm_graph_input_attn_no_cache;

-    auto inp = std::make_unique<llm_graph_input_attn_no_cache>(hparams, cparams_causal);
+    void set_input(const llama_ubatch * ubatch) override {
+        const int64_t n_tokens = ubatch->n_tokens;
+
+        const auto fill_mask = [&](auto * data, auto zero, auto ninf) {
+            for (int64_t i1 = 0; i1 < n_tokens; ++i1) {
+                for (int64_t i0 = 0; i0 < n_tokens; ++i0) {
+                    const bool visible = i0 <= i1 && ubatch->seq_id[i0][0] == ubatch->seq_id[i1][0];
+                    data[i1 * n_tokens + i0] = visible ? zero : ninf;
+                }
+            }
+        };
+
+        GGML_ASSERT(ggml_backend_buffer_is_host(self_kq_mask->buffer));
+        if (self_kq_mask->type == GGML_TYPE_F16) {
+            fill_mask((ggml_fp16_t *) self_kq_mask->data, ggml_fp32_to_fp16(0.0f), ggml_fp32_to_fp16(-INFINITY));
+        } else {
+            fill_mask((float *) self_kq_mask->data, 0.0f, -INFINITY);
+        }
+    }
+};
+
+// same as build_attn_inp_no_cache(), with a causal mask even if the batch is processed by the encoder path
+llm_graph_input_attn_no_cache * llama_model_clef::graph::build_attn_inp_causal() {
+    auto inp = std::make_unique<llm_graph_input_attn_clef>(hparams, cparams);

     const auto type_mask = cparams.flash_attn ? GGML_TYPE_F16 : GGML_TYPE_F32;

diff --git a/tools/server/README.md b/tools/server/README.md
index 4a37bc2f3..d2d6ab2be 100644
--- a/tools/server/README.md
+++ b/tools/server/README.md
@@ -1696,7 +1696,7 @@ For laya and clef, the whole prompt is evaluated in one batch: it must fit in `-

 *Image input:*

-Image input needs a model that supports it (for example: openjev) and its multimodal projector, see `--mmproj`. Image input is not supported yet for clef.
+Image input needs a model that supports it (for example: openjev, clef) and its multimodal projector, see `--mmproj`.

 Images can be given in two ways, and both can be used in the same request:

diff --git a/tools/server/server-context.cpp b/tools/server/server-context.cpp
index 809be94cd..13be63acd 100644
--- a/tools/server/server-context.cpp
+++ b/tools/server/server-context.cpp
@@ -140,10 +140,11 @@ struct server_batch {
     struct token {
         int32_t id_slot;
         llama_token token;
-        llama_pos pos;
+        std::array<llama_pos, GGML_MROPE_SECTIONS> pos; // only pos[0] is used for text tokens
         bool output;
         bool is_prompt; // for stats tracking
         int32_t decision_order = 0;
+        int32_t i_embd = -1; // row in embd, -1 if this is a token
     };
     std::vector<token> tokens;
     int32_t n_tokens_alloc = 0;
@@ -152,8 +153,8 @@ struct server_batch {
     // track if given slot can be batched with slots already in the batch
     server_slot * slot_batched = nullptr;

-    bool has_embd = false;
-    std::vector<float> embd;
+    // embedding entries, they can be mixed with tokens if the context supports it
+    std::vector<float> embd; // n_embd per row

     float  alora_scale       = -1.0f;
     size_t alora_disabled_id = 0;
@@ -166,24 +167,28 @@ struct server_batch {
     }

     bool add(int32_t id_slot, llama_token token, llama_pos pos, bool output, bool is_prompt) {
-        GGML_ASSERT(!has_embd); // cannot mix tokens + embd in same batch
         if ((int32_t)tokens.size() >= n_tokens_alloc) {
             return false;
         }
-        tokens.push_back({ id_slot, token, pos, output, is_prompt });
+        tokens.push_back({ id_slot, token, { pos, 0, 0, 0 }, output, is_prompt });
         return true;
     }

-    bool add(int32_t id_slot, const std::vector<float> & embd_in, llama_pos pos, bool output, bool is_prompt) {
+    // embd_in has n_embd values, pos has GGML_MROPE_SECTIONS values
+    bool add_embd(int32_t id_slot, const float * embd_in, const llama_pos * pos, bool output, bool is_prompt) {
         if ((int32_t)tokens.size() >= n_tokens_alloc) {
             return false;
         }
-        tokens.push_back({ id_slot, LLAMA_TOKEN_NULL, pos, output, is_prompt });
-        has_embd = true;
-        embd.insert(embd.end(), embd_in.begin(), embd_in.end());
+        tokens.push_back({ id_slot, LLAMA_TOKEN_NULL, { pos[0], pos[1], pos[2], pos[3] }, output, is_prompt });
+        tokens.back().i_embd = (int32_t) (embd.size() / n_embd);
+        embd.insert(embd.end(), embd_in, embd_in + n_embd);
         return true;
     }

+    bool has_embd() const {
+        return !embd.empty();
+    }
+
     void clear() {
         tokens.clear();
         embd.clear();
@@ -191,13 +196,24 @@ struct server_batch {
         slot_batched      = nullptr;
         alora_scale       = -1.0f;
         alora_disabled_id = 0;
-        has_embd          = false;
     }

     int32_t size() const {
         return (int32_t)tokens.size();
     }

+    // remove the entries after the first n
+    void truncate(int32_t n) {
+        GGML_ASSERT(n >= 0 && n <= size());
+        for (int32_t i = n; i < size(); i++) {
+            if (tokens[i].i_embd >= 0) {
+                embd.resize((size_t) tokens[i].i_embd * n_embd);
+                break;
+            }
+        }
+        tokens.resize(n);
+    }
+
     void set_output(int32_t idx, bool output) {
         GGML_ASSERT(idx >= 0 && idx < (int32_t)tokens.size());
         tokens[idx].output = output;
@@ -216,12 +232,10 @@ struct server_batch {
         view.clear();
         for (int32_t i = off; i < off + n_tokens; i++) {
             const auto & t = tokens[i];
-            if (has_embd) {
-                // text embeddings broadcast the same position across the M-RoPE sections
-                const llama_pos pos[GGML_MROPE_SECTIONS] = { t.pos, t.pos, t.pos, 0 };
-                view.add_embd({ embd.data() + (size_t) i * n_embd, 1, (size_t) n_embd }, pos, t.id_slot, t.output);
+            if (t.i_embd >= 0) {
+                view.add_embd({ embd.data() + (size_t) t.i_embd * n_embd, 1, (size_t) n_embd }, t.pos.data(), t.id_slot, t.output);
             } else {
-                view.add(t.token, t.pos, t.id_slot, t.output);
+                view.add(t.token, t.pos[0], t.id_slot, t.output);
             }
             view.tokens.back().decision_order = t.decision_order;
         }
@@ -529,7 +543,11 @@ struct server_slot {
             i_batch = batch.size();

             if (!inp_embd.empty()) {
-                add_ok &= batch.add(id, inp_embd, prompt.tokens.pos_next(), true, false);
+                // text embeddings broadcast the same position across the M-RoPE sections
+                const llama_pos p = prompt.tokens.pos_next();
+                const llama_pos pos[GGML_MROPE_SECTIONS] = { p, p, p, 0 };
+                GGML_ASSERT((int32_t) inp_embd.size() == batch.n_embd);
+                add_ok &= batch.add_embd(id, inp_embd.data(), pos, true, false);
             } else {
                 add_ok &= batch.add(id, sampled, prompt.tokens.pos_next(), true, false);
             }
@@ -762,6 +780,44 @@ struct server_slot {
     }
 };

+// encode the mtmd chunk at idx, batched with as many of the next chunks as possible
+// returns 0 on success
+static int encode_mtmd_chunk(const server_slot & slot, mtmd::batch_ptr & mbatch, size_t idx) {
+    const auto & mctx = slot.mctx;
+    const auto & input_tokens = slot.task->tokens;
+    const auto & chunk = input_tokens.find_chunk(idx);
+
+    mbatch.reset(mtmd_batch_init(mctx));
+    int32_t res = mtmd_batch_add_chunk(mbatch.get(), chunk.get());
+    GGML_ASSERT(res == 0); // we should never have an empty batch
+
+    // try batching as much as possible
+    int n_added = 1;
+    size_t idx_cur = idx;
+    while (res == 0) {
+        auto [next_chunk, next_idx] = input_tokens.find_next_media_chunk(idx_cur);
+        if (next_chunk == nullptr) {
+            break;
+        }
+        res = mtmd_batch_add_chunk(mbatch.get(), next_chunk->get());
+        n_added += (res == 0 ? 1 : 0);
+        idx_cur = next_idx;
+        SLT_DBG(slot, "try adding media chunk idx = %zu to batch, res = %d\n", next_idx, res);
+        // if res != 0, batch is full or chunk is not compatible -> this loop breaks
+    }
+
+    // TODO @ngxson : move this log line to debug when it become more stable
+    SLT_TRC(slot, "encoding mtmd batch from idx = %zu, n_chunks = %d\n", idx, n_added);
+
+    res = mtmd_batch_encode(mbatch.get());
+    if (res != 0) {
+        SLT_ERR(slot, "failed to encode mtmd batch for chunk idx = %zu, res = %d\n", idx, res);
+        return -1;
+    }
+
+    return 0;
+}
+
 // returns 0 on success
 // caller need to update prompt.tokens after a successful call to keep track of the processing progress
 // note: this is not a member of server_slot because we want to run it inside yield_to_queue
@@ -834,37 +890,41 @@ static int process_mtmd_chunk(const server_slot & slot, mtmd::batch_ptr & mbatch

     // otherwise, the batch is either uninitialized or is used up
     // we need to create & encode a new batch
-    mbatch.reset(mtmd_batch_init(mctx));
-    res = mtmd_batch_add_chunk(mbatch.get(), chunk.get());
-    GGML_ASSERT(res == 0); // we should never have an empty batch
-
-    // try batching as much as possible
-    int n_added = 1;
-    size_t idx_cur = idx;
-    while (res == 0) {
-        auto [next_chunk, next_idx] = input_tokens.find_next_media_chunk(idx_cur);
-        if (next_chunk == nullptr) {
-            break;
-        }
-        res = mtmd_batch_add_chunk(mbatch.get(), next_chunk->get());
-        n_added += (res == 0 ? 1 : 0);
-        idx_cur = next_idx;
-        SLT_DBG(slot, "try adding media chunk idx = %zu to batch, res = %d\n", next_idx, res);
-        // if res != 0, batch is full or chunk is not compatible -> this loop breaks
-    }
-
-    // TODO @ngxson : move this log line to debug when it become more stable
-    SLT_TRC(slot, "encoding mtmd batch from idx = %zu, n_chunks = %d\n", idx, n_added);
-
-    res = mtmd_batch_encode(mbatch.get());
-    if (res != 0) {
-        SLT_ERR(slot, "failed to encode mtmd batch for chunk idx = %zu, res = %d\n", idx, res);
+    if (encode_mtmd_chunk(slot, mbatch, idx) != 0) {
         return -1;
     }

     return try_decode();
 }

+// add an encoded mtmd chunk to the batch of the text tokens, instead of decoding it on its own like process_mtmd_chunk()
+// embd is the output of the encoder for this chunk
+// returns false if the batch is full
+static bool add_mtmd_chunk(const server_slot & slot, const mtmd_input_chunk * chunk, const float * embd, server_batch & batch) {
+    // positions are the ones of mtmd_helper_decode_image_chunk()
+    const auto * image    = mtmd_input_chunk_get_tokens_image(chunk);
+    const bool   is_mrope = mtmd_decode_use_mrope(slot.mctx);
+    const size_t n_tokens = mtmd_input_chunk_get_n_tokens(chunk);
+    const size_t n_embd   = batch.n_embd;
+    const llama_pos pos_0 = slot.prompt.tokens.pos_next();
+
+    for (size_t i = 0; i < n_tokens; i++) {
+        const llama_pos p = pos_0 + (llama_pos) i;
+        llama_pos pos[GGML_MROPE_SECTIONS] = { p, p, p, p };
+        if (is_mrope && image) {
+            const mtmd_decoder_pos rel = mtmd_image_tokens_get_decoder_pos(image, pos_0, i);
+            pos[0] = rel.t;
+            pos[1] = rel.y;
+            pos[2] = rel.x;
+            pos[3] = rel.z;
+        }
+        if (!batch.add_embd(slot.id, embd + i * n_embd, pos, slot.need_embd(), /* is_prompt */ true)) {
+            return false;
+        }
+    }
+    return true;
+}
+
 //
 // server_context_impl (private implementation)
 //
@@ -1215,14 +1275,20 @@ private:
                 mtmd_helper_log_set(common_log_default_callback, nullptr);
             }

-            // non-causal models need the whole image in one ubatch
+            // the image must fit in one ubatch if the model needs non-causal attention on it
+            // a non-causal model also has the text of the prompt in that ubatch, leave half of it to the text
             {
                 const int n_ubatch = llama_n_ubatch(ctx_tgt);
-                if (mmproj_usage.use_non_causal && mmproj_usage.image_max_tokens > n_ubatch) {
-                    SRV_WRN("cap image_max_tokens (original=%d) to n_ubatch (%d) because model needs non-causal attention on image\n", mmproj_usage.image_max_tokens, n_ubatch);
-                    SRV_WRN("%s\n", "increase n_ubatch (-ub) to increase vision token budget");
-                    mparams.image_max_tokens = n_ubatch;
-                    mparams.image_min_tokens = std::min(mparams.image_min_tokens, n_ubatch);
+                const bool is_mixed = use_mixed_batch();
+                if (is_mixed || mmproj_usage.use_non_causal) {
+                    const int n_max = is_mixed ? n_ubatch / 2 : n_ubatch;
+                    if (mmproj_usage.image_max_tokens > n_max) {
+                        SRV_WRN("cap image_max_tokens (original=%d) to %d (n_ubatch = %d) because %s\n", mmproj_usage.image_max_tokens, n_max, n_ubatch,
+                                is_mixed ? "the model processes the prompt in one ubatch" : "model needs non-causal attention on image");
+                        SRV_WRN("%s\n", "increase n_ubatch (-ub) to increase vision token budget");
+                        mparams.image_max_tokens = n_max;
+                        mparams.image_min_tokens = std::min(mparams.image_min_tokens, n_max);
+                    }
                 }
             }

@@ -3654,6 +3720,13 @@ private:
                             ctx_tgt_seq_rm_type == COMMON_CONTEXT_SEQ_RM_TYPE_RS ||
                             n_swa > 0);

+                    // TODO: do the same for all models, then remove process_mtmd_chunk()
+                    if (use_mixed_batch() && !slot.can_split() && input_tokens.has_mtmd) {
+                        if (!add_prompt_mixed(slot)) {
+                            return;
+                        }
+                    }
+
                     bool has_mtmd = false;

                     // check if we should process the mtmd chunk
@@ -3839,6 +3912,64 @@ private:
         }
     }

+    // see https://github.com/ggml-org/llama.cpp/pull/29969
+    // TODO @ngxson : maybe remove this once we use "mixed" batch everywhere
+    bool use_mixed_batch() const {
+        return !llama_get_memory(ctx_tgt) || !llama_get_causal_attn(ctx_tgt);
+    }
+
+    // add the rest of the prompt to the batch, the mtmd chunks are added as embeddings next to the text tokens
+    // the caller makes sure that it fits in the batch
+    // returns false on error, the slot is then released
+    bool add_prompt_mixed(server_slot & slot) {
+        const auto & input_tokens = slot.task->tokens;
+        const auto n_tokens_prev = batch.size();
+
+        while (slot.prompt.n_tokens() < slot.task->n_tokens()) {
+            const auto cur_token_idx = slot.prompt.n_tokens();
+            const llama_token cur_tok = input_tokens[cur_token_idx];
+
+            if (cur_tok != LLAMA_TOKEN_NULL) {
+                const bool add_ok = batch.add(slot.id,
+                    cur_tok,
+                    /* pos       = */ slot.prompt.tokens.pos_next(),
+                    /* output    = */ slot.need_embd(),
+                    /* is_prompt = */ true);
+                GGML_ASSERT(add_ok);
+                if (!slot.task->decision.order.empty()) {
+                    batch.set_decision_order(batch.size() - 1, slot.task->decision.order[cur_token_idx]);
+                }
+                slot.prompt.tokens.push_back(cur_tok);
+                continue;
+            }
+
+            const auto & chunk = input_tokens.find_chunk(cur_token_idx);
+
+            float * embd = slot.mbatch ? mtmd_batch_get_output_embd(slot.mbatch.get(), chunk.get()) : nullptr;
+            if (!embd) {
+                // encode on the worker thread, so we can still handle metrics tasks
+                int32_t res = 0;
+                queue_tasks.yield_to_queue([&]() {
+                    res = encode_mtmd_chunk(slot, slot.mbatch, cur_token_idx);
+                });
+                embd = res == 0 ? mtmd_batch_get_output_embd(slot.mbatch.get(), chunk.get()) : nullptr;
+            }
+
+            if (!embd || !add_mtmd_chunk(slot, chunk.get(), embd, batch)) {
+                SLT_ERR(slot, "%s", "failed to process mtmd chunk\n");
+                // the batch must not keep the entries of a released slot
+                batch.truncate(n_tokens_prev);
+                send_error(slot, "failed to process mtmd chunk", ERROR_TYPE_SERVER);
+                slot.release();
+                return false;
+            }
+
+            slot.prompt.tokens.push_back_placeholder(chunk.get());
+        }
+
+        return true;
+    }
+
     // returns true = success ; false = retry with smaller batch size
     // throw std::runtime_error on fatal error
     bool decode(int32_t & n_batch, int32_t off) {
@@ -3860,7 +3991,7 @@ private:

         // TODO @ngxson : dft model may have different n_embd than the tgt model, so we check & reject if that's the case
         // this case is not currently used by any models, but may need to be supported in the future
-        if (spec && batch.has_embd) {
+        if (spec && batch.has_embd()) {
             if (llama_model_n_embd_inp(model_dft) != llama_model_n_embd_inp(model_tgt)) {
                 SRV_ERR("%s", "unsupported batch.has_embd + spec case\n");
                 throw std::runtime_error("unsupported batch.has_embd + spec case");
@@ -5468,7 +5599,7 @@ void server_routes::init_routes() {
             if (decision.is_joint()) {
                 server_task task = server_task(SERVER_TASK_TYPE_DECISION);
                 task.id = rd.get_new_id();
-                decision.fill_task_joint(state, questions, task);
+                decision.fill_task_joint(state, questions, files, ctx_server.mctx, ctx_server.init_opt, task);
                 tasks.push_back(std::move(task));
             } else {
                 for (const auto & question : questions) {
diff --git a/tools/server/server-decision.cpp b/tools/server/server-decision.cpp
index 522aa2bed..cc9533182 100644
--- a/tools/server/server-decision.cpp
+++ b/tools/server/server-decision.cpp
@@ -223,6 +223,9 @@ static void decision_load_image(const json & url, std::vector<raw_buffer> & file
 }

 json server_decision_context::parse_state(const json & body, std::vector<raw_buffer> & files) const {
+    if (body.contains("videos") && !body.at("videos").is_null() && !body.at("videos").empty()) {
+        throw std::invalid_argument("\"videos\" is not supported");
+    }
     if (body.contains("images") && !body.at("images").is_null()) {
         if (!body.at("images").is_array()) {
             throw std::invalid_argument("\"images\" must be an array");
@@ -563,7 +566,13 @@ static const std::string CLEF_SEP           = "<<clef:sep>>";
 static const std::string CLEF_MARK_QUESTION = "<<clef:question>>";
 static const std::string CLEF_MARK_OPTION   = "<<clef:option>>";

-void server_decision_context::fill_task_joint(const json & state, const std::vector<server_decision_question> & questions, server_task & task) const {
+void server_decision_context::fill_task_joint(
+        const json & state,
+        const std::vector<server_decision_question> & questions,
+        const std::vector<raw_buffer> & files,
+        mtmd_context * mctx,
+        const mtmd_helper_init_opt & init_opt,
+        server_task & task) const {
     json inp_questions = json::array();
     for (const auto & question : questions) {
         json options = json::array();
@@ -587,6 +596,16 @@ void server_decision_context::fill_task_joint(const json & state, const std::vec
         {"questions", inp_questions},
     };
     inp = decision_replace_text(decision_sort_keys(inp), CLEF_MARKER, "<<clef ");
+
+    // the template puts one media marker per image
+    json images = json::array();
+    if (!files.empty()) {
+        inp = decision_replace_text(inp, get_media_marker(), " ");
+        for (size_t i = 0; i < files.size(); i++) {
+            images.push_back(get_media_marker());
+        }
+    }
+    inp["images"]        = images;
     inp["sep"]           = CLEF_SEP;
     inp["mark_question"] = CLEF_MARK_QUESTION;
     inp["mark_option"]   = CLEF_MARK_OPTION;
@@ -597,15 +616,34 @@ void server_decision_context::fill_task_joint(const json & state, const std::vec
     const jinja::value results = runtime.execute(tmpl->prog);
     const std::string prompt   = jinja::runtime::gather_string_parts(results)->as_string().str();

+    const auto invalid = std::runtime_error("unexpected layout of the decision prompt");
+
     // the model was trained with the pieces tokenized one by one
-    llama_tokens tokens;
+    const std::vector<std::string> pieces = string_split(prompt, CLEF_SEP);
+    size_t i_piece = 0;
+
+    task.tokens = server_tokens(llama_tokens(), false);
+    if (!files.empty()) {
+        // the text before the images is tokenized with them, a vision token separates the pieces anyway
+        std::string head;
+        while (i_piece < pieces.size() && head.find(get_media_marker()) == std::string::npos) {
+            head += pieces[i_piece++];
+        }
+        if (head.find(get_media_marker()) == std::string::npos || head.find(CLEF_MARKER) != std::string::npos) {
+            throw invalid;
+        }
+        task.tokens = process_mtmd_prompt(mctx, head, files, init_opt);
+        task.decision.order.resize(task.tokens.size(), LLAMA_DECISION_ORDER_NONE);
+    }
+
     size_t i_question = 0;
-    for (std::string piece : string_split(prompt, CLEF_SEP)) {
+    for (; i_piece < pieces.size(); i_piece++) {
+        std::string piece = pieces[i_piece];
         int32_t order = LLAMA_DECISION_ORDER_NONE;
         if (string_starts_with(piece, CLEF_MARK_QUESTION)) {
             piece = piece.substr(CLEF_MARK_QUESTION.size());
             if (i_question >= questions.size()) {
-                throw std::runtime_error("unexpected layout of the decision prompt");
+                throw invalid;
             }
             switch (questions[i_question++].type) {
                 case SERVER_DECISION_QUESTION_NOUL:   order = LLAMA_DECISION_ORDER_QUESTION_NOUL;   break;
@@ -622,8 +660,10 @@ void server_decision_context::fill_task_joint(const json & state, const std::vec
         if (order != LLAMA_DECISION_ORDER_NONE && piece_tokens.empty()) {
             throw std::invalid_argument("the instructions and the options of a question must not be empty");
         }
-        tokens.insert(tokens.end(), piece_tokens.begin(), piece_tokens.end());
-        task.decision.order.resize(tokens.size(), order);
+        for (const llama_token token : piece_tokens) {
+            task.tokens.push_back(token);
+        }
+        task.decision.order.resize(task.tokens.size(), order);
     }

     size_t n_options = 0;
@@ -631,10 +671,8 @@ void server_decision_context::fill_task_joint(const json & state, const std::vec
         n_options += question.options.size();
     }
     if (i_question != questions.size() || (size_t) task.decision.n_scores != n_options) {
-        throw std::runtime_error("unexpected layout of the decision prompt");
+        throw invalid;
     }
-
-    task.tokens = server_tokens(tokens, false);
 }

 //
diff --git a/tools/server/server-decision.h b/tools/server/server-decision.h
index 93480ba26..ade52fd00 100644
--- a/tools/server/server-decision.h
+++ b/tools/server/server-decision.h
@@ -54,10 +54,10 @@ struct server_decision_context {
     }

     // true if the prompt of the model has a place for images
-    // TODO: clef needs token and embedding entries in the same batch, see https://github.com/ggml-org/llama.cpp/pull/29622
     bool can_use_images() const {
         switch (type) {
             case COMMON_DECISION_TYPE_OPENJEV:
+            case COMMON_DECISION_TYPE_CLEF:
                 return true;
             default:
                 return false;
@@ -87,7 +87,14 @@ struct server_decision_context {
             server_task & task) const;

     // set the prompt of all the questions, the result has the scores of all their options, in order
-    void fill_task_joint(const json & state, const std::vector<server_decision_question> & questions, server_task & task) const;
+    // mctx is only used if there are files
+    void fill_task_joint(
+            const json & state,
+            const std::vector<server_decision_question> & questions,
+            const std::vector<raw_buffer> & files,
+            mtmd_context * mctx,
+            const mtmd_helper_init_opt & init_opt,
+            server_task & task) const;

     // scores: the raw model outputs of each variant
     json format_answer(const server_decision_question & question, const std::vector<std::vector<float>> & scores) const;