Commit 23b0202a1 for llama.cpp

commit 23b0202a189c44a54625aadcb37a946dd1d6278d
Author: Pascal <admin@serveurperso.com>
Date:   Sat Oct 10 21:25:38 2026 +0200

    server: leave a busy slot untouched when a request pins it (#30295)

    A request asking for a busy id_slot still ran the prompt cache update
    on that slot before being deferred. When the RAM cache held a better
    match, it was loaded into the slot while another request was still
    generating there, and that generation continued on the wrong context.

    The busy slot is now returned as is and the request waits for it.

diff --git a/tools/server/server-context.cpp b/tools/server/server-context.cpp
index 9a9c0c90d..44aaaddea 100644
--- a/tools/server/server-context.cpp
+++ b/tools/server/server-context.cpp
@@ -1686,6 +1686,11 @@ private:
         if (task.id_slot != -1) {
             ret = get_slot_by_id(task.id_slot);
             if (ret) {
+                // a busy slot is returned untouched, the caller defers the task
+                if (ret->is_processing()) {
+                    return ret;
+                }
+
                 SLT_INF(*ret, "selected slot by id (%d)\n", task.id_slot);
             }
         }
diff --git a/tools/server/tests/unit/test_completion.py b/tools/server/tests/unit/test_completion.py
index 09482b75c..b62abc661 100644
--- a/tools/server/tests/unit/test_completion.py
+++ b/tools/server/tests/unit/test_completion.py
@@ -237,6 +237,29 @@ def test_nocache_long_input_prompt():
     })
     assert res.status_code == 400

+
+# a request pinned to a busy slot leaves the generation running on it untouched
+def test_pinned_request_on_busy_slot():
+    global server
+    server.n_ctx = 4096
+    server.start()
+    story = "Once upon a time a dragon named Ember guarded a golden key in a deep cave. " * 8
+
+    def run(pin_busy: bool) -> str:
+        server.make_request("POST", "/completion", data={"prompt": story, "id_slot": 0, "n_predict": 4, "temperature": 0.0})
+        res = server.make_stream_request("POST", "/completion", data={
+            "prompt": "To bake bread, mix flour, water and salt, then",
+            "id_slot": 0, "n_predict": 1024, "ignore_eos": True, "temperature": 0.0, "stream": True,
+        })
+        content = next(res)["content"]
+        if pin_busy:
+            server.make_request("POST", "/completion", data={"prompt": story + "The dragon", "id_slot": 0, "n_predict": 4, "temperature": 0.0})
+        return content + "".join(chunk["content"] for chunk in res)
+
+    baseline = run(pin_busy=False)
+    assert run(pin_busy=True) == baseline
+
+
 def test_json_prompt_no_mtmd():
     global server
     server.start()