Commit e94acad85 for llama.cpp

commit e94acad853c0d11a3b47974e3167ec12f263d86a
Author: Pascal <admin@serveurperso.com>
Date:   Fri Oct 9 12:27:32 2026 +0200

    ci: fix Models Backend webgpu by supporting GGML_OP_DUP (#30216)

    The mixed batch path of PR-29622 writes the token rows with set_rows
    into a dup of the embeddings. WebGPU did not support DUP, so the dup
    ran on the CPU while the set_rows writing into it was scheduled on
    WebGPU, which then bound a CPU buffer and crashed. DUP is the same copy
    as CPY and CONT and now goes through the same path.

diff --git a/ggml/src/ggml-webgpu/ggml-webgpu.cpp b/ggml/src/ggml-webgpu/ggml-webgpu.cpp
index 11a4fc46f..b14a1fe2b 100644
--- a/ggml/src/ggml-webgpu/ggml-webgpu.cpp
+++ b/ggml/src/ggml-webgpu/ggml-webgpu.cpp
@@ -3356,6 +3356,7 @@ static std::optional<webgpu_encoded_op> ggml_webgpu_encode(webgpu_context ctx,
         case GGML_OP_TRANSPOSE:
         case GGML_OP_RESHAPE:
             return std::nullopt;
+        case GGML_OP_DUP:
         case GGML_OP_CPY:
         case GGML_OP_CONT:
             return ggml_webgpu_cpy(ctx, src0, node);
@@ -4419,6 +4420,7 @@ static bool ggml_backend_webgpu_device_supports_op(ggml_backend_dev_t dev, const
             supports_op = (src0->type == GGML_TYPE_F32 || src0->type == GGML_TYPE_F16 || src0->type == GGML_TYPE_I32 ||
                            src0->type == GGML_TYPE_I16);
             break;
+        case GGML_OP_DUP:
         case GGML_OP_CPY:
         case GGML_OP_CONT:
             supports_op = (op->type == GGML_TYPE_F16 || op->type == GGML_TYPE_F32 || op->type == GGML_TYPE_I32) &&