Commit 631109b34 for llama.cpp

commit 631109b34da437a3c4a5ebd75091d677671392e3
Author: Georgi Gerganov <ggerganov@gmail.com>
Date:   Fri Oct 2 11:08:08 2026 +0300

    ggml : add `alloc_buffer_n` to buffer type interface (#23671)

    * ggml : add `alloc_buffer_n` to buffer type interface

    Add alloc_buffer_n method to ggml_backend_buffer_type_i
    interface, with a public API ggml_backend_buft_alloc_buffer_n.

    - Default implementation in ggml-backend.cpp handles multi-buffer
      splitting and tensor allocation via ggml_tallocr
    - Meta buffer type provides custom implementation that creates
      per-device sub-contexts and delegates to simple buffer types
    - ggml_backend_alloc_ctx_tensors_from_buft now collects tensors
      into a list and delegates to the new API
    - Remove temporary ggml_backend_meta_alloc_ctx_tensors_from_buft
    - Add NULL alloc_buffer_n to all existing buffer type
      interfaces (cpu, metal, openvino, hexagon, webgpu, zdnn, virtgpu, repack)

    Assisted-by: llama.cpp:local pi

    * cont : fix `cur_buf_size` init after flushing a buffer

    * ggml : add TODO tag for shared buffer split logic

    Assisted-by: pi:llama.cpp/DeepSeek-V4-Flash-Vision-Exp

    * tests : add alloc_buffer_n coverage

    Assisted-by: pi:llama.cpp/DeepSeek-V4-Flash-Vision-Exp

    * cont : fix compile warnings

    * tests : add descriptions for alloc_buffer_n tests

    Assisted-by: pi:llama.cpp/Qwen3.8-27B

    * ggml : address review comments on alloc_buffer_n

    - restore GGML_LOG_ERROR on buffer alloc / tensor init failure in the
      default impl (name the failing tensor)
    - check the malloc result and drop the _impl indirection in
      ggml_backend_alloc_ctx_tensors_from_buft
    - remove comments that restate the code
    - fix the TAG_ALLOC_SHARED_BUFFER_SPLIT typo

    Assisted-by: pi:llama.cpp/Qwen3.8-27B

    * ggml : add get_alloc_size_n to buffer type interface

    - Add ggml_backend_buft_get_alloc_size_n public API
    - Add optional get_alloc_size_n callback to ggml_backend_buffer_type_i
    - Share tensor->buffer planning between alloc_buffer_n default and get_alloc_size_n default
    - Replace unchecked realloc with std::vector in alloc_buffer_n default
    - Make ggml_backend_alloc_ctx_tensors_from_buft_size use the new API
    - Add test-alloc coverage for get_alloc_size_n

    Assisted-by: pi:llama.cpp/DeepSeek-V4-Flash-Vision-Exp

    * cont : report malloc failure

diff --git a/ggml/include/ggml-alloc.h b/ggml/include/ggml-alloc.h
index a7926a21a..81cce2e2a 100644
--- a/ggml/include/ggml-alloc.h
+++ b/ggml/include/ggml-alloc.h
@@ -75,9 +75,10 @@ GGML_API size_t ggml_gallocr_get_buffer_size(ggml_gallocr_t galloc, int buffer_i

 // Utils
 // Create a buffer and allocate all the tensors in a ggml_context
-// ggml_backend_alloc_ctx_tensors_from_buft_size returns the size of the buffer that would be allocated by ggml_backend_alloc_ctx_tensors_from_buft
-// ggml_backend_alloc_ctx_tensors_from_buft returns NULL on failure or if all tensors in ctx are already allocated or zero-sized
+
+// returns the size of the buffer that would be allocated by ggml_backend_alloc_ctx_tensors_from_buft. returns 0 on failure
 GGML_API size_t                       ggml_backend_alloc_ctx_tensors_from_buft_size(struct ggml_context * ctx, ggml_backend_buffer_type_t buft);
+// returns NULL on failure or if all tensors in ctx are already allocated or zero-sized
 GGML_API struct ggml_backend_buffer * ggml_backend_alloc_ctx_tensors_from_buft(struct ggml_context * ctx, ggml_backend_buffer_type_t buft);
 GGML_API struct ggml_backend_buffer * ggml_backend_alloc_ctx_tensors(struct ggml_context * ctx, ggml_backend_t backend);

diff --git a/ggml/include/ggml-backend.h b/ggml/include/ggml-backend.h
index 2a0957436..5ef80da30 100644
--- a/ggml/include/ggml-backend.h
+++ b/ggml/include/ggml-backend.h
@@ -34,13 +34,15 @@ extern "C" {
     // Backend buffer type
     //

-    GGML_API const char *          ggml_backend_buft_name          (ggml_backend_buffer_type_t buft);
-    GGML_API ggml_backend_buffer_t ggml_backend_buft_alloc_buffer  (ggml_backend_buffer_type_t buft, size_t size);
-    GGML_API size_t                ggml_backend_buft_get_alignment (ggml_backend_buffer_type_t buft);
-    GGML_API size_t                ggml_backend_buft_get_max_size  (ggml_backend_buffer_type_t buft);
-    GGML_API size_t                ggml_backend_buft_get_alloc_size(ggml_backend_buffer_type_t buft, const struct ggml_tensor * tensor);
-    GGML_API bool                  ggml_backend_buft_is_host       (ggml_backend_buffer_type_t buft);
-    GGML_API ggml_backend_dev_t    ggml_backend_buft_get_device    (ggml_backend_buffer_type_t buft);
+    GGML_API const char *          ggml_backend_buft_name            (ggml_backend_buffer_type_t buft);
+    GGML_API ggml_backend_buffer_t ggml_backend_buft_alloc_buffer    (ggml_backend_buffer_type_t buft, size_t size);
+    GGML_API ggml_backend_buffer_t ggml_backend_buft_alloc_buffer_n  (ggml_backend_buffer_type_t buft, struct ggml_tensor ** tensors, int n_tensors);
+    GGML_API size_t                ggml_backend_buft_get_alignment   (ggml_backend_buffer_type_t buft);
+    GGML_API size_t                ggml_backend_buft_get_max_size    (ggml_backend_buffer_type_t buft);
+    GGML_API size_t                ggml_backend_buft_get_alloc_size  (ggml_backend_buffer_type_t buft, const struct ggml_tensor * tensor);
+    GGML_API size_t                ggml_backend_buft_get_alloc_size_n(ggml_backend_buffer_type_t buft, struct ggml_tensor ** tensors, int n_tensors);
+    GGML_API bool                  ggml_backend_buft_is_host         (ggml_backend_buffer_type_t buft);
+    GGML_API ggml_backend_dev_t    ggml_backend_buft_get_device      (ggml_backend_buffer_type_t buft);

     //
     // Backend buffer
diff --git a/ggml/src/ggml-alloc.c b/ggml/src/ggml-alloc.c
index a71838eaf..b0c067b49 100644
--- a/ggml/src/ggml-alloc.c
+++ b/ggml/src/ggml-alloc.c
@@ -1117,131 +1117,55 @@ size_t ggml_gallocr_get_buffer_size(ggml_gallocr_t galloc, int buffer_id) {

 // utils

-static void free_buffers(ggml_backend_buffer_t ** buffers, const size_t * n_buffers) {
-    for (size_t i = 0; i < *n_buffers; i++) {
-        ggml_backend_buffer_free((*buffers)[i]);
+static struct ggml_tensor ** ggml_backend_alloc_ctx_tensors_from_buft_collect(
+        struct ggml_context * ctx, int * n_tensors) {
+    int n = 0;
+    for (struct ggml_tensor * t = ggml_get_first_tensor(ctx); t != NULL; t = ggml_get_next_tensor(ctx, t)) {
+        n++;
     }
-    free(*buffers);
-}
-
-static bool alloc_tensor_range(struct ggml_context * ctx,
-        struct ggml_tensor * first, struct ggml_tensor * last,
-        ggml_backend_buffer_type_t buft, size_t size,
-        ggml_backend_buffer_t ** buffers, size_t * n_buffers) {
-
-    ggml_backend_buffer_t buffer = ggml_backend_buft_alloc_buffer(buft, size);
-    if (buffer == NULL) {
-        GGML_LOG_ERROR("%s: failed to allocate %s buffer of size %zu\n", __func__, ggml_backend_buft_name(buft), size);
-        free_buffers(buffers, n_buffers);
-        return false;
+    *n_tensors = n;
+    if (n == 0) {
+        return NULL;
     }

-    *buffers = realloc(*buffers, sizeof(ggml_backend_buffer_t) * (*n_buffers + 1));
-    (*buffers)[(*n_buffers)++] = buffer;
-
-    struct ggml_tallocr tallocr = ggml_tallocr_new(buffer);
-
-    for (struct ggml_tensor * t = first; t != last; t = ggml_get_next_tensor(ctx, t)) {
-        enum ggml_status status = GGML_STATUS_SUCCESS;
-        if (t->data == NULL) {
-            if (t->view_src == NULL) {
-                status = ggml_tallocr_alloc(&tallocr, t);
-            } else if (t->buffer == NULL) {
-                status = ggml_backend_view_init(t);
-            }
-        } else {
-            if (t->view_src != NULL && t->buffer == NULL) {
-                // view of a pre-allocated tensor
-                status = ggml_backend_view_init(t);
-            }
-        }
-        if (status != GGML_STATUS_SUCCESS) {
-            GGML_LOG_ERROR("%s: failed to initialize tensor %s\n", __func__, t->name);
-            free_buffers(buffers, n_buffers);
-            return false;
-        }
+    struct ggml_tensor ** tensors = (struct ggml_tensor **) malloc(n * sizeof(struct ggml_tensor *));
+    if (tensors == NULL) {
+        GGML_LOG_ERROR("%s: failed to allocate %zu bytes\n", __func__, n * sizeof(struct ggml_tensor *));
+        return NULL;
     }
-
-    return true;
+    int i = 0;
+    for (struct ggml_tensor * t = ggml_get_first_tensor(ctx); t != NULL; t = ggml_get_next_tensor(ctx, t)) {
+        tensors[i++] = t;
+    }
+    return tensors;
 }

-static ggml_backend_buffer_t ggml_backend_alloc_ctx_tensors_from_buft_impl(
-        struct ggml_context * ctx, ggml_backend_buffer_type_t buft, size_t * nbytes_total, bool no_alloc) {
+ggml_backend_buffer_t ggml_backend_alloc_ctx_tensors_from_buft(struct ggml_context * ctx, ggml_backend_buffer_type_t buft) {
     GGML_ASSERT(ggml_get_no_alloc(ctx) == true);

-    size_t alignment = ggml_backend_buft_get_alignment(buft);
-    size_t max_size = ggml_backend_buft_get_max_size(buft);
-
-    ggml_backend_buffer_t * buffers = NULL;
-    size_t n_buffers = 0;
-    *nbytes_total = 0;
-
-    size_t cur_buf_size = 0;
-    struct ggml_tensor * first = ggml_get_first_tensor(ctx);
-    for (struct ggml_tensor * t = first; t != NULL; t = ggml_get_next_tensor(ctx, t)) {
-        size_t this_size = 0;
-        if (t->data == NULL && t->view_src == NULL) {
-            this_size = GGML_PAD(ggml_backend_buft_get_alloc_size(buft, t), alignment);
-        }
-
-        if (cur_buf_size > 0 && (cur_buf_size + this_size) > max_size) {
-            // allocate tensors in the current buffer
-            if (!no_alloc && !alloc_tensor_range(ctx, first, t, buft, cur_buf_size, &buffers, &n_buffers)) {
-                return NULL;
-            }
-            first = t;
-            *nbytes_total += cur_buf_size;
-            cur_buf_size = this_size;
-        } else {
-            cur_buf_size += this_size;
-        }
-    }
-
-    // allocate remaining tensors
-    if (cur_buf_size > 0) {
-        *nbytes_total += cur_buf_size;
-        if (!no_alloc && !alloc_tensor_range(ctx, first, NULL, buft, cur_buf_size, &buffers, &n_buffers)) {
-            return NULL;
-        }
-    }
-
-    if (no_alloc) {
-        return NULL;
-    }
-
-    if (n_buffers == 0) {
-#ifndef NDEBUG
-        GGML_LOG_DEBUG("%s: all tensors in the context are already allocated\n", __func__);
-#endif
-        GGML_ASSERT(!buffers);
+    int n_tensors = 0;
+    struct ggml_tensor ** tensors = ggml_backend_alloc_ctx_tensors_from_buft_collect(ctx, &n_tensors);
+    if (tensors == NULL) {
         return NULL;
     }

-    ggml_backend_buffer_t buffer;
-    if (n_buffers == 1) {
-        buffer = buffers[0];
-    } else {
-        buffer = ggml_backend_multi_buffer_alloc_buffer(buffers, n_buffers);
-    }
-    if (buffers) {
-        free(buffers); // can be NULL if context is empty or no_alloc
-    }
+    ggml_backend_buffer_t buffer = ggml_backend_buft_alloc_buffer_n(buft, tensors, n_tensors);
+    free(tensors);
     return buffer;
 }

 size_t ggml_backend_alloc_ctx_tensors_from_buft_size(struct ggml_context * ctx, ggml_backend_buffer_type_t buft) {
-    size_t nbytes_total = 0;
-    ggml_backend_buffer_t buf = ggml_backend_alloc_ctx_tensors_from_buft_impl(ctx, buft, &nbytes_total, /*no_alloc=*/ true);
-    GGML_ASSERT(!buf);
-    return nbytes_total;
-}
+    GGML_ASSERT(ggml_get_no_alloc(ctx) == true);

-ggml_backend_buffer_t ggml_backend_alloc_ctx_tensors_from_buft(struct ggml_context * ctx, ggml_backend_buffer_type_t buft) {
-    size_t nbytes_total = 0;
-    if (ggml_backend_buft_is_meta(buft)) {
-        return ggml_backend_meta_alloc_ctx_tensors_from_buft(ctx, buft);
+    int n_tensors = 0;
+    struct ggml_tensor ** tensors = ggml_backend_alloc_ctx_tensors_from_buft_collect(ctx, &n_tensors);
+    if (tensors == NULL) {
+        return 0;
     }
-    return ggml_backend_alloc_ctx_tensors_from_buft_impl(ctx, buft, &nbytes_total, /*no_alloc =*/ false);
+
+    size_t nbytes_total = ggml_backend_buft_get_alloc_size_n(buft, tensors, n_tensors);
+    free(tensors);
+    return nbytes_total;
 }

 ggml_backend_buffer_t ggml_backend_alloc_ctx_tensors(struct ggml_context * ctx, ggml_backend_t backend) {
diff --git a/ggml/src/ggml-backend-impl.h b/ggml/src/ggml-backend-impl.h
index ef05905cf..dc7367dec 100644
--- a/ggml/src/ggml-backend-impl.h
+++ b/ggml/src/ggml-backend-impl.h
@@ -8,24 +8,28 @@
 extern "C" {
 #endif

-    #define GGML_BACKEND_API_VERSION 2
+    #define GGML_BACKEND_API_VERSION 3

     //
     // Backend buffer type
     //

     struct ggml_backend_buffer_type_i {
-        const char *          (*get_name)      (ggml_backend_buffer_type_t buft);
+        const char *          (*get_name)        (ggml_backend_buffer_type_t buft);
         // allocate a buffer of this type
-        ggml_backend_buffer_t (*alloc_buffer)  (ggml_backend_buffer_type_t buft, size_t size);
+        ggml_backend_buffer_t (*alloc_buffer)    (ggml_backend_buffer_type_t buft, size_t size);
+        // (optional) allocate tensors from a list into a buffer of this type (defaults to alloc_buffer + linear allocator)
+        ggml_backend_buffer_t (*alloc_buffer_n)  (ggml_backend_buffer_type_t buft, struct ggml_tensor ** tensors, int n_tensors);
         // tensor alignment
-        size_t                (*get_alignment) (ggml_backend_buffer_type_t buft);
+        size_t                (*get_alignment)   (ggml_backend_buffer_type_t buft);
         // (optional) max buffer size that can be allocated (defaults to SIZE_MAX)
-        size_t                (*get_max_size)  (ggml_backend_buffer_type_t buft);
+        size_t                (*get_max_size)    (ggml_backend_buffer_type_t buft);
         // (optional) data size needed to allocate the tensor, including padding (defaults to ggml_nbytes)
-        size_t                (*get_alloc_size)(ggml_backend_buffer_type_t buft, const struct ggml_tensor * tensor);
+        size_t                (*get_alloc_size)  (ggml_backend_buffer_type_t buft, const struct ggml_tensor * tensor);
+        // (optional) total data size needed to allocate the given tensors, including padding and splitting (defaults to per-tensor get_alloc_size)
+        size_t                (*get_alloc_size_n)(ggml_backend_buffer_type_t buft, struct ggml_tensor ** tensors, int n_tensors);
         // (optional) check if tensor data is in host memory and uses standard ggml tensor layout (defaults to false)
-        bool                  (*is_host)       (ggml_backend_buffer_type_t buft);
+        bool                  (*is_host)         (ggml_backend_buffer_type_t buft);
     };

     struct ggml_backend_buffer_type {
@@ -101,9 +105,6 @@ extern "C" {
     GGML_API size_t         ggml_backend_meta_n_backends    (ggml_backend_t meta_backend);
     GGML_API ggml_backend_t ggml_backend_meta_simple_backend(ggml_backend_t meta_backend, size_t index);

-    // temporary workaround to statically allocate tensors from a context in a deduplicated way:
-    GGML_API struct ggml_backend_buffer * ggml_backend_meta_alloc_ctx_tensors_from_buft(struct ggml_context * ctx, ggml_backend_buffer_type_t buft);
-
     //
     // Backend (stream)
     //
diff --git a/ggml/src/ggml-backend-meta.cpp b/ggml/src/ggml-backend-meta.cpp
index 39c5e478c..0394433c0 100644
--- a/ggml/src/ggml-backend-meta.cpp
+++ b/ggml/src/ggml-backend-meta.cpp
@@ -290,6 +290,8 @@ static ggml_backend_buffer_type_t ggml_backend_meta_buft_simple_buft(ggml_backen

 static ggml_backend_buffer_t ggml_backend_meta_buffer_type_alloc_buffer(ggml_backend_buffer_type_t buft, size_t size);

+static ggml_backend_buffer_t ggml_backend_meta_buffer_type_alloc_buffer_n(ggml_backend_buffer_type_t buft, ggml_tensor ** tensors, int n_tensors);
+
 static size_t ggml_backend_meta_buffer_type_get_alignment(ggml_backend_buffer_type_t buft) {
     const size_t n_simple_bufts = ggml_backend_meta_buft_n_bufts(buft);
     size_t max_alignment = 1;
@@ -331,12 +333,14 @@ static bool ggml_backend_meta_buffer_type_is_host(ggml_backend_buffer_type_t buf
 }

 static const struct ggml_backend_buffer_type_i ggml_backend_meta_buffer_type_iface = {
-    /* .get_name         = */ ggml_backend_meta_buffer_type_get_name,
-    /* .alloc_buffer     = */ ggml_backend_meta_buffer_type_alloc_buffer,
-    /* .get_alignment    = */ ggml_backend_meta_buffer_type_get_alignment,
-    /* .get_max_size     = */ ggml_backend_meta_buffer_type_get_max_size,
-    /* .get_alloc_size   = */ ggml_backend_meta_buffer_type_get_alloc_size,
-    /* .is_host          = */ ggml_backend_meta_buffer_type_is_host,
+    /* .get_name            = */ ggml_backend_meta_buffer_type_get_name,
+    /* .alloc_buffer        = */ ggml_backend_meta_buffer_type_alloc_buffer,
+    /* .alloc_buffer_n      = */ ggml_backend_meta_buffer_type_alloc_buffer_n,
+    /* .get_alignment       = */ ggml_backend_meta_buffer_type_get_alignment,
+    /* .get_max_size        = */ ggml_backend_meta_buffer_type_get_max_size,
+    /* .get_alloc_size      = */ ggml_backend_meta_buffer_type_get_alloc_size,
+    /* .get_alloc_size_n    = */ NULL,
+    /* .is_host             = */ ggml_backend_meta_buffer_type_is_host,
 };

 bool ggml_backend_buft_is_meta(ggml_backend_buffer_type_t buft) {
@@ -1715,17 +1719,17 @@ static ggml_backend_buffer_t ggml_backend_meta_buffer_type_alloc_buffer(ggml_bac
     return ggml_backend_buffer_init(buft, ggml_backend_meta_buffer_iface, buf_ctx, max_size);
 }

-struct ggml_backend_buffer * ggml_backend_meta_alloc_ctx_tensors_from_buft(struct ggml_context * ctx, ggml_backend_buffer_type_t buft) {
+static ggml_backend_buffer_t ggml_backend_meta_buffer_type_alloc_buffer_n(ggml_backend_buffer_type_t buft, ggml_tensor ** tensors, int n_tensors) {
     const size_t n_simple_bufts = ggml_backend_meta_buft_n_bufts(buft);

     constexpr size_t compute_headroom = 16; // Maximum number of views per statically allocated tensor that can be created between evals.
     const ggml_init_params params_static = {
-        /*.mem_size   =*/ ggml_get_mem_size(ctx),
+        /*.mem_size   =*/ n_tensors * ggml_tensor_overhead(),
         /*.mem_buffer =*/ nullptr,
         /*.no_alloc   =*/ true,
     };
     const ggml_init_params params_compute = {
-        /*.mem_size   =*/ compute_headroom*ggml_get_mem_size(ctx),
+        /*.mem_size   =*/ compute_headroom * n_tensors * ggml_tensor_overhead(),
         /*.mem_buffer =*/ nullptr,
         /*.no_alloc   =*/ true,
     };
@@ -1737,7 +1741,8 @@ struct ggml_backend_buffer * ggml_backend_meta_alloc_ctx_tensors_from_buft(struc
     ggml_backend_meta_buffer_context * meta_buf_ctx = new ggml_backend_meta_buffer_context(stc_static, stc_compute_0, stc_compute_1, bufs);

     ggml_backend_buffer_t meta_buf = ggml_backend_buffer_init(buft, ggml_backend_meta_buffer_iface, meta_buf_ctx, 0);
-    for (ggml_tensor * t = ggml_get_first_tensor(ctx); t != nullptr; t = ggml_get_next_tensor(ctx, t)) {
+    for (int i = 0; i < n_tensors; i++) {
+        ggml_tensor * t = tensors[i];
         t->buffer = meta_buf;
         ggml_backend_meta_buffer_init_tensor_impl(meta_buf_ctx->stc_static, t);
         t->data = (void *) 0x2000000000000000; // FIXME
diff --git a/ggml/src/ggml-backend.cpp b/ggml/src/ggml-backend.cpp
index 273dc92b2..5bdade6de 100644
--- a/ggml/src/ggml-backend.cpp
+++ b/ggml/src/ggml-backend.cpp
@@ -45,6 +45,138 @@ ggml_backend_buffer_t ggml_backend_buft_alloc_buffer(ggml_backend_buffer_type_t
     return buft->iface.alloc_buffer(buft, size);
 }

+// shared planning logic for allocating a list of tensors into one or more buffers of the given type
+struct ggml_backend_buft_alloc_buffer_n_plan_item {
+    size_t size;  // total bytes for this buffer
+    int    first; // first tensor index (inclusive)
+    int    last;  // last tensor index (exclusive)
+};
+
+using ggml_backend_buft_alloc_buffer_n_plan_t = std::vector<ggml_backend_buft_alloc_buffer_n_plan_item>;
+
+static ggml_backend_buft_alloc_buffer_n_plan_t ggml_backend_buft_alloc_buffer_n_plan(
+        ggml_backend_buffer_type_t buft, struct ggml_tensor ** tensors, int n_tensors) {
+    ggml_backend_buft_alloc_buffer_n_plan_t plan;
+
+    const size_t alignment = ggml_backend_buft_get_alignment(buft);
+    const size_t max_size  = ggml_backend_buft_get_max_size(buft);
+
+    size_t cur_buf_size = 0;
+    int    first        = 0;
+
+    for (int i = 0; i < n_tensors; i++) {
+        size_t this_size = 0;
+        struct ggml_tensor * t = tensors[i];
+        if (t->data == NULL && t->view_src == NULL) {
+            this_size = GGML_PAD(ggml_backend_buft_get_alloc_size(buft, t), alignment);
+        }
+
+        // flush the current buffer if adding this tensor would exceed max_size
+        if (cur_buf_size > 0 && (cur_buf_size + this_size) > max_size) {
+            plan.push_back({ cur_buf_size, first, i });
+            cur_buf_size = this_size;
+            first        = i;
+        } else {
+            cur_buf_size += this_size;
+        }
+    }
+
+    if (cur_buf_size > 0) {
+        plan.push_back({ cur_buf_size, first, n_tensors });
+    }
+
+    return plan;
+}
+
+// default implementation of alloc_buffer_n
+// allocates tensors from a list into one or more buffers of the given type
+static ggml_backend_buffer_t ggml_backend_buft_alloc_buffer_n_default(ggml_backend_buffer_type_t buft, struct ggml_tensor ** tensors, int n_tensors) {
+    const ggml_backend_buft_alloc_buffer_n_plan_t plan = ggml_backend_buft_alloc_buffer_n_plan(buft, tensors, n_tensors);
+
+    std::vector<ggml_backend_buffer_t> buffers;
+    buffers.reserve(plan.size());
+
+    for (const ggml_backend_buft_alloc_buffer_n_plan_item & item : plan) {
+        ggml_backend_buffer_t buffer = ggml_backend_buft_alloc_buffer(buft, item.size);
+        if (buffer == NULL) {
+            GGML_LOG_ERROR("%s: failed to allocate %s buffer of size %zu\n", __func__, ggml_backend_buft_name(buft), item.size);
+            for (ggml_backend_buffer_t b : buffers) {
+                ggml_backend_buffer_free(b);
+            }
+            return NULL;
+        }
+
+        struct ggml_tallocr tallocr = ggml_tallocr_new(buffer);
+
+        // allocate tensors in the current buffer
+        struct ggml_tensor * t_failed = NULL;
+        for (int j = item.first; j < item.last; j++) {
+            struct ggml_tensor * t = tensors[j];
+            if (t->data == NULL) {
+                if (t->view_src == NULL) {
+                    if (ggml_tallocr_alloc(&tallocr, t) != GGML_STATUS_SUCCESS) {
+                        t_failed = t;
+                        break;
+                    }
+                } else if (t->buffer == NULL) {
+                    if (ggml_backend_view_init(t) != GGML_STATUS_SUCCESS) {
+                        t_failed = t;
+                        break;
+                    }
+                }
+            } else {
+                if (t->view_src != NULL && t->buffer == NULL) {
+                    // view of a pre-allocated tensor
+                    if (ggml_backend_view_init(t) != GGML_STATUS_SUCCESS) {
+                        t_failed = t;
+                        break;
+                    }
+                }
+            }
+        }
+        if (t_failed != NULL) {
+            GGML_LOG_ERROR("%s: failed to initialize tensor %s\n", __func__, t_failed->name);
+            for (ggml_backend_buffer_t b : buffers) {
+                ggml_backend_buffer_free(b);
+            }
+            ggml_backend_buffer_free(buffer);
+            return NULL;
+        }
+
+        buffers.push_back(buffer);
+    }
+
+    if (buffers.empty()) {
+        return NULL;
+    }
+
+    if (buffers.size() == 1) {
+        return buffers[0];
+    }
+
+    return ggml_backend_multi_buffer_alloc_buffer(buffers.data(), buffers.size());
+}
+
+// default implementation of get_alloc_size_n
+// returns the total size that alloc_buffer_n_default would allocate for the given tensors
+static size_t ggml_backend_buft_get_alloc_size_n_default(ggml_backend_buffer_type_t buft, struct ggml_tensor ** tensors, int n_tensors) {
+    const ggml_backend_buft_alloc_buffer_n_plan_t plan = ggml_backend_buft_alloc_buffer_n_plan(buft, tensors, n_tensors);
+
+    size_t total = 0;
+    for (const ggml_backend_buft_alloc_buffer_n_plan_item & item : plan) {
+        total += item.size;
+    }
+    return total;
+}
+
+ggml_backend_buffer_t ggml_backend_buft_alloc_buffer_n(ggml_backend_buffer_type_t buft, struct ggml_tensor ** tensors, int n_tensors) {
+    GGML_ASSERT(buft);
+    if (buft->iface.alloc_buffer_n) {
+        return buft->iface.alloc_buffer_n(buft, tensors, n_tensors);
+    }
+    return ggml_backend_buft_alloc_buffer_n_default(buft, tensors, n_tensors);
+}
+
 size_t ggml_backend_buft_get_alignment(ggml_backend_buffer_type_t buft) {
     GGML_ASSERT(buft);
     return buft->iface.get_alignment(buft);
@@ -78,6 +210,14 @@ size_t ggml_backend_buft_get_alloc_size(ggml_backend_buffer_type_t buft, const s
     return ggml_nbytes(tensor);
 }

+size_t ggml_backend_buft_get_alloc_size_n(ggml_backend_buffer_type_t buft, struct ggml_tensor ** tensors, int n_tensors) {
+    GGML_ASSERT(buft);
+    if (buft->iface.get_alloc_size_n) {
+        return buft->iface.get_alloc_size_n(buft, tensors, n_tensors);
+    }
+    return ggml_backend_buft_get_alloc_size_n_default(buft, tensors, n_tensors);
+}
+
 bool ggml_backend_buft_is_host(ggml_backend_buffer_type_t buft) {
     GGML_ASSERT(buft);
     if (buft->iface.is_host) {
@@ -2486,12 +2626,14 @@ static bool ggml_backend_cpu_buffer_type_is_host(ggml_backend_buffer_type_t buft
 ggml_backend_buffer_type_t ggml_backend_cpu_buffer_type(void) {
     static struct ggml_backend_buffer_type ggml_backend_cpu_buffer_type = {
         /* .iface   = */ {
-            /* .get_name         = */ ggml_backend_cpu_buffer_type_get_name,
-            /* .alloc_buffer     = */ ggml_backend_cpu_buffer_type_alloc_buffer,
-            /* .get_alignment    = */ ggml_backend_cpu_buffer_type_get_alignment,
-            /* .get_max_size     = */ NULL, // defaults to SIZE_MAX
-            /* .get_alloc_size   = */ NULL, // defaults to ggml_nbytes
-            /* .is_host          = */ ggml_backend_cpu_buffer_type_is_host,
+            /* .get_name            = */ ggml_backend_cpu_buffer_type_get_name,
+            /* .alloc_buffer        = */ ggml_backend_cpu_buffer_type_alloc_buffer,
+            /* .alloc_buffer_n      = */ NULL,
+            /* .get_alignment       = */ ggml_backend_cpu_buffer_type_get_alignment,
+            /* .get_max_size        = */ NULL, // defaults to SIZE_MAX
+            /* .get_alloc_size      = */ NULL, // defaults to ggml_nbytes
+            /* .get_alloc_size_n    = */ NULL,
+            /* .is_host             = */ ggml_backend_cpu_buffer_type_is_host,
         },
         /* .device  = */ NULL, // FIXME ggml_backend_reg_dev_get(ggml_backend_cpu_reg(), 0),
         /* .context = */ NULL,
@@ -2509,12 +2651,14 @@ static const char * ggml_backend_cpu_buffer_from_ptr_type_get_name(ggml_backend_
 static ggml_backend_buffer_type_t ggml_backend_cpu_buffer_from_ptr_type(void) {
     static struct ggml_backend_buffer_type ggml_backend_cpu_buffer_type = {
         /* .iface   = */ {
-            /* .get_name         = */ ggml_backend_cpu_buffer_from_ptr_type_get_name,
-            /* .alloc_buffer     = */ ggml_backend_cpu_buffer_type_alloc_buffer,
-            /* .get_alignment    = */ ggml_backend_cpu_buffer_type_get_alignment,
-            /* .get_max_size     = */ NULL, // defaults to SIZE_MAX
-            /* .get_alloc_size   = */ NULL, // defaults to ggml_nbytes
-            /* .is_host          = */ ggml_backend_cpu_buffer_type_is_host,
+            /* .get_name            = */ ggml_backend_cpu_buffer_from_ptr_type_get_name,
+            /* .alloc_buffer        = */ ggml_backend_cpu_buffer_type_alloc_buffer,
+            /* .alloc_buffer_n      = */ NULL,
+            /* .get_alignment       = */ ggml_backend_cpu_buffer_type_get_alignment,
+            /* .get_max_size        = */ NULL, // defaults to SIZE_MAX
+            /* .get_alloc_size      = */ NULL, // defaults to ggml_nbytes
+            /* .get_alloc_size_n    = */ NULL,
+            /* .is_host             = */ ggml_backend_cpu_buffer_type_is_host,
         },
         /* .device  = */ NULL, // FIXME ggml_backend_reg_dev_get(ggml_backend_cpu_reg(), 0),
         /* .context = */ NULL,
diff --git a/ggml/src/ggml-cann/ggml-cann.cpp b/ggml/src/ggml-cann/ggml-cann.cpp
index c2745014a..28e17f3f7 100644
--- a/ggml/src/ggml-cann/ggml-cann.cpp
+++ b/ggml/src/ggml-cann/ggml-cann.cpp
@@ -1595,12 +1595,14 @@ static bool ggml_backend_cann_buffer_type_is_host(ggml_backend_buffer_type_t buf
  * memory for CANN buffer types in the GGML backend.
  */
 static const ggml_backend_buffer_type_i ggml_backend_cann_buffer_type_interface = {
-    /* .get_name         = */ ggml_backend_cann_buffer_type_name,
-    /* .alloc_buffer     = */ ggml_backend_cann_buffer_type_alloc_buffer,
-    /* .get_alignment    = */ ggml_backend_cann_buffer_type_get_alignment,
-    /* .get_max_size     = */ NULL,  // defaults to SIZE_MAX
-    /* .get_alloc_size   = */ ggml_backend_cann_buffer_type_get_alloc_size,
-    /* .is_host          = */ ggml_backend_cann_buffer_type_is_host,
+    /* .get_name            = */ ggml_backend_cann_buffer_type_name,
+    /* .alloc_buffer        = */ ggml_backend_cann_buffer_type_alloc_buffer,
+    /* .alloc_buffer_n      = */ NULL,
+    /* .get_alignment       = */ ggml_backend_cann_buffer_type_get_alignment,
+    /* .get_max_size        = */ NULL,  // defaults to SIZE_MAX
+    /* .get_alloc_size      = */ ggml_backend_cann_buffer_type_get_alloc_size,
+    /* .get_alloc_size_n    = */ NULL,
+    /* .is_host             = */ ggml_backend_cann_buffer_type_is_host,
 };

 /**
@@ -1742,12 +1744,14 @@ static ggml_backend_buffer_t ggml_backend_cann_host_buffer_type_alloc_buffer(ggm
 ggml_backend_buffer_type_t ggml_backend_cann_host_buffer_type() {
     static struct ggml_backend_buffer_type ggml_backend_cann_buffer_type_host = {
         /* .iface    = */ {
-                           /* .get_name         = */ ggml_backend_cann_host_buffer_type_name,
-                           /* .alloc_buffer     = */ ggml_backend_cann_host_buffer_type_alloc_buffer,
-                           /* .get_alignment    = */ ggml_backend_cpu_buffer_type()->iface.get_alignment,
-                           /* .get_max_size     = */ NULL,  // defaults to SIZE_MAX
-            /* .get_alloc_size   = */ ggml_backend_cpu_buffer_type()->iface.get_alloc_size,
-                           /* .is_host          = */ ggml_backend_cpu_buffer_type()->iface.is_host,
+                           /* .get_name             = */ ggml_backend_cann_host_buffer_type_name,
+                           /* .alloc_buffer         = */ ggml_backend_cann_host_buffer_type_alloc_buffer,
+                           /* .alloc_buffer_n       = */ NULL,
+                           /* .get_alignment        = */ ggml_backend_cpu_buffer_type()->iface.get_alignment,
+                           /* .get_max_size         = */ NULL,  // defaults to SIZE_MAX
+                           /* .get_alloc_size       = */ ggml_backend_cpu_buffer_type()->iface.get_alloc_size,
+                           /* .get_alloc_size_n     = */ NULL,
+                           /* .is_host              = */ ggml_backend_cpu_buffer_type()->iface.is_host,
                            },
         /* .device   = */
         ggml_backend_reg_dev_get(ggml_backend_cann_reg(), 0),
diff --git a/ggml/src/ggml-cpu/amx/amx.cpp b/ggml/src/ggml-cpu/amx/amx.cpp
index 1118f7169..e90c2797f 100644
--- a/ggml/src/ggml-cpu/amx/amx.cpp
+++ b/ggml/src/ggml-cpu/amx/amx.cpp
@@ -228,12 +228,14 @@ static bool ggml_amx_init() {
 ggml_backend_buffer_type_t ggml_backend_amx_buffer_type() {
     static struct ggml_backend_buffer_type ggml_backend_buffer_type_amx = {
         /* .iface = */ {
-                        /* .get_name         = */ ggml_backend_amx_buffer_type_get_name,
-                        /* .alloc_buffer     = */ ggml_backend_amx_buffer_type_alloc_buffer,
-                        /* .get_alignment    = */ ggml_backend_amx_buffer_type_get_alignment,
-                        /* .get_max_size     = */ nullptr,  // defaults to SIZE_MAX
-                        /* .get_alloc_size   = */ ggml_backend_amx_buffer_type_get_alloc_size,
-                        /* .is_host          = */ nullptr,
+                        /* .get_name            = */ ggml_backend_amx_buffer_type_get_name,
+                        /* .alloc_buffer        = */ ggml_backend_amx_buffer_type_alloc_buffer,
+                        /* .alloc_buffer_n      = */ nullptr,
+                        /* .get_alignment       = */ ggml_backend_amx_buffer_type_get_alignment,
+                        /* .get_max_size        = */ nullptr,  // defaults to SIZE_MAX
+                        /* .get_alloc_size      = */ ggml_backend_amx_buffer_type_get_alloc_size,
+                        /* .get_alloc_size_n    = */ NULL,
+                        /* .is_host             = */ nullptr,
                         },
         /* .device  = */ ggml_backend_reg_dev_get(ggml_backend_cpu_reg(), 0),
         /* .context = */ new ggml::cpu::amx::extra_buffer_type(),
diff --git a/ggml/src/ggml-cpu/hbm.cpp b/ggml/src/ggml-cpu/hbm.cpp
index a4073c15e..0ca5c8647 100644
--- a/ggml/src/ggml-cpu/hbm.cpp
+++ b/ggml/src/ggml-cpu/hbm.cpp
@@ -40,12 +40,14 @@ static ggml_backend_buffer_t ggml_backend_cpu_hbm_buffer_type_alloc_buffer(ggml_
 ggml_backend_buffer_type_t ggml_backend_cpu_hbm_buffer_type(void) {
     static struct ggml_backend_buffer_type ggml_backend_cpu_buffer_type_hbm = {
         /* .iface    = */ {
-                           /* .get_name         = */ ggml_backend_cpu_hbm_buffer_type_get_name,
-                           /* .alloc_buffer     = */ ggml_backend_cpu_hbm_buffer_type_alloc_buffer,
-                           /* .get_alignment    = */ ggml_backend_cpu_buffer_type_get_alignment,
-                           /* .get_max_size     = */ nullptr,  // defaults to SIZE_MAX
-                           /* .get_alloc_size   = */ nullptr,  // defaults to ggml_nbytes
-                           /* .is_host          = */ ggml_backend_cpu_buffer_type_is_host,
+                           /* .get_name             = */ ggml_backend_cpu_hbm_buffer_type_get_name,
+                           /* .alloc_buffer         = */ ggml_backend_cpu_hbm_buffer_type_alloc_buffer,
+                           /* .alloc_buffer_n       = */ nullptr,
+                           /* .get_alignment        = */ ggml_backend_cpu_buffer_type_get_alignment,
+                           /* .get_max_size         = */ nullptr,  // defaults to SIZE_MAX
+                           /* .get_alloc_size       = */ nullptr,  // defaults to ggml_nbytes
+                           /* .get_alloc_size_n     = */ NULL,
+                           /* .is_host              = */ ggml_backend_cpu_buffer_type_is_host,
                            },
         /* .context  = */ nullptr,
     };
diff --git a/ggml/src/ggml-cpu/kleidiai/kleidiai.cpp b/ggml/src/ggml-cpu/kleidiai/kleidiai.cpp
index dbd198780..1384ca927 100644
--- a/ggml/src/ggml-cpu/kleidiai/kleidiai.cpp
+++ b/ggml/src/ggml-cpu/kleidiai/kleidiai.cpp
@@ -1902,12 +1902,14 @@ ggml_backend_buffer_type_t ggml_backend_cpu_kleidiai_buffer_type(void) {
     static ggml::cpu::kleidiai::extra_buffer_type ctx;
     static struct ggml_backend_buffer_type ggml_backend_cpu_buffer_type_kleidiai = {
         /* .iface    = */ {
-                           /* .get_name         = */ ggml_backend_cpu_kleidiai_buffer_type_get_name,
-                           /* .alloc_buffer     = */ ggml_backend_cpu_kleidiai_buffer_type_alloc_buffer,
-                           /* .get_alignment    = */ ggml_backend_cpu_kleidiai_buffer_type_get_alignment,
-                           /* .get_max_size     = */ nullptr,  // defaults to SIZE_MAX
-                           /* .get_alloc_size   = */ ggml_backend_cpu_kleidiai_buffer_type_get_alloc_size,
-                           /* .is_host          = */ nullptr,
+                           /* .get_name             = */ ggml_backend_cpu_kleidiai_buffer_type_get_name,
+                           /* .alloc_buffer         = */ ggml_backend_cpu_kleidiai_buffer_type_alloc_buffer,
+                           /* .alloc_buffer_n       = */ nullptr,
+                           /* .get_alignment        = */ ggml_backend_cpu_kleidiai_buffer_type_get_alignment,
+                           /* .get_max_size         = */ nullptr,  // defaults to SIZE_MAX
+                           /* .get_alloc_size       = */ ggml_backend_cpu_kleidiai_buffer_type_get_alloc_size,
+                           /* .get_alloc_size_n     = */ NULL,
+                           /* .is_host              = */ nullptr,
                            },
         /* .device  = */ ggml_backend_reg_dev_get(ggml_backend_cpu_reg(), 0),
         /* .context = */ &ctx,
diff --git a/ggml/src/ggml-cpu/repack.cpp b/ggml/src/ggml-cpu/repack.cpp
index d56db9802..2632ae3a0 100644
--- a/ggml/src/ggml-cpu/repack.cpp
+++ b/ggml/src/ggml-cpu/repack.cpp
@@ -5238,12 +5238,14 @@ class extra_buffer_type : ggml::cpu::extra_buffer_type {
 ggml_backend_buffer_type_t ggml_backend_cpu_repack_buffer_type(void) {
     static struct ggml_backend_buffer_type ggml_backend_cpu_buffer_type_repack = {
         /* .iface    = */ {
-                           /* .get_name         = */ ggml_backend_cpu_repack_buffer_type_get_name,
-                           /* .alloc_buffer     = */ ggml_backend_cpu_repack_buffer_type_alloc_buffer,
-                           /* .get_alignment    = */ ggml_backend_cpu_repack_buffer_type_get_alignment,
-                           /* .get_max_size     = */ nullptr,  // defaults to SIZE_MAX
-                           /* .get_alloc_size   = */ nullptr,  // defaults to ggml_nbytes
-                           /* .is_host          = */ nullptr,
+                           /* .get_name             = */ ggml_backend_cpu_repack_buffer_type_get_name,
+                           /* .alloc_buffer         = */ ggml_backend_cpu_repack_buffer_type_alloc_buffer,
+                           /* .alloc_buffer_n       = */ nullptr,
+                           /* .get_alignment        = */ ggml_backend_cpu_repack_buffer_type_get_alignment,
+                           /* .get_max_size         = */ nullptr,  // defaults to SIZE_MAX
+                           /* .get_alloc_size       = */ nullptr,  // defaults to ggml_nbytes
+                           /* .get_alloc_size_n     = */ NULL,
+                           /* .is_host              = */ nullptr,
                            },
         /* .device  = */ ggml_backend_reg_dev_get(ggml_backend_cpu_reg(), 0),
         /* .context = */ new ggml::cpu::repack::extra_buffer_type(),
diff --git a/ggml/src/ggml-cpu/spacemit/ime.cpp b/ggml/src/ggml-cpu/spacemit/ime.cpp
index 29d683270..450d2d1aa 100644
--- a/ggml/src/ggml-cpu/spacemit/ime.cpp
+++ b/ggml/src/ggml-cpu/spacemit/ime.cpp
@@ -1650,12 +1650,14 @@ ggml_backend_buffer_type_t ggml_backend_cpu_riscv64_spacemit_buffer_type(void) {
     static ggml_backend_buffer_type ggml_backend_cpu_buffer_type_riscv64_spacemit = {
   /* .iface    = */
         {
-         /* .get_name         = */ ggml_backend_cpu_riscv64_spacemit_buffer_type_get_name,
-         /* .alloc_buffer     = */ ggml_backend_cpu_riscv64_spacemit_buffer_type_alloc_buffer,
-         /* .get_alignment    = */ ggml_backend_cpu_riscv64_spacemit_buffer_type_get_alignment,
-         /* .get_max_size     = */ nullptr,
-         /* .get_alloc_size   = */ ggml_backend_cpu_riscv64_spacemit_nbytes,
-         /* .is_host          = */ nullptr,
+         /* .get_name           = */ ggml_backend_cpu_riscv64_spacemit_buffer_type_get_name,
+         /* .alloc_buffer       = */ ggml_backend_cpu_riscv64_spacemit_buffer_type_alloc_buffer,
+         /* .alloc_buffer_n     = */ NULL,
+         /* .get_alignment      = */ ggml_backend_cpu_riscv64_spacemit_buffer_type_get_alignment,
+         /* .get_max_size       = */ nullptr,
+         /* .get_alloc_size     = */ ggml_backend_cpu_riscv64_spacemit_nbytes,
+         /* .get_alloc_size_n   = */ NULL,
+         /* .is_host            = */ nullptr,
          },
  /* .device  = */
         ggml_backend_reg_dev_get(ggml_backend_cpu_reg(), 0),
diff --git a/ggml/src/ggml-cuda/ggml-cuda.cu b/ggml/src/ggml-cuda/ggml-cuda.cu
index f9fb46c2a..043511b72 100644
--- a/ggml/src/ggml-cuda/ggml-cuda.cu
+++ b/ggml/src/ggml-cuda/ggml-cuda.cu
@@ -920,12 +920,14 @@ static size_t ggml_backend_cuda_buffer_type_get_alloc_size(ggml_backend_buffer_t
 }

 static const ggml_backend_buffer_type_i ggml_backend_cuda_buffer_type_interface = {
-    /* .get_name         = */ ggml_backend_cuda_buffer_type_get_name,
-    /* .alloc_buffer     = */ ggml_backend_cuda_buffer_type_alloc_buffer,
-    /* .get_alignment    = */ ggml_backend_cuda_buffer_type_get_alignment,
-    /* .get_max_size     = */ NULL, // defaults to SIZE_MAX
-    /* .get_alloc_size   = */ ggml_backend_cuda_buffer_type_get_alloc_size,
-    /* .is_host          = */ NULL,
+    /* .get_name            = */ ggml_backend_cuda_buffer_type_get_name,
+    /* .alloc_buffer        = */ ggml_backend_cuda_buffer_type_alloc_buffer,
+    /* .alloc_buffer_n      = */ NULL,
+    /* .get_alignment       = */ ggml_backend_cuda_buffer_type_get_alignment,
+    /* .get_max_size        = */ NULL, // defaults to SIZE_MAX
+    /* .get_alloc_size      = */ ggml_backend_cuda_buffer_type_get_alloc_size,
+    /* .get_alloc_size_n    = */ NULL,
+    /* .is_host             = */ NULL,
 };

 ggml_backend_buffer_type_t ggml_backend_cuda_buffer_type(int device) {
@@ -1304,12 +1306,14 @@ static ggml_backend_buffer_t ggml_backend_cuda_host_buffer_type_alloc_buffer(ggm
 ggml_backend_buffer_type_t ggml_backend_cuda_host_buffer_type() {
     static struct ggml_backend_buffer_type ggml_backend_cuda_buffer_type_host = {
         /* .iface    = */ {
-            /* .get_name         = */ ggml_backend_cuda_host_buffer_type_name,
-            /* .alloc_buffer     = */ ggml_backend_cuda_host_buffer_type_alloc_buffer,
-            /* .get_alignment    = */ ggml_backend_cpu_buffer_type()->iface.get_alignment,
-            /* .get_max_size     = */ NULL, // defaults to SIZE_MAX
-            /* .get_alloc_size   = */ ggml_backend_cpu_buffer_type()->iface.get_alloc_size,
-            /* .is_host          = */ ggml_backend_cpu_buffer_type()->iface.is_host,
+            /* .get_name            = */ ggml_backend_cuda_host_buffer_type_name,
+            /* .alloc_buffer        = */ ggml_backend_cuda_host_buffer_type_alloc_buffer,
+            /* .alloc_buffer_n      = */ NULL,
+            /* .get_alignment       = */ ggml_backend_cpu_buffer_type()->iface.get_alignment,
+            /* .get_max_size        = */ NULL, // defaults to SIZE_MAX
+            /* .get_alloc_size      = */ ggml_backend_cpu_buffer_type()->iface.get_alloc_size,
+            /* .get_alloc_size_n    = */ NULL,
+            /* .is_host             = */ ggml_backend_cpu_buffer_type()->iface.is_host,
         },
         /* .device   = */ ggml_backend_reg_dev_get(ggml_backend_cuda_reg(), 0),
         /* .context  = */ nullptr,
diff --git a/ggml/src/ggml-et/ggml-et.cpp b/ggml/src/ggml-et/ggml-et.cpp
index 755077fad..54a6d166d 100644
--- a/ggml/src/ggml-et/ggml-et.cpp
+++ b/ggml/src/ggml-et/ggml-et.cpp
@@ -439,12 +439,14 @@ static bool ggml_backend_et_buffer_type_is_host(ggml_backend_buffer_type_t buft)
 }

 static const struct ggml_backend_buffer_type_i ggml_backend_et_buffer_type_i = {
-    /* .get_name         = */ ggml_backend_et_buffer_type_get_name,
-    /* .alloc_buffer     = */ ggml_backend_et_buffer_type_alloc_buffer,
-    /* .get_alignment    = */ ggml_backend_et_buffer_type_get_alignment,
-    /* .get_max_size     = */ ggml_backend_et_buffer_type_get_max_size,
-    /* .get_alloc_size   = */ ggml_backend_et_buffer_type_get_alloc_size,
-    /* .is_host          = */ ggml_backend_et_buffer_type_is_host,
+    /* .get_name            = */ ggml_backend_et_buffer_type_get_name,
+    /* .alloc_buffer        = */ ggml_backend_et_buffer_type_alloc_buffer,
+    /* .alloc_buffer_n      = */ NULL,
+    /* .get_alignment       = */ ggml_backend_et_buffer_type_get_alignment,
+    /* .get_max_size        = */ ggml_backend_et_buffer_type_get_max_size,
+    /* .get_alloc_size      = */ ggml_backend_et_buffer_type_get_alloc_size,
+    /* .get_alloc_size_n    = */ NULL,
+    /* .is_host             = */ ggml_backend_et_buffer_type_is_host,
 };

 static const char * ggml_backend_et_get_name(ggml_backend_t backend) {
diff --git a/ggml/src/ggml-hexagon/ggml-hexagon.cpp b/ggml/src/ggml-hexagon/ggml-hexagon.cpp
index f59c49c46..454594472 100644
--- a/ggml/src/ggml-hexagon/ggml-hexagon.cpp
+++ b/ggml/src/ggml-hexagon/ggml-hexagon.cpp
@@ -2795,21 +2795,25 @@ static bool ggml_backend_hexagon_host_buffer_type_is_host(ggml_backend_buffer_ty
 }

 static ggml_backend_buffer_type_i ggml_backend_hexagon_buffer_type_interface = {
-    /* .get_name         = */ ggml_backend_hexagon_buffer_type_name,
-    /* .alloc_buffer     = */ ggml_backend_hexagon_buffer_type_alloc_buffer,
-    /* .get_alignment    = */ ggml_backend_hexagon_buffer_type_get_alignment,
-    /* .get_max_size     = */ ggml_backend_hexagon_buffer_type_get_max_size,
-    /* .get_alloc_size   = */ ggml_backend_hexagon_buffer_type_get_alloc_size,
-    /* .is_host          = */ ggml_backend_hexagon_buffer_type_is_host,
+    /* .get_name            = */ ggml_backend_hexagon_buffer_type_name,
+    /* .alloc_buffer        = */ ggml_backend_hexagon_buffer_type_alloc_buffer,
+    /* .alloc_buffer_n      = */ NULL,
+    /* .get_alignment       = */ ggml_backend_hexagon_buffer_type_get_alignment,
+    /* .get_max_size        = */ ggml_backend_hexagon_buffer_type_get_max_size,
+    /* .get_alloc_size      = */ ggml_backend_hexagon_buffer_type_get_alloc_size,
+    /* .get_alloc_size_n    = */ NULL,
+    /* .is_host             = */ ggml_backend_hexagon_buffer_type_is_host,
 };

 static ggml_backend_buffer_type_i ggml_backend_hexagon_host_buffer_type_interface = {
-    /* .get_name         = */ ggml_backend_hexagon_buffer_type_name,
-    /* .alloc_buffer     = */ ggml_backend_hexagon_host_buffer_type_alloc_buffer,
-    /* .get_alignment    = */ ggml_backend_hexagon_buffer_type_get_alignment,
-    /* .get_max_size     = */ ggml_backend_hexagon_buffer_type_get_max_size,
-    /* .get_alloc_size   = */ ggml_backend_hexagon_buffer_type_get_alloc_size,
-    /* .is_host          = */ ggml_backend_hexagon_host_buffer_type_is_host,
+    /* .get_name            = */ ggml_backend_hexagon_buffer_type_name,
+    /* .alloc_buffer        = */ ggml_backend_hexagon_host_buffer_type_alloc_buffer,
+    /* .alloc_buffer_n      = */ NULL,
+    /* .get_alignment       = */ ggml_backend_hexagon_buffer_type_get_alignment,
+    /* .get_max_size        = */ ggml_backend_hexagon_buffer_type_get_max_size,
+    /* .get_alloc_size      = */ ggml_backend_hexagon_buffer_type_get_alloc_size,
+    /* .get_alloc_size_n    = */ NULL,
+    /* .is_host             = */ ggml_backend_hexagon_host_buffer_type_is_host,
 };

 ggml_backend_hexagon_device_context::ggml_backend_hexagon_device_context(int dev_id, const ggml_hexagon_device_config & config, ggml_backend_dev_t dev)
diff --git a/ggml/src/ggml-metal/ggml-metal.cpp b/ggml/src/ggml-metal/ggml-metal.cpp
index c6c8ce836..344882e04 100644
--- a/ggml/src/ggml-metal/ggml-metal.cpp
+++ b/ggml/src/ggml-metal/ggml-metal.cpp
@@ -310,12 +310,14 @@ static ggml_backend_buffer_type_t ggml_backend_metal_buffer_type_shared(int devi

             ggml_backend_buffer_type buft = {
                 /* .iface = */ {
-                    /* .get_name         = */ ggml_backend_metal_buffer_type_shared_get_name,
-                    /* .alloc_buffer     = */ ggml_backend_metal_buffer_type_shared_alloc_buffer,
-                    /* .get_alignment    = */ ggml_backend_metal_buffer_type_shared_get_alignment,
-                    /* .get_max_size     = */ ggml_backend_metal_buffer_type_shared_get_max_size,
-                    /* .get_alloc_size   = */ ggml_backend_metal_buffer_type_shared_get_alloc_size,
-                    /* .is_host          = */ ggml_backend_metal_buffer_type_shared_is_host,
+                    /* .get_name            = */ ggml_backend_metal_buffer_type_shared_get_name,
+                    /* .alloc_buffer        = */ ggml_backend_metal_buffer_type_shared_alloc_buffer,
+                    /* .alloc_buffer_n      = */ NULL,
+                    /* .get_alignment       = */ ggml_backend_metal_buffer_type_shared_get_alignment,
+                    /* .get_max_size        = */ ggml_backend_metal_buffer_type_shared_get_max_size,
+                    /* .get_alloc_size      = */ ggml_backend_metal_buffer_type_shared_get_alloc_size,
+                    /* .get_alloc_size_n    = */ NULL,
+                    /* .is_host             = */ ggml_backend_metal_buffer_type_shared_is_host,
                 },
                 /* .device  = */ ggml_backend_reg_dev_get(ggml_backend_metal_reg(), i),
                 /* .context = */ raw_ctx,
@@ -385,12 +387,14 @@ static ggml_backend_buffer_type_t ggml_backend_metal_buffer_type_private(int dev

             ggml_backend_buffer_type buft = {
                 /* .iface = */ {
-                    /* .get_name         = */ ggml_backend_metal_buffer_type_private_get_name,
-                    /* .alloc_buffer     = */ ggml_backend_metal_buffer_type_private_alloc_buffer,
-                    /* .get_alignment    = */ ggml_backend_metal_buffer_type_private_get_alignment,
-                    /* .get_max_size     = */ ggml_backend_metal_buffer_type_private_get_max_size,
-                    /* .get_alloc_size   = */ ggml_backend_metal_buffer_type_private_get_alloc_size,
-                    /* .is_host          = */ ggml_backend_metal_buffer_type_private_is_host,
+                    /* .get_name            = */ ggml_backend_metal_buffer_type_private_get_name,
+                    /* .alloc_buffer        = */ ggml_backend_metal_buffer_type_private_alloc_buffer,
+                    /* .alloc_buffer_n      = */ NULL,
+                    /* .get_alignment       = */ ggml_backend_metal_buffer_type_private_get_alignment,
+                    /* .get_max_size        = */ ggml_backend_metal_buffer_type_private_get_max_size,
+                    /* .get_alloc_size      = */ ggml_backend_metal_buffer_type_private_get_alloc_size,
+                    /* .get_alloc_size_n    = */ NULL,
+                    /* .is_host             = */ ggml_backend_metal_buffer_type_private_is_host,
                 },
                 /* .device  = */ ggml_backend_reg_dev_get(ggml_backend_metal_reg(), i),
                 /* .context = */ raw_ctx,
@@ -463,12 +467,14 @@ static ggml_backend_buffer_type_t ggml_backend_metal_buffer_type_mapped(int devi
             //       https://github.com/ggml-org/llama.cpp/pull/15832#discussion_r2333177099
             ggml_backend_buffer_type buft = {
                 /* .iface = */ {
-                    /* .get_name         = */ ggml_backend_metal_buffer_type_mapped_get_name,
-                    /* .alloc_buffer     = */ ggml_backend_metal_buffer_type_mapped_alloc_buffer,
-                    /* .get_alignment    = */ ggml_backend_metal_buffer_type_mapped_get_alignment,
-                    /* .get_max_size     = */ ggml_backend_metal_buffer_type_mapped_get_max_size,
-                    /* .get_alloc_size   = */ ggml_backend_metal_buffer_type_mapped_get_alloc_size,
-                    /* .is_host          = */ ggml_backend_metal_buffer_type_mapped_is_host,
+                    /* .get_name            = */ ggml_backend_metal_buffer_type_mapped_get_name,
+                    /* .alloc_buffer        = */ ggml_backend_metal_buffer_type_mapped_alloc_buffer,
+                    /* .alloc_buffer_n      = */ NULL,
+                    /* .get_alignment       = */ ggml_backend_metal_buffer_type_mapped_get_alignment,
+                    /* .get_max_size        = */ ggml_backend_metal_buffer_type_mapped_get_max_size,
+                    /* .get_alloc_size      = */ ggml_backend_metal_buffer_type_mapped_get_alloc_size,
+                    /* .get_alloc_size_n    = */ NULL,
+                    /* .is_host             = */ ggml_backend_metal_buffer_type_mapped_is_host,
                 },
                 /* .device  = */ ggml_backend_reg_dev_get(ggml_backend_metal_reg(), i),
                 /* .context = */ raw_ctx,
diff --git a/ggml/src/ggml-opencl/ggml-opencl.cpp b/ggml/src/ggml-opencl/ggml-opencl.cpp
index acad7a67f..715a19404 100644
--- a/ggml/src/ggml-opencl/ggml-opencl.cpp
+++ b/ggml/src/ggml-opencl/ggml-opencl.cpp
@@ -12886,12 +12886,14 @@ static size_t ggml_backend_opencl_buffer_type_get_alloc_size(ggml_backend_buffer
 }

 static ggml_backend_buffer_type_i ggml_backend_opencl_buffer_type_interface = {
-    /* .get_name         = */ ggml_backend_opencl_buffer_type_get_name,
-    /* .alloc_buffer     = */ ggml_backend_opencl_buffer_type_alloc_buffer,
-    /* .get_alignment    = */ ggml_backend_opencl_buffer_type_get_alignment,
-    /* .get_max_size     = */ ggml_backend_opencl_buffer_type_get_max_size,
-    /* .get_alloc_size   = */ ggml_backend_opencl_buffer_type_get_alloc_size,
-    /* .is_host          = */ NULL,
+    /* .get_name            = */ ggml_backend_opencl_buffer_type_get_name,
+    /* .alloc_buffer        = */ ggml_backend_opencl_buffer_type_alloc_buffer,
+    /* .alloc_buffer_n      = */ NULL,
+    /* .get_alignment       = */ ggml_backend_opencl_buffer_type_get_alignment,
+    /* .get_max_size        = */ ggml_backend_opencl_buffer_type_get_max_size,
+    /* .get_alloc_size      = */ ggml_backend_opencl_buffer_type_get_alloc_size,
+    /* .get_alloc_size_n    = */ NULL,
+    /* .is_host             = */ NULL,
 };

 //
diff --git a/ggml/src/ggml-openvino/ggml-openvino.cpp b/ggml/src/ggml-openvino/ggml-openvino.cpp
index d5050bff7..336c793a0 100644
--- a/ggml/src/ggml-openvino/ggml-openvino.cpp
+++ b/ggml/src/ggml-openvino/ggml-openvino.cpp
@@ -631,12 +631,14 @@ static size_t ggml_backend_openvino_buffer_type_get_alloc_size(ggml_backend_buff
 }

 static const ggml_backend_buffer_type_i ggml_backend_openvino_buffer_type_interface = {
-    /* .get_name         = */ ggml_backend_openvino_buffer_type_get_name,
-    /* .alloc_buffer     = */ ggml_backend_openvino_buffer_type_alloc_buffer,
-    /* .get_alignment    = */ ggml_backend_openvino_buffer_type_get_alignment,
-    /* .get_max_size     = */ ggml_backend_openvino_buffer_type_get_max_size,
-    /* .get_alloc_size   = */ ggml_backend_openvino_buffer_type_get_alloc_size,
-    /* .is_host          = */ nullptr,
+    /* .get_name            = */ ggml_backend_openvino_buffer_type_get_name,
+    /* .alloc_buffer        = */ ggml_backend_openvino_buffer_type_alloc_buffer,
+    /* .alloc_buffer_n      = */ NULL,
+    /* .get_alignment       = */ ggml_backend_openvino_buffer_type_get_alignment,
+    /* .get_max_size        = */ ggml_backend_openvino_buffer_type_get_max_size,
+    /* .get_alloc_size      = */ ggml_backend_openvino_buffer_type_get_alloc_size,
+    /* .get_alloc_size_n    = */ NULL,
+    /* .is_host             = */ nullptr,
 };

 // Get buffer type for a specific device
@@ -684,12 +686,14 @@ static bool ggml_backend_openvino_host_buffer_type_is_host(ggml_backend_buffer_t
 }

 static const ggml_backend_buffer_type_i ggml_backend_openvino_host_buffer_type_interface = {
-    /* .get_name         = */ ggml_backend_openvino_host_buffer_type_get_name,
-    /* .alloc_buffer     = */ ggml_backend_openvino_buffer_type_alloc_buffer,
-    /* .get_alignment    = */ ggml_backend_openvino_buffer_type_get_alignment,
-    /* .get_max_size     = */ ggml_backend_openvino_buffer_type_get_max_size,
-    /* .get_alloc_size   = */ ggml_backend_openvino_buffer_type_get_alloc_size,
-    /* .is_host          = */ ggml_backend_openvino_host_buffer_type_is_host,
+    /* .get_name            = */ ggml_backend_openvino_host_buffer_type_get_name,
+    /* .alloc_buffer        = */ ggml_backend_openvino_buffer_type_alloc_buffer,
+    /* .alloc_buffer_n      = */ NULL,
+    /* .get_alignment       = */ ggml_backend_openvino_buffer_type_get_alignment,
+    /* .get_max_size        = */ ggml_backend_openvino_buffer_type_get_max_size,
+    /* .get_alloc_size      = */ ggml_backend_openvino_buffer_type_get_alloc_size,
+    /* .get_alloc_size_n    = */ NULL,
+    /* .is_host             = */ ggml_backend_openvino_host_buffer_type_is_host,
 };

 GGML_BACKEND_API ggml_backend_buffer_type_t ggml_backend_openvino_host_buffer_type(int device) {
diff --git a/ggml/src/ggml-rpc/ggml-rpc.cpp b/ggml/src/ggml-rpc/ggml-rpc.cpp
index 353b79b07..158a15bb8 100644
--- a/ggml/src/ggml-rpc/ggml-rpc.cpp
+++ b/ggml/src/ggml-rpc/ggml-rpc.cpp
@@ -929,12 +929,14 @@ static size_t ggml_backend_rpc_buffer_type_get_alloc_size(ggml_backend_buffer_ty
 }

 static ggml_backend_buffer_type_i ggml_backend_rpc_buffer_type_interface = {
-    /* .get_name         = */ ggml_backend_rpc_buffer_type_name,
-    /* .alloc_buffer     = */ ggml_backend_rpc_buffer_type_alloc_buffer,
-    /* .get_alignment    = */ ggml_backend_rpc_buffer_type_get_alignment,
-    /* .get_max_size     = */ ggml_backend_rpc_get_max_size,
-    /* .get_alloc_size   = */ ggml_backend_rpc_buffer_type_get_alloc_size,
-    /* .is_host          = */ NULL,
+    /* .get_name            = */ ggml_backend_rpc_buffer_type_name,
+    /* .alloc_buffer        = */ ggml_backend_rpc_buffer_type_alloc_buffer,
+    /* .alloc_buffer_n      = */ NULL,
+    /* .get_alignment       = */ ggml_backend_rpc_buffer_type_get_alignment,
+    /* .get_max_size        = */ ggml_backend_rpc_get_max_size,
+    /* .get_alloc_size      = */ ggml_backend_rpc_buffer_type_get_alloc_size,
+    /* .get_alloc_size_n    = */ NULL,
+    /* .is_host             = */ NULL,
 };

 static const char * ggml_backend_rpc_name(ggml_backend_t backend) {
diff --git a/ggml/src/ggml-sycl/ggml-sycl.cpp b/ggml/src/ggml-sycl/ggml-sycl.cpp
index 2bc2aaa32..6495ed433 100644
--- a/ggml/src/ggml-sycl/ggml-sycl.cpp
+++ b/ggml/src/ggml-sycl/ggml-sycl.cpp
@@ -1065,12 +1065,14 @@ static size_t ggml_backend_sycl_buffer_type_get_alloc_size(ggml_backend_buffer_t
 }

 static const ggml_backend_buffer_type_i ggml_backend_sycl_buffer_type_interface = {
-    /* .get_name         = */ ggml_backend_sycl_buffer_type_get_name,
-    /* .alloc_buffer     = */ ggml_backend_sycl_buffer_type_alloc_buffer,
-    /* .get_alignment    = */ ggml_backend_sycl_buffer_type_get_alignment,
-    /* .get_max_size     = */ ggml_backend_sycl_buffer_type_get_max_size,
-    /* .get_alloc_size   = */ ggml_backend_sycl_buffer_type_get_alloc_size,
-    /* .is_host          = */ NULL,
+    /* .get_name            = */ ggml_backend_sycl_buffer_type_get_name,
+    /* .alloc_buffer        = */ ggml_backend_sycl_buffer_type_alloc_buffer,
+    /* .alloc_buffer_n      = */ NULL,
+    /* .get_alignment       = */ ggml_backend_sycl_buffer_type_get_alignment,
+    /* .get_max_size        = */ ggml_backend_sycl_buffer_type_get_max_size,
+    /* .get_alloc_size      = */ ggml_backend_sycl_buffer_type_get_alloc_size,
+    /* .get_alloc_size_n    = */ NULL,
+    /* .is_host             = */ NULL,
 };

 ggml_backend_buffer_type_t ggml_backend_sycl_buffer_type(int device) {
@@ -1501,12 +1503,14 @@ static bool ggml_backend_sycl_split_buffer_type_is_host(ggml_backend_buffer_type
 }

 static ggml_backend_buffer_type_i ggml_backend_sycl_split_buffer_type_interface = {
-    /* .get_name         = */ ggml_backend_sycl_split_buffer_type_get_name,
-    /* .alloc_buffer     = */ ggml_backend_sycl_split_buffer_type_alloc_buffer,
-    /* .get_alignment    = */ ggml_backend_sycl_split_buffer_type_get_alignment,
-    /* .get_max_size     = */ NULL, // defaults to SIZE_MAX
-    /* .get_alloc_size   = */ ggml_backend_sycl_split_buffer_type_get_alloc_size,
-    /* .is_host          = */ ggml_backend_sycl_split_buffer_type_is_host,
+    /* .get_name            = */ ggml_backend_sycl_split_buffer_type_get_name,
+    /* .alloc_buffer        = */ ggml_backend_sycl_split_buffer_type_alloc_buffer,
+    /* .alloc_buffer_n      = */ NULL,
+    /* .get_alignment       = */ ggml_backend_sycl_split_buffer_type_get_alignment,
+    /* .get_max_size        = */ NULL, // defaults to SIZE_MAX
+    /* .get_alloc_size      = */ ggml_backend_sycl_split_buffer_type_get_alloc_size,
+    /* .get_alloc_size_n    = */ NULL,
+    /* .is_host             = */ ggml_backend_sycl_split_buffer_type_is_host,
 };

 ggml_backend_buffer_type_t ggml_backend_sycl_split_buffer_type(int main_device, const float * tensor_split) {
@@ -1645,12 +1649,14 @@ static ggml_backend_buffer_type_t ggml_backend_sycl_host_buffer_type_for_device(
         for (size_t i = 0; i < bufts.size(); i++) {
             bufts[i] = {
                 /* .iface    = */ {
-                    /* .get_name         = */ ggml_backend_sycl_host_buffer_type_name,
-                    /* .alloc_buffer     = */ ggml_backend_sycl_host_buffer_type_alloc_buffer,
-                    /* .get_alignment    = */ ggml_backend_cpu_buffer_type()->iface.get_alignment,
-                    /* .get_max_size     = */ ggml_backend_sycl_host_buffer_type_get_max_size,
-                    /* .get_alloc_size   = */ ggml_backend_cpu_buffer_type()->iface.get_alloc_size,
-                    /* .is_host          = */ ggml_backend_cpu_buffer_type()->iface.is_host,
+                    /* .get_name            = */ ggml_backend_sycl_host_buffer_type_name,
+                    /* .alloc_buffer        = */ ggml_backend_sycl_host_buffer_type_alloc_buffer,
+                    /* .alloc_buffer_n      = */ NULL,
+                    /* .get_alignment       = */ ggml_backend_cpu_buffer_type()->iface.get_alignment,
+                    /* .get_max_size        = */ ggml_backend_sycl_host_buffer_type_get_max_size,
+                    /* .get_alloc_size      = */ ggml_backend_cpu_buffer_type()->iface.get_alloc_size,
+                    /* .get_alloc_size_n    = */ NULL,
+                    /* .is_host             = */ ggml_backend_cpu_buffer_type()->iface.is_host,
                 },
                 /* .device   = */ ggml_backend_reg_dev_get(ggml_backend_sycl_reg(), i),
                 /* .context  = */ nullptr,
diff --git a/ggml/src/ggml-virtgpu/ggml-backend-buffer-type.cpp b/ggml/src/ggml-virtgpu/ggml-backend-buffer-type.cpp
index d5bdc993b..d105a2f46 100644
--- a/ggml/src/ggml-virtgpu/ggml-backend-buffer-type.cpp
+++ b/ggml/src/ggml-virtgpu/ggml-backend-buffer-type.cpp
@@ -63,19 +63,23 @@ static size_t ggml_backend_remoting_buffer_type_get_alloc_size(ggml_backend_buff
 }

 const ggml_backend_buffer_type_i ggml_backend_remoting_buffer_type_interface = {
-    /* .get_name         = */ ggml_backend_remoting_buffer_type_get_name,
-    /* .alloc_buffer     = */ ggml_backend_remoting_buffer_type_alloc_buffer,
-    /* .get_alignment    = */ ggml_backend_remoting_buffer_type_get_alignment,
-    /* .get_max_size     = */ ggml_backend_remoting_buffer_type_get_max_size,
-    /* .get_alloc_size   = */ ggml_backend_remoting_buffer_type_get_alloc_size,
-    /* .is_host          = */ NULL,
+    /* .get_name            = */ ggml_backend_remoting_buffer_type_get_name,
+    /* .alloc_buffer        = */ ggml_backend_remoting_buffer_type_alloc_buffer,
+    /* .alloc_buffer_n      = */ NULL,
+    /* .get_alignment       = */ ggml_backend_remoting_buffer_type_get_alignment,
+    /* .get_max_size        = */ ggml_backend_remoting_buffer_type_get_max_size,
+    /* .get_alloc_size      = */ ggml_backend_remoting_buffer_type_get_alloc_size,
+    /* .get_alloc_size_n    = */ NULL,
+    /* .is_host             = */ NULL,
 };

 const ggml_backend_buffer_type_i ggml_backend_remoting_buffer_from_ptr_type_interface = {
-    /* .get_name         = */ ggml_backend_remoting_buffer_type_get_name,
-    /* .alloc_buffer     = */ NULL,
-    /* .get_alignment    = */ ggml_backend_remoting_buffer_type_get_alignment,
-    /* .get_max_size     = */ ggml_backend_remoting_buffer_type_get_max_size,
-    /* .get_alloc_size   = */ ggml_backend_remoting_buffer_type_get_alloc_size,
-    /* .is_host          = */ NULL,
+    /* .get_name            = */ ggml_backend_remoting_buffer_type_get_name,
+    /* .alloc_buffer        = */ NULL,
+    /* .alloc_buffer_n      = */ NULL,
+    /* .get_alignment       = */ ggml_backend_remoting_buffer_type_get_alignment,
+    /* .get_max_size        = */ ggml_backend_remoting_buffer_type_get_max_size,
+    /* .get_alloc_size      = */ ggml_backend_remoting_buffer_type_get_alloc_size,
+    /* .get_alloc_size_n    = */ NULL,
+    /* .is_host             = */ NULL,
 };
diff --git a/ggml/src/ggml-vulkan/ggml-vulkan-buffers.cpp b/ggml/src/ggml-vulkan/ggml-vulkan-buffers.cpp
index 4d4c84951..f7a21dd27 100644
--- a/ggml/src/ggml-vulkan/ggml-vulkan-buffers.cpp
+++ b/ggml/src/ggml-vulkan/ggml-vulkan-buffers.cpp
@@ -1,12 +1,14 @@
 #include "ggml-vulkan-common.h"

 ggml_backend_buffer_type_i ggml_backend_vk_buffer_type_interface = {
-    /* .get_name         = */ ggml_backend_vk_buffer_type_name,
-    /* .alloc_buffer     = */ ggml_backend_vk_buffer_type_alloc_buffer,
-    /* .get_alignment    = */ ggml_backend_vk_buffer_type_get_alignment,
-    /* .get_max_size     = */ ggml_backend_vk_buffer_type_get_max_size,
-    /* .get_alloc_size   = */ ggml_backend_vk_buffer_type_get_alloc_size,
-    /* .is_host          = */ NULL,
+    /* .get_name            = */ ggml_backend_vk_buffer_type_name,
+    /* .alloc_buffer        = */ ggml_backend_vk_buffer_type_alloc_buffer,
+    /* .alloc_buffer_n      = */ NULL,
+    /* .get_alignment       = */ ggml_backend_vk_buffer_type_get_alignment,
+    /* .get_max_size        = */ ggml_backend_vk_buffer_type_get_max_size,
+    /* .get_alloc_size      = */ ggml_backend_vk_buffer_type_get_alloc_size,
+    /* .get_alloc_size_n    = */ NULL,
+    /* .is_host             = */ NULL,
 };

 static std::vector<uint32_t> ggml_vk_find_memory_properties(const vk::PhysicalDeviceMemoryProperties* mem_props, vk::MemoryRequirements* mem_req, vk::MemoryPropertyFlags flags) {
diff --git a/ggml/src/ggml-vulkan/ggml-vulkan.cpp b/ggml/src/ggml-vulkan/ggml-vulkan.cpp
index 0c0f64e2b..a45d5e1f0 100644
--- a/ggml/src/ggml-vulkan/ggml-vulkan.cpp
+++ b/ggml/src/ggml-vulkan/ggml-vulkan.cpp
@@ -13114,12 +13114,14 @@ static size_t ggml_backend_vk_host_buffer_type_get_max_size(ggml_backend_buffer_
 ggml_backend_buffer_type_t ggml_backend_vk_host_buffer_type() {
     static struct ggml_backend_buffer_type ggml_backend_vk_buffer_type_host = {
         /* .iface    = */ {
-            /* .get_name         = */ ggml_backend_vk_host_buffer_type_name,
-            /* .alloc_buffer     = */ ggml_backend_vk_host_buffer_type_alloc_buffer,
-            /* .get_alignment    = */ ggml_backend_vk_host_buffer_type_get_alignment,
-            /* .get_max_size     = */ ggml_backend_vk_host_buffer_type_get_max_size,
-            /* .get_alloc_size   = */ ggml_backend_cpu_buffer_type()->iface.get_alloc_size,
-            /* .is_host          = */ ggml_backend_cpu_buffer_type()->iface.is_host,
+            /* .get_name            = */ ggml_backend_vk_host_buffer_type_name,
+            /* .alloc_buffer        = */ ggml_backend_vk_host_buffer_type_alloc_buffer,
+            /* .alloc_buffer_n      = */ nullptr,
+            /* .get_alignment       = */ ggml_backend_vk_host_buffer_type_get_alignment,
+            /* .get_max_size        = */ ggml_backend_vk_host_buffer_type_get_max_size,
+            /* .get_alloc_size      = */ ggml_backend_cpu_buffer_type()->iface.get_alloc_size,
+            /* .get_alloc_size_n    = */ NULL,
+            /* .is_host             = */ ggml_backend_cpu_buffer_type()->iface.is_host,
         },
         /* .device   = */ ggml_backend_reg_dev_get(ggml_backend_vk_reg(), 0),
         /* .context  = */ nullptr,
diff --git a/ggml/src/ggml-webgpu/ggml-webgpu.cpp b/ggml/src/ggml-webgpu/ggml-webgpu.cpp
index c5750ebbe..3feefd499 100644
--- a/ggml/src/ggml-webgpu/ggml-webgpu.cpp
+++ b/ggml/src/ggml-webgpu/ggml-webgpu.cpp
@@ -4325,12 +4325,14 @@ static ggml_backend_buffer_type_t ggml_backend_webgpu_device_get_buffer_type(ggm

     static struct ggml_backend_buffer_type ggml_backend_webgpu_buffer_type = {
         /* .iface = */ {
-                        /* .get_name       = */ ggml_backend_webgpu_buffer_type_get_name,
-                        /* .alloc_buffer   = */ ggml_backend_webgpu_buffer_type_alloc_buffer,
-                        /* .get_alignment  = */ ggml_backend_webgpu_buffer_type_get_alignment,
-                        /* .get_max_size   = */ ggml_backend_webgpu_buffer_type_get_max_size,
-                        /* .get_alloc_size = */ ggml_backend_webgpu_buffer_type_get_alloc_size,
-                        /* .is_host        = */ NULL,  // defaults to false
+                        /* .get_name            = */ ggml_backend_webgpu_buffer_type_get_name,
+                        /* .alloc_buffer        = */ ggml_backend_webgpu_buffer_type_alloc_buffer,
+                        /* .alloc_buffer_n      = */ NULL,
+                        /* .get_alignment       = */ ggml_backend_webgpu_buffer_type_get_alignment,
+                        /* .get_max_size        = */ ggml_backend_webgpu_buffer_type_get_max_size,
+                        /* .get_alloc_size      = */ ggml_backend_webgpu_buffer_type_get_alloc_size,
+                        /* .get_alloc_size_n    = */ NULL,
+                        /* .is_host             = */ NULL,  // defaults to false
         },
         /* .device  = */
         dev,
diff --git a/ggml/src/ggml-zdnn/ggml-zdnn.cpp b/ggml/src/ggml-zdnn/ggml-zdnn.cpp
index 46b37c05e..5d1209357 100644
--- a/ggml/src/ggml-zdnn/ggml-zdnn.cpp
+++ b/ggml/src/ggml-zdnn/ggml-zdnn.cpp
@@ -414,12 +414,14 @@ static bool ggml_backend_zdnn_buffer_type_is_host(ggml_backend_buffer_type_t buf
 ggml_backend_buffer_type_t ggml_backend_zdnn_buffer_type(void) {
     static ggml_backend_buffer_type ggml_backend_buffer_type_zdnn = {
         /* .iface   = */ {
-            /* .get_name       = */ ggml_backend_zdnn_buffer_type_get_name,
-            /* .alloc_buffer   = */ ggml_backend_zdnn_buffer_type_alloc_buffer,
-            /* .get_alignment  = */ ggml_backend_zdnn_buffer_type_get_alignment,
-            /* .get_max_size   = */ NULL,
-            /* .get_alloc_size = */ NULL,  // defaults to ggml_nbytes
-            /* .is_host        = */ ggml_backend_zdnn_buffer_type_is_host,
+            /* .get_name            = */ ggml_backend_zdnn_buffer_type_get_name,
+            /* .alloc_buffer        = */ ggml_backend_zdnn_buffer_type_alloc_buffer,
+            /* .alloc_buffer_n      = */ NULL,
+            /* .get_alignment       = */ ggml_backend_zdnn_buffer_type_get_alignment,
+            /* .get_max_size        = */ NULL,
+            /* .get_alloc_size      = */ NULL,  // defaults to ggml_nbytes
+            /* .get_alloc_size_n    = */ NULL,
+            /* .is_host             = */ ggml_backend_zdnn_buffer_type_is_host,
         },
         /* .device  = */ &g_ggml_backend_zdnn_device,
         /* .context = */ NULL,
diff --git a/tests/test-alloc.cpp b/tests/test-alloc.cpp
index 8f1a98aa0..45d9a3eda 100644
--- a/tests/test-alloc.cpp
+++ b/tests/test-alloc.cpp
@@ -5,7 +5,6 @@
 #include "ggml.h"

 #include <algorithm>
-#include <exception>
 #include <memory>
 #include <vector>

@@ -23,6 +22,9 @@ struct dummy_backend_context {
     ggml_backend                       backend;
     std::vector<ggml_backend_buffer_t> buffers;

+    bool custom_alloc_buffer_n_called   = false;
+    bool custom_get_alloc_size_n_called = false;
+
     size_t allocated_total() const {
         size_t n = 0;
         for (ggml_backend_buffer_t buf : buffers) {
@@ -34,7 +36,7 @@ struct dummy_backend_context {

 // ggml_backend_buffer_type interface

-static const char * dummy_backend_buffer_type_get_name(ggml_backend_buffer_type_t) {
+static const char * dummy_backend_buffer_type_get_name(ggml_backend_buffer_type_t /*buft*/) {
     return "dummy_buffer_type";
 }

@@ -55,7 +57,32 @@ static size_t dummy_backend_buffer_type_get_max_size(ggml_backend_buffer_type_t
     return ctx->max_buffer_size;
 }

-static bool dummy_backend_buffer_type_is_host(ggml_backend_buffer_type_t) {
+static ggml_backend_buffer_t dummy_backend_buffer_type_alloc_buffer_n_custom(
+        ggml_backend_buffer_type_t buft, ggml_tensor ** tensors, int n_tensors) {
+    dummy_backend_context * ctx = (dummy_backend_context *) buft->context;
+    ctx->custom_alloc_buffer_n_called = true;
+
+    ggml_backend_buffer_t buffer = ggml_backend_buft_alloc_buffer(buft, 64);
+    if (buffer == nullptr) {
+        return nullptr;
+    }
+    for (int i = 0; i < n_tensors; i++) {
+        tensors[i]->buffer = buffer;
+        tensors[i]->data   = (char *) ggml_backend_buffer_get_base(buffer);
+    }
+    return buffer;
+}
+
+static size_t dummy_backend_buffer_type_get_alloc_size_n_custom(
+        ggml_backend_buffer_type_t buft, ggml_tensor ** tensors, int n_tensors) {
+    dummy_backend_context * ctx = (dummy_backend_context *) buft->context;
+    GGML_UNUSED(tensors);
+    GGML_UNUSED(n_tensors);
+    ctx->custom_get_alloc_size_n_called = true;
+    return 64;
+}
+
+static bool dummy_backend_buffer_type_is_host(ggml_backend_buffer_type_t /*buft*/) {
     return true;
 }

@@ -69,29 +96,29 @@ static void dummy_backend_buffer_free_buffer(ggml_backend_buffer_t buffer) {
     ctx->buffers.erase(i);
 }

-static void * dummy_backend_buffer_get_base(ggml_backend_buffer_t) {
+static void * dummy_backend_buffer_get_base(ggml_backend_buffer_t /*buft*/) {
     return alloc_base;
 }

-static ggml_status dummy_backend_buffer_init_tensor(ggml_backend_buffer_t, ggml_tensor *) {
+static ggml_status dummy_backend_buffer_init_tensor(ggml_backend_buffer_t /*buft*/, ggml_tensor * /*x*/) {
     return GGML_STATUS_SUCCESS;
 }

-static void dummy_backend_buffer_memset_tensor(ggml_backend_buffer_t, ggml_tensor *, uint8_t, size_t, size_t) {}
+static void dummy_backend_buffer_memset_tensor(ggml_backend_buffer_t /*buft*/, ggml_tensor * /*x*/, uint8_t, size_t, size_t) {}

-static void dummy_backend_buffer_set_tensor(ggml_backend_buffer_t, ggml_tensor *, const void *, size_t, size_t) {}
+static void dummy_backend_buffer_set_tensor(ggml_backend_buffer_t /*buft*/, ggml_tensor * /*x*/, const void *, size_t, size_t) {}

-static void dummy_backend_buffer_get_tensor(ggml_backend_buffer_t, const ggml_tensor *, void *, size_t, size_t) {}
+static void dummy_backend_buffer_get_tensor(ggml_backend_buffer_t /*buft*/, const ggml_tensor * /*x*/, void *, size_t, size_t) {}

-static void dummy_backend_buffer_clear(ggml_backend_buffer_t, uint8_t) {}
+static void dummy_backend_buffer_clear(ggml_backend_buffer_t /*buft*/, uint8_t /*val*/) {}

 // ggml_backend_device interface

-static enum ggml_backend_dev_type dummy_backend_device_get_type(ggml_backend_dev_t) {
+static enum ggml_backend_dev_type dummy_backend_device_get_type(ggml_backend_dev_t /*dev*/) {
     return GGML_BACKEND_DEVICE_TYPE_CPU;
 }

-static bool dummy_backend_device_supports_op(ggml_backend_dev_t, const ggml_tensor *) {
+static bool dummy_backend_device_supports_op(ggml_backend_dev_t /*dev*/, const ggml_tensor * /*x*/) {
     return true;
 }

@@ -101,7 +128,7 @@ static bool dummy_backend_device_supports_buft(ggml_backend_dev_t device, ggml_b

 // ggml_backend interface

-static const char * dummy_backend_get_name(ggml_backend_t) {
+static const char * dummy_backend_get_name(ggml_backend_t /*backend*/) {
     return "dummy_backend";
 }

@@ -226,7 +253,7 @@ static void check_all_allocated(ggml_cgraph * graph) {

 static void check_max_size(ggml_context * ctx) {
     for (ggml_tensor * t = ggml_get_first_tensor(ctx); t; t = ggml_get_next_tensor(ctx, t)) {
-        auto   buft     = ggml_backend_buffer_get_type(t->buffer);
+        auto * buft     = ggml_backend_buffer_get_type(t->buffer);
         size_t max_size = ggml_backend_buft_get_max_size(buft);
         size_t offset   = (char *) t->data - (char *) ggml_backend_buffer_get_base(t->buffer);
         GGML_ASSERT(t->data >= ggml_backend_buffer_get_base(t->buffer));
@@ -615,7 +642,7 @@ static void test_reallocation() {
     }
 }

-static void test_backend_graph_optimize(ggml_backend_t, ggml_cgraph * graph, ggml_backend_graph_optimize_params * params) {
+static void test_backend_graph_optimize(ggml_backend_t /*backend*/, ggml_cgraph * graph, ggml_backend_graph_optimize_params * params) {
     GGML_ASSERT(graph->n_nodes == 3);
     params->add_alloc_dep(params->user_data, graph->nodes[0], graph->nodes[2]);
 }
@@ -650,6 +677,428 @@ static void test_graph_optimize_alloc_dep() {
     GGML_ASSERT(!graph_reuses_allocation(true));
 }

+// Check that the size reported by ggml_backend_alloc_ctx_tensors_from_buft_size
+// matches the actual size of the buffer allocated for the ctx tensors
+static ggml_backend_buffer_ptr check_size_matches(ggml_backend_buffer_type_t buft, ggml_context * ctx) {
+    std::vector<ggml_tensor *> tensors;
+    for (ggml_tensor * t = ggml_get_first_tensor(ctx); t != NULL; t = ggml_get_next_tensor(ctx, t)) {
+        tensors.push_back(t);
+    }
+
+    const size_t expected_size = ggml_backend_alloc_ctx_tensors_from_buft_size(ctx, buft);
+    GGML_ASSERT(ggml_backend_buft_get_alloc_size_n(buft, tensors.data(), (int) tensors.size()) == expected_size);
+
+    ggml_backend_buffer_ptr buffer(ggml_backend_alloc_ctx_tensors_from_buft(ctx, buft));
+    GGML_ASSERT((buffer != nullptr) == (expected_size != 0));
+    if (buffer) {
+        GGML_ASSERT(ggml_backend_buffer_get_size(buffer.get()) == expected_size);
+    }
+    return buffer;
+}
+
+// Check that all tensors are placed in a single buffer when they fit within the
+// backend's max size
+static void test_buft_alloc_buffer_n_single_buffer() {
+    dummy_backend backend      = dummy_backend_init(SIZE_MAX);
+    auto [ctx, graph, ctx_ptr] = make_context();
+
+    ggml_tensor * x[3];
+    x[0] = make_input_with_size(ctx, 8);
+    x[1] = make_input_with_size(ctx, 8);
+    x[2] = ggml_add(ctx, x[0], x[1]);
+    assign_names(ctx);
+
+    ggml_tensor * tensors[3] = { x[0], x[1], x[2] };
+    ggml_backend_buffer_ptr buffer(ggml_backend_buft_alloc_buffer_n(&backend.buffer_type, tensors, 3));
+    GGML_ASSERT(buffer != nullptr);
+    GGML_ASSERT(!ggml_backend_buffer_is_multi_buffer(buffer.get()));
+    GGML_ASSERT(backend.context->buffers.size() == 1);
+    GGML_ASSERT(ggml_backend_buffer_get_size(buffer.get()) == 24);
+    for (ggml_tensor * t : tensors) {
+        GGML_ASSERT(t->buffer == buffer.get());
+        GGML_ASSERT(t->data != nullptr);
+    }
+    for (int i = 0; i < 3; i++) {
+        for (int j = i + 1; j < 3; j++) {
+            GGML_ASSERT(!memory_overlap(tensors[i], tensors[j]));
+        }
+    }
+}
+
+// Check that tensors are split across multiple underlying buffers when they don't
+// fit within the backend's max size
+static void test_buft_alloc_buffer_n_multi_buffer() {
+    dummy_backend backend      = dummy_backend_init(16);
+    auto [ctx, graph, ctx_ptr] = make_context();
+
+    ggml_tensor * x[4];
+    x[0] = make_input_with_size(ctx, 8);
+    x[1] = make_input_with_size(ctx, 8);
+    x[2] = make_input_with_size(ctx, 8);
+    x[3] = make_input_with_size(ctx, 8);
+    assign_names(ctx);
+
+    ggml_tensor * tensors[4] = { x[0], x[1], x[2], x[3] };
+    ggml_backend_buffer_ptr buffer(ggml_backend_buft_alloc_buffer_n(&backend.buffer_type, tensors, 4));
+    GGML_ASSERT(buffer != nullptr);
+    GGML_ASSERT(ggml_backend_buffer_is_multi_buffer(buffer.get()));
+    GGML_ASSERT(backend.context->buffers.size() == 2);
+    GGML_ASSERT(ggml_backend_buffer_get_size(buffer.get()) == 32);
+    for (ggml_tensor * t : tensors) {
+        GGML_ASSERT(t->buffer != nullptr);
+        GGML_ASSERT(t->data != nullptr);
+    }
+    for (int i = 0; i < 4; i++) {
+        for (int j = i + 1; j < 4; j++) {
+            GGML_ASSERT(!memory_overlap(tensors[i], tensors[j]));
+        }
+    }
+}
+
+// Check that allocating tensors with zero total size returns nullptr
+static void test_buft_alloc_buffer_n_zero_size() {
+    dummy_backend backend      = dummy_backend_init(SIZE_MAX);
+    auto [ctx, graph, ctx_ptr] = make_context();
+
+    ggml_tensor * x[2];
+    x[0] = make_input_1d(ctx, 0);
+    x[1] = make_input_1d(ctx, 0);
+    assign_names(ctx);
+
+    ggml_tensor * tensors[2] = { x[0], x[1] };
+    GGML_ASSERT(ggml_backend_buft_alloc_buffer_n(&backend.buffer_type, tensors, 2) == nullptr);
+    GGML_ASSERT(ggml_backend_alloc_ctx_tensors_from_buft(ctx, &backend.buffer_type) == nullptr);
+    GGML_ASSERT(ggml_backend_alloc_ctx_tensors_from_buft_size(ctx, &backend.buffer_type) == 0);
+}
+
+// Check that tensors already allocated to a buffer are left in place, and only the
+// remaining tensors are allocated to a new buffer
+static void test_buft_alloc_buffer_n_already_allocated() {
+    dummy_backend backend      = dummy_backend_init(SIZE_MAX);
+    auto [ctx, graph, ctx_ptr] = make_context();
+
+    ggml_tensor * a[2];
+    a[0] = make_input_with_size(ctx, 8);
+    a[1] = make_input_with_size(ctx, 8);
+    assign_names(ctx, "a");
+
+    ggml_backend_buffer_ptr buf_a(ggml_backend_alloc_ctx_tensors_from_buft(ctx, &backend.buffer_type));
+    GGML_ASSERT(buf_a != nullptr);
+
+    ggml_tensor * b = make_input_with_size(ctx, 8);
+    assign_names(ctx, "b");
+
+    GGML_ASSERT(ggml_backend_alloc_ctx_tensors_from_buft_size(ctx, &backend.buffer_type) == 8);
+    GGML_ASSERT(ggml_backend_buft_alloc_buffer_n(&backend.buffer_type, a, 2) == nullptr);
+
+    ggml_tensor * tensors[3] = { a[0], a[1], b };
+    ggml_backend_buffer_ptr buf_b(ggml_backend_buft_alloc_buffer_n(&backend.buffer_type, tensors, 3));
+    GGML_ASSERT(buf_b != nullptr);
+    GGML_ASSERT(backend.context->buffers.size() == 2);
+    GGML_ASSERT(a[0]->buffer == buf_a.get());
+    GGML_ASSERT(a[1]->buffer == buf_a.get());
+    GGML_ASSERT(b->buffer == buf_b.get());
+    GGML_ASSERT(ggml_backend_buffer_get_size(buf_b.get()) == 8);
+    GGML_ASSERT(ggml_backend_alloc_ctx_tensors_from_buft(ctx, &backend.buffer_type) == nullptr);
+}
+
+// Check that views don't require any extra memory and share the base tensor's data
+static void test_buft_alloc_buffer_n_views() {
+    dummy_backend backend      = dummy_backend_init(SIZE_MAX);
+    auto [ctx, graph, ctx_ptr] = make_context();
+
+    ggml_tensor * base  = make_input_1d(ctx, 4);
+    ggml_tensor * view  = ggml_view_1d(ctx, base, 2, 0);
+    ggml_tensor * extra = make_input_1d(ctx, 2);
+    assign_names(ctx);
+
+    ggml_tensor * tensors[3] = { base, view, extra };
+    ggml_backend_buffer_ptr buffer(ggml_backend_buft_alloc_buffer_n(&backend.buffer_type, tensors, 3));
+    GGML_ASSERT(buffer != nullptr);
+    GGML_ASSERT(backend.context->buffers.size() == 1);
+    GGML_ASSERT(ggml_backend_buffer_get_size(buffer.get()) == 24);
+    GGML_ASSERT(base->buffer == buffer.get());
+    GGML_ASSERT(view->buffer == buffer.get());
+    GGML_ASSERT(extra->buffer == buffer.get());
+    GGML_ASSERT(view->data == base->data);
+}
+
+// Check that the reported size matches the allocated size in various scenarios:
+// single buffer, multi buffer, and views
+static void test_alloc_ctx_tensors_from_buft_size_matches() {
+    {
+        dummy_backend backend      = dummy_backend_init(SIZE_MAX, 8);
+        auto [ctx, graph, ctx_ptr] = make_context();
+
+        ggml_tensor * x[3];
+        x[0] = make_input_with_size(ctx, 4);
+        x[1] = make_input_with_size(ctx, 4);
+        x[2] = make_input_with_size(ctx, 8);
+        assign_names(ctx);
+
+        GGML_UNUSED(x);
+
+        ggml_backend_buffer_ptr buffer = check_size_matches(&backend.buffer_type, ctx);
+        GGML_ASSERT(buffer != nullptr);
+        GGML_ASSERT(backend.context->allocated_total() == 24);
+    }
+    {
+        dummy_backend backend      = dummy_backend_init(16);
+        auto [ctx, graph, ctx_ptr] = make_context();
+
+        ggml_tensor * x[4];
+        x[0] = make_input_with_size(ctx, 8);
+        x[1] = make_input_with_size(ctx, 8);
+        x[2] = make_input_with_size(ctx, 8);
+        x[3] = make_input_with_size(ctx, 8);
+        assign_names(ctx);
+
+        GGML_UNUSED(x);
+
+        ggml_backend_buffer_ptr buffer = check_size_matches(&backend.buffer_type, ctx);
+        GGML_ASSERT(buffer != nullptr);
+        GGML_ASSERT(backend.context->allocated_total() == 32);
+    }
+    {
+        dummy_backend backend      = dummy_backend_init(SIZE_MAX);
+        auto [ctx, graph, ctx_ptr] = make_context();
+
+        ggml_tensor * base  = make_input_1d(ctx, 4);
+        ggml_tensor * view  = ggml_view_1d(ctx, base, 2, 0);
+        ggml_tensor * extra = make_input_1d(ctx, 2);
+        assign_names(ctx);
+
+        GGML_UNUSED(view);
+        GGML_UNUSED(extra);
+
+        ggml_backend_buffer_ptr buffer = check_size_matches(&backend.buffer_type, ctx);
+        GGML_ASSERT(buffer != nullptr);
+        GGML_ASSERT(backend.context->allocated_total() == 24);
+    }
+}
+
+// Check that a backend-provided alloc_buffer_n implementation takes precedence over
+// the default one
+static void test_buft_alloc_buffer_n_custom_override() {
+    dummy_backend backend = dummy_backend_init(SIZE_MAX);
+    backend.buffer_type.iface.alloc_buffer_n = dummy_backend_buffer_type_alloc_buffer_n_custom;
+
+    auto [ctx, graph, ctx_ptr] = make_context();
+
+    ggml_tensor * x[2];
+    x[0] = make_input_with_size(ctx, 8);
+    x[1] = make_input_with_size(ctx, 8);
+    assign_names(ctx);
+
+    ggml_tensor * tensors[2] = { x[0], x[1] };
+    ggml_backend_buffer_ptr buffer(ggml_backend_buft_alloc_buffer_n(&backend.buffer_type, tensors, 2));
+    GGML_ASSERT(buffer != nullptr);
+    GGML_ASSERT(backend.context->custom_alloc_buffer_n_called);
+    GGML_ASSERT(backend.context->buffers.size() == 1);
+    GGML_ASSERT(ggml_backend_buffer_get_size(buffer.get()) == 64);
+    GGML_ASSERT(x[0]->buffer == buffer.get());
+    GGML_ASSERT(x[1]->buffer == buffer.get());
+}
+
+// Check that get_alloc_size_n predicts a single buffer allocation
+static void test_buft_get_alloc_size_n_single_buffer() {
+    dummy_backend backend      = dummy_backend_init(SIZE_MAX);
+    auto [ctx, graph, ctx_ptr] = make_context();
+
+    ggml_tensor * x[2];
+    x[0] = make_input_with_size(ctx, 8);
+    x[1] = make_input_with_size(ctx, 8);
+    assign_names(ctx);
+
+    ggml_tensor * tensors[2] = { x[0], x[1] };
+    GGML_ASSERT(ggml_backend_buft_get_alloc_size_n(&backend.buffer_type, tensors, 2) == 16);
+
+    ggml_backend_buffer_ptr buffer(ggml_backend_buft_alloc_buffer_n(&backend.buffer_type, tensors, 2));
+    GGML_ASSERT(buffer != nullptr);
+    GGML_ASSERT(ggml_backend_buffer_get_size(buffer.get()) == 16);
+}
+
+// Check that get_alloc_size_n accounts for splitting into multiple buffers
+static void test_buft_get_alloc_size_n_multi_buffer() {
+    dummy_backend backend      = dummy_backend_init(16);
+    auto [ctx, graph, ctx_ptr] = make_context();
+
+    ggml_tensor * x[4];
+    x[0] = make_input_with_size(ctx, 8);
+    x[1] = make_input_with_size(ctx, 8);
+    x[2] = make_input_with_size(ctx, 8);
+    x[3] = make_input_with_size(ctx, 8);
+    assign_names(ctx);
+
+    ggml_tensor * tensors[4] = { x[0], x[1], x[2], x[3] };
+    GGML_ASSERT(ggml_backend_buft_get_alloc_size_n(&backend.buffer_type, tensors, 4) == 32);
+
+    ggml_backend_buffer_ptr buffer(ggml_backend_buft_alloc_buffer_n(&backend.buffer_type, tensors, 4));
+    GGML_ASSERT(buffer != nullptr);
+    GGML_ASSERT(ggml_backend_buffer_is_multi_buffer(buffer.get()));
+    GGML_ASSERT(ggml_backend_buffer_get_size(buffer.get()) == 32);
+}
+
+// Check that a tensor larger than max_size is still accounted for
+static void test_buft_get_alloc_size_n_single_tensor_exceeds_max() {
+    dummy_backend backend      = dummy_backend_init(8);
+    auto [ctx, graph, ctx_ptr] = make_context();
+
+    ggml_tensor * x = make_input_with_size(ctx, 16);
+    assign_names(ctx);
+
+    ggml_tensor * tensors[1] = { x };
+    GGML_ASSERT(ggml_backend_buft_get_alloc_size_n(&backend.buffer_type, tensors, 1) == 16);
+
+    ggml_backend_buffer_ptr buffer(ggml_backend_buft_alloc_buffer_n(&backend.buffer_type, tensors, 1));
+    GGML_ASSERT(buffer != nullptr);
+    GGML_ASSERT(ggml_backend_buffer_get_size(buffer.get()) == 16);
+}
+
+// Check that zero-size tensors report a total allocation size of 0
+static void test_buft_get_alloc_size_n_zero_size() {
+    dummy_backend backend      = dummy_backend_init(SIZE_MAX);
+    auto [ctx, graph, ctx_ptr] = make_context();
+
+    ggml_tensor * x[2];
+    x[0] = make_input_1d(ctx, 0);
+    x[1] = make_input_1d(ctx, 0);
+    assign_names(ctx);
+
+    ggml_tensor * tensors[2] = { x[0], x[1] };
+    GGML_ASSERT(ggml_backend_buft_get_alloc_size_n(&backend.buffer_type, tensors, 2) == 0);
+    GGML_ASSERT(ggml_backend_buft_alloc_buffer_n(&backend.buffer_type, tensors, 2) == nullptr);
+}
+
+// Check that already-allocated tensors are not counted again
+static void test_buft_get_alloc_size_n_already_allocated() {
+    dummy_backend backend      = dummy_backend_init(SIZE_MAX);
+    auto [ctx, graph, ctx_ptr] = make_context();
+
+    ggml_tensor * a[2];
+    a[0] = make_input_with_size(ctx, 8);
+    a[1] = make_input_with_size(ctx, 8);
+    assign_names(ctx, "a");
+
+    ggml_backend_buffer_ptr buf_a(ggml_backend_alloc_ctx_tensors_from_buft(ctx, &backend.buffer_type));
+    GGML_ASSERT(buf_a != nullptr);
+
+    ggml_tensor * b = make_input_with_size(ctx, 8);
+    assign_names(ctx, "b");
+
+    ggml_tensor * tensors[3] = { a[0], a[1], b };
+    GGML_ASSERT(ggml_backend_buft_get_alloc_size_n(&backend.buffer_type, tensors, 3) == 8);
+
+    ggml_backend_buffer_ptr buf_b(ggml_backend_buft_alloc_buffer_n(&backend.buffer_type, tensors, 3));
+    GGML_ASSERT(buf_b != nullptr);
+    GGML_ASSERT(backend.context->buffers.size() == 2);
+    GGML_ASSERT(a[0]->buffer == buf_a.get());
+    GGML_ASSERT(a[1]->buffer == buf_a.get());
+    GGML_ASSERT(b->buffer == buf_b.get());
+    GGML_ASSERT(ggml_backend_buffer_get_size(buf_b.get()) == 8);
+}
+
+// Check that views do not add to the projected allocation size
+static void test_buft_get_alloc_size_n_views() {
+    dummy_backend backend      = dummy_backend_init(SIZE_MAX);
+    auto [ctx, graph, ctx_ptr] = make_context();
+
+    ggml_tensor * base  = make_input_1d(ctx, 4);
+    ggml_tensor * view  = ggml_view_1d(ctx, base, 2, 0);
+    ggml_tensor * extra = make_input_1d(ctx, 2);
+    assign_names(ctx);
+
+    ggml_tensor * tensors[3] = { base, view, extra };
+    GGML_ASSERT(ggml_backend_buft_get_alloc_size_n(&backend.buffer_type, tensors, 3) == 24);
+
+    ggml_backend_buffer_ptr buffer(ggml_backend_buft_alloc_buffer_n(&backend.buffer_type, tensors, 3));
+    GGML_ASSERT(buffer != nullptr);
+    GGML_ASSERT(ggml_backend_buffer_get_size(buffer.get()) == 24);
+}
+
+// Check that n_tensors == 0 is handled
+static void test_buft_get_alloc_size_n_n_tensors_zero() {
+    dummy_backend backend = dummy_backend_init(SIZE_MAX);
+    GGML_ASSERT(ggml_backend_buft_get_alloc_size_n(&backend.buffer_type, nullptr, 0) == 0);
+    GGML_ASSERT(ggml_backend_buft_alloc_buffer_n(&backend.buffer_type, nullptr, 0) == nullptr);
+}
+
+// Check that get_alloc_size_n respects the backend alignment
+static void test_buft_get_alloc_size_n_alignment() {
+    dummy_backend backend      = dummy_backend_init(SIZE_MAX, 16);
+    auto [ctx, graph, ctx_ptr] = make_context();
+
+    ggml_tensor * x[2];
+    x[0] = make_input_with_size(ctx, 4);
+    x[1] = make_input_with_size(ctx, 4);
+    assign_names(ctx);
+
+    ggml_tensor * tensors[2] = { x[0], x[1] };
+    GGML_ASSERT(ggml_backend_buft_get_alloc_size_n(&backend.buffer_type, tensors, 2) == 32);
+
+    ggml_backend_buffer_ptr buffer(ggml_backend_buft_alloc_buffer_n(&backend.buffer_type, tensors, 2));
+    GGML_ASSERT(buffer != nullptr);
+    GGML_ASSERT(ggml_backend_buffer_get_size(buffer.get()) == 32);
+}
+
+// Check that a backend-provided get_alloc_size_n implementation takes precedence over
+// the default one
+static void test_buft_get_alloc_size_n_custom_override() {
+    dummy_backend backend = dummy_backend_init(SIZE_MAX);
+    backend.buffer_type.iface.alloc_buffer_n   = dummy_backend_buffer_type_alloc_buffer_n_custom;
+    backend.buffer_type.iface.get_alloc_size_n = dummy_backend_buffer_type_get_alloc_size_n_custom;
+
+    auto [ctx, graph, ctx_ptr] = make_context();
+
+    ggml_tensor * x[2];
+    x[0] = make_input_with_size(ctx, 8);
+    x[1] = make_input_with_size(ctx, 8);
+    assign_names(ctx);
+
+    ggml_tensor * tensors[2] = { x[0], x[1] };
+    GGML_ASSERT(ggml_backend_buft_get_alloc_size_n(&backend.buffer_type, tensors, 2) == 64);
+    GGML_ASSERT(backend.context->custom_get_alloc_size_n_called);
+
+    ggml_backend_buffer_ptr buffer(ggml_backend_buft_alloc_buffer_n(&backend.buffer_type, tensors, 2));
+    GGML_ASSERT(buffer != nullptr);
+    GGML_ASSERT(backend.context->custom_alloc_buffer_n_called);
+    GGML_ASSERT(ggml_backend_buffer_get_size(buffer.get()) == 64);
+}
+
+// Check that the custom get_alloc_size_n override is used by the context size helper
+static void test_buft_get_alloc_size_n_custom_override_ctx_size() {
+    dummy_backend backend = dummy_backend_init(SIZE_MAX);
+    backend.buffer_type.iface.get_alloc_size_n = dummy_backend_buffer_type_get_alloc_size_n_custom;
+
+    auto [ctx, graph, ctx_ptr] = make_context();
+
+    ggml_tensor * x[2];
+    x[0] = make_input_with_size(ctx, 8);
+    x[1] = make_input_with_size(ctx, 8);
+    assign_names(ctx);
+
+    GGML_UNUSED(x);
+
+    GGML_ASSERT(ggml_backend_alloc_ctx_tensors_from_buft_size(ctx, &backend.buffer_type) == 64);
+}
+
+// Check that querying the projected allocation size does not allocate anything
+static void test_buft_get_alloc_size_n_does_not_allocate() {
+    dummy_backend backend      = dummy_backend_init(SIZE_MAX);
+    auto [ctx, graph, ctx_ptr] = make_context();
+
+    ggml_tensor * x[2];
+    x[0] = make_input_with_size(ctx, 8);
+    x[1] = make_input_with_size(ctx, 8);
+    assign_names(ctx);
+
+    ggml_tensor * tensors[2] = { x[0], x[1] };
+    GGML_ASSERT(ggml_backend_buft_get_alloc_size_n(&backend.buffer_type, tensors, 2) == 16);
+    GGML_ASSERT(backend.context->buffers.empty());
+    GGML_ASSERT(x[0]->data == nullptr);
+    GGML_ASSERT(x[1]->data == nullptr);
+}
+
 static void run(const char * name, void (*f)()) {
     printf("%s ", name);
     fflush(stdout);
@@ -672,5 +1121,23 @@ int main() {
     run("test_buffer_size_zero", test_buffer_size_zero);
     run("test_reallocation", test_reallocation);
     run("test_graph_optimize_alloc_dep", test_graph_optimize_alloc_dep);
+    run("test_buft_alloc_buffer_n_single_buffer", test_buft_alloc_buffer_n_single_buffer);
+    run("test_buft_alloc_buffer_n_multi_buffer", test_buft_alloc_buffer_n_multi_buffer);
+    run("test_buft_alloc_buffer_n_zero_size", test_buft_alloc_buffer_n_zero_size);
+    run("test_buft_alloc_buffer_n_already_allocated", test_buft_alloc_buffer_n_already_allocated);
+    run("test_buft_alloc_buffer_n_views", test_buft_alloc_buffer_n_views);
+    run("test_alloc_ctx_tensors_from_buft_size_matches", test_alloc_ctx_tensors_from_buft_size_matches);
+    run("test_buft_alloc_buffer_n_custom_override", test_buft_alloc_buffer_n_custom_override);
+    run("test_buft_get_alloc_size_n_single_buffer", test_buft_get_alloc_size_n_single_buffer);
+    run("test_buft_get_alloc_size_n_multi_buffer", test_buft_get_alloc_size_n_multi_buffer);
+    run("test_buft_get_alloc_size_n_single_tensor_exceeds_max", test_buft_get_alloc_size_n_single_tensor_exceeds_max);
+    run("test_buft_get_alloc_size_n_zero_size", test_buft_get_alloc_size_n_zero_size);
+    run("test_buft_get_alloc_size_n_already_allocated", test_buft_get_alloc_size_n_already_allocated);
+    run("test_buft_get_alloc_size_n_views", test_buft_get_alloc_size_n_views);
+    run("test_buft_get_alloc_size_n_n_tensors_zero", test_buft_get_alloc_size_n_n_tensors_zero);
+    run("test_buft_get_alloc_size_n_alignment", test_buft_get_alloc_size_n_alignment);
+    run("test_buft_get_alloc_size_n_custom_override", test_buft_get_alloc_size_n_custom_override);
+    run("test_buft_get_alloc_size_n_custom_override_ctx_size", test_buft_get_alloc_size_n_custom_override_ctx_size);
+    run("test_buft_get_alloc_size_n_does_not_allocate", test_buft_get_alloc_size_n_does_not_allocate);
     return 0;
 }