Commit 4d756bc72 for llama.cpp

commit 4d756bc72bf00a4aacf410ae15a2d315f3db400d
Author: Ruben Ortlam <rortlam@redhat.com>
Date:   Wed Oct 7 08:20:04 2026 +0200

    vulkan: fix amd iGPU slow checkpoint read (#30049)

diff --git a/ggml/src/ggml-vulkan/ggml-vulkan-buffers.cpp b/ggml/src/ggml-vulkan/ggml-vulkan-buffers.cpp
index f7a21dd27..351951638 100644
--- a/ggml/src/ggml-vulkan/ggml-vulkan-buffers.cpp
+++ b/ggml/src/ggml-vulkan/ggml-vulkan-buffers.cpp
@@ -623,7 +623,10 @@ void ggml_vk_buffer_read_2d(vk_buffer& src, size_t offset, void * dst, size_t sp
     // If the device is not an UMA device the memory is host-accessible through rebar. While writing
     // through PCIe is sufficient fast reading back data from PCIe is slower than going through
     // the HW device to host copy path.
-    if(src->memory_property_flags & vk::MemoryPropertyFlagBits::eHostVisible && src->device->uma) {
+    // AMD UMA: uncached host-visible memory is write-combined, CPU reads are slow
+    const bool slow_host_read = src->device->vendor_id == VK_VENDOR_ID_AMD &&
+                                !(src->memory_property_flags & vk::MemoryPropertyFlagBits::eHostCached);
+    if(src->memory_property_flags & vk::MemoryPropertyFlagBits::eHostVisible && src->device->uma && !slow_host_read) {
         GGML_ASSERT(src->memory_property_flags & vk::MemoryPropertyFlagBits::eHostCoherent);

         std::lock_guard<std::recursive_mutex> guard(src->device->mutex);