Commit 4d756bc72 for llama.cpp
commit 4d756bc72bf00a4aacf410ae15a2d315f3db400d
Author: Ruben Ortlam <rortlam@redhat.com>
Date: Wed Oct 7 08:20:04 2026 +0200
vulkan: fix amd iGPU slow checkpoint read (#30049)
diff --git a/ggml/src/ggml-vulkan/ggml-vulkan-buffers.cpp b/ggml/src/ggml-vulkan/ggml-vulkan-buffers.cpp
index f7a21dd27..351951638 100644
--- a/ggml/src/ggml-vulkan/ggml-vulkan-buffers.cpp
+++ b/ggml/src/ggml-vulkan/ggml-vulkan-buffers.cpp
@@ -623,7 +623,10 @@ void ggml_vk_buffer_read_2d(vk_buffer& src, size_t offset, void * dst, size_t sp
// If the device is not an UMA device the memory is host-accessible through rebar. While writing
// through PCIe is sufficient fast reading back data from PCIe is slower than going through
// the HW device to host copy path.
- if(src->memory_property_flags & vk::MemoryPropertyFlagBits::eHostVisible && src->device->uma) {
+ // AMD UMA: uncached host-visible memory is write-combined, CPU reads are slow
+ const bool slow_host_read = src->device->vendor_id == VK_VENDOR_ID_AMD &&
+ !(src->memory_property_flags & vk::MemoryPropertyFlagBits::eHostCached);
+ if(src->memory_property_flags & vk::MemoryPropertyFlagBits::eHostVisible && src->device->uma && !slow_host_read) {
GGML_ASSERT(src->memory_property_flags & vk::MemoryPropertyFlagBits::eHostCoherent);
std::lock_guard<std::recursive_mutex> guard(src->device->mutex);