Commit 16c163d56 for llama.cpp
commit 16c163d561b976d8375a17db5660105abe47cda1
Author: Ruben Ortlam <rortlam@redhat.com>
Date: Sun Oct 4 14:05:21 2026 +0200
vulkan: fix rdna4 mat_vec tuning (#29934)
diff --git a/ggml/src/ggml-vulkan/ggml-vulkan.cpp b/ggml/src/ggml-vulkan/ggml-vulkan.cpp
index bfc65bed8..a4c7bcc8c 100644
--- a/ggml/src/ggml-vulkan/ggml-vulkan.cpp
+++ b/ggml/src/ggml-vulkan/ggml-vulkan.cpp
@@ -2889,11 +2889,11 @@ void ggml_vk_load_shaders(vk_device& device, vk_pipeline requested) {
rm_stdq = 2;
rm_stdq_int = 2;
}
- // RDNA3: above four columns, static 4 rows for all types bench faster than the default
- const bool is_rdna3 = device->vendor_id == VK_VENDOR_ID_AMD && device->architecture == AMD_RDNA3;
- auto const &rm_int_n = [&](uint32_t rows, uint32_t i) { return (is_rdna3 && i >= 4) ? 4u : rows; };
- // RDNA3: Static 4 rows for all types bench faster than the default
- auto const &rm_id = [&](uint32_t rows) { return is_rdna3 ? 4u : rows; };
+ // RDNA3/4: above four columns, static 4 rows for all types bench faster than the default
+ const bool is_rdna3_or_4 = device->vendor_id == VK_VENDOR_ID_AMD && (device->architecture == AMD_RDNA3 || device->architecture == AMD_RDNA4);
+ auto const &rm_int_n = [&](uint32_t rows, uint32_t i) { return (is_rdna3_or_4 && i >= 4) ? 4u : rows; };
+ // RDNA3/4: Static 4 rows for all types bench faster than the default
+ auto const &rm_id = [&](uint32_t rows) { return is_rdna3_or_4 ? 4u : rows; };
uint32_t rm_iq = 2 * rm_kq;
const bool use_subgroups = device->subgroup_arithmetic;
@@ -3081,7 +3081,7 @@ void ggml_vk_load_shaders(vk_device& device, vk_pipeline requested) {
#if !defined(GGML_VULKAN_INTEGER_DOT_GLSLC_SUPPORT)
GGML_UNUSED(rm_stdq_int);
GGML_UNUSED(rm_kq_int);
- GGML_UNUSED(is_rdna3);
+ GGML_UNUSED(is_rdna3_or_4);
GGML_UNUSED(rm_int_n);
GGML_UNUSED(rm_id);
GGML_UNUSED(rm_iq_int);