Commit c83f3058b for llama.cpp
commit c83f3058b6f876f2f7e6ccbada858b2446860c85
Author: lhez <lih@qti.qualcomm.com>
Date: Sun Oct 11 09:41:29 2026 -0700
opencl: fix image limit for q4_k dense bin kernels, q4_0 bcast and rms_norm (#30310)
* opencl: fallback when Q4_K weight images exceed device limits
Co-authored-by: Hongqiang Wang <wangh@qti.qualcomm.com>
* opencl: avoid incomplete subgroup in rms_norm
* opencl: refine conditions for dense bin kernels and fix bcast for q4_0
Co-authored-by: Hongqiang Wang <wangh@qti.qualcomm.com>
---------
Co-authored-by: Hongqiang Wang <wangh@qti.qualcomm.com>
diff --git a/ggml/src/ggml-opencl/ggml-opencl.cpp b/ggml/src/ggml-opencl/ggml-opencl.cpp
index ea85d99bc..625b12f54 100644
--- a/ggml/src/ggml-opencl/ggml-opencl.cpp
+++ b/ggml/src/ggml-opencl/ggml-opencl.cpp
@@ -8583,6 +8583,11 @@ inline bool use_adreno_kernels(const ggml_backend_opencl_context *backend_ctx, c
bool threashold_ok = tensor->ne[0] >= threshold_ne0 && tensor->ne[1] >= threshold_ne1 &&
tensor->ne[2] == 1 && tensor->ne[3] == 1;
+ // the transposed layout needs K % 32 == 0 and M % 4 == 0
+ if (tensor->ne[0] % 32 != 0 || tensor->ne[1] % 4 != 0) {
+ return false;
+ }
+
// The noshuffle layout packs 2 rows per 32-bit texel and the GEMV reads it at an
// ne1/2 texel stride with an exact-cover dispatch, so it is only addressable when
// ne1 is a multiple of 64; an unaligned ne1 truncates the stride and the weight is
@@ -8731,7 +8736,10 @@ inline bool use_q4_0_bin_kernels(const ggml_backend_opencl_context *backend_ctx,
!backend_ctx->kernel_gemm_noshuffle_q4_0_f32_32b_trans_ila_a8_bin) {
return false;
}
- return (tensor->ne[0] % 32 == 0) && (tensor->ne[1] % 64 == 0);
+
+ // bin kernels require ne2 == 1 and ne3 == 1 for weights
+ return tensor->ne[2] == 1 && tensor->ne[3] == 1 &&
+ (tensor->ne[0] % 32 == 0) && (tensor->ne[1] % 64 == 0);
#else
GGML_UNUSED(backend_ctx);
GGML_UNUSED(tensor);
@@ -8779,22 +8787,30 @@ static inline bool flat_large_m_enabled() {
return en;
}
+// The noshuffle Q4_K weight image stores eight weights per uint texel.
+static inline bool q4_K_weight_image_fits(const ggml_backend_opencl_context *backend_ctx, const ggml_tensor *tensor) {
+ const size_t texels = (size_t) ggml_nelements(tensor) / 8;
+ return texels != 0 && texels <= backend_ctx->image_max_buffer_size;
+}
+
static inline bool use_flat_gemv_for_large_m_q4_K(const ggml_backend_opencl_context *backend_ctx, const ggml_tensor *tensor) {
if (tensor->ne[1] % 4 != 0 && tensor->ne[2] == 1 && tensor->ne[3] == 1) {
return true;
}
- if (!flat_large_m_enabled()) {
+ if (tensor->ne[2] != 1 || tensor->ne[3] != 1 || use_q4k_tiled(backend_ctx, tensor)) {
return false;
}
+
+ // The image limit is a correctness guard, independent of the large-M opt-in.
+ if (!q4_K_weight_image_fits(backend_ctx, tensor)) {
+ return true;
+ }
+
// gemv_noshuffle variant perf drops for large M, use flat variant for large M.
// threshold is well above typical hidden/FFN dims, but below typical vocab sizes.
// note that this forces large M weights to use LM GEMM.
- // EXCEPT when this branch's tiled-canonical lm_head/embed layout is active: the
- // weight is converted to the 64-row tiled layout, which the flat gemv would
- // misread as garbage. use_q4k_tiled owns these large-M weights, so defer to it.
- return tensor->ne[1] >= 32768 && tensor->ne[2] == 1 && tensor->ne[3] == 1
- && !use_q4k_tiled(backend_ctx, tensor);
+ return flat_large_m_enabled() && tensor->ne[1] >= 32768;
}
static inline bool use_flat_gemv_for_large_m_q6_K(const ggml_backend_opencl_context *backend_ctx, const ggml_tensor *tensor) {
@@ -8846,7 +8862,9 @@ inline bool use_q6_k_bin_kernels(const ggml_backend_opencl_context *backend_ctx,
!backend_ctx->kernel_gemm_noshuffle_q6_k_f32_32b_trans_ila_a8_bin) {
return false;
}
- return (tensor->ne[0] % 256 == 0) && (tensor->ne[1] % 64 == 0) &&
+ // bin kernels require ne2 == 1 and ne3 == 1 for weights
+ return tensor->ne[2] == 1 && tensor->ne[3] == 1 &&
+ (tensor->ne[0] % 256 == 0) && (tensor->ne[1] % 64 == 0) &&
!use_q6k_tiled(backend_ctx, tensor) && !use_flat_gemv_for_large_m_q6_K(backend_ctx, tensor);
#else
GGML_UNUSED(backend_ctx);
@@ -8861,7 +8879,9 @@ inline bool use_q4_k_bin_kernels(const ggml_backend_opencl_context *backend_ctx,
!backend_ctx->kernel_gemm_noshuffle_q4_k_f32_32b_trans_ila_a8_bin) {
return false;
}
- return (tensor->ne[0] % 256 == 0) && (tensor->ne[1] % 64 == 0) &&
+ // bin kernels require ne2 == 1 and ne3 == 1 for weights
+ return tensor->ne[2] == 1 && tensor->ne[3] == 1 &&
+ (tensor->ne[0] % 256 == 0) && (tensor->ne[1] % 64 == 0) &&
!use_q4k_tiled(backend_ctx, tensor) && !use_flat_gemv_for_large_m_q4_K(backend_ctx, tensor);
#else
GGML_UNUSED(backend_ctx);
@@ -15071,7 +15091,7 @@ static void ggml_cl_rms_norm(ggml_backend_t backend, const ggml_tensor * src0, c
GGML_ASSERT(ne00 % 4 == 0);
- const int nth = MIN(64, ne00);
+ const int nth = 64;
size_t global_work_size[] = {(size_t)ne01*nth, (size_t)ne02, (size_t)ne03};
size_t local_work_size[] = {(size_t)nth, 1, 1};
@@ -23453,7 +23473,17 @@ static void ggml_cl_mul_mat(ggml_backend_t backend, const ggml_tensor * src0, co
// quant kv without FA
// used for non-contiguous src0 (the usual head-major permuted K view when n_head_kv>1)
// AND for the contiguous case that occurs when n_head_kv==1 (e.g. Gemma-4 E2B)
- if ((src0t == GGML_TYPE_Q4_0 || src0t == GGML_TYPE_Q8_0) &&
+ // Q4_0 bin kernels use a special weight layout, which the restore below does not support.
+ bool q4_0_bin_layout = false;
+#ifdef GGML_OPENCL_USE_ADRENO_KERNELS
+ q4_0_bin_layout = src0t == GGML_TYPE_Q4_0 && src0->view_src == nullptr &&
+ ggml_is_contiguous(src0) &&
+ src0->ne[2] == 1 && src0->ne[3] == 1 &&
+ use_adreno_kernels(backend_ctx, src0) &&
+ !use_adreno_moe_kernels(backend_ctx, src0) &&
+ use_q4_0_bin_kernels(backend_ctx, src0);
+#endif
+ if ((src0t == GGML_TYPE_Q4_0 || src0t == GGML_TYPE_Q8_0) && !q4_0_bin_layout &&
(!ggml_is_contiguous(src0) || src1->ne[2] > src0->ne[2])) {
cl_mem f16_buf = ggml_cl_mul_mat_dequant_quant_to_f16(backend_ctx, src0, nullptr);