Commit 64df9183f for llama.cpp
commit 64df9183f5ca178c049a392f35e795d8e97c97b3
Author: jingzhou <jinzho@qti.qualcomm.com>
Date: Fri Oct 9 09:19:20 2026 -0700
opencl: fix kernel compilation for a6x GPUs (#30176)
* opencl: skip kernel_cpy_f32_f32_pack on A6X to avoid shader compiler crash
* The A6x compiler backend found in iot device with a623 (E031.50.31.01)
cannot handle kernels with a large number of arguments. Skip this
kernel for A6x to avoid compiler crash
* opencl: A6X constant-fold workaround for get_local_size in GEMV kernels
* opencl: add Adreno 623 to A6X GPU detection list
diff --git a/ggml/src/ggml-opencl/ggml-opencl.cpp b/ggml/src/ggml-opencl/ggml-opencl.cpp
index 347c76931..45b791edc 100644
--- a/ggml/src/ggml-opencl/ggml-opencl.cpp
+++ b/ggml/src/ggml-opencl/ggml-opencl.cpp
@@ -280,7 +280,8 @@ static ADRENO_GPU_GEN get_adreno_gpu_gen(const char *device_name) {
strstr(device_name, "613") || strstr(device_name, "615") ||
strstr(device_name, "616") || strstr(device_name, "618") ||
strstr(device_name, "619") || strstr(device_name, "620") ||
- strstr(device_name, "630") || strstr(device_name, "640") ||
+ strstr(device_name, "623") || strstr(device_name, "630") ||
+ strstr(device_name, "640") ||
strstr(device_name, "642") || strstr(device_name, "643") ||
strstr(device_name, "644") || strstr(device_name, "650") ||
strstr(device_name, "660") || strstr(device_name, "663") ||
@@ -863,7 +864,8 @@ struct ggml_backend_opencl_context {
cl_kernel kernel_set_rows_q4_0_soa_i64, kernel_set_rows_q4_0_soa_i32;
cl_kernel kernel_rope_norm_f32, kernel_rope_norm_f16, kernel_rope_neox_f32, kernel_rope_neox_f16;
cl_kernel kernel_rope_multi_f32, kernel_rope_multi_f16, kernel_rope_vision_f32, kernel_rope_vision_f16;
- cl_kernel kernel_cpy_f16_f16, kernel_cpy_f16_f32, kernel_cpy_f32_f16, kernel_cpy_f32_f32, kernel_cpy_f32_f32_pack, kernel_cpy_i32_i32;
+ cl_kernel kernel_cpy_f16_f16, kernel_cpy_f16_f32, kernel_cpy_f32_f16, kernel_cpy_f32_f32, kernel_cpy_i32_i32;
+ cl_kernel kernel_cpy_f32_f32_pack = nullptr;
cl_kernel kernel_cpy_f32_f32_flat = nullptr;
cl_kernel kernel_mul_mat_f32_f32;
cl_kernel kernel_mul_mat_f16_f16;
@@ -1604,14 +1606,18 @@ static void load_cl_kernels(ggml_backend_opencl_context *backend_ctx) {
#else
const std::string kernel_src = read_file("cpy.cl");
#endif
- cl_program prog =
- build_program_from_source(backend_ctx, kernel_src.c_str(), compile_opts);
+ const bool no_cpy_pack = backend_ctx->adreno_gen == ADRENO_GPU_GEN::A6X;
+
+ cl_program prog = build_program_from_source(backend_ctx, kernel_src.c_str(),
+ no_cpy_pack ? compile_opts + " -DGGML_CL_NO_CPY_PACK" : compile_opts);
CL_CHECK((backend_ctx->kernel_cpy_f16_f16 = clCreateKernel(prog, "kernel_cpy_f16_f16", &err), err));
CL_CHECK((backend_ctx->kernel_cpy_f16_f32 = clCreateKernel(prog, "kernel_cpy_f16_f32", &err), err));
CL_CHECK((backend_ctx->kernel_cpy_f32_f16 = clCreateKernel(prog, "kernel_cpy_f32_f16", &err), err));
CL_CHECK((backend_ctx->kernel_cpy_f32_f32 = clCreateKernel(prog, "kernel_cpy_f32_f32", &err), err));
- CL_CHECK((backend_ctx->kernel_cpy_f32_f32_pack = clCreateKernel(prog, "kernel_cpy_f32_f32_pack", &err), err));
+ if (!no_cpy_pack) {
+ CL_CHECK((backend_ctx->kernel_cpy_f32_f32_pack = clCreateKernel(prog, "kernel_cpy_f32_f32_pack", &err), err));
+ }
{ // optional: without it ggml_cl_cpy keeps the row-mapped kernel
cl_int err_flat = CL_SUCCESS;
cl_kernel k = clCreateKernel(prog, "kernel_cpy_f32_f32_flat", &err_flat);
@@ -3770,6 +3776,9 @@ static void load_cl_kernels(ggml_backend_opencl_context *backend_ctx) {
if (backend_ctx->has_vector_subgroup_broadcast) {
CL_gemv_compile_opts += " -DVECTOR_SUB_GROUP_BROADCAST ";
}
+ if (backend_ctx->adreno_gen == ADRENO_GPU_GEN::A6X) {
+ CL_gemv_compile_opts += " -DGGML_CL_A6X_CONSTFOLD_FIX";
+ }
#ifdef GGML_OPENCL_EMBED_KERNELS
const std::string kernel_src_CL_gemv_general {
@@ -4306,6 +4315,9 @@ static void load_cl_kernels(ggml_backend_opencl_context *backend_ctx) {
if (backend_ctx->has_vector_subgroup_broadcast) {
CL_gemv_compile_opts += " -DVECTOR_SUB_GROUP_BROADCAST ";
}
+ if (backend_ctx->adreno_gen == ADRENO_GPU_GEN::A6X) {
+ CL_gemv_compile_opts += " -DGGML_CL_A6X_CONSTFOLD_FIX";
+ }
// Opt-in: dequant-once-per-block mc3 verify GEMV (factors q4_K dequant
// out of the 3-column loop; byte-identical, lower spill). A/B vs the
// shipped inline mc3 in the same binary.
@@ -28335,7 +28347,8 @@ static void ggml_cl_cpy(ggml_backend_t backend, const ggml_tensor * src0, const
kernel = backend_ctx->kernel_cpy_f32_f16;
break;
case GGML_TYPE_F32:
- kernel = ne00 < 32 ? backend_ctx->kernel_cpy_f32_f32_pack
+ kernel = (ne00 < 32 && backend_ctx->kernel_cpy_f32_f32_pack)
+ ? backend_ctx->kernel_cpy_f32_f32_pack
: backend_ctx->kernel_cpy_f32_f32;
break;
default:
diff --git a/ggml/src/ggml-opencl/kernels/cpy.cl b/ggml/src/ggml-opencl/kernels/cpy.cl
index e875bfaf7..11d7a57c1 100644
--- a/ggml/src/ggml-opencl/kernels/cpy.cl
+++ b/ggml/src/ggml-opencl/kernels/cpy.cl
@@ -183,6 +183,7 @@ kernel void kernel_cpy_f32_f32(
}
}
+#ifndef GGML_CL_NO_CPY_PACK
kernel void kernel_cpy_f32_f32_pack(
global float * src0,
ulong offset0,
@@ -241,6 +242,7 @@ kernel void kernel_cpy_f32_f32_pack(
dst_data[i00] = src[0];
}
}
+#endif // GGML_CL_NO_CPY_PACK
kernel void kernel_cpy_i32_i32(
global int * src0,
diff --git a/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q4_0_f32.cl b/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q4_0_f32.cl
index 023e848f7..0dcc929a6 100644
--- a/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q4_0_f32.cl
+++ b/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q4_0_f32.cl
@@ -7,6 +7,14 @@
#define REQD_SUBGROUP_SIZE_64 __attribute__((qcom_reqd_sub_group_size("half")))
#endif
+// A6X compiler incorrectly constant-folds get_local_size() results;
+// force runtime materialization via a no-op ALU round-trip.
+#ifdef GGML_CL_A6X_CONSTFOLD_FIX
+#define MATERIALIZE_WG(x) do { (x) *= 2u; if ((x) > 1u) (x) /= 2u; } while(0)
+#else
+#define MATERIALIZE_WG(x)
+#endif
+
// assume
#define QK4_0 32
#define N_SIMDGROUP 4
@@ -327,6 +335,7 @@ __kernel void kernel_gemv_noshuffle_q4_0_f32_mc3(
uint BLOCK_STRIDE_A = N_SIMDGROUP * M; // = 4 * M (N_SIMDGROUP is the #define 4)
uint COL_STRIDE = K / 4; // float4 pixels per activation column
uint nsg = get_local_size(1); // runtime K-split (4 default, 8 small-M)
+ MATERIALIZE_WG(nsg);
__private uint4 regA_hi, regA_lo;
__private half2 regS;
diff --git a/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q4_k_f32.cl b/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q4_k_f32.cl
index c0078131e..916e00118 100644
--- a/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q4_k_f32.cl
+++ b/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q4_k_f32.cl
@@ -11,6 +11,14 @@
#define NSUBGROUPS 4
#define SUBGROUP_SIZE 64
+// A6X compiler incorrectly constant-folds get_local_size() results;
+// force runtime materialization via a no-op ALU round-trip.
+#ifdef GGML_CL_A6X_CONSTFOLD_FIX
+#define MATERIALIZE_WG(x) do { (x) *= 2u; if ((x) > 1u) (x) /= 2u; } while(0)
+#else
+#define MATERIALIZE_WG(x)
+#endif
+
// scales are transposed: consecutive codes of a row are `stride` apart
inline void get_scale_min_k4(
int j,
@@ -233,6 +241,7 @@ kernel void kernel_gemv_noshuffle_q4_k_f32(
// K-split (more waves/SP -> latency hiding) while large-M keeps 4. The
// physical weight layout stride below is INDEPENDENT of this (see BLOCK_STRIDE_A).
uint nsg = get_local_size(1);
+ MATERIALIZE_WG(nsg);
uint K = ne00;
uint M = ne01;
@@ -400,6 +409,7 @@ kernel void kernel_gemv_noshuffle_q4_k_f32_glu(
uint gid = get_global_id(0);
ushort slid = get_sub_group_local_id();
uint nsg = get_local_size(1);
+ MATERIALIZE_WG(nsg);
uint K = ne00;
uint M = ne01;
@@ -514,6 +524,7 @@ kernel void kernel_gemv_noshuffle_q4_k_f32_splitk(
uint gid = get_global_id(0);
ushort slid = get_sub_group_local_id();
uint nsg = get_local_size(1);
+ MATERIALIZE_WG(nsg);
uint ksplit = get_num_groups(1);
uint kslice = get_group_id(1);