Commit 83078fec0 for llama.cpp

commit 83078fec0db82d6b5a00d9599062c38c39145755
Author: uvos <carl@uvos.xyz>
Date:   Wed Sep 16 13:46:21 2026 +0200

    CUDA/HIP: improve access patterns in im2col (#28013)

diff --git a/ggml/src/ggml-cuda/im2col.cu b/ggml/src/ggml-cuda/im2col.cu
index 28c79ab46..d377f2856 100644
--- a/ggml/src/ggml-cuda/im2col.cu
+++ b/ggml/src/ggml-cuda/im2col.cu
@@ -7,40 +7,41 @@ template <typename T>
 static  __global__ void im2col_kernel(
         const float * x, T * dst,
         int64_t IC, int64_t IW, int64_t IH, int64_t OH, int64_t OW, int64_t KW, int64_t KH,
-        int64_t IC_IH_IW, int64_t IH_IW, int64_t N_OH, int64_t KH_KW, int64_t IC_KH_KW,
+        int64_t N, int64_t IC_IH_IW, int64_t IH_IW, int64_t N_OH, int64_t KH_KW, int64_t IC_KH_KW,
         int s0, int s1, int p0, int p1, int d0, int d1) {
-    const int64_t i = threadIdx.x + blockIdx.x * blockDim.x;
-    if (i >= IC_KH_KW) {
-        return;
-    }
-
-    const int64_t iic = i / (KH_KW);
-    const int64_t rem = i - iic * KH_KW;
-    const int64_t ikh = rem / KW;
-    const int64_t ikw = rem - ikh * KW;
-
-    for (int64_t iow = blockIdx.y; iow < OW; iow += MAX_GRIDDIM_Y) {
-        for (int64_t iz = blockIdx.z; iz < N_OH; iz += MAX_GRIDDIM_Z) {
-            const int64_t  in = iz / OH;
-            const int64_t  ioh = iz - in * OH;
-
-            const int64_t iiw = iow * s0 + ikw * d0 - p0;
-            const int64_t iih = ioh * s1 + ikh * d1 - p1;
-
-            const int64_t offset_dst =
-                ((in * OH + ioh) * OW + iow) * IC_KH_KW + iic * KH_KW + ikh * KW + ikw;
-
-            if (iih < 0 || iih >= IH || iiw < 0 || iiw >= IW) {
-                dst[offset_dst] = 0.0f;
-            } else {
-                const int64_t offset_src = iic * IC_IH_IW + in * IH_IW;
-                dst[offset_dst] = x[offset_src + iih * IW + iiw];
+    const int tid = threadIdx.x;
+
+    const int64_t total_channels = IC * KH * KW;
+    const int threads_per_pos = blockDim.x;
+    const int64_t start_ch = tid;
+    const int64_t stride_ch = threads_per_pos;
+
+    for (int64_t iow = blockIdx.x; iow < OW; iow += MAX_GRIDDIM_Y) {
+        for (int64_t iz = blockIdx.y; iz < N_OH; iz += MAX_GRIDDIM_Z) {
+            const int64_t in = iz / OH;
+            const int64_t ioh = iz - in * OH;
+
+            for (int64_t iic_khw = start_ch; iic_khw < total_channels; iic_khw += stride_ch) {
+                const int64_t iic = iic_khw / KH_KW;
+                const int64_t rem = iic_khw - iic * KH_KW;
+                const int64_t ikh = rem / KW;
+                const int64_t ikw = rem - ikh * KW;
+
+                const int64_t iiw = iow * s0 + ikw * d0 - p0;
+                const int64_t iih = ioh * s1 + ikh * d1 - p1;
+
+                const int64_t offset_dst =
+                    ((in * OH + ioh) * OW + iow) * IC_KH_KW + iic * KH_KW + ikh * KW + ikw;
+
+                if (iih < 0 || iih >= IH || iiw < 0 || iiw >= IW) {
+                    dst[offset_dst] = 0.0f;
+                } else {
+                    const int64_t offset_src = iic * IC_IH_IW + in * IH_IW;
+                    dst[offset_dst] = x[offset_src + iih * IW + iiw];
+                }
             }
         }
     }
-
-    GGML_UNUSED(IC);
-    GGML_UNUSED(KH);
 }

 // im2col: [N, IC, IH, IW] => [N, OH, OW, IC*KH*KW]
@@ -50,13 +51,15 @@ static void im2col_cuda(const float * x, T* dst,
     int64_t N, int64_t IC_IH_IW, int64_t IH_IW,
     int s0,int s1,int p0,int p1,int d0,int d1, cudaStream_t stream) {
     const int64_t IC_KH_KW = IC * KH * KW;
-    const int64_t num_blocks = (IC_KH_KW + CUDA_IM2COL_BLOCK_SIZE - 1) / CUDA_IM2COL_BLOCK_SIZE;
     const int64_t N_OH = N * OH;
     const int64_t KH_KW = KW*KH;
-    dim3 block_nums(num_blocks, MIN(OW, MAX_GRIDDIM_Y), MIN(N_OH, MAX_GRIDDIM_Z));
-    im2col_kernel<<<block_nums, MIN(IC_KH_KW, CUDA_IM2COL_BLOCK_SIZE) , 0, stream>>>(x, dst, IC, IW, IH, OH, OW, KW, KH,
-                                                                                     IC_IH_IW, IH_IW, N_OH, KH_KW, IC_KH_KW,
-                                                                                     s0, s1, p0, p1, d0, d1);
+    const int threads_per_block = MIN((int)IC_KH_KW, CUDA_IM2COL_BLOCK_SIZE);
+    dim3 block_nums(MIN(OW, MAX_GRIDDIM_Y), MIN(N_OH, MAX_GRIDDIM_Z));
+
+    im2col_kernel<<<block_nums, threads_per_block, 0, stream>>>(
+        x, dst, IC, IW, IH, OH, OW, KW, KH,
+        N, IC_IH_IW, IH_IW, N_OH, KH_KW, IC_KH_KW,
+        s0, s1, p0, p1, d0, d1);
 }

 static void im2col_cuda_f16(const float * x, half * dst,