Commit 8c6ca4ab for whisper.cpp
commit 8c6ca4ab76d48022d8639c0c8b75ac907d2e3c6b
Author: Konrad Moren <kmoren@nvidia.com>
Date: Mon Oct 5 13:43:07 2026 +0200
CUDA: Optimize accumulation in mmq for NVFP4 type (llama/29857)
* ggml_cuda: optimize accumulation in mmq_vec_dot_fp4_fp4_mma for better performance
* remove whitespace
* fix: correct indentation in mma_block_scaled_fp4 loop
diff --git a/ggml/src/ggml-cuda/mmq-vec-dot.cuh b/ggml/src/ggml-cuda/mmq-vec-dot.cuh
index 4ca6542d..b39e4d57 100644
--- a/ggml/src/ggml-cuda/mmq-vec-dot.cuh
+++ b/ggml/src/ggml-cuda/mmq-vec-dot.cuh
@@ -1225,14 +1225,11 @@ template <ggml_type type, int J, bool fallback> static __device__ __forceinline_
#pragma unroll
for (int n = 0; n < ntx; ++n) {
+ // accumulate in place into the output sum array
+ tile_C & C = *reinterpret_cast<tile_C *>(sum + (j0 / tile_C::J + n) * tile_C::ne);
#pragma unroll
for (int frag = 0; frag < nfrags; ++frag) {
- tile_C C = {};
mma_block_scaled_fp4<type>(C, A[n][frag], B[frag], scaleA[n][frag], scaleB[frag]);
-#pragma unroll
- for (int l = 0; l < tile_C::ne; ++l) {
- sum[(j0 / tile_C::J + n) * tile_C::ne + l] += C.x[l];
- }
}
}
}