Commit 0eb6d9a81 for llama.cpp
commit 0eb6d9a8137ff13deb1b3a755b01b1e5b7f89229
Author: Adrien Gallouët <angt@huggingface.co>
Date: Sun Oct 4 21:21:31 2026 +0200
cuda : move neu_padded to where it is used (#29940)
Signed-off-by: Adrien Gallouët <angt@huggingface.co>
diff --git a/ggml/src/ggml-cuda/mmid.cu b/ggml/src/ggml-cuda/mmid.cu
index 0b222e63a..12ae14ce9 100644
--- a/ggml/src/ggml-cuda/mmid.cu
+++ b/ggml/src/ggml-cuda/mmid.cu
@@ -37,9 +37,6 @@ static __global__ void mm_ids_helper(
const int n_expert_used = n_expert_used_template == 0 ? n_expert_used_var : n_expert_used_template;
const int expert = blockIdx.x;
- // token slots per warp lane group, padded to a power of 2 so a warp divides evenly
- constexpr int neu_padded = mm_ids_pow2<n_expert_used_template>::value;
-
extern __shared__ char data_mm_ids_helper[];
mm_ids_helper_store * store = (mm_ids_helper_store *) data_mm_ids_helper;
@@ -69,6 +66,7 @@ static __global__ void mm_ids_helper(
} else {
// Implementation optimized for specific numbers of experts used:
// a warp holds a whole number of token slots, so the slot count is padded to a power of 2
+ constexpr int neu_padded = mm_ids_pow2<n_expert_used_template>::value;
static_assert(neu_padded <= warp_size && warp_size % neu_padded == 0, "bad n_expert_used");
for (int it0 = 0; it0 < n_tokens; it0 += warp_size/neu_padded) {
const int it = it0 + threadIdx.x / neu_padded;