Commit 25453c0a for whisper.cpp
commit 25453c0adf831dce54d47ff0c549f92d8da22817
Author: Adrien Gallouët <angt@huggingface.co>
Date: Sun Oct 4 21:21:31 2026 +0200
cuda : move neu_padded to where it is used (llama/29940)
Signed-off-by: Adrien Gallouët <angt@huggingface.co>
diff --git a/ggml/src/ggml-cuda/mmid.cu b/ggml/src/ggml-cuda/mmid.cu
index 0b222e63..12ae14ce 100644
--- a/ggml/src/ggml-cuda/mmid.cu
+++ b/ggml/src/ggml-cuda/mmid.cu
@@ -37,9 +37,6 @@ static __global__ void mm_ids_helper(
const int n_expert_used = n_expert_used_template == 0 ? n_expert_used_var : n_expert_used_template;
const int expert = blockIdx.x;
- // token slots per warp lane group, padded to a power of 2 so a warp divides evenly
- constexpr int neu_padded = mm_ids_pow2<n_expert_used_template>::value;
-
extern __shared__ char data_mm_ids_helper[];
mm_ids_helper_store * store = (mm_ids_helper_store *) data_mm_ids_helper;
@@ -69,6 +66,7 @@ static __global__ void mm_ids_helper(
} else {
// Implementation optimized for specific numbers of experts used:
// a warp holds a whole number of token slots, so the slot count is padded to a power of 2
+ constexpr int neu_padded = mm_ids_pow2<n_expert_used_template>::value;
static_assert(neu_padded <= warp_size && warp_size % neu_padded == 0, "bad n_expert_used");
for (int it0 = 0; it0 < n_tokens; it0 += warp_size/neu_padded) {
const int it = it0 + threadIdx.x / neu_padded;