Commit 6421d2c1 for whisper.cpp
commit 6421d2c1f2723e93d5989f5f2d7990dd50a50a16
Author: Johannes Gäßler <johannesg@5d6.de>
Date: Sun Oct 4 14:10:43 2026 +0200
CUDA: fix MMQ memory fault if n_expert >> n_ubatch (llama/29941)
diff --git a/ggml/src/ggml-cuda/mmq.cu b/ggml/src/ggml-cuda/mmq.cu
index 3e1721bb..43603bb8 100644
--- a/ggml/src/ggml-cuda/mmq.cu
+++ b/ggml/src/ggml-cuda/mmq.cu
@@ -255,7 +255,7 @@ void ggml_cuda_mul_mat_q(
}
const size_t nbytes_src1_q8_1 = ne12*n_expert_used*ne10_padded * y_block_size/y_values_per_block +
- ggml_cuda_mmq_get_J_max(src0->type, fallback, cc, ne11) * sizeof(block_q8_1_mmq);
+ ggml_cuda_mmq_get_J_max(src0->type, fallback, cc, ne12) * sizeof(block_q8_1_mmq);
ggml_cuda_pool_alloc<char> src1_q8_1(ctx.pool(), nbytes_src1_q8_1);
ggml_cuda_pool_alloc<float> src1_scale(ctx.pool());
if (src0->type == GGML_TYPE_NVFP4 && use_native_fp4) {