Commit f872b5911 for llama.cpp
commit f872b591121761ac7b2af18283bd99bdc092a63a
Author: R0CKSTAR <yeahdongcn@gmail.com>
Date: Thu Oct 1 04:33:22 2026 +0800
cuda: guard the iq4_nl dequantize row kernel against short rows (#29683)
dequantize_block_iq4_nl writes QK_K values per block, but a row can be shorter than that (an IQ4_NL row is only guaranteed to be a multiple of QK4_NL). Threads whose 32-value sub-block starts at or past k currently read and write past the end of the row. Skip those sub-blocks; for rows that are a multiple of QK_K the check never fires.
diff --git a/ggml/src/ggml-cuda/convert.cu b/ggml/src/ggml-cuda/convert.cu
index 0619f4760..77fd28894 100644
--- a/ggml/src/ggml-cuda/convert.cu
+++ b/ggml/src/ggml-cuda/convert.cu
@@ -223,9 +223,14 @@ static __global__ void dequantize_block_iq1_m(const void * __restrict__ vx, dst_
}
template<typename dst_t>
-static __global__ void dequantize_block_iq4_nl(const void * __restrict__ vx, dst_t * __restrict__ yy) {
+static __global__ void dequantize_block_iq4_nl(const void * __restrict__ vx, dst_t * __restrict__ yy, int nb32) {
const int64_t i = blockIdx.x;
+ const int64_t ib = 8*i + threadIdx.x%8;
+ if (ib >= nb32) {
+ return;
+ }
+
dequantize_iq4_nl(vx, i, yy + i*QK_K, threadIdx.x);
}
@@ -352,8 +357,9 @@ static void dequantize_row_iq1_s_cuda(const void * vx, dst_t * y, const int64_t
template<typename dst_t>
static void dequantize_row_iq4_nl_cuda(const void * vx, dst_t * y, const int64_t k, cudaStream_t stream) {
+ const int nb32 = k / QK4_NL;
const int nb = (k + QK_K - 1) / QK_K;
- dequantize_block_iq4_nl<<<nb, 32, 0, stream>>>(vx, y);
+ dequantize_block_iq4_nl<<<nb, 32, 0, stream>>>(vx, y, nb32);
}
template<typename dst_t>