Commit f872b5911 for llama.cpp

commit f872b591121761ac7b2af18283bd99bdc092a63a
Author: R0CKSTAR <yeahdongcn@gmail.com>
Date:   Thu Oct 1 04:33:22 2026 +0800

    cuda: guard the iq4_nl dequantize row kernel against short rows (#29683)

    dequantize_block_iq4_nl writes QK_K values per block, but a row can be shorter than that (an IQ4_NL row is only guaranteed to be a multiple of QK4_NL). Threads whose 32-value sub-block starts at or past k currently read and write past the end of the row. Skip those sub-blocks; for rows that are a multiple of QK_K the check never fires.

diff --git a/ggml/src/ggml-cuda/convert.cu b/ggml/src/ggml-cuda/convert.cu
index 0619f4760..77fd28894 100644
--- a/ggml/src/ggml-cuda/convert.cu
+++ b/ggml/src/ggml-cuda/convert.cu
@@ -223,9 +223,14 @@ static __global__ void dequantize_block_iq1_m(const void * __restrict__ vx, dst_
 }

 template<typename dst_t>
-static __global__ void dequantize_block_iq4_nl(const void * __restrict__ vx, dst_t * __restrict__ yy) {
+static __global__ void dequantize_block_iq4_nl(const void * __restrict__ vx, dst_t * __restrict__ yy, int nb32) {
     const int64_t i = blockIdx.x;

+    const int64_t ib = 8*i + threadIdx.x%8;
+    if (ib >= nb32) {
+        return;
+    }
+
     dequantize_iq4_nl(vx, i, yy + i*QK_K, threadIdx.x);
 }

@@ -352,8 +357,9 @@ static void dequantize_row_iq1_s_cuda(const void * vx, dst_t * y, const int64_t

 template<typename dst_t>
 static void dequantize_row_iq4_nl_cuda(const void * vx, dst_t * y, const int64_t k, cudaStream_t stream) {
+    const int nb32 = k / QK4_NL;
     const int nb = (k + QK_K - 1) / QK_K;
-    dequantize_block_iq4_nl<<<nb, 32, 0, stream>>>(vx, y);
+    dequantize_block_iq4_nl<<<nb, 32, 0, stream>>>(vx, y, nb32);
 }

 template<typename dst_t>