Commit 6a2743f02 for llama.cpp
commit 6a2743f028f78bfb88a7189607b49bde30df3769
Author: Pascal <admin@serveurperso.com>
Date: Tue Sep 29 20:09:10 2026 +0200
CUDA: bitonic argsort handles rows wider than one block (#28957)
Without CUB (HIP, MUSA) argsort ran the bitonic kernel with one thread
per padded column, so any row above 1024 entries launched an invalid
block configuration. Each thread now owns several columns, every stage
of the network runs all owned columns before the barrier, and the block
is capped at 1024 threads. Shared memory becomes the only bound, which
supports_op checks against the device instead of a fixed 1024.
Rows up to 1024 run the same work as before. Bit-exact with the CUB
path on rows of 2048.
diff --git a/ggml/src/ggml-cuda/argsort.cu b/ggml/src/ggml-cuda/argsort.cu
index 24115da09..f6a850dda 100644
--- a/ggml/src/ggml-cuda/argsort.cu
+++ b/ggml/src/ggml-cuda/argsort.cu
@@ -166,52 +166,62 @@ static inline __device__ void ggml_cuda_swap(T & a, T & b) {
b = tmp;
}
+// One compare-exchange of the bitonic network at (k, j) for column col.
template<ggml_sort_order order>
-static __global__ void k_argsort_f32_i32(const float * x, int * dst, const int ncols, int ncols_pad) {
- // bitonic sort
- int col = threadIdx.x;
- int row = blockIdx.x;
-
- if (col >= ncols_pad) {
+static inline __device__ void bitonic_step(const float * x_row, int * dst_row, const int ncols, const int col, const int k, const int j) {
+ const int ixj = col ^ j;
+ if (ixj <= col) {
return;
}
+ if ((col & k) == 0) {
+ if (dst_row[col] >= ncols ||
+ (dst_row[ixj] < ncols && (order == GGML_SORT_ORDER_ASC ?
+ x_row[dst_row[col]] > x_row[dst_row[ixj]] :
+ x_row[dst_row[col]] < x_row[dst_row[ixj]]))
+ ) {
+ ggml_cuda_swap(dst_row[col], dst_row[ixj]);
+ }
+ } else {
+ if (dst_row[ixj] >= ncols ||
+ (dst_row[col] < ncols && (order == GGML_SORT_ORDER_ASC ?
+ x_row[dst_row[col]] < x_row[dst_row[ixj]] :
+ x_row[dst_row[col]] > x_row[dst_row[ixj]]))
+ ) {
+ ggml_cuda_swap(dst_row[col], dst_row[ixj]);
+ }
+ }
+}
+
+// Bitonic sort of one row per block. Each thread owns the columns
+// threadIdx.x + i * blockDim.x, so rows wider than the block (up to the
+// shared memory limit) sort with several columns per thread. Every
+// (k, j) stage runs all owned columns before the barrier; a pair
+// (col, col ^ j) is exchanged by the owner of its lower index only.
+template<ggml_sort_order order>
+static __global__ void k_argsort_f32_i32(const float * x, int * dst, const int ncols, int ncols_pad) {
+ const int row = blockIdx.x;
const float * x_row = x + row * ncols;
extern __shared__ int dst_row[];
// initialize indices
- dst_row[col] = col;
+ for (int col = threadIdx.x; col < ncols_pad; col += blockDim.x) {
+ dst_row[col] = col;
+ }
__syncthreads();
for (int k = 2; k <= ncols_pad; k *= 2) {
for (int j = k / 2; j > 0; j /= 2) {
- int ixj = col ^ j;
- if (ixj > col) {
- if ((col & k) == 0) {
- if (dst_row[col] >= ncols ||
- (dst_row[ixj] < ncols && (order == GGML_SORT_ORDER_ASC ?
- x_row[dst_row[col]] > x_row[dst_row[ixj]] :
- x_row[dst_row[col]] < x_row[dst_row[ixj]]))
- ) {
- ggml_cuda_swap(dst_row[col], dst_row[ixj]);
- }
- } else {
- if (dst_row[ixj] >= ncols ||
- (dst_row[col] < ncols && (order == GGML_SORT_ORDER_ASC ?
- x_row[dst_row[col]] < x_row[dst_row[ixj]] :
- x_row[dst_row[col]] > x_row[dst_row[ixj]]))
- ) {
- ggml_cuda_swap(dst_row[col], dst_row[ixj]);
- }
- }
+ for (int col = threadIdx.x; col < ncols_pad; col += blockDim.x) {
+ bitonic_step<order>(x_row, dst_row, ncols, col, k, j);
}
__syncthreads();
}
}
// copy the result to dst without the padding
- if (col < ncols) {
+ for (int col = threadIdx.x; col < ncols; col += blockDim.x) {
dst[row * ncols + col] = dst_row[col];
}
}
@@ -233,7 +243,9 @@ void argsort_f32_i32_cuda_bitonic(const float * x,
// bitonic sort requires ncols to be power of 2
const int ncols_pad = next_power_of_2(ncols);
- const dim3 block_dims(ncols_pad, 1, 1);
+ // one thread per column up to the block limit, several columns per
+ // thread beyond it; shared memory is the remaining bound
+ const dim3 block_dims(ncols_pad < CUDA_ARGSORT_BLOCK_SIZE ? ncols_pad : CUDA_ARGSORT_BLOCK_SIZE, 1, 1);
const dim3 block_nums(nrows, 1, 1);
const size_t shared_mem = ncols_pad * sizeof(int);
diff --git a/ggml/src/ggml-cuda/argsort.cuh b/ggml/src/ggml-cuda/argsort.cuh
index 3abb6448a..c9adfcb98 100644
--- a/ggml/src/ggml-cuda/argsort.cuh
+++ b/ggml/src/ggml-cuda/argsort.cuh
@@ -1,5 +1,7 @@
#include "common.cuh"
+#define CUDA_ARGSORT_BLOCK_SIZE 1024
+
void ggml_cuda_op_argsort(ggml_backend_cuda_context & ctx, ggml_tensor * dst);
#ifdef GGML_CUDA_USE_CUB
diff --git a/ggml/src/ggml-cuda/ggml-cuda.cu b/ggml/src/ggml-cuda/ggml-cuda.cu
index b50d05d5f..9afc2aa4a 100644
--- a/ggml/src/ggml-cuda/ggml-cuda.cu
+++ b/ggml/src/ggml-cuda/ggml-cuda.cu
@@ -5553,7 +5553,14 @@ static bool ggml_backend_cuda_device_supports_op(ggml_backend_dev_t dev, const g
#endif // defined(GGML_USE_HIP) || defined(GGML_CUDA_USE_CUB)
case GGML_OP_ARGSORT:
#ifndef GGML_CUDA_USE_CUB
- return op->src[0]->ne[0] <= 1024;
+ {
+ // bitonic path: the padded row must fit in shared memory
+ int64_t ncols_pad = 1;
+ while (ncols_pad < op->src[0]->ne[0]) {
+ ncols_pad *= 2;
+ }
+ return ncols_pad * sizeof(int) <= ggml_cuda_info().devices[dev_ctx->device].smpb;
+ }
#else
return true;
#endif