Commit 767767850 for llama.cpp
commit 7677678503e921f98da87e77bde531752977848e
Author: Oliver Simons <osimons@nvidia.com>
Date: Thu Oct 1 13:05:29 2026 +0200
CUDA: Make CCCL configurable + pin it to 3.4.3 for CI jobs (#29792)
Pinning to >= 3.4.3 is required to enable DeviceTopK, which was affected by
a race condition https://github.com/NVIDIA/cccl/pull/10627.
We will relax this for future CTK versions which will bundle CCCL >
3.4.X (CTK 13.5 will bundle CCCL 3.5.0 for example)
diff --git a/.github/workflows/build-cuda-ubuntu.yml b/.github/workflows/build-cuda-ubuntu.yml
index 68b6c0190..e76a2acf5 100644
--- a/.github/workflows/build-cuda-ubuntu.yml
+++ b/.github/workflows/build-cuda-ubuntu.yml
@@ -68,7 +68,7 @@ jobs:
hf_bucket: ggml-org/cache
- name: Build with CMake
- # TODO: Remove GGML_CUDA_CUB_3DOT2 flag once CCCL 3.2 is bundled within CTK and that CTK version is used in this project
+ # TODO: Drop GGML_CUDA_CCCL_VERSION when this job uses CTK >= 13.5, which bundles CCCL >= 3.5.
run: |
cmake -S . -B build -G Ninja \
-DLLAMA_FATAL_WARNINGS=ON \
@@ -77,7 +77,7 @@ jobs:
-DCMAKE_EXE_LINKER_FLAGS=-Wl,--allow-shlib-undefined \
-DGGML_NATIVE=OFF \
-DGGML_CUDA=ON \
- -DGGML_CUDA_CUB_3DOT2=ON
+ -DGGML_CUDA_CCCL_VERSION=v3.4.3
cmake --build build
- name: ccache-buckets-save
diff --git a/.github/workflows/build-cuda-windows.yml b/.github/workflows/build-cuda-windows.yml
index e08553e6c..f722874bd 100644
--- a/.github/workflows/build-cuda-windows.yml
+++ b/.github/workflows/build-cuda-windows.yml
@@ -31,15 +31,16 @@ jobs:
strategy:
matrix:
include:
+ # CTK >= 13.5 bundles CCCL >= 3.5; omit GGML_CUDA_CCCL_VERSION for those versions.
- cuda: '12.4'
arch: x64
- defines: '-DGGML_CUDA_CUB_3DOT2=ON'
+ defines: '-DGGML_CUDA_CCCL_VERSION=v3.4.3'
- cuda: '13.4'
arch: x64
- defines: ''
+ defines: '-DGGML_CUDA_CCCL_VERSION=v3.4.3'
- cuda: '13.4'
arch: arm64
- defines: '-DCMAKE_TOOLCHAIN_FILE=cmake/arm64-windows-msvc-cuda.cmake'
+ defines: '-DGGML_CUDA_CCCL_VERSION=v3.4.3 -DCMAKE_TOOLCHAIN_FILE=cmake/arm64-windows-msvc-cuda.cmake'
steps:
- name: Clone
diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml
index b5dba37de..4784bb718 100644
--- a/.github/workflows/release.yml
+++ b/.github/workflows/release.yml
@@ -323,21 +323,22 @@ jobs:
include:
# label = short version used in artifact names / release body
# cuda = full container image tag
+ # CTK >= 13.5 bundles CCCL >= 3.5; omit GGML_CUDA_CCCL_VERSION for those versions.
- build: 'x64'
os: ubuntu-24.04
cuda: '12.8.2'
label: '12.8'
- defines: '-DGGML_CUDA_CUB_3DOT2=ON'
+ defines: '-DGGML_CUDA_CCCL_VERSION=v3.4.3'
- build: 'x64'
os: ubuntu-24.04
cuda: '13.4.1'
label: '13.4'
- defines: ''
+ defines: '-DGGML_CUDA_CCCL_VERSION=v3.4.3'
- build: 'arm64'
os: ubuntu-24.04-arm
cuda: '13.4.1'
label: '13.4'
- defines: ''
+ defines: '-DGGML_CUDA_CCCL_VERSION=v3.4.3'
runs-on: ${{ matrix.os }}
container: nvidia/cuda:${{ matrix.cuda }}-devel-ubuntu24.04
@@ -1244,15 +1245,16 @@ jobs:
strategy:
matrix:
include:
+ # CTK >= 13.5 bundles CCCL >= 3.5; omit GGML_CUDA_CCCL_VERSION for those versions.
- cuda: '12.4'
arch: x64
- defines: '-DGGML_CUDA_CUB_3DOT2=ON'
+ defines: '-DGGML_CUDA_CCCL_VERSION=v3.4.3'
- cuda: '13.4'
arch: x64
- defines: ''
+ defines: '-DGGML_CUDA_CCCL_VERSION=v3.4.3'
- cuda: '13.4'
arch: arm64
- defines: '-DCMAKE_TOOLCHAIN_FILE=cmake/arm64-windows-msvc-cuda.cmake'
+ defines: '-DGGML_CUDA_CCCL_VERSION=v3.4.3 -DCMAKE_TOOLCHAIN_FILE=cmake/arm64-windows-msvc-cuda.cmake'
steps:
- name: Clone
@@ -1279,7 +1281,6 @@ jobs:
- name: Build
id: cmake_build
shell: cmd
- # TODO: Remove GGML_CUDA_CUB_3DOT2 flag once CCCL 3.2 is bundled within CTK and that CTK version is used in this project
run: |
call "C:\Program Files\Microsoft Visual Studio\2022\Enterprise\VC\Auxiliary\Build\vcvarsall.bat" ${{ matrix.arch == 'x64' && 'x64' || 'amd64_arm64' }}
cmake -S . -B build -G "Ninja Multi-Config" ^
diff --git a/ci/run.sh b/ci/run.sh
index ccfc0562f..43dccc5c1 100755
--- a/ci/run.sh
+++ b/ci/run.sh
@@ -72,8 +72,8 @@ else
fi
if [ ! -z ${GG_BUILD_CUDA} ]; then
- # TODO: Remove GGML_CUDA_CUB_3DOT2 flag once CCCL 3.2 is bundled within CTK and that CTK version is used in this project
- CMAKE_EXTRA="${CMAKE_EXTRA} -DGGML_CUDA=ON -DGGML_CUDA_CUB_3DOT2=ON"
+ # TODO: Drop GGML_CUDA_CCCL_VERSION when CUDA CI uses CTK >= 13.5, which bundles CCCL >= 3.5.
+ CMAKE_EXTRA="${CMAKE_EXTRA} -DGGML_CUDA=ON -DGGML_CUDA_CCCL_VERSION=v3.4.3"
if command -v nvidia-smi >/dev/null 2>&1; then
CUDA_ARCH=$(nvidia-smi --query-gpu=compute_cap --format=csv,noheader,nounits 2>/dev/null | head -1 | tr -d '.')
diff --git a/docs/build.md b/docs/build.md
index 45364281b..08a8f3333 100644
--- a/docs/build.md
+++ b/docs/build.md
@@ -195,6 +195,8 @@ cmake -B build -DGGML_CUDA=ON
cmake --build build --config Release
```
+To use a specific CCCL version instead of the one bundled with the installed CUDA Toolkit, add `-DGGML_CUDA_CCCL_VERSION=vMAJOR.MINOR.PATCH`. CUB DeviceTopK requires CCCL 3.4.3 or newer; older versions use the sort fallback.
+
Note that this also builds the CPU backend by default. On Windows on ARM, MSVC's
support for the ARM NEON intrinsics used by the CPU backend may be incomplete, so
a CUDA build produced entirely with MSVC might have a slower CPU backend. If CPU
diff --git a/ggml/CMakeLists.txt b/ggml/CMakeLists.txt
index 752dabbb6..4af7fc881 100644
--- a/ggml/CMakeLists.txt
+++ b/ggml/CMakeLists.txt
@@ -197,6 +197,7 @@ set(GGML_BLAS_VENDOR ${GGML_BLAS_VENDOR_DEFAULT} CACHE STRING
option(GGML_LLAMAFILE "ggml: use LLAMAFILE" ${GGML_LLAMAFILE_DEFAULT})
option(GGML_CUDA "ggml: use CUDA" OFF)
+set (GGML_CUDA_CCCL_VERSION "" CACHE STRING "ggml: CCCL git tag to fetch, empty to use the version bundled with the installed CUDA Toolkit")
option(GGML_MUSA "ggml: use MUSA" OFF)
option(GGML_CUDA_FORCE_MMQ "ggml: use mmq kernels instead of cuBLAS" OFF)
option(GGML_CUDA_FORCE_CUBLAS "ggml: always use cuBLAS instead of mmq kernels" OFF)
diff --git a/ggml/src/ggml-cuda/CMakeLists.txt b/ggml/src/ggml-cuda/CMakeLists.txt
index dd57ac423..297d9ffc9 100644
--- a/ggml/src/ggml-cuda/CMakeLists.txt
+++ b/ggml/src/ggml-cuda/CMakeLists.txt
@@ -71,14 +71,13 @@ if (CUDAToolkit_FOUND)
enable_language(CUDA)
- # TODO: Remove once CCCL 3.2 has been released and bundled with CUDA Toolkit
- if (GGML_CUDA_CUB_3DOT2)
+ if (GGML_CUDA_CCCL_VERSION)
include(FetchContent)
FetchContent_Declare(
CCCL
GIT_REPOSITORY https://github.com/nvidia/cccl.git
- GIT_TAG v3.2.0
+ GIT_TAG "${GGML_CUDA_CCCL_VERSION}"
GIT_SHALLOW TRUE
)
@@ -157,14 +156,15 @@ if (CUDAToolkit_FOUND)
add_compile_definitions(GGML_CUDA_NO_PEER_COPY)
endif()
+ if (GGML_CUDA_CCCL_VERSION)
+ target_link_libraries(ggml-cuda PRIVATE CCCL::CCCL)
+ endif()
+
if (GGML_STATIC)
if (WIN32)
# As of 12.3.1 CUDA Toolkit for Windows does not offer a static cublas library
target_link_libraries(ggml-cuda PRIVATE CUDA::cudart_static CUDA::cublas)
else ()
- if (GGML_CUDA_CUB_3DOT2)
- target_link_libraries(ggml-cuda PRIVATE CCCL::CCCL)
- endif()
if (CUDAToolkit_VERSION VERSION_GREATER_EQUAL "10.1")
target_link_libraries(ggml-cuda PRIVATE CUDA::cudart_static CUDA::cublas_static CUDA::cublasLt_static)
else()
@@ -172,9 +172,6 @@ if (CUDAToolkit_FOUND)
endif()
endif()
else()
- if (GGML_CUDA_CUB_3DOT2)
- target_link_libraries(ggml-cuda PRIVATE CCCL::CCCL)
- endif()
target_link_libraries(ggml-cuda PRIVATE CUDA::cudart CUDA::cublas)
endif()
diff --git a/ggml/src/ggml-cuda/top-k.cu b/ggml/src/ggml-cuda/top-k.cu
index c7a0c8317..3ffbba839 100644
--- a/ggml/src/ggml-cuda/top-k.cu
+++ b/ggml/src/ggml-cuda/top-k.cu
@@ -3,11 +3,15 @@
#ifdef GGML_CUDA_USE_CUB
# include <cub/cub.cuh>
-# if (CCCL_MAJOR_VERSION >= 3 && CCCL_MINOR_VERSION >= 2)
+// DeviceTopK has a race condition before CCCL 3.4.3.
+// https://github.com/NVIDIA/cccl/pull/10627
+# if (CCCL_MAJOR_VERSION > 3 || \
+ (CCCL_MAJOR_VERSION == 3 && CCCL_MINOR_VERSION > 4) || \
+ (CCCL_MAJOR_VERSION == 3 && CCCL_MINOR_VERSION == 4 && CCCL_PATCH_VERSION >= 3))
# define CUB_TOP_K_AVAILABLE
# include <cuda/iterator>
using namespace cub;
-# endif // CCCL_MAJOR_VERSION >= 3 && CCCL_MINOR_VERSION >= 2
+# endif // CCCL >= 3.4.3
#endif // GGML_CUDA_USE_CUB
#ifdef CUB_TOP_K_AVAILABLE