Commit 767767850 for llama.cpp

commit 7677678503e921f98da87e77bde531752977848e
Author: Oliver Simons <osimons@nvidia.com>
Date:   Thu Oct 1 13:05:29 2026 +0200

    CUDA: Make CCCL configurable + pin it to 3.4.3 for CI jobs (#29792)

    Pinning to >= 3.4.3 is required to enable DeviceTopK, which was affected by
    a race condition https://github.com/NVIDIA/cccl/pull/10627.

    We will relax this for future CTK versions which will bundle CCCL >
    3.4.X (CTK 13.5 will bundle CCCL 3.5.0 for example)

diff --git a/.github/workflows/build-cuda-ubuntu.yml b/.github/workflows/build-cuda-ubuntu.yml
index 68b6c0190..e76a2acf5 100644
--- a/.github/workflows/build-cuda-ubuntu.yml
+++ b/.github/workflows/build-cuda-ubuntu.yml
@@ -68,7 +68,7 @@ jobs:
           hf_bucket: ggml-org/cache

       - name: Build with CMake
-        # TODO: Remove GGML_CUDA_CUB_3DOT2 flag once CCCL 3.2 is bundled within CTK and that CTK version is used in this project
+        # TODO: Drop GGML_CUDA_CCCL_VERSION when this job uses CTK >= 13.5, which bundles CCCL >= 3.5.
         run: |
           cmake -S . -B build -G Ninja \
             -DLLAMA_FATAL_WARNINGS=ON \
@@ -77,7 +77,7 @@ jobs:
             -DCMAKE_EXE_LINKER_FLAGS=-Wl,--allow-shlib-undefined \
             -DGGML_NATIVE=OFF \
             -DGGML_CUDA=ON \
-            -DGGML_CUDA_CUB_3DOT2=ON
+            -DGGML_CUDA_CCCL_VERSION=v3.4.3
           cmake --build build

       - name: ccache-buckets-save
diff --git a/.github/workflows/build-cuda-windows.yml b/.github/workflows/build-cuda-windows.yml
index e08553e6c..f722874bd 100644
--- a/.github/workflows/build-cuda-windows.yml
+++ b/.github/workflows/build-cuda-windows.yml
@@ -31,15 +31,16 @@ jobs:
     strategy:
       matrix:
         include:
+          # CTK >= 13.5 bundles CCCL >= 3.5; omit GGML_CUDA_CCCL_VERSION for those versions.
           - cuda: '12.4'
             arch: x64
-            defines: '-DGGML_CUDA_CUB_3DOT2=ON'
+            defines: '-DGGML_CUDA_CCCL_VERSION=v3.4.3'
           - cuda: '13.4'
             arch: x64
-            defines: ''
+            defines: '-DGGML_CUDA_CCCL_VERSION=v3.4.3'
           - cuda: '13.4'
             arch: arm64
-            defines: '-DCMAKE_TOOLCHAIN_FILE=cmake/arm64-windows-msvc-cuda.cmake'
+            defines: '-DGGML_CUDA_CCCL_VERSION=v3.4.3 -DCMAKE_TOOLCHAIN_FILE=cmake/arm64-windows-msvc-cuda.cmake'

     steps:
       - name: Clone
diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml
index b5dba37de..4784bb718 100644
--- a/.github/workflows/release.yml
+++ b/.github/workflows/release.yml
@@ -323,21 +323,22 @@ jobs:
         include:
           # label = short version used in artifact names / release body
           # cuda  = full container image tag
+          # CTK >= 13.5 bundles CCCL >= 3.5; omit GGML_CUDA_CCCL_VERSION for those versions.
           - build: 'x64'
             os: ubuntu-24.04
             cuda: '12.8.2'
             label: '12.8'
-            defines: '-DGGML_CUDA_CUB_3DOT2=ON'
+            defines: '-DGGML_CUDA_CCCL_VERSION=v3.4.3'
           - build: 'x64'
             os: ubuntu-24.04
             cuda: '13.4.1'
             label: '13.4'
-            defines: ''
+            defines: '-DGGML_CUDA_CCCL_VERSION=v3.4.3'
           - build: 'arm64'
             os: ubuntu-24.04-arm
             cuda: '13.4.1'
             label: '13.4'
-            defines: ''
+            defines: '-DGGML_CUDA_CCCL_VERSION=v3.4.3'

     runs-on: ${{ matrix.os }}
     container: nvidia/cuda:${{ matrix.cuda }}-devel-ubuntu24.04
@@ -1244,15 +1245,16 @@ jobs:
     strategy:
       matrix:
         include:
+          # CTK >= 13.5 bundles CCCL >= 3.5; omit GGML_CUDA_CCCL_VERSION for those versions.
           - cuda: '12.4'
             arch: x64
-            defines: '-DGGML_CUDA_CUB_3DOT2=ON'
+            defines: '-DGGML_CUDA_CCCL_VERSION=v3.4.3'
           - cuda: '13.4'
             arch: x64
-            defines: ''
+            defines: '-DGGML_CUDA_CCCL_VERSION=v3.4.3'
           - cuda: '13.4'
             arch: arm64
-            defines: '-DCMAKE_TOOLCHAIN_FILE=cmake/arm64-windows-msvc-cuda.cmake'
+            defines: '-DGGML_CUDA_CCCL_VERSION=v3.4.3 -DCMAKE_TOOLCHAIN_FILE=cmake/arm64-windows-msvc-cuda.cmake'

     steps:
       - name: Clone
@@ -1279,7 +1281,6 @@ jobs:
       - name: Build
         id: cmake_build
         shell: cmd
-        # TODO: Remove GGML_CUDA_CUB_3DOT2 flag once CCCL 3.2 is bundled within CTK and that CTK version is used in this project
         run: |
           call "C:\Program Files\Microsoft Visual Studio\2022\Enterprise\VC\Auxiliary\Build\vcvarsall.bat" ${{ matrix.arch == 'x64' && 'x64' || 'amd64_arm64' }}
           cmake -S . -B build -G "Ninja Multi-Config" ^
diff --git a/ci/run.sh b/ci/run.sh
index ccfc0562f..43dccc5c1 100755
--- a/ci/run.sh
+++ b/ci/run.sh
@@ -72,8 +72,8 @@ else
 fi

 if [ ! -z ${GG_BUILD_CUDA} ]; then
-    # TODO: Remove GGML_CUDA_CUB_3DOT2 flag once CCCL 3.2 is bundled within CTK and that CTK version is used in this project
-    CMAKE_EXTRA="${CMAKE_EXTRA} -DGGML_CUDA=ON -DGGML_CUDA_CUB_3DOT2=ON"
+    # TODO: Drop GGML_CUDA_CCCL_VERSION when CUDA CI uses CTK >= 13.5, which bundles CCCL >= 3.5.
+    CMAKE_EXTRA="${CMAKE_EXTRA} -DGGML_CUDA=ON -DGGML_CUDA_CCCL_VERSION=v3.4.3"

     if command -v nvidia-smi >/dev/null 2>&1; then
         CUDA_ARCH=$(nvidia-smi --query-gpu=compute_cap --format=csv,noheader,nounits 2>/dev/null | head -1 | tr -d '.')
diff --git a/docs/build.md b/docs/build.md
index 45364281b..08a8f3333 100644
--- a/docs/build.md
+++ b/docs/build.md
@@ -195,6 +195,8 @@ cmake -B build -DGGML_CUDA=ON
 cmake --build build --config Release
 ```

+To use a specific CCCL version instead of the one bundled with the installed CUDA Toolkit, add `-DGGML_CUDA_CCCL_VERSION=vMAJOR.MINOR.PATCH`. CUB DeviceTopK requires CCCL 3.4.3 or newer; older versions use the sort fallback.
+
 Note that this also builds the CPU backend by default. On Windows on ARM, MSVC's
 support for the ARM NEON intrinsics used by the CPU backend may be incomplete, so
 a CUDA build produced entirely with MSVC might have a slower CPU backend. If CPU
diff --git a/ggml/CMakeLists.txt b/ggml/CMakeLists.txt
index 752dabbb6..4af7fc881 100644
--- a/ggml/CMakeLists.txt
+++ b/ggml/CMakeLists.txt
@@ -197,6 +197,7 @@ set(GGML_BLAS_VENDOR ${GGML_BLAS_VENDOR_DEFAULT} CACHE STRING
 option(GGML_LLAMAFILE                       "ggml: use LLAMAFILE"                             ${GGML_LLAMAFILE_DEFAULT})

 option(GGML_CUDA                            "ggml: use CUDA"                                  OFF)
+set   (GGML_CUDA_CCCL_VERSION "" CACHE STRING "ggml: CCCL git tag to fetch, empty to use the version bundled with the installed CUDA Toolkit")
 option(GGML_MUSA                            "ggml: use MUSA"                                  OFF)
 option(GGML_CUDA_FORCE_MMQ                  "ggml: use mmq kernels instead of cuBLAS"         OFF)
 option(GGML_CUDA_FORCE_CUBLAS               "ggml: always use cuBLAS instead of mmq kernels"  OFF)
diff --git a/ggml/src/ggml-cuda/CMakeLists.txt b/ggml/src/ggml-cuda/CMakeLists.txt
index dd57ac423..297d9ffc9 100644
--- a/ggml/src/ggml-cuda/CMakeLists.txt
+++ b/ggml/src/ggml-cuda/CMakeLists.txt
@@ -71,14 +71,13 @@ if (CUDAToolkit_FOUND)

     enable_language(CUDA)

-    # TODO: Remove once CCCL 3.2 has been released and bundled with CUDA Toolkit
-    if (GGML_CUDA_CUB_3DOT2)
+    if (GGML_CUDA_CCCL_VERSION)
         include(FetchContent)

         FetchContent_Declare(
             CCCL
             GIT_REPOSITORY https://github.com/nvidia/cccl.git
-            GIT_TAG        v3.2.0
+            GIT_TAG        "${GGML_CUDA_CCCL_VERSION}"
             GIT_SHALLOW    TRUE
         )

@@ -157,14 +156,15 @@ if (CUDAToolkit_FOUND)
         add_compile_definitions(GGML_CUDA_NO_PEER_COPY)
     endif()

+    if (GGML_CUDA_CCCL_VERSION)
+        target_link_libraries(ggml-cuda PRIVATE CCCL::CCCL)
+    endif()
+
     if (GGML_STATIC)
         if (WIN32)
             # As of 12.3.1 CUDA Toolkit for Windows does not offer a static cublas library
             target_link_libraries(ggml-cuda PRIVATE CUDA::cudart_static CUDA::cublas)
         else ()
-            if (GGML_CUDA_CUB_3DOT2)
-                target_link_libraries(ggml-cuda PRIVATE  CCCL::CCCL)
-            endif()
             if (CUDAToolkit_VERSION VERSION_GREATER_EQUAL "10.1")
                 target_link_libraries(ggml-cuda PRIVATE  CUDA::cudart_static CUDA::cublas_static CUDA::cublasLt_static)
             else()
@@ -172,9 +172,6 @@ if (CUDAToolkit_FOUND)
             endif()
         endif()
     else()
-        if (GGML_CUDA_CUB_3DOT2)
-            target_link_libraries(ggml-cuda PRIVATE  CCCL::CCCL)
-        endif()
         target_link_libraries(ggml-cuda PRIVATE CUDA::cudart CUDA::cublas)
     endif()

diff --git a/ggml/src/ggml-cuda/top-k.cu b/ggml/src/ggml-cuda/top-k.cu
index c7a0c8317..3ffbba839 100644
--- a/ggml/src/ggml-cuda/top-k.cu
+++ b/ggml/src/ggml-cuda/top-k.cu
@@ -3,11 +3,15 @@

 #ifdef GGML_CUDA_USE_CUB
 #    include <cub/cub.cuh>
-#    if (CCCL_MAJOR_VERSION >= 3 && CCCL_MINOR_VERSION >= 2)
+// DeviceTopK has a race condition before CCCL 3.4.3.
+// https://github.com/NVIDIA/cccl/pull/10627
+#    if (CCCL_MAJOR_VERSION > 3 || \
+         (CCCL_MAJOR_VERSION == 3 && CCCL_MINOR_VERSION > 4) || \
+         (CCCL_MAJOR_VERSION == 3 && CCCL_MINOR_VERSION == 4 && CCCL_PATCH_VERSION >= 3))
 #        define CUB_TOP_K_AVAILABLE
 #        include <cuda/iterator>
 using namespace cub;
-#    endif  // CCCL_MAJOR_VERSION >= 3 && CCCL_MINOR_VERSION >= 2
+#    endif  // CCCL >= 3.4.3
 #endif      // GGML_CUDA_USE_CUB

 #ifdef CUB_TOP_K_AVAILABLE