Commit 5ad1c5da0 for llama.cpp

commit 5ad1c5da0ad7f6176256b823925aad19134f0263
Author: qiao_px <84374584+Qiao12-pixel@users.noreply.github.com>
Date:   Wed Oct 7 04:36:21 2026 +0800

    cuda : add BF16 support for XIELU (#29955)

    The XIELU CUDA kernel template is already generic over the element
    type; only the F32/F16 type assertion and the else-if dispatch were
    missing. Add the nv_bfloat16 branch to the launcher, and drop the
    temporary supports_op gate in ggml-cuda.cu that rejected BF16+XIELU.

    test-backend-ops gains two BF16 cases ([10,5,4,3] and [512,16,1,1]).
    docs/ops/CUDA.csv and docs/ops.md are regenerated; the F32 xIELU row
    flips from no to yes as well, i.e. the previous record was stale.

    Tested:
    - Mac CPU: xIELU F32/F16/BF16, 6/6
    - Mac Metal: existing F32/F16, 4/4; BF16 still unsupported
    - RTX 4090 CUDA: xIELU F32/F16/BF16, 6/6
    - RTX 4090 CUDA BF16-only: 2/2
    - git diff --check passes

diff --git a/docs/ops.md b/docs/ops.md
index d47dda94a..8a594fa4a 100644
--- a/docs/ops.md
+++ b/docs/ops.md
@@ -129,4 +129,4 @@ Legend:
 |                              TRI | ❌ | ❌ | ✅ | ✅ | ✅ | ✅ | ✅ | ❌ | ✅ | ✅ | ✅ | ❌ | ❌ |
 |                            TRUNC | ❌ | ❌ | ✅ | 🟡 | 🟡 | ❌ | ✅ | ❌ | ✅ | ✅ | ✅ | ❌ | ❌ |
 |                          UPSCALE | ❌ | 🟡 | ✅ | ✅ | ❌ | ❌ | ✅ | 🟡 | ✅ | ✅ | ✅ | ❌ | ❌ |
-|                            XIELU | ❌ | ❌ | ✅ | ❌ | ❌ | ❌ | ✅ | ❌ | ✅ | ✅ | ✅ | ❌ | ❌ |
+|                            XIELU | ❌ | ❌ | ✅ | ✅ | ❌ | ❌ | ✅ | ❌ | ✅ | ✅ | ✅ | ❌ | ❌ |
diff --git a/docs/ops/CUDA.csv b/docs/ops/CUDA.csv
index 598aee957..7499a4927 100644
--- a/docs/ops/CUDA.csv
+++ b/docs/ops/CUDA.csv
@@ -9934,7 +9934,12 @@
 "CUDA0","CUMSUM","type=f32,ne=[2048,5,4,3]","support","1","yes","CUDA"
 "CUDA0","CUMSUM","type=f32,ne=[242004,1,1,1]","support","1","yes","CUDA"
 "CUDA0","CUMSUM","type=f32,ne=[375960,1,1,1]","support","1","yes","CUDA"
-"CUDA0","XIELU","type=f32,ne=[10,5,4,3]","support","0","no","CUDA"
+"CUDA0","XIELU","type=f32,ne=[10,5,4,3]","support","1","yes","CUDA"
+"CUDA0","XIELU","type=f16,ne=[10,5,4,3]","support","1","yes","CUDA"
+"CUDA0","XIELU","type=bf16,ne=[10,5,4,3]","support","1","yes","CUDA"
+"CUDA0","XIELU","type=f32,ne=[512,16,1,1]","support","1","yes","CUDA"
+"CUDA0","XIELU","type=f16,ne=[512,16,1,1]","support","1","yes","CUDA"
+"CUDA0","XIELU","type=bf16,ne=[512,16,1,1]","support","1","yes","CUDA"
 "CUDA0","TRI","type=f32,ne=[10,10,4,3],tri_type=3","support","1","yes","CUDA"
 "CUDA0","TRI","type=f32,ne=[10,10,4,3],tri_type=2","support","1","yes","CUDA"
 "CUDA0","TRI","type=f32,ne=[10,10,4,3],tri_type=1","support","1","yes","CUDA"
diff --git a/ggml/src/ggml-cuda/ggml-cuda.cu b/ggml/src/ggml-cuda/ggml-cuda.cu
index 0f88dfe88..33ebfdfa0 100644
--- a/ggml/src/ggml-cuda/ggml-cuda.cu
+++ b/ggml/src/ggml-cuda/ggml-cuda.cu
@@ -5314,9 +5314,6 @@ static bool ggml_backend_cuda_device_supports_op(ggml_backend_dev_t dev, const g
                 case GGML_UNARY_OP_CEIL:
                 case GGML_UNARY_OP_ROUND:
                 case GGML_UNARY_OP_TRUNC:
-                    if (op->src[0]->type == GGML_TYPE_BF16 && ggml_get_unary_op(op) == GGML_UNARY_OP_XIELU) {
-                        return false;
-                    }
                     // TODO: should become:
                     //return ggml_is_contiguous_rows(op->src[0]);
                     return ggml_is_contiguous(op->src[0]);
diff --git a/ggml/src/ggml-cuda/unary.cu b/ggml/src/ggml-cuda/unary.cu
index 84788c5dc..c9a8cf6b0 100644
--- a/ggml/src/ggml-cuda/unary.cu
+++ b/ggml/src/ggml-cuda/unary.cu
@@ -547,8 +547,8 @@ void ggml_cuda_op_xielu(ggml_backend_cuda_context & ctx, ggml_tensor * dst) {

     GGML_ASSERT(ggml_is_contiguous(src0));

-    GGML_ASSERT(src0->type == GGML_TYPE_F32 || src0->type == GGML_TYPE_F16);
-    GGML_ASSERT( dst->type == GGML_TYPE_F32 ||  dst->type == GGML_TYPE_F16);
+    GGML_ASSERT(src0->type == GGML_TYPE_F32 || src0->type == GGML_TYPE_F16 || src0->type == GGML_TYPE_BF16);
+    GGML_ASSERT( dst->type == GGML_TYPE_F32 ||  dst->type == GGML_TYPE_F16 ||  dst->type == GGML_TYPE_BF16);
     GGML_ASSERT(src0->type == dst->type);

     const float alpha_n = ggml_get_op_params_f32(dst, 1);
@@ -558,6 +558,8 @@ void ggml_cuda_op_xielu(ggml_backend_cuda_context & ctx, ggml_tensor * dst) {

     if (src0->type == GGML_TYPE_F16) {
         xielu_cuda((const half *)src0_d, (half *)dst_d, ggml_nelements(src0), alpha_n, alpha_p, beta, eps, stream);
+    } else if (src0->type == GGML_TYPE_BF16) {
+        xielu_cuda((const nv_bfloat16 *)src0_d, (nv_bfloat16 *)dst_d, ggml_nelements(src0), alpha_n, alpha_p, beta, eps, stream);
     } else {
         xielu_cuda((const float *)src0_d, (float *)dst_d, ggml_nelements(src0), alpha_n, alpha_p, beta, eps, stream);
     }
diff --git a/tests/test-backend-ops.cpp b/tests/test-backend-ops.cpp
index cbe59178d..b080ab496 100644
--- a/tests/test-backend-ops.cpp
+++ b/tests/test-backend-ops.cpp
@@ -11189,8 +11189,10 @@ static std::vector<std::unique_ptr<test_case>> make_test_cases_eval() {

     test_cases.emplace_back(new test_xielu());
     test_cases.emplace_back(new test_xielu(GGML_TYPE_F16));
+    test_cases.emplace_back(new test_xielu(GGML_TYPE_BF16));
     test_cases.emplace_back(new test_xielu(GGML_TYPE_F32, { 512, 16, 1, 1 }));
     test_cases.emplace_back(new test_xielu(GGML_TYPE_F16, { 512, 16, 1, 1 }));
+    test_cases.emplace_back(new test_xielu(GGML_TYPE_BF16, { 512, 16, 1, 1 }));

     test_cases.emplace_back(new test_tri(GGML_TRI_TYPE_LOWER));
     test_cases.emplace_back(new test_tri(GGML_TRI_TYPE_LOWER_DIAG));