Commit 90c908d06 for llama.cpp
commit 90c908d06d82ab02d0204a44c4a46fd3348f0535
Author: Pascal <admin@serveurperso.com>
Date: Wed Sep 30 14:14:45 2026 +0200
cpu: accept BF16 in src1 of mul_mat (#28937)
* cpu: accept BF16 in src1 of mul_mat
ggml_conv_1d_dw builds its im2col in F32 when the kernel is BF16, then
calls ggml_mul_mat(im2col, kernel), which puts F32 in src0 and BF16 in
src1. The CPU backend refused that combination, so it was reported as
unsupported on every backend and never compared against anything.
Widen BF16 into the F32 work buffer, next to the existing packing of F32
into vec_dot_type. This is the arithmetic the Metal mat vec kernel
already uses, both operands promoted to float and accumulated in float,
so the two agree exactly rather than approximately.
Cover it with a conv_1d_dw test over F32, F16 and BF16 kernels, plus
three mul_mat cases with BF16 in src1.
* vulkan: reject BF16 in src1 of mul_mat unless src0 is BF16
supports_op only checked the src1 type for non contiguous tensors, so
a contiguous BF16 src1 was accepted and the pipeline lookup asserted.
The only BF16 src1 path is the BF16 x BF16 multiply, every other src0
type now reports the op as unsupported and the scheduler keeps it on
the CPU.
The BF16 kernel case of the conv_1d_dw test needs the f32 x bf16
mat vec variants of the Metal backend, which land separately.
diff --git a/ggml/src/ggml-cpu/ggml-cpu.c b/ggml/src/ggml-cpu/ggml-cpu.c
index 24c47569c..8620c6c7a 100644
--- a/ggml/src/ggml-cpu/ggml-cpu.c
+++ b/ggml/src/ggml-cpu/ggml-cpu.c
@@ -1334,9 +1334,11 @@ UseGgmlGemm1:;
const size_t nbw3 = nbw2*ne12;
assert(params->wsize >= ne13*nbw3);
- GGML_ASSERT(src1->type == GGML_TYPE_F32 || src1->type == GGML_TYPE_F16);
- // the F16 path below writes plain floats into wdata, so it needs an F32 vec_dot_type
- GGML_ASSERT(src1->type == GGML_TYPE_F32 || vec_dot_type == GGML_TYPE_F32);
+ // src1 is either packed from F32 into vec_dot_type, or widened from F16 or BF16 into the F32 work buffer
+ const bool widen = src1->type != GGML_TYPE_F32;
+
+ GGML_ASSERT(!widen || vec_dot_type == GGML_TYPE_F32);
+ GGML_ASSERT(!widen || src1->type == GGML_TYPE_F16 || src1->type == GGML_TYPE_BF16);
#if 0
for (int64_t i13 = 0; i13 < ne13; ++i13) {
@@ -1349,20 +1351,24 @@ UseGgmlGemm1:;
}
}
#else
+ const int64_t bs = ggml_blck_size(vec_dot_type);
+
for (int64_t i13 = 0; i13 < ne13; ++i13) {
for (int64_t i12 = 0; i12 < ne12; ++i12) {
for (int64_t i11 = 0; i11 < ne11; ++i11) {
- size_t bs = ggml_blck_size(vec_dot_type);
int64_t ne10_block_start = (ith * ne10/bs) / nth;
int64_t ne10_block_end = ((ith + 1) * ne10/bs) / nth;
- const char * src1_block = (const char *) src1->data + i13*nb13 + i12*nb12 + i11*nb11 + ne10_block_start*bs*nb10;
- char * dst_block = wdata + i13*nbw3 + i12*nbw2 + i11*nbw1 + ne10_block_start*nbw0;
- const int64_t n_block = (ne10_block_end - ne10_block_start) * bs;
+ const void * src1_row = (const char *) src1->data + i13*nb13 + i12*nb12 + i11*nb11 + ne10_block_start*bs*nb10;
+ void * wdata_row = wdata + i13*nbw3 + i12*nbw2 + i11*nbw1 + ne10_block_start*nbw0;
+
+ const int64_t ne10_block_size = (ne10_block_end - ne10_block_start) * bs;
- if (src1->type == GGML_TYPE_F32) {
- from_float((const float *) src1_block, dst_block, n_block);
+ if (src1->type == GGML_TYPE_F16) {
+ ggml_cpu_fp16_to_fp32((const ggml_fp16_t *) src1_row, (float *) wdata_row, ne10_block_size);
+ } else if (src1->type == GGML_TYPE_BF16) {
+ ggml_cpu_bf16_to_fp32((const ggml_bf16_t *) src1_row, (float *) wdata_row, ne10_block_size);
} else {
- ggml_cpu_fp16_to_fp32((const ggml_fp16_t *) src1_block, (float *) dst_block, n_block);
+ from_float((const float *) src1_row, wdata_row, ne10_block_size);
}
}
}
diff --git a/ggml/src/ggml-cpu/ggml-cpu.cpp b/ggml/src/ggml-cpu/ggml-cpu.cpp
index 1df0f2bb9..81ff5d79f 100644
--- a/ggml/src/ggml-cpu/ggml-cpu.cpp
+++ b/ggml/src/ggml-cpu/ggml-cpu.cpp
@@ -455,7 +455,10 @@ static bool ggml_backend_cpu_device_supports_op(ggml_backend_dev_t dev, const st
src0->type == GGML_TYPE_F32 && op->type == GGML_TYPE_F32) {
return src1->type == GGML_TYPE_F32 || src1->type == GGML_TYPE_F16;
}
- return src1->type == GGML_TYPE_F32 || src1->type == ggml_get_type_traits_cpu(src0->type)->vec_dot_type;
+ // BF16 in src1 is widened into the F32 work buffer
+ return src1->type == GGML_TYPE_F32 ||
+ src1->type == ggml_get_type_traits_cpu(src0->type)->vec_dot_type ||
+ (src1->type == GGML_TYPE_BF16 && ggml_get_type_traits_cpu(src0->type)->vec_dot_type == GGML_TYPE_F32);
case GGML_OP_SOFT_MAX_BACK: {
if (op->src[0]->type != GGML_TYPE_F32 || op->src[1]->type != GGML_TYPE_F32) {
return false;
diff --git a/ggml/src/ggml-vulkan/ggml-vulkan.cpp b/ggml/src/ggml-vulkan/ggml-vulkan.cpp
index e23d5b433..cbd3697ef 100644
--- a/ggml/src/ggml-vulkan/ggml-vulkan.cpp
+++ b/ggml/src/ggml-vulkan/ggml-vulkan.cpp
@@ -15441,6 +15441,10 @@ static bool ggml_backend_vk_device_supports_op(ggml_backend_dev_t dev, const ggm
// So don't support this combination for now.
return false;
}
+ if (op->src[1]->type == GGML_TYPE_BF16 && op->src[0]->type != GGML_TYPE_BF16) {
+ // BF16 in src1 is only served by the BF16 x BF16 pipelines
+ return false;
+ }
return true;
}
diff --git a/tests/test-backend-ops.cpp b/tests/test-backend-ops.cpp
index 17ac9dafe..23b23147d 100644
--- a/tests/test-backend-ops.cpp
+++ b/tests/test-backend-ops.cpp
@@ -6465,6 +6465,49 @@ struct test_conv_2d : public test_case {
}
};
+// GGML_OP_IM2COL + GGML_OP_MUL_MAT
+// the im2col of a BF16 kernel is F32, which puts F32 in src0 and BF16 in src1 of the mul_mat
+struct test_conv_1d_dw : public test_case {
+ const std::array<int64_t, 4> ne_input; // T, C
+ const std::array<int64_t, 4> ne_kernel; // K, 1, C
+ const ggml_type type_kernel;
+ const int stride;
+ const int padding;
+ const int dilation;
+
+ std::string op_desc(ggml_tensor * t) override {
+ GGML_UNUSED(t);
+ return "CONV_1D_DW";
+ }
+
+ std::string vars() override {
+ return VARS_TO_STR6(ne_input, ne_kernel, type_kernel, stride, padding, dilation);
+ }
+
+ double max_nmse_err() override {
+ return 5e-4;
+ }
+
+ test_conv_1d_dw(
+ std::array<int64_t, 4> ne_input = {64, 16, 1, 1},
+ std::array<int64_t, 4> ne_kernel = {3, 1, 16, 1},
+ ggml_type type_kernel = GGML_TYPE_F32,
+ int stride = 1, int padding = 0, int dilation = 1)
+ : ne_input(ne_input), ne_kernel(ne_kernel), type_kernel(type_kernel), stride(stride), padding(padding), dilation(dilation) {}
+
+ ggml_tensor * build_graph(ggml_context * ctx) override {
+ ggml_tensor * input = ggml_new_tensor(ctx, GGML_TYPE_F32, 4, ne_input.data());
+ ggml_set_name(input, "input");
+
+ ggml_tensor * kernel = ggml_new_tensor(ctx, type_kernel, 4, ne_kernel.data());
+ ggml_set_name(kernel, "kernel");
+
+ ggml_tensor * out = ggml_conv_1d_dw(ctx, kernel, input, stride, padding, dilation);
+ ggml_set_name(out, "out");
+ return out;
+ }
+};
+
// GGML_OP_CONV_2D_DW
struct test_conv_2d_dw : public test_case {
const std::array<int64_t, 4> ne_input;
@@ -9535,6 +9578,12 @@ static std::vector<std::unique_ptr<test_case>> make_test_cases_eval() {
// test_cases.emplace_back(new test_im2col(GGML_TYPE_F32, GGML_TYPE_F16, GGML_TYPE_F16, {1024, 1024, 256, 1}, {3, 3, 256, 1}, 1, 1, 1, 1, 1, 1, true));
// test_cases.emplace_back(new test_im2col(GGML_TYPE_F32, GGML_TYPE_F16, GGML_TYPE_F32, {1024, 1024, 256, 1}, {3, 3, 256, 1}, 1, 1, 1, 1, 1, 1, true));
+ for (ggml_type kernel_type : {GGML_TYPE_F32, GGML_TYPE_F16, GGML_TYPE_BF16}) {
+ test_cases.emplace_back(new test_conv_1d_dw({64, 16, 1, 1}, {3, 1, 16, 1}, kernel_type, 1, 0, 1));
+ test_cases.emplace_back(new test_conv_1d_dw({64, 16, 1, 1}, {7, 1, 16, 1}, kernel_type, 1, 3, 1));
+ test_cases.emplace_back(new test_conv_1d_dw({97, 33, 1, 1}, {5, 1, 33, 1}, kernel_type, 2, 2, 2));
+ }
+
test_cases.emplace_back(new test_conv_2d_dw({17, 34, 9, 1}, {3, 3, 1, 9}, GGML_TYPE_F32, 1, 0, 1, false));
test_cases.emplace_back(new test_conv_2d_dw({17, 34, 9, 1}, {3, 3, 1, 9}, GGML_TYPE_F32, 1, 0, 1, true));
test_cases.emplace_back(new test_conv_2d_dw({32, 8, 64, 1}, {3, 3, 1, 64}, GGML_TYPE_F32, 2, 1, 1, false));
@@ -10230,6 +10279,11 @@ static std::vector<std::unique_ptr<test_case>> make_test_cases_eval() {
test_cases.emplace_back(new test_mul_mat(GGML_TYPE_BF16, GGML_TYPE_F32, 16, 16, 256, {2, 3}, {1, 1}, {0, 1, 3, 2}));
test_cases.emplace_back(new test_mul_mat(GGML_TYPE_BF16, GGML_TYPE_F32, 16, 16, 256, {2, 3}, {1, 1}, {0, 3, 2, 1}));
+ // BF16 in src1, as ggml_conv_1d_dw emits it for a BF16 kernel
+ test_cases.emplace_back(new test_mul_mat(GGML_TYPE_F32, GGML_TYPE_BF16, 16, 1, 256, {1, 1}, {1, 1}));
+ test_cases.emplace_back(new test_mul_mat(GGML_TYPE_F32, GGML_TYPE_BF16, 16, 1, 256, {3, 2}, {2, 2}));
+ test_cases.emplace_back(new test_mul_mat(GGML_TYPE_F32, GGML_TYPE_BF16, 16, 8, 256, {1, 1}, {1, 1}));
+
// token-tile boundary coverage. With n_used == n_mats every token routes to every expert, so
// each expert receives exactly n rows, with no dependence on the random draw. mul_mm_id is used
// from 32 tokens up: n = 32, 33, 47, 48, 49 reach it, leaving a last tile of 32, 1, 15, 16 and