Commit 3d65c90d0 for llama.cpp
commit 3d65c90d04d337e88f2b1f7f0061f40a5324e662
Author: Nicolas Mowen <nickmowen213@gmail.com>
Date: Thu Oct 8 20:19:32 2026 -0600
sycl : Q5_K reorder-layout MMVQ and fused GLU (#29375)
diff --git a/ggml/src/ggml-sycl/ggml-sycl.cpp b/ggml/src/ggml-sycl/ggml-sycl.cpp
index 2a5184e01..d2c1a1a44 100644
--- a/ggml/src/ggml-sycl/ggml-sycl.cpp
+++ b/ggml/src/ggml-sycl/ggml-sycl.cpp
@@ -5102,12 +5102,19 @@ static bool ggml_sycl_mul_mat_glu_mmvq_fused(ggml_backend_sycl_context & ctx, gg
return false;
}
- // quant pairs the reorder kernel cannot serve (mixed gate/up types) take the
- // standard-layout fused path instead; q4_K keeps the reorder path below
- if (wg->type != GGML_TYPE_Q4_K || wu->type != GGML_TYPE_Q4_K) {
+ // quant pairs the reorder kernel does not serve (mixed gate/up types, q5_K off BMG) take the
+ // standard-layout fused path instead; same-type q4_K / q5_K keep the reorder path below
+ const bool reorder_pair = wg->type == wu->type &&
+ (wu->type == GGML_TYPE_Q4_K || (wu->type == GGML_TYPE_Q5_K && ggml_sycl_q5_k_mmvq_reuse(ctx.device)));
+ if (!reorder_pair) {
return ggml_sycl_mul_mat_glu_mmvq_plain(ctx, glu, gate, up, wu, wg, act);
}
+ // past 5 columns the two unfused q5_K GEMVs are faster than the fused kernel
+ if (wu->type == GGML_TYPE_Q5_K && act->ne[1] > 5) {
+ return false;
+ }
+
// install the reorder (SoA) layout the fused kernel needs, as the unfused mmvq path would;
// a no-op once done. after the bail checks so a declined op does not pay for it.
opt_for_reorder(&ctx, wu, act, up, mul_mat_algo::MMVQ);
diff --git a/ggml/src/ggml-sycl/mmvq.cpp b/ggml/src/ggml-sycl/mmvq.cpp
index ceaacb59e..e8b293a97 100644
--- a/ggml/src/ggml-sycl/mmvq.cpp
+++ b/ggml/src/ggml-sycl/mmvq.cpp
@@ -110,7 +110,8 @@ static void mul_mat_vec_q_reorder(const void * __restrict__ vx, const void * __r
// With has_fusion, `vgate` is a second weight matrix sharing vx's shape, stride and reorder
// layout: one pass computes both row dot products and the epilogue writes glu(gate, up).
-template <typename reorder_vec_dot_q_sycl, int ncols_dst, bool has_fusion = false, int rows_per_sg = 1>
+template <typename reorder_vec_dot_q_sycl, int ncols_dst, bool has_fusion = false, int rows_per_sg = 1,
+ bool shared_weights = reorder_vec_dot_shared_weights<reorder_vec_dot_q_sycl::gtype>::value>
static void mul_mat_vec_q_reorder_ncols(const void * __restrict__ vx, const void * __restrict__ vgate,
const void * __restrict__ vy, float * __restrict__ dst, const int ncols,
const int nrows, const int stride_col_y_bytes, const int stride_col_dst,
@@ -181,7 +182,7 @@ static void mul_mat_vec_q_reorder_ncols(const void * __restrict__ vx, const void
}
}
}
- } else if constexpr (reorder_vec_dot_shared_weights<reorder_vec_dot_q_sycl::gtype>::value) {
+ } else if constexpr (shared_weights) {
const int ibx = row0 * blocks_per_row + i;
const auto bx_offset = block_type::get_block_offset(ibx, nblocks);
const auto d_offset = block_type::get_d_offset(nrows, ncols, ibx);
@@ -1945,8 +1946,8 @@ static void reorder_mul_mat_vec_q5_k_q8_1_sycl(const void * vx, const void * vy,
});
}
-template <int ncols_dst>
-static void reorder_mul_mat_vec_q5_k_q8_1_sycl_ncols(
+template <int ncols_dst, int rows_per_sg, bool shared_weights>
+static void reorder_mul_mat_vec_q5_k_q8_1_sycl_ncols_impl(
const void * vx, const void * vy, float * dst,
const int ncols, const int nrows,
const int stride_col_y_bytes, const int stride_col_dst,
@@ -1954,20 +1955,35 @@ static void reorder_mul_mat_vec_q5_k_q8_1_sycl_ncols(
GGML_ASSERT(ncols % QK_K == 0);
constexpr size_t num_subgroups = WARP_SIZE;
- const int block_num_y = ceil_div(nrows, GGML_SYCL_MMV_Y * (int) num_subgroups);
+ const int block_num_y = ceil_div(nrows, GGML_SYCL_MMV_Y * (int) num_subgroups * rows_per_sg);
const sycl::range<3> block_nums(1, 1, block_num_y);
const sycl::range<3> block_dims(1, GGML_SYCL_MMV_Y, num_subgroups * WARP_SIZE);
stream->submit([&](sycl::handler & cgh) {
cgh.parallel_for(sycl::nd_range<3>(block_nums * block_dims, block_dims),
[=](sycl::nd_item<3> nd_item) [[sycl::reqd_sub_group_size(WARP_SIZE)]] {
- mul_mat_vec_q_reorder_ncols<reorder_vec_dot_q_sycl<GGML_TYPE_Q5_K>, ncols_dst>(
+ mul_mat_vec_q_reorder_ncols<reorder_vec_dot_q_sycl<GGML_TYPE_Q5_K>, ncols_dst,
+ /*has_fusion=*/ false, rows_per_sg, shared_weights>(
vx, /*vgate=*/ nullptr, vy, dst, ncols, nrows, stride_col_y_bytes, stride_col_dst,
/*glu_op=*/ GGML_GLU_OP_SWIGLU, nd_item);
});
});
}
+template <int ncols_dst>
+static void reorder_mul_mat_vec_q5_k_q8_1_sycl_ncols(
+ const void * vx, const void * vy, float * dst,
+ const int ncols, const int nrows,
+ const int stride_col_y_bytes, const int stride_col_dst,
+ dpct::queue_ptr stream) {
+ if (ggml_sycl_q5_k_mmvq_reuse(ggml_sycl_get_device())) {
+ constexpr int rows_per_sg = ncols_dst >= 3 ? 2 : 1;
+ reorder_mul_mat_vec_q5_k_q8_1_sycl_ncols_impl<ncols_dst, rows_per_sg, true>(vx, vy, dst, ncols, nrows, stride_col_y_bytes, stride_col_dst, stream);
+ } else {
+ reorder_mul_mat_vec_q5_k_q8_1_sycl_ncols_impl<ncols_dst, 1, false>(vx, vy, dst, ncols, nrows, stride_col_y_bytes, stride_col_dst, stream);
+ }
+}
+
static void reorder_mul_mat_vec_q5_k_q8_1_sycl_switch_ncols(
const void * vx, const void * vy, float * dst,
const int ncols, const int nrows, const int ncols_dst,
@@ -3129,8 +3145,11 @@ static void launch_mul_mat_vec_q_reorder_glu(const void * vx, const void * vgate
const int ncols, const int nrows, const int stride_col_y_bytes,
const int stride_col_dst, const ggml_glu_op glu_op,
dpct::queue_ptr stream) {
+ // q4_K pairs rows for 3..4 columns, q5_K for 3..5
+ constexpr int row_pair_max = reorder_vec_dot_q_sycl::gtype == GGML_TYPE_Q5_K ? 5 : 4;
constexpr int rows_per_sg =
- reorder_vec_dot_shared_activations<reorder_vec_dot_q_sycl::gtype>::value && ncols_dst >= 3 && ncols_dst <= 4
+ reorder_vec_dot_shared_activations<reorder_vec_dot_q_sycl::gtype>::value && ncols_dst >= 3 &&
+ ncols_dst <= row_pair_max
? 2
: 1;
launch_mul_mat_vec_q_reorder_glu_impl<reorder_vec_dot_q_sycl, ncols_dst, rows_per_sg>(vx, vgate, vy, dst, ncols, nrows, stride_col_y_bytes, stride_col_dst, glu_op, stream);
@@ -3321,55 +3340,48 @@ bool ggml_sycl_mul_mat_vec_q_glu_plain(enum ggml_type gate_type, enum ggml_type
return false;
}
+template <ggml_type type, int... Ns>
+static bool mul_mat_vec_q_glu_reorder_ncols(enum ggml_glu_op glu_op, const void * vx, const void * vgate,
+ const void * vy, float * dst, int ncols, int nrows, int ncols_dst,
+ int stride_col_y_bytes, int stride_col_dst, dpct::queue_ptr stream) {
+ using vec_dot = reorder_vec_dot_q_sycl<type>;
+
+ auto launch = [&](auto I) -> bool {
+ constexpr int n = decltype(I)::value;
+ if (ncols_dst != n) {
+ return false;
+ }
+ if constexpr (type == GGML_TYPE_Q4_K && n == 2) {
+ if (nrows >= Q4_K_MMVQ_ROW_PAIR_MIN_NROWS) {
+ launch_mul_mat_vec_q_reorder_glu_impl<vec_dot, 2, 2>(vx, vgate, vy, dst, ncols, nrows, stride_col_y_bytes, stride_col_dst, glu_op, stream);
+ return true;
+ }
+ }
+ launch_mul_mat_vec_q_reorder_glu<vec_dot, n>(vx, vgate, vy, dst, ncols, nrows, stride_col_y_bytes,
+ stride_col_dst, glu_op, stream);
+ return true;
+ };
+
+ // unary fold over launch
+ return (launch(std::integral_constant<int, Ns>{}) || ...);
+}
+
bool ggml_sycl_mul_mat_vec_q_glu_reorder(enum ggml_type src0_type, enum ggml_glu_op glu_op, const void * vx,
const void * vgate, const void * vy, float * dst, int ncols, int nrows,
int ncols_dst, int stride_col_y_bytes, int stride_col_dst,
dpct::queue_ptr stream) {
- if (src0_type != GGML_TYPE_Q4_K) {
- return false;
- }
if (glu_op != GGML_GLU_OP_SWIGLU && glu_op != GGML_GLU_OP_GEGLU) {
return false;
}
- using vec_dot = reorder_vec_dot_q_sycl<GGML_TYPE_Q4_K>;
-
- switch (ncols_dst) {
- case 1:
- launch_mul_mat_vec_q_reorder_glu<vec_dot, 1>(vx, vgate, vy, dst, ncols, nrows, stride_col_y_bytes,
- stride_col_dst, glu_op, stream);
- return true;
- case 2:
- if (nrows >= Q4_K_MMVQ_ROW_PAIR_MIN_NROWS) {
- launch_mul_mat_vec_q_reorder_glu_impl<vec_dot, 2, 2>(vx, vgate, vy, dst, ncols, nrows, stride_col_y_bytes, stride_col_dst, glu_op, stream);
- } else {
- launch_mul_mat_vec_q_reorder_glu_impl<vec_dot, 2, 1>(vx, vgate, vy, dst, ncols, nrows, stride_col_y_bytes, stride_col_dst, glu_op, stream);
- }
- return true;
- case 3:
- launch_mul_mat_vec_q_reorder_glu<vec_dot, 3>(vx, vgate, vy, dst, ncols, nrows, stride_col_y_bytes,
- stride_col_dst, glu_op, stream);
- return true;
- case 4:
- launch_mul_mat_vec_q_reorder_glu<vec_dot, 4>(vx, vgate, vy, dst, ncols, nrows, stride_col_y_bytes,
- stride_col_dst, glu_op, stream);
- return true;
- case 5:
- launch_mul_mat_vec_q_reorder_glu<vec_dot, 5>(vx, vgate, vy, dst, ncols, nrows, stride_col_y_bytes,
- stride_col_dst, glu_op, stream);
- return true;
- case 6:
- launch_mul_mat_vec_q_reorder_glu<vec_dot, 6>(vx, vgate, vy, dst, ncols, nrows, stride_col_y_bytes,
- stride_col_dst, glu_op, stream);
- return true;
- case 7:
- launch_mul_mat_vec_q_reorder_glu<vec_dot, 7>(vx, vgate, vy, dst, ncols, nrows, stride_col_y_bytes,
- stride_col_dst, glu_op, stream);
- return true;
- case 8:
- launch_mul_mat_vec_q_reorder_glu<vec_dot, 8>(vx, vgate, vy, dst, ncols, nrows, stride_col_y_bytes,
- stride_col_dst, glu_op, stream);
- return true;
+ switch (src0_type) {
+ case GGML_TYPE_Q4_K:
+ return mul_mat_vec_q_glu_reorder_ncols<GGML_TYPE_Q4_K, 1, 2, 3, 4, 5, 6, 7, 8>(
+ glu_op, vx, vgate, vy, dst, ncols, nrows, ncols_dst, stride_col_y_bytes, stride_col_dst, stream);
+ case GGML_TYPE_Q5_K:
+ // fusion declines q5_K past 5 columns
+ return mul_mat_vec_q_glu_reorder_ncols<GGML_TYPE_Q5_K, 1, 2, 3, 4, 5>(
+ glu_op, vx, vgate, vy, dst, ncols, nrows, ncols_dst, stride_col_y_bytes, stride_col_dst, stream);
default:
return false;
}
diff --git a/ggml/src/ggml-sycl/mmvq.hpp b/ggml/src/ggml-sycl/mmvq.hpp
index 7fb9cf6f8..3a0ef07d6 100644
--- a/ggml/src/ggml-sycl/mmvq.hpp
+++ b/ggml/src/ggml-sycl/mmvq.hpp
@@ -15,6 +15,12 @@
#include "common.hpp"
+// q5_K multi-column MMVQ shares weights across columns, pairs rows and fuses gate/up in the reorder
+// layout: faster on Xe2 (BMG), so untested archs keep the per-column kernel
+inline bool ggml_sycl_q5_k_mmvq_reuse(int device) {
+ const gpu_arch arch = ggml_sycl_info().devices[device].hw_info.arch;
+ return arch == gpu_arch::intel_gpu_bmg_g21 || arch == gpu_arch::intel_gpu_bmg_g31;
+}
void ggml_sycl_op_mul_mat_vec_q(
ggml_backend_sycl_context & ctx,
diff --git a/ggml/src/ggml-sycl/vecdotq.hpp b/ggml/src/ggml-sycl/vecdotq.hpp
index cc6ae6a8d..6ae951525 100644
--- a/ggml/src/ggml-sycl/vecdotq.hpp
+++ b/ggml/src/ggml-sycl/vecdotq.hpp
@@ -362,6 +362,10 @@ template <> struct reorder_vec_dot_shared_weights<GGML_TYPE_Q4_K> {
static constexpr bool value = true;
};
+template <> struct reorder_vec_dot_shared_weights<GGML_TYPE_Q5_K> {
+ static constexpr bool value = true;
+};
+
template <ggml_type T> struct reorder_vec_dot_shared_activations {
static constexpr bool value = false;
};
@@ -370,6 +374,10 @@ template <> struct reorder_vec_dot_shared_activations<GGML_TYPE_Q4_K> {
static constexpr bool value = true;
};
+template <> struct reorder_vec_dot_shared_activations<GGML_TYPE_Q5_K> {
+ static constexpr bool value = true;
+};
+
template <> struct reorder_vec_dot_q_sycl<GGML_TYPE_Q4_0> {
static constexpr ggml_type gtype = GGML_TYPE_Q4_0;
@@ -676,56 +684,72 @@ template <> struct reorder_vec_dot_q_sycl<GGML_TYPE_Q5_K> {
using q5_k_block = ggml_sycl_reordered::block_q_t<GGML_TYPE_Q5_K>;
using q5_k_traits = typename q5_k_block::traits;
- __dpct_inline__ float operator()(const void * __restrict__ vbq, const std::pair<int, int> ibx_offset,
- const std::pair<int, int> d_offset, const int8_t * q8_1_quant_ptr,
- const sycl::half2 * q8_1_ds, const int & iqs) {
- const uint8_t * base = static_cast<const uint8_t *>(vbq);
- const uint8_t * qs = base + ibx_offset.first; // low 4 bits
- const uint8_t * qh_base = base + ibx_offset.second; // high bit
- const uint8_t * scs = base + d_offset.first;
- const ggml_half2 * dms = reinterpret_cast<const ggml_half2 *>(base + d_offset.second);
+ struct weights {
+ int vl[2];
+ int vh[2];
+ uint16_t aux[2];
+ ggml_half2 dm;
+ };
+
+ // same activation layout as Q4_K
+ static_assert(QR5_K == QR4_K);
+ using activations = reorder_vec_dot_q_sycl<GGML_TYPE_Q4_K>::activations;
+
+ __dpct_inline__ static weights load(const void * __restrict__ vbq, const std::pair<int, int> ibx_offset,
+ const std::pair<int, int> d_offset, const int & iqs) {
+ const uint8_t * base = static_cast<const uint8_t *>(vbq);
+ const uint8_t * qs = base + ibx_offset.first; // low 4 bits
+ const uint8_t * qh_base = base + ibx_offset.second; // high bit
+ const uint8_t * scs = base + d_offset.first;
+ const ggml_half2 * dms = reinterpret_cast<const ggml_half2 *>(base + d_offset.second);
const int bq8_offset = QR5_K * ((iqs / 2) / (QI8_1 / 2));
const int * ql_ptr = (const int *) (qs + 16 * bq8_offset + 4 * ((iqs / 2) % 4));
const int * qh_ptr = (const int *) (qh_base + 4 * ((iqs / 2) % 4));
const uint16_t * scales = (const uint16_t *) scs;
- int vl[2];
- int vh[2];
- int u[2 * QR5_K];
- float d8[QR5_K];
-
- vl[0] = ql_ptr[0];
- vl[1] = ql_ptr[4];
+ weights w;
+ w.vl[0] = ql_ptr[0];
+ w.vl[1] = ql_ptr[4];
- vh[0] = qh_ptr[0] >> bq8_offset;
- vh[1] = qh_ptr[4] >> bq8_offset;
+ w.vh[0] = qh_ptr[0] >> bq8_offset;
+ w.vh[1] = qh_ptr[4] >> bq8_offset;
- uint16_t aux[2];
const int j = (QR5_K * ((iqs / 2) / (QI8_1 / 2))) / 2;
if (j < 2) {
- aux[0] = scales[j + 0] & 0x3f3f;
- aux[1] = scales[j + 2] & 0x3f3f;
+ w.aux[0] = scales[j + 0] & 0x3f3f;
+ w.aux[1] = scales[j + 2] & 0x3f3f;
} else {
- aux[0] = ((scales[j + 2] >> 0) & 0x0f0f) | ((scales[j - 2] & 0xc0c0) >> 2);
- aux[1] = ((scales[j + 2] >> 4) & 0x0f0f) | ((scales[j - 0] & 0xc0c0) >> 2);
+ w.aux[0] = ((scales[j + 2] >> 0) & 0x0f0f) | ((scales[j - 2] & 0xc0c0) >> 2);
+ w.aux[1] = ((scales[j + 2] >> 4) & 0x0f0f) | ((scales[j - 0] & 0xc0c0) >> 2);
}
- const uint8_t * sc = (const uint8_t *) aux;
- const uint8_t * m = sc + 2;
+ w.dm = *dms;
- for (int i = 0; i < QR5_K; ++i) {
- const int8_t* quant_base_ptr = q8_1_quant_ptr + (bq8_offset + i) * QK8_1;
- sycl::half2 ds_values = *(q8_1_ds + bq8_offset + i);
+ return w;
+ }
- d8[i] = ds_values[0];
+ __dpct_inline__ static activations load_activations(const int8_t * q8_1_quant_ptr,
+ const sycl::half2 * q8_1_ds, const int & iqs) {
+ return reorder_vec_dot_q_sycl<GGML_TYPE_Q4_K>::load_activations(q8_1_quant_ptr, q8_1_ds, iqs);
+ }
- const int * q8 = (const int *) quant_base_ptr + ((iqs / 2) % 4);
- u[2 * i + 0] = q8[0];
- u[2 * i + 1] = q8[4];
- }
+ __dpct_inline__ static float apply(const weights & w, const activations & a) {
+ const uint8_t * sc = (const uint8_t *) w.aux;
+ const uint8_t * m = sc + 2;
+
+ return vec_dot_q5_K_q8_1_impl_vmmq(w.vl, w.vh, a.u, sc, m, w.dm, a.d8);
+ }
- return vec_dot_q5_K_q8_1_impl_vmmq(vl, vh, u, sc, m, *dms, d8);
+ __dpct_inline__ static float dot(const weights & w, const int8_t * q8_1_quant_ptr,
+ const sycl::half2 * q8_1_ds, const int & iqs) {
+ return apply(w, load_activations(q8_1_quant_ptr, q8_1_ds, iqs));
+ }
+
+ __dpct_inline__ float operator()(const void * __restrict__ vbq, const std::pair<int, int> ibx_offset,
+ const std::pair<int, int> d_offset, const int8_t * q8_1_quant_ptr,
+ const sycl::half2 * q8_1_ds, const int & iqs) {
+ return dot(load(vbq, ibx_offset, d_offset, iqs), q8_1_quant_ptr, q8_1_ds, iqs);
}
};