Commit 3d65c90d0 for llama.cpp

commit 3d65c90d04d337e88f2b1f7f0061f40a5324e662
Author: Nicolas Mowen <nickmowen213@gmail.com>
Date:   Thu Oct 8 20:19:32 2026 -0600

    sycl : Q5_K reorder-layout MMVQ and fused GLU (#29375)

diff --git a/ggml/src/ggml-sycl/ggml-sycl.cpp b/ggml/src/ggml-sycl/ggml-sycl.cpp
index 2a5184e01..d2c1a1a44 100644
--- a/ggml/src/ggml-sycl/ggml-sycl.cpp
+++ b/ggml/src/ggml-sycl/ggml-sycl.cpp
@@ -5102,12 +5102,19 @@ static bool ggml_sycl_mul_mat_glu_mmvq_fused(ggml_backend_sycl_context & ctx, gg
         return false;
     }

-    // quant pairs the reorder kernel cannot serve (mixed gate/up types) take the
-    // standard-layout fused path instead; q4_K keeps the reorder path below
-    if (wg->type != GGML_TYPE_Q4_K || wu->type != GGML_TYPE_Q4_K) {
+    // quant pairs the reorder kernel does not serve (mixed gate/up types, q5_K off BMG) take the
+    // standard-layout fused path instead; same-type q4_K / q5_K keep the reorder path below
+    const bool reorder_pair = wg->type == wu->type &&
+        (wu->type == GGML_TYPE_Q4_K || (wu->type == GGML_TYPE_Q5_K && ggml_sycl_q5_k_mmvq_reuse(ctx.device)));
+    if (!reorder_pair) {
         return ggml_sycl_mul_mat_glu_mmvq_plain(ctx, glu, gate, up, wu, wg, act);
     }

+    // past 5 columns the two unfused q5_K GEMVs are faster than the fused kernel
+    if (wu->type == GGML_TYPE_Q5_K && act->ne[1] > 5) {
+        return false;
+    }
+
     // install the reorder (SoA) layout the fused kernel needs, as the unfused mmvq path would;
     // a no-op once done. after the bail checks so a declined op does not pay for it.
     opt_for_reorder(&ctx, wu, act, up, mul_mat_algo::MMVQ);
diff --git a/ggml/src/ggml-sycl/mmvq.cpp b/ggml/src/ggml-sycl/mmvq.cpp
index ceaacb59e..e8b293a97 100644
--- a/ggml/src/ggml-sycl/mmvq.cpp
+++ b/ggml/src/ggml-sycl/mmvq.cpp
@@ -110,7 +110,8 @@ static void mul_mat_vec_q_reorder(const void * __restrict__ vx, const void * __r

 // With has_fusion, `vgate` is a second weight matrix sharing vx's shape, stride and reorder
 // layout: one pass computes both row dot products and the epilogue writes glu(gate, up).
-template <typename reorder_vec_dot_q_sycl, int ncols_dst, bool has_fusion = false, int rows_per_sg = 1>
+template <typename reorder_vec_dot_q_sycl, int ncols_dst, bool has_fusion = false, int rows_per_sg = 1,
+          bool shared_weights = reorder_vec_dot_shared_weights<reorder_vec_dot_q_sycl::gtype>::value>
 static void mul_mat_vec_q_reorder_ncols(const void * __restrict__ vx, const void * __restrict__ vgate,
                                         const void * __restrict__ vy, float * __restrict__ dst, const int ncols,
                                         const int nrows, const int stride_col_y_bytes, const int stride_col_dst,
@@ -181,7 +182,7 @@ static void mul_mat_vec_q_reorder_ncols(const void * __restrict__ vx, const void
                         }
                     }
                 }
-            } else if constexpr (reorder_vec_dot_shared_weights<reorder_vec_dot_q_sycl::gtype>::value) {
+            } else if constexpr (shared_weights) {
                 const int ibx = row0 * blocks_per_row + i;
                 const auto bx_offset = block_type::get_block_offset(ibx, nblocks);
                 const auto d_offset  = block_type::get_d_offset(nrows, ncols, ibx);
@@ -1945,8 +1946,8 @@ static void reorder_mul_mat_vec_q5_k_q8_1_sycl(const void * vx, const void * vy,
     });
 }

-template <int ncols_dst>
-static void reorder_mul_mat_vec_q5_k_q8_1_sycl_ncols(
+template <int ncols_dst, int rows_per_sg, bool shared_weights>
+static void reorder_mul_mat_vec_q5_k_q8_1_sycl_ncols_impl(
         const void * vx, const void * vy, float * dst,
         const int ncols, const int nrows,
         const int stride_col_y_bytes, const int stride_col_dst,
@@ -1954,20 +1955,35 @@ static void reorder_mul_mat_vec_q5_k_q8_1_sycl_ncols(
     GGML_ASSERT(ncols % QK_K == 0);

     constexpr size_t num_subgroups = WARP_SIZE;
-    const int block_num_y = ceil_div(nrows, GGML_SYCL_MMV_Y * (int) num_subgroups);
+    const int block_num_y = ceil_div(nrows, GGML_SYCL_MMV_Y * (int) num_subgroups * rows_per_sg);
     const sycl::range<3> block_nums(1, 1, block_num_y);
     const sycl::range<3> block_dims(1, GGML_SYCL_MMV_Y, num_subgroups * WARP_SIZE);

     stream->submit([&](sycl::handler & cgh) {
         cgh.parallel_for(sycl::nd_range<3>(block_nums * block_dims, block_dims),
                          [=](sycl::nd_item<3> nd_item) [[sycl::reqd_sub_group_size(WARP_SIZE)]] {
-                             mul_mat_vec_q_reorder_ncols<reorder_vec_dot_q_sycl<GGML_TYPE_Q5_K>, ncols_dst>(
+                             mul_mat_vec_q_reorder_ncols<reorder_vec_dot_q_sycl<GGML_TYPE_Q5_K>, ncols_dst,
+                                                        /*has_fusion=*/ false, rows_per_sg, shared_weights>(
                                  vx, /*vgate=*/ nullptr, vy, dst, ncols, nrows, stride_col_y_bytes, stride_col_dst,
                                  /*glu_op=*/ GGML_GLU_OP_SWIGLU, nd_item);
                          });
     });
 }

+template <int ncols_dst>
+static void reorder_mul_mat_vec_q5_k_q8_1_sycl_ncols(
+        const void * vx, const void * vy, float * dst,
+        const int ncols, const int nrows,
+        const int stride_col_y_bytes, const int stride_col_dst,
+        dpct::queue_ptr stream) {
+    if (ggml_sycl_q5_k_mmvq_reuse(ggml_sycl_get_device())) {
+        constexpr int rows_per_sg = ncols_dst >= 3 ? 2 : 1;
+        reorder_mul_mat_vec_q5_k_q8_1_sycl_ncols_impl<ncols_dst, rows_per_sg, true>(vx, vy, dst, ncols, nrows, stride_col_y_bytes, stride_col_dst, stream);
+    } else {
+        reorder_mul_mat_vec_q5_k_q8_1_sycl_ncols_impl<ncols_dst, 1, false>(vx, vy, dst, ncols, nrows, stride_col_y_bytes, stride_col_dst, stream);
+    }
+}
+
 static void reorder_mul_mat_vec_q5_k_q8_1_sycl_switch_ncols(
         const void * vx, const void * vy, float * dst,
         const int ncols, const int nrows, const int ncols_dst,
@@ -3129,8 +3145,11 @@ static void launch_mul_mat_vec_q_reorder_glu(const void * vx, const void * vgate
                                              const int ncols, const int nrows, const int stride_col_y_bytes,
                                              const int stride_col_dst, const ggml_glu_op glu_op,
                                              dpct::queue_ptr stream) {
+    // q4_K pairs rows for 3..4 columns, q5_K for 3..5
+    constexpr int row_pair_max = reorder_vec_dot_q_sycl::gtype == GGML_TYPE_Q5_K ? 5 : 4;
     constexpr int rows_per_sg =
-        reorder_vec_dot_shared_activations<reorder_vec_dot_q_sycl::gtype>::value && ncols_dst >= 3 && ncols_dst <= 4
+        reorder_vec_dot_shared_activations<reorder_vec_dot_q_sycl::gtype>::value && ncols_dst >= 3 &&
+                ncols_dst <= row_pair_max
             ? 2
             : 1;
     launch_mul_mat_vec_q_reorder_glu_impl<reorder_vec_dot_q_sycl, ncols_dst, rows_per_sg>(vx, vgate, vy, dst, ncols, nrows, stride_col_y_bytes, stride_col_dst, glu_op, stream);
@@ -3321,55 +3340,48 @@ bool ggml_sycl_mul_mat_vec_q_glu_plain(enum ggml_type gate_type, enum ggml_type
     return false;
 }

+template <ggml_type type, int... Ns>
+static bool mul_mat_vec_q_glu_reorder_ncols(enum ggml_glu_op glu_op, const void * vx, const void * vgate,
+                                            const void * vy, float * dst, int ncols, int nrows, int ncols_dst,
+                                            int stride_col_y_bytes, int stride_col_dst, dpct::queue_ptr stream) {
+    using vec_dot = reorder_vec_dot_q_sycl<type>;
+
+    auto launch = [&](auto I) -> bool {
+        constexpr int n = decltype(I)::value;
+        if (ncols_dst != n) {
+            return false;
+        }
+        if constexpr (type == GGML_TYPE_Q4_K && n == 2) {
+            if (nrows >= Q4_K_MMVQ_ROW_PAIR_MIN_NROWS) {
+                launch_mul_mat_vec_q_reorder_glu_impl<vec_dot, 2, 2>(vx, vgate, vy, dst, ncols, nrows, stride_col_y_bytes, stride_col_dst, glu_op, stream);
+                return true;
+            }
+        }
+        launch_mul_mat_vec_q_reorder_glu<vec_dot, n>(vx, vgate, vy, dst, ncols, nrows, stride_col_y_bytes,
+                                                     stride_col_dst, glu_op, stream);
+        return true;
+    };
+
+    // unary fold over launch
+    return (launch(std::integral_constant<int, Ns>{}) || ...);
+}
+
 bool ggml_sycl_mul_mat_vec_q_glu_reorder(enum ggml_type src0_type, enum ggml_glu_op glu_op, const void * vx,
                                          const void * vgate, const void * vy, float * dst, int ncols, int nrows,
                                          int ncols_dst, int stride_col_y_bytes, int stride_col_dst,
                                          dpct::queue_ptr stream) {
-    if (src0_type != GGML_TYPE_Q4_K) {
-        return false;
-    }
     if (glu_op != GGML_GLU_OP_SWIGLU && glu_op != GGML_GLU_OP_GEGLU) {
         return false;
     }

-    using vec_dot = reorder_vec_dot_q_sycl<GGML_TYPE_Q4_K>;
-
-    switch (ncols_dst) {
-        case 1:
-            launch_mul_mat_vec_q_reorder_glu<vec_dot, 1>(vx, vgate, vy, dst, ncols, nrows, stride_col_y_bytes,
-                                                         stride_col_dst, glu_op, stream);
-            return true;
-        case 2:
-            if (nrows >= Q4_K_MMVQ_ROW_PAIR_MIN_NROWS) {
-                launch_mul_mat_vec_q_reorder_glu_impl<vec_dot, 2, 2>(vx, vgate, vy, dst, ncols, nrows, stride_col_y_bytes, stride_col_dst, glu_op, stream);
-            } else {
-                launch_mul_mat_vec_q_reorder_glu_impl<vec_dot, 2, 1>(vx, vgate, vy, dst, ncols, nrows, stride_col_y_bytes, stride_col_dst, glu_op, stream);
-            }
-            return true;
-        case 3:
-            launch_mul_mat_vec_q_reorder_glu<vec_dot, 3>(vx, vgate, vy, dst, ncols, nrows, stride_col_y_bytes,
-                                                         stride_col_dst, glu_op, stream);
-            return true;
-        case 4:
-            launch_mul_mat_vec_q_reorder_glu<vec_dot, 4>(vx, vgate, vy, dst, ncols, nrows, stride_col_y_bytes,
-                                                         stride_col_dst, glu_op, stream);
-            return true;
-        case 5:
-            launch_mul_mat_vec_q_reorder_glu<vec_dot, 5>(vx, vgate, vy, dst, ncols, nrows, stride_col_y_bytes,
-                                                         stride_col_dst, glu_op, stream);
-            return true;
-        case 6:
-            launch_mul_mat_vec_q_reorder_glu<vec_dot, 6>(vx, vgate, vy, dst, ncols, nrows, stride_col_y_bytes,
-                                                         stride_col_dst, glu_op, stream);
-            return true;
-        case 7:
-            launch_mul_mat_vec_q_reorder_glu<vec_dot, 7>(vx, vgate, vy, dst, ncols, nrows, stride_col_y_bytes,
-                                                         stride_col_dst, glu_op, stream);
-            return true;
-        case 8:
-            launch_mul_mat_vec_q_reorder_glu<vec_dot, 8>(vx, vgate, vy, dst, ncols, nrows, stride_col_y_bytes,
-                                                         stride_col_dst, glu_op, stream);
-            return true;
+    switch (src0_type) {
+        case GGML_TYPE_Q4_K:
+            return mul_mat_vec_q_glu_reorder_ncols<GGML_TYPE_Q4_K, 1, 2, 3, 4, 5, 6, 7, 8>(
+                glu_op, vx, vgate, vy, dst, ncols, nrows, ncols_dst, stride_col_y_bytes, stride_col_dst, stream);
+        case GGML_TYPE_Q5_K:
+            // fusion declines q5_K past 5 columns
+            return mul_mat_vec_q_glu_reorder_ncols<GGML_TYPE_Q5_K, 1, 2, 3, 4, 5>(
+                glu_op, vx, vgate, vy, dst, ncols, nrows, ncols_dst, stride_col_y_bytes, stride_col_dst, stream);
         default:
             return false;
     }
diff --git a/ggml/src/ggml-sycl/mmvq.hpp b/ggml/src/ggml-sycl/mmvq.hpp
index 7fb9cf6f8..3a0ef07d6 100644
--- a/ggml/src/ggml-sycl/mmvq.hpp
+++ b/ggml/src/ggml-sycl/mmvq.hpp
@@ -15,6 +15,12 @@

 #include "common.hpp"

+// q5_K multi-column MMVQ shares weights across columns, pairs rows and fuses gate/up in the reorder
+// layout: faster on Xe2 (BMG), so untested archs keep the per-column kernel
+inline bool ggml_sycl_q5_k_mmvq_reuse(int device) {
+    const gpu_arch arch = ggml_sycl_info().devices[device].hw_info.arch;
+    return arch == gpu_arch::intel_gpu_bmg_g21 || arch == gpu_arch::intel_gpu_bmg_g31;
+}

 void ggml_sycl_op_mul_mat_vec_q(
     ggml_backend_sycl_context & ctx,
diff --git a/ggml/src/ggml-sycl/vecdotq.hpp b/ggml/src/ggml-sycl/vecdotq.hpp
index cc6ae6a8d..6ae951525 100644
--- a/ggml/src/ggml-sycl/vecdotq.hpp
+++ b/ggml/src/ggml-sycl/vecdotq.hpp
@@ -362,6 +362,10 @@ template <> struct reorder_vec_dot_shared_weights<GGML_TYPE_Q4_K> {
     static constexpr bool value = true;
 };

+template <> struct reorder_vec_dot_shared_weights<GGML_TYPE_Q5_K> {
+    static constexpr bool value = true;
+};
+
 template <ggml_type T> struct reorder_vec_dot_shared_activations {
     static constexpr bool value = false;
 };
@@ -370,6 +374,10 @@ template <> struct reorder_vec_dot_shared_activations<GGML_TYPE_Q4_K> {
     static constexpr bool value = true;
 };

+template <> struct reorder_vec_dot_shared_activations<GGML_TYPE_Q5_K> {
+    static constexpr bool value = true;
+};
+
 template <> struct reorder_vec_dot_q_sycl<GGML_TYPE_Q4_0> {
     static constexpr ggml_type gtype = GGML_TYPE_Q4_0;

@@ -676,56 +684,72 @@ template <> struct reorder_vec_dot_q_sycl<GGML_TYPE_Q5_K> {
     using q5_k_block  = ggml_sycl_reordered::block_q_t<GGML_TYPE_Q5_K>;
     using q5_k_traits = typename q5_k_block::traits;

-    __dpct_inline__ float operator()(const void * __restrict__ vbq, const std::pair<int, int> ibx_offset,
-                                     const std::pair<int, int> d_offset, const int8_t * q8_1_quant_ptr,
-                                     const sycl::half2 * q8_1_ds, const int & iqs) {
-        const uint8_t *    base           = static_cast<const uint8_t *>(vbq);
-        const uint8_t *    qs             = base + ibx_offset.first;   // low 4 bits
-        const uint8_t *    qh_base        = base + ibx_offset.second;  // high bit
-        const uint8_t *    scs            = base + d_offset.first;
-        const ggml_half2 * dms            = reinterpret_cast<const ggml_half2 *>(base + d_offset.second);
+    struct weights {
+        int        vl[2];
+        int        vh[2];
+        uint16_t   aux[2];
+        ggml_half2 dm;
+    };
+
+    // same activation layout as Q4_K
+    static_assert(QR5_K == QR4_K);
+    using activations = reorder_vec_dot_q_sycl<GGML_TYPE_Q4_K>::activations;
+
+    __dpct_inline__ static weights load(const void * __restrict__ vbq, const std::pair<int, int> ibx_offset,
+                                        const std::pair<int, int> d_offset, const int & iqs) {
+        const uint8_t *    base    = static_cast<const uint8_t *>(vbq);
+        const uint8_t *    qs      = base + ibx_offset.first;   // low 4 bits
+        const uint8_t *    qh_base = base + ibx_offset.second;  // high bit
+        const uint8_t *    scs     = base + d_offset.first;
+        const ggml_half2 * dms     = reinterpret_cast<const ggml_half2 *>(base + d_offset.second);

         const int        bq8_offset = QR5_K * ((iqs / 2) / (QI8_1 / 2));
         const int *      ql_ptr     = (const int *) (qs + 16 * bq8_offset + 4 * ((iqs / 2) % 4));
         const int *      qh_ptr     = (const int *) (qh_base + 4 * ((iqs / 2) % 4));
         const uint16_t * scales     = (const uint16_t *) scs;

-        int   vl[2];
-        int   vh[2];
-        int   u[2 * QR5_K];
-        float d8[QR5_K];
-
-        vl[0] = ql_ptr[0];
-        vl[1] = ql_ptr[4];
+        weights w;
+        w.vl[0] = ql_ptr[0];
+        w.vl[1] = ql_ptr[4];

-        vh[0] = qh_ptr[0] >> bq8_offset;
-        vh[1] = qh_ptr[4] >> bq8_offset;
+        w.vh[0] = qh_ptr[0] >> bq8_offset;
+        w.vh[1] = qh_ptr[4] >> bq8_offset;

-        uint16_t  aux[2];
         const int j = (QR5_K * ((iqs / 2) / (QI8_1 / 2))) / 2;
         if (j < 2) {
-            aux[0] = scales[j + 0] & 0x3f3f;
-            aux[1] = scales[j + 2] & 0x3f3f;
+            w.aux[0] = scales[j + 0] & 0x3f3f;
+            w.aux[1] = scales[j + 2] & 0x3f3f;
         } else {
-            aux[0] = ((scales[j + 2] >> 0) & 0x0f0f) | ((scales[j - 2] & 0xc0c0) >> 2);
-            aux[1] = ((scales[j + 2] >> 4) & 0x0f0f) | ((scales[j - 0] & 0xc0c0) >> 2);
+            w.aux[0] = ((scales[j + 2] >> 0) & 0x0f0f) | ((scales[j - 2] & 0xc0c0) >> 2);
+            w.aux[1] = ((scales[j + 2] >> 4) & 0x0f0f) | ((scales[j - 0] & 0xc0c0) >> 2);
         }

-        const uint8_t * sc = (const uint8_t *) aux;
-        const uint8_t * m  = sc + 2;
+        w.dm = *dms;

-        for (int i = 0; i < QR5_K; ++i) {
-            const int8_t* quant_base_ptr = q8_1_quant_ptr + (bq8_offset + i) * QK8_1;
-            sycl::half2 ds_values = *(q8_1_ds + bq8_offset + i);
+        return w;
+    }

-            d8[i]                   = ds_values[0];
+    __dpct_inline__ static activations load_activations(const int8_t * q8_1_quant_ptr,
+                                                        const sycl::half2 * q8_1_ds, const int & iqs) {
+        return reorder_vec_dot_q_sycl<GGML_TYPE_Q4_K>::load_activations(q8_1_quant_ptr, q8_1_ds, iqs);
+    }

-            const int * q8 = (const int *) quant_base_ptr + ((iqs / 2) % 4);
-            u[2 * i + 0]   = q8[0];
-            u[2 * i + 1]   = q8[4];
-        }
+    __dpct_inline__ static float apply(const weights & w, const activations & a) {
+        const uint8_t * sc = (const uint8_t *) w.aux;
+        const uint8_t * m  = sc + 2;
+
+        return vec_dot_q5_K_q8_1_impl_vmmq(w.vl, w.vh, a.u, sc, m, w.dm, a.d8);
+    }

-        return vec_dot_q5_K_q8_1_impl_vmmq(vl, vh, u, sc, m, *dms, d8);
+    __dpct_inline__ static float dot(const weights & w, const int8_t * q8_1_quant_ptr,
+                                     const sycl::half2 * q8_1_ds, const int & iqs) {
+        return apply(w, load_activations(q8_1_quant_ptr, q8_1_ds, iqs));
+    }
+
+    __dpct_inline__ float operator()(const void * __restrict__ vbq, const std::pair<int, int> ibx_offset,
+                                     const std::pair<int, int> d_offset, const int8_t * q8_1_quant_ptr,
+                                     const sycl::half2 * q8_1_ds, const int & iqs) {
+        return dot(load(vbq, ibx_offset, d_offset, iqs), q8_1_quant_ptr, q8_1_ds, iqs);
     }
 };