Commit 83dd71f86 for llama.cpp

commit 83dd71f869ac753e0a14058f9e81736c196f22c8
Author: Matt Corallo <649246+TheBlueMatt@users.noreply.github.com>
Date:   Tue Sep 29 17:39:36 2026 +0000

    vulkan : Load F32 A matrix 2 at a time when its 2-aligned (#29254)

    It turns out Intel doesn't particularly like loading F32s one at a
    time and we already have the _2aliagned load logic in mul_mat_vec,
    so here we use it.

    While we do already check all the requirements to load elements 4
    at a time across [B]F16 and F32, it turns out [B]F16 loading 4 at a
    time is sometimes slower on very specific shapes on Intel BMG.
    Loading 4 at a time is a bit faster on F32, but its not material
    and I assume might be slower on other platforms.

    Note that we also need to validate `a_offset` is 2-aligned in
    `mul_mat_vec.comp`, which was missing in the original 2-way-load
    patch.

    Some selected speedups from `test-backend-ops perf` on a B60.

      MUL_MAT(type_a=f32,type_b=f32,m=4096,n=1,k=14336,bs=[1,1],nr=[1,1],per=[0,1,2,3],k_v=0,o=1,src_overlap=0):                    1704 runs -   767.17 us/run - 117.44 MFLOP/run - 153.08 GFLOPS
      MUL_MAT(type_a=f32,type_b=f32,m=4096,n=1,k=14336,bs=[1,1],nr=[1,1],per=[0,1,2,3],k_v=0,o=1,src_overlap=0):                    2556 runs -   529.81 us/run - 117.44 MFLOP/run - 221.66 GFLOPS

      MUL_MAT(type_a=f32,type_b=f32,m=4096,n=2,k=14336,bs=[1,1],nr=[1,1],per=[0,1,2,3],k_v=0,o=1,src_overlap=0):                    1704 runs -   727.13 us/run - 234.88 MFLOP/run - 323.03 GFLOPS
      MUL_MAT(type_a=f32,type_b=f32,m=4096,n=2,k=14336,bs=[1,1],nr=[1,1],per=[0,1,2,3],k_v=0,o=1,src_overlap=0):                    2130 runs -   528.84 us/run - 234.88 MFLOP/run - 444.15 GFLOPS

      MUL_MAT(type_a=f32,type_b=f32,m=4096,n=3,k=14336,bs=[1,1],nr=[1,1],per=[0,1,2,3],k_v=0,o=1,src_overlap=0):                    1704 runs -   702.19 us/run - 352.32 MFLOP/run - 501.74 GFLOPS
      MUL_MAT(type_a=f32,type_b=f32,m=4096,n=3,k=14336,bs=[1,1],nr=[1,1],per=[0,1,2,3],k_v=0,o=1,src_overlap=0):                    1988 runs -   532.14 us/run - 352.32 MFLOP/run - 662.08 GFLOPS

      MUL_MAT(type_a=f32,type_b=f32,m=4096,n=4,k=14336,bs=[1,1],nr=[1,1],per=[0,1,2,3],k_v=0,o=1,src_overlap=0):                    1278 runs -   919.50 us/run - 469.76 MFLOP/run - 510.89 GFLOPS
      MUL_MAT(type_a=f32,type_b=f32,m=4096,n=4,k=14336,bs=[1,1],nr=[1,1],per=[0,1,2,3],k_v=0,o=1,src_overlap=0):                    1917 runs -   543.69 us/run - 469.76 MFLOP/run - 864.03 GFLOPS

      MUL_MAT(type_a=f32,type_b=f32,m=4096,n=5,k=14336,bs=[1,1],nr=[1,1],per=[0,1,2,3],k_v=0,o=1,src_overlap=0):                    1197 runs -   892.12 us/run - 587.20 MFLOP/run - 658.21 GFLOPS
      MUL_MAT(type_a=f32,type_b=f32,m=4096,n=5,k=14336,bs=[1,1],nr=[1,1],per=[0,1,2,3],k_v=0,o=1,src_overlap=0):                    1881 runs -   575.17 us/run - 587.20 MFLOP/run -   1.02 TFLOPS

      MUL_MAT(type_a=f32,type_b=f32,m=4096,n=8,k=14336,bs=[1,1],nr=[1,1],per=[0,1,2,3],k_v=0,o=1,src_overlap=0):                    1498 runs -   716.40 us/run - 939.52 MFLOP/run -   1.31 TFLOPS
      MUL_MAT(type_a=f32,type_b=f32,m=4096,n=8,k=14336,bs=[1,1],nr=[1,1],per=[0,1,2,3],k_v=0,o=1,src_overlap=0):                    1819 runs -   576.36 us/run - 939.52 MFLOP/run -   1.63 TFLOPS

      MUL_MAT(type_a=f32,type_b=f32,m=4096,n=512,k=14336,bs=[1,1],nr=[1,1],per=[0,1,2,3],k_v=0,o=1,src_overlap=0):                   134 runs -  7467.09 us/run -  60.13 GFLOP/run -   8.05 TFLOPS
      MUL_MAT(type_a=f32,type_b=f32,m=4096,n=512,k=14336,bs=[1,1],nr=[1,1],per=[0,1,2,3],k_v=0,o=1,src_overlap=0):                   134 runs -  7478.12 us/run -  60.13 GFLOP/run -   8.04 TFLOPS

diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/dequant_funcs.glsl b/ggml/src/ggml-vulkan/vulkan-shaders/dequant_funcs.glsl
index 9df66cb44..911ceac22 100644
--- a/ggml/src/ggml-vulkan/vulkan-shaders/dequant_funcs.glsl
+++ b/ggml/src/ggml-vulkan/vulkan-shaders/dequant_funcs.glsl
@@ -16,8 +16,9 @@ vec4 dequantize4(uint ib, uint iqs, uint a_offset) {
                 data_a[a_offset + ib + 2], data_a[a_offset + ib + 3]);
 }
 vec4 dequantize4_2aligned(uint ib, uint iqs, uint a_offset) {
-    return vec4(data_a[a_offset + ib    ], data_a[a_offset + ib + 1],
-                data_a[a_offset + ib + 2], data_a[a_offset + ib + 3]);
+    const vec2 a = data_a_packed64[(a_offset + ib)/2];
+    const vec2 b = data_a_packed64[(a_offset + ib)/2 + 1];
+    return vec4(a, b);
 }

 #endif
diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vec.comp b/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vec.comp
index 5a9d0e778..34be1f72b 100644
--- a/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vec.comp
+++ b/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vec.comp
@@ -143,9 +143,9 @@ void compute_outputs(const uint32_t first_row, const uint32_t num_rows) {

     get_offsets(a_offset, b_offset, d_offset);
     const bool is_aligned_nonquant =
-        p.batch_stride_b % 4 == 0 && b_offset % 4 == 0 &&
-        p.ncols % 4 == 0 && BLOCK_SIZE % 4 == 0 &&
-        K_PER_ITER == 4;
+        p.batch_stride_b % 4 == 0 && p.ncols % 4 == 0 &&
+        a_offset % 4 == 0 && b_offset % 4 == 0 &&
+        BLOCK_SIZE % 4 == 0 && K_PER_ITER == 4;

     y_offset = QUANT_R == 1 ? 1 : QUANT_K/2;

diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vec_iface.glsl b/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vec_iface.glsl
index e8d053cdd..a2b7cef64 100644
--- a/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vec_iface.glsl
+++ b/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vec_iface.glsl
@@ -15,6 +15,9 @@ layout (binding = 0) readonly buffer A_PACKED16 {A_TYPE_PACKED16 data_a_packed16
 #if defined(A_TYPE_PACKED32)
 layout (binding = 0) readonly buffer A_PACKED32 {A_TYPE_PACKED32 data_a_packed32[];};
 #endif
+#if defined(A_TYPE_PACKED64)
+layout (binding = 0) readonly buffer A_PACKED64 {A_TYPE_PACKED64 data_a_packed64[];};
+#endif

 layout (binding = 1) readonly buffer B {B_TYPE data_b[];};
 #ifdef B_TYPEV2
diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/types.glsl b/ggml/src/ggml-vulkan/vulkan-shaders/types.glsl
index 7d62e92e3..ec0a80c7e 100644
--- a/ggml/src/ggml-vulkan/vulkan-shaders/types.glsl
+++ b/ggml/src/ggml-vulkan/vulkan-shaders/types.glsl
@@ -23,6 +23,7 @@
 #else
 #define A_TYPE float
 #endif
+#define A_TYPE_PACKED64 vec2
 #endif

 #if defined(DATA_A_F16)