Commit f1b6fbf35 for llama.cpp

commit f1b6fbf35cfa010b0a8d6301fdfccbb7f41bd903
Author: Aaron Teo <aaron.teo1@ibm.com>
Date:   Thu Sep 10 14:50:28 2026 +0800

    ggml-cpu(s390x): add Q1_0 vector intrinsic support (#28606)

    * ggml-cpu: add `ggml_vec_dot_q1_0_q8_0` support

    Signed-off-by: Aaron Teo <aaron.teo1@ibm.com>

    * ggml-cpu: clean up variable naming for understanding

    Signed-off-by: Aaron Teo <aaron.teo1@ibm.com>

    * docs: update support for Q1_0

    Signed-off-by: Aaron Teo <aaron.teo1@ibm.com>

    ---------

    Signed-off-by: Aaron Teo <aaron.teo1@ibm.com>

diff --git a/docs/build-s390x.md b/docs/build-s390x.md
index 4568d5010..005dd2983 100644
--- a/docs/build-s390x.md
+++ b/docs/build-s390x.md
@@ -243,6 +243,7 @@ IBM VXE/VXE2 SIMD acceleration depends on the BLAS implementation. It is strongl
 | FP32       | ✅           | ✅    | ❓     |
 | FP16       | ✅           | ✅    | ❓     |
 | BF16       | ✅           | ✅    | ❓     |
+| Q1_0       | ✅           | ❓    | ❓     |
 | Q4_0       | ✅           | ❓    | ❓     |
 | Q4_1       | ✅           | ❓    | ❓     |
 | MXFP4      | ✅           | ❓    | ❓     |
@@ -272,4 +273,4 @@ IBM VXE/VXE2 SIMD acceleration depends on the BLAS implementation. It is strongl
 -   🚫 - acceleration unavailable, will still run using scalar implementation
 -   ❓ - acceleration unknown, please contribute if you can test it yourself

-Last Updated by **Aaron Teo (aaron.teo1@ibm.com)** on Feb 15, 2026.
+Last Updated by **Aaron Teo (aaron.teo1@ibm.com)** on Sep 8, 2026.
diff --git a/ggml/src/ggml-cpu/arch-fallback.h b/ggml/src/ggml-cpu/arch-fallback.h
index 152e0bac9..98ef5e140 100644
--- a/ggml/src/ggml-cpu/arch-fallback.h
+++ b/ggml/src/ggml-cpu/arch-fallback.h
@@ -247,7 +247,6 @@
 // quants.c
 #define quantize_row_q8_K_generic quantize_row_q8_K
 #define ggml_vec_dot_nvfp4_q8_0_generic ggml_vec_dot_nvfp4_q8_0
-#define ggml_vec_dot_q1_0_q8_0_generic ggml_vec_dot_q1_0_q8_0
 #define ggml_vec_dot_q2_0_q8_0_generic ggml_vec_dot_q2_0_q8_0
 #define ggml_vec_dot_tq1_0_q8_K_generic ggml_vec_dot_tq1_0_q8_K
 #define ggml_vec_dot_tq2_0_q8_K_generic ggml_vec_dot_tq2_0_q8_K
diff --git a/ggml/src/ggml-cpu/arch/s390/quants.c b/ggml/src/ggml-cpu/arch/s390/quants.c
index d3436c24b..70f2882d8 100644
--- a/ggml/src/ggml-cpu/arch/s390/quants.c
+++ b/ggml/src/ggml-cpu/arch/s390/quants.c
@@ -146,6 +146,74 @@ void quantize_row_q8_1(const float * GGML_RESTRICT x, void * GGML_RESTRICT vy, i

 //===================================== Dot products =================================

+void ggml_vec_dot_q1_0_q8_0(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, size_t bx, const void * GGML_RESTRICT vy, size_t by, int nrc) {
+    const int qk = QK1_0;  // 128
+    const int nb = n / qk;
+
+    assert(n % qk == 0);
+    assert(nrc == 1);
+    UNUSED(nrc);
+    UNUSED(bx);
+    UNUSED(by);
+    UNUSED(bs);
+
+    const block_q1_0 * GGML_RESTRICT x = vx;
+    const block_q8_0 * GGML_RESTRICT y = vy;
+
+#if defined(__VXE__) || defined(__VXE2__)
+    float32x4_t v_sumf = vec_splats(0.0f);
+
+    const uint8x16_t v_zero = vec_splats((uint8_t)0x00);  // zero
+    const uint8x16_t v_bias = vec_splats((uint8_t)0x80);  // bias from signed to unsigned
+                                                          // v ^ 0x80 == v + 128
+
+    const uint8x16_t v_idx = (const uint8x16_t){ 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 1, 1, 1, 1, 1, 1 };
+    const uint8x16_t v_bit = (const uint8x16_t){ 1, 2, 4, 8, 16, 32, 64, 128, 1, 2, 4, 8, 16, 32, 64, 128 };
+
+    for (int i = 0; i < nb; ++i) {
+        const uint8x16_t  v_x  = vec_xl(0, (const uint8_t *)x[i].qs);
+        const float32x4_t v_xd = vec_splats(GGML_CPU_FP16_TO_FP32(x[i].d));
+
+        for (int k = 0; k < 4; ++k) {
+            // sub-block k holds elements 32k .. 32k+31
+            const block_q8_0 * GGML_RESTRICT yb = &y[i*4 + k];
+            const float32x4_t v_yd = vec_splats(GGML_CPU_FP16_TO_FP32(yb->d));
+
+            const uint8x16_t v_xrl = vec_perm(v_x, v_x, vec_add(v_idx, vec_splats((uint8_t)(k*4 + 0))));
+            const uint8x16_t v_xrh = vec_perm(v_x, v_x, vec_add(v_idx, vec_splats((uint8_t)(k*4 + 2))));
+
+            // isolate each lane's bit, then set all ones where that bit is clear, the -d case
+            const int8x16_t v_ml = (int8x16_t)vec_cmpeq(vec_and(v_xrl, v_bit), v_zero);
+            const int8x16_t v_mh = (int8x16_t)vec_cmpeq(vec_and(v_xrh, v_bit), v_zero);
+
+            const int8x16_t v_yl = vec_xl(0,       (const int8_t *)yb->qs);
+            const int8x16_t v_yh = vec_xl(QK8_0/2, (const int8_t *)yb->qs);
+
+            // weights are only +1 or -1, so negate y
+            const int8x16_t v_ysl = vec_sub(vec_xor(v_yl, v_ml), v_ml);
+            const int8x16_t v_ysh = vec_sub(vec_xor(v_yh, v_mh), v_mh);
+
+            // bias to unsigned, then vec_sum4 adds each group of 4 bytes into one word
+            const uint32x4_t v_p = vec_add(vec_sum4(vec_xor((uint8x16_t)v_ysl, v_bias), v_zero),
+                                           vec_sum4(vec_xor((uint8x16_t)v_ysh, v_bias), v_zero));
+
+            // each word summed 8 biased bytes, so take back 8 * 128
+            const int32x4_t v_xy = vec_sub((int32x4_t)v_p, vec_splats((int32_t)1024));
+
+            // apply both block scales and add into the running total
+            v_sumf = vec_madd(vec_float(v_xy), vec_mul(v_xd, v_yd), v_sumf);
+        }
+    }
+
+    *s = vec_hsum_f32x4(v_sumf);
+#else
+    UNUSED(nb);
+    UNUSED(x);
+    UNUSED(y);
+    ggml_vec_dot_q1_0_q8_0_generic(n, s, bs, vx, bx, vy, by, nrc);
+#endif
+}
+
 void ggml_vec_dot_q4_0_q8_0(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, size_t bx, const void * GGML_RESTRICT vy, size_t by, int nrc) {
     const int qk = QK8_0;
     const int nb = n / qk;