Commit 982937a33 for llama.cpp

commit 982937a3337f7e97ef08fd5603f4157575ece7e1
Author: Rohanjames1997 <rohan.james4@gmail.com>
Date:   Fri Sep 11 13:19:37 2026 -0500

    tests: extend test-quantize-fns to test nrc=2 (i8mm) kernels (#16234)

    * Test for nrc=2 as well | i8mm kernels

    * Trigger only on supported HW

    * Remove trailing whitespace

    * Address review comment

    * test: properly prepare nrc=2 inputs with independent data per row

    * tests : make nrc=2 dot product inputs distinct

    Assisted-by: Kiro

    * tests : use non-trivial strides in nrc=2 dot product test

    * tests : fail nrc=2 dot product test on non-finite errors

diff --git a/tests/test-quantize-fns.cpp b/tests/test-quantize-fns.cpp
index 9510ac14c..570fca89a 100644
--- a/tests/test-quantize-fns.cpp
+++ b/tests/test-quantize-fns.cpp
@@ -5,6 +5,8 @@

 #undef NDEBUG
 #include <assert.h>
+#include <algorithm>
+#include <cmath>
 #include <math.h>
 #include <stdio.h>
 #include <string>
@@ -32,9 +34,9 @@ static const char* RESULT_STR[] = {"ok", "FAILED"};


 // Generate synthetic data
-static void generate_data(float offset, size_t n, float * dst) {
+static void generate_data(float offset, size_t n, float * dst, float amplitude = 2.0f) {
     for (size_t i = 0; i < n; i++) {
-        dst[i] = 0.1 + 2*cosf(i + offset);
+        dst[i] = 0.1 + amplitude*cosf(i + offset);
     }
 }

@@ -83,23 +85,50 @@ static float dot_product(const float * a1, const float * a2, size_t test_size) {
 }

 // Total dot product error
-static float dot_product_error(const ggml_type_traits * qfns, const ggml_type_traits_cpu * qfns_cpu, size_t test_size, const float * test_data1, const float * test_data2) {
-    GGML_UNUSED(qfns);
-
-    std::vector<uint8_t> tmp_q1(2*test_size);
-    std::vector<uint8_t> tmp_q2(2*test_size);
-
+static float dot_product_error(const ggml_type_traits_cpu * qfns_cpu, ggml_type src0_type, size_t test_size,
+                               const float * test_data1, const float * test_data2,
+                               const float * test_data3, const float * test_data4,
+                               const int nrc) {
     const auto * vdot = ggml_get_type_traits_cpu(qfns_cpu->vec_dot_type);
+    const size_t pad  = 64;
+    const size_t bx   = ggml_row_size(src0_type, test_size) + pad;
+    const size_t by   = ggml_row_size(qfns_cpu->vec_dot_type, test_size) + pad;
+
+    std::vector<uint8_t> tmp_q1(bx * nrc);
+    std::vector<uint8_t> tmp_q2(by * nrc);

     qfns_cpu->from_float(test_data1, tmp_q1.data(), test_size);
     vdot->from_float(test_data2, tmp_q2.data(), test_size);

-    float result = INFINITY;
-    qfns_cpu->vec_dot(test_size, &result, 0, tmp_q1.data(), 0, tmp_q2.data(), 0, 1);
+    if (nrc == 1) {
+        float result = INFINITY;
+        qfns_cpu->vec_dot(test_size, &result, 0, tmp_q1.data(), 0, tmp_q2.data(), 0, 1);
+
+        const float dot_ref = dot_product(test_data1, test_data2, test_size);
+        return fabsf(result - dot_ref) / test_size;
+    }
+
+    // nrc == 2: kernel computes a 2x2 dot product matrix
+    // Output layout: s[0]=dot(vx0,vy0), s[1]=dot(vx1,vy0), s[bs]=dot(vx0,vy1), s[bs+1]=dot(vx1,vy1)
+    // row and output strides are padded, same as in the mul_mat path
+    qfns_cpu->from_float(test_data3, tmp_q1.data() + bx, test_size);
+    vdot->from_float(test_data4, tmp_q2.data() + by, test_size);
+
+    const size_t bs = 16;
+    std::vector<float> result(bs + 2, INFINITY);
+    qfns_cpu->vec_dot(test_size, result.data(), bs, tmp_q1.data(), bx, tmp_q2.data(), by, 2);
+
+    const float ref00 = dot_product(test_data1, test_data2, test_size);
+    const float ref10 = dot_product(test_data3, test_data2, test_size);
+    const float ref01 = dot_product(test_data1, test_data4, test_size);
+    const float ref11 = dot_product(test_data3, test_data4, test_size);

-    const float dot_ref = dot_product(test_data1, test_data2, test_size);
+    const auto err = [test_size](float val, float ref) {
+        const float e = fabsf(val - ref) / test_size;
+        return std::isfinite(e) ? e : INFINITY;
+    };

-    return fabsf(result - dot_ref) / test_size;
+    return std::max({err(result[0], ref00), err(result[1], ref10), err(result[bs], ref01), err(result[bs + 1], ref11)});
 }

 static int test_vec_dot_f32(bool verbose) {
@@ -133,9 +162,13 @@ static int test_vec_dot_q(bool verbose) {

     std::vector<float> test_data(test_size);
     std::vector<float> test_data2(test_size);
+    std::vector<float> test_data3(test_size);
+    std::vector<float> test_data4(test_size);

     generate_data(0.0, test_data.size(), test_data.data());
     generate_data(1.0, test_data2.size(), test_data2.data());
+    generate_data(3.0, test_data3.size(), test_data3.data(), 1.0f);
+    generate_data(4.0, test_data4.size(), test_data4.data(), 1.5f);

     for (int i = 0; i < GGML_TYPE_COUNT; i++) {
         ggml_type type = (ggml_type) i;
@@ -178,7 +211,7 @@ static int test_vec_dot_q(bool verbose) {
                 printf("%5s reference implementation error: %s (%f)\n", ggml_type_name(type), RESULT_STR[failed], reference_error);
             }

-            const float vec_dot_error = dot_product_error(qfns, qfns_cpu, test_size, test_data.data(), test_data2.data());
+            const float vec_dot_error = dot_product_error(qfns_cpu, type, test_size, test_data.data(), test_data2.data(), nullptr, nullptr, 1);
             const float max_allowed_error = type == GGML_TYPE_Q2_K || type == GGML_TYPE_IQ2_XS || type == GGML_TYPE_IQ2_XXS ||
                 type == GGML_TYPE_IQ3_XXS || type == GGML_TYPE_IQ3_S || type == GGML_TYPE_IQ2_S
                 ? MAX_DOT_PRODUCT_ERROR_LOWBIT
@@ -194,6 +227,16 @@ static int test_vec_dot_q(bool verbose) {
             if (failed || verbose) {
                 printf("%5s dot product error:              %s (%f)\n", ggml_type_name(type), RESULT_STR[failed], vec_dot_error);
             }
+
+            // Test nrc=2 path for types that support it
+            if (qfns_cpu->nrows == 2) {
+                const float vec_dot_error_nrc2 = dot_product_error(qfns_cpu, type, test_size, test_data.data(), test_data2.data(), test_data3.data(), test_data4.data(), 2);
+                failed = !(vec_dot_error_nrc2 < max_allowed_error);
+                num_failed += failed;
+                if (failed || verbose) {
+                    printf("%5s dot product error (nrc=2):    %s (%f)\n", ggml_type_name(type), RESULT_STR[failed], vec_dot_error_nrc2);
+                }
+            }
         }
     }