Commit 982937a33 for llama.cpp
commit 982937a3337f7e97ef08fd5603f4157575ece7e1
Author: Rohanjames1997 <rohan.james4@gmail.com>
Date: Fri Sep 11 13:19:37 2026 -0500
tests: extend test-quantize-fns to test nrc=2 (i8mm) kernels (#16234)
* Test for nrc=2 as well | i8mm kernels
* Trigger only on supported HW
* Remove trailing whitespace
* Address review comment
* test: properly prepare nrc=2 inputs with independent data per row
* tests : make nrc=2 dot product inputs distinct
Assisted-by: Kiro
* tests : use non-trivial strides in nrc=2 dot product test
* tests : fail nrc=2 dot product test on non-finite errors
diff --git a/tests/test-quantize-fns.cpp b/tests/test-quantize-fns.cpp
index 9510ac14c..570fca89a 100644
--- a/tests/test-quantize-fns.cpp
+++ b/tests/test-quantize-fns.cpp
@@ -5,6 +5,8 @@
#undef NDEBUG
#include <assert.h>
+#include <algorithm>
+#include <cmath>
#include <math.h>
#include <stdio.h>
#include <string>
@@ -32,9 +34,9 @@ static const char* RESULT_STR[] = {"ok", "FAILED"};
// Generate synthetic data
-static void generate_data(float offset, size_t n, float * dst) {
+static void generate_data(float offset, size_t n, float * dst, float amplitude = 2.0f) {
for (size_t i = 0; i < n; i++) {
- dst[i] = 0.1 + 2*cosf(i + offset);
+ dst[i] = 0.1 + amplitude*cosf(i + offset);
}
}
@@ -83,23 +85,50 @@ static float dot_product(const float * a1, const float * a2, size_t test_size) {
}
// Total dot product error
-static float dot_product_error(const ggml_type_traits * qfns, const ggml_type_traits_cpu * qfns_cpu, size_t test_size, const float * test_data1, const float * test_data2) {
- GGML_UNUSED(qfns);
-
- std::vector<uint8_t> tmp_q1(2*test_size);
- std::vector<uint8_t> tmp_q2(2*test_size);
-
+static float dot_product_error(const ggml_type_traits_cpu * qfns_cpu, ggml_type src0_type, size_t test_size,
+ const float * test_data1, const float * test_data2,
+ const float * test_data3, const float * test_data4,
+ const int nrc) {
const auto * vdot = ggml_get_type_traits_cpu(qfns_cpu->vec_dot_type);
+ const size_t pad = 64;
+ const size_t bx = ggml_row_size(src0_type, test_size) + pad;
+ const size_t by = ggml_row_size(qfns_cpu->vec_dot_type, test_size) + pad;
+
+ std::vector<uint8_t> tmp_q1(bx * nrc);
+ std::vector<uint8_t> tmp_q2(by * nrc);
qfns_cpu->from_float(test_data1, tmp_q1.data(), test_size);
vdot->from_float(test_data2, tmp_q2.data(), test_size);
- float result = INFINITY;
- qfns_cpu->vec_dot(test_size, &result, 0, tmp_q1.data(), 0, tmp_q2.data(), 0, 1);
+ if (nrc == 1) {
+ float result = INFINITY;
+ qfns_cpu->vec_dot(test_size, &result, 0, tmp_q1.data(), 0, tmp_q2.data(), 0, 1);
+
+ const float dot_ref = dot_product(test_data1, test_data2, test_size);
+ return fabsf(result - dot_ref) / test_size;
+ }
+
+ // nrc == 2: kernel computes a 2x2 dot product matrix
+ // Output layout: s[0]=dot(vx0,vy0), s[1]=dot(vx1,vy0), s[bs]=dot(vx0,vy1), s[bs+1]=dot(vx1,vy1)
+ // row and output strides are padded, same as in the mul_mat path
+ qfns_cpu->from_float(test_data3, tmp_q1.data() + bx, test_size);
+ vdot->from_float(test_data4, tmp_q2.data() + by, test_size);
+
+ const size_t bs = 16;
+ std::vector<float> result(bs + 2, INFINITY);
+ qfns_cpu->vec_dot(test_size, result.data(), bs, tmp_q1.data(), bx, tmp_q2.data(), by, 2);
+
+ const float ref00 = dot_product(test_data1, test_data2, test_size);
+ const float ref10 = dot_product(test_data3, test_data2, test_size);
+ const float ref01 = dot_product(test_data1, test_data4, test_size);
+ const float ref11 = dot_product(test_data3, test_data4, test_size);
- const float dot_ref = dot_product(test_data1, test_data2, test_size);
+ const auto err = [test_size](float val, float ref) {
+ const float e = fabsf(val - ref) / test_size;
+ return std::isfinite(e) ? e : INFINITY;
+ };
- return fabsf(result - dot_ref) / test_size;
+ return std::max({err(result[0], ref00), err(result[1], ref10), err(result[bs], ref01), err(result[bs + 1], ref11)});
}
static int test_vec_dot_f32(bool verbose) {
@@ -133,9 +162,13 @@ static int test_vec_dot_q(bool verbose) {
std::vector<float> test_data(test_size);
std::vector<float> test_data2(test_size);
+ std::vector<float> test_data3(test_size);
+ std::vector<float> test_data4(test_size);
generate_data(0.0, test_data.size(), test_data.data());
generate_data(1.0, test_data2.size(), test_data2.data());
+ generate_data(3.0, test_data3.size(), test_data3.data(), 1.0f);
+ generate_data(4.0, test_data4.size(), test_data4.data(), 1.5f);
for (int i = 0; i < GGML_TYPE_COUNT; i++) {
ggml_type type = (ggml_type) i;
@@ -178,7 +211,7 @@ static int test_vec_dot_q(bool verbose) {
printf("%5s reference implementation error: %s (%f)\n", ggml_type_name(type), RESULT_STR[failed], reference_error);
}
- const float vec_dot_error = dot_product_error(qfns, qfns_cpu, test_size, test_data.data(), test_data2.data());
+ const float vec_dot_error = dot_product_error(qfns_cpu, type, test_size, test_data.data(), test_data2.data(), nullptr, nullptr, 1);
const float max_allowed_error = type == GGML_TYPE_Q2_K || type == GGML_TYPE_IQ2_XS || type == GGML_TYPE_IQ2_XXS ||
type == GGML_TYPE_IQ3_XXS || type == GGML_TYPE_IQ3_S || type == GGML_TYPE_IQ2_S
? MAX_DOT_PRODUCT_ERROR_LOWBIT
@@ -194,6 +227,16 @@ static int test_vec_dot_q(bool verbose) {
if (failed || verbose) {
printf("%5s dot product error: %s (%f)\n", ggml_type_name(type), RESULT_STR[failed], vec_dot_error);
}
+
+ // Test nrc=2 path for types that support it
+ if (qfns_cpu->nrows == 2) {
+ const float vec_dot_error_nrc2 = dot_product_error(qfns_cpu, type, test_size, test_data.data(), test_data2.data(), test_data3.data(), test_data4.data(), 2);
+ failed = !(vec_dot_error_nrc2 < max_allowed_error);
+ num_failed += failed;
+ if (failed || verbose) {
+ printf("%5s dot product error (nrc=2): %s (%f)\n", ggml_type_name(type), RESULT_STR[failed], vec_dot_error_nrc2);
+ }
+ }
}
}