Commit 2923cf286 for llama.cpp
commit 2923cf2862ad0afa159444cf07fec7600d755fe1
Author: Yash Raj Pandey <55940078+devYRPauli@users.noreply.github.com>
Date: Fri Oct 2 10:37:35 2026 -0400
ggml-quants : avoid invalid rounding in qkx3 scale search (#29817)
* ggml-quants : avoid invalid rounding in qkx3 scale search
The imatrix scale search can produce an infinite, NaN, or otherwise out-of-range value when the fitted minimum collapses to the maximum or makes the range extremely small. That value is then passed to nearest_int and can trip its assertion in Debug builds.
Clamp the quantization level to [0, nmax] before rounding so valid in-range values behave the same as before while invalid scale-search results no longer reach nearest_int.
Add regression coverage for degenerate imatrix groups across q2_K, q4_K, q5_K, q4_1, and q5_1.
Fixes #29804.
Assisted-by: Claude Opus 5.5
* tests: print degenerate imatrix quant types
diff --git a/ggml/src/ggml-quants.c b/ggml/src/ggml-quants.c
index 55db802c0..7750a72ce 100644
--- a/ggml/src/ggml-quants.c
+++ b/ggml/src/ggml-quants.c
@@ -1036,8 +1036,9 @@ static float make_qkx3_quants(int n, int nmax, const float * GGML_RESTRICT x, co
iscale = (rmin + rdelta*is + nmax)/(max - min);
float sum_l = 0, sum_l2 = 0, sum_xl = 0;
for (int i = 0; i < n; ++i) {
- int l = nearest_int(iscale*(x[i] - min));
- l = MAX(0, MIN(nmax, l));
+ // min is the best fit so far and can be at or near max, so v can be inf, nan or out of range for nearest_int
+ const float v = iscale*(x[i] - min);
+ const int l = v > 0 ? nearest_int(MIN(v, nmax)) : 0;
Laux[i] = l;
float w = weights ? weights[i] : x[i]*x[i];
sum_l += w*l;
diff --git a/tests/test-quantize-fns.cpp b/tests/test-quantize-fns.cpp
index 570fca89a..1930a09b2 100644
--- a/tests/test-quantize-fns.cpp
+++ b/tests/test-quantize-fns.cpp
@@ -243,6 +243,42 @@ static int test_vec_dot_q(bool verbose) {
return num_failed;
}
+// In every group, all values with importance are equal, and the max (0) has none.
+// The scale search in make_qkx3_quants then fits min == max and passes inf/nan to nearest_int (#29804).
+static int test_quantize_imatrix_degenerate(bool verbose) {
+ const int64_t n = 256;
+ std::vector<float> x(n);
+ std::vector<float> imatrix(n);
+ std::vector<float> out(n);
+ int num_failed = 0;
+
+ printf("Testing degenerate imatrix:\n");
+ for (ggml_type type : {GGML_TYPE_Q2_K, GGML_TYPE_Q4_K, GGML_TYPE_Q5_K, GGML_TYPE_Q4_1, GGML_TYPE_Q5_1}) {
+ const int64_t group = type == GGML_TYPE_Q2_K ? 16 : 32;
+ for (int64_t i = 0; i < n; ++i) {
+ const int64_t g = i / group;
+ const int64_t p = i % group;
+ const bool important = p % 3 == 1;
+ x[i] = important ? -0.02f*(g + 1) : (p % 2 ? -1.0f : 0.0f);
+ imatrix[i] = important ? 1.0f : 0.0f;
+ }
+ printf(" - %s\n", ggml_type_name(type));
+
+ std::vector<uint8_t> q(ggml_row_size(type, n));
+ ggml_quantize_init(type);
+ ggml_quantize_chunk(type, x.data(), q.data(), 0, 1, n, imatrix.data());
+ ggml_get_type_traits(type)->to_float(q.data(), out.data(), n);
+
+ const bool failed = !std::all_of(out.begin(), out.end(), [](float v) { return std::isfinite(v); });
+ num_failed += failed;
+ if (failed || verbose) {
+ printf("%5s imatrix degenerate groups: %s\n", ggml_type_name(type), RESULT_STR[failed]);
+ }
+ }
+
+ return num_failed;
+}
+
int main(int argc, char * argv[]) {
bool verbose = false;
@@ -264,6 +300,7 @@ int main(int argc, char * argv[]) {
num_failed += test_vec_dot_f32(verbose);
num_failed += test_vec_dot_q(verbose);
+ num_failed += test_quantize_imatrix_degenerate(verbose);
if (num_failed || verbose) {
printf("%d tests failed\n", num_failed);