From 6b704713cc6e23056770b0769c8e1399f4c5c41c Mon Sep 17 00:00:00 2001 From: Yash Raj Pandey <55940078+devYRPauli@users.noreply.github.com> Date: Fri, 2 Oct 2026 10:37:35 -0400 Subject: [PATCH] ggml-quants : avoid invalid rounding in qkx3 scale search (llama/29817) * ggml-quants : avoid invalid rounding in qkx3 scale search The imatrix scale search can produce an infinite, NaN, or otherwise out-of-range value when the fitted minimum collapses to the maximum or makes the range extremely small. That value is then passed to nearest_int and can trip its assertion in Debug builds. Clamp the quantization level to [0, nmax] before rounding so valid in-range values behave the same as before while invalid scale-search results no longer reach nearest_int. Add regression coverage for degenerate imatrix groups across q2_K, q4_K, q5_K, q4_1, and q5_1. Fixes #29804. Assisted-by: Claude Opus 5.5 * tests: print degenerate imatrix quant types --- ggml/src/ggml-quants.c | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/ggml/src/ggml-quants.c b/ggml/src/ggml-quants.c index 55db802c0..7750a72ce 100644 --- a/ggml/src/ggml-quants.c +++ b/ggml/src/ggml-quants.c @@ -1036,8 +1036,9 @@ static float make_qkx3_quants(int n, int nmax, const float * GGML_RESTRICT x, co iscale = (rmin + rdelta*is + nmax)/(max - min); float sum_l = 0, sum_l2 = 0, sum_xl = 0; for (int i = 0; i < n; ++i) { - int l = nearest_int(iscale*(x[i] - min)); - l = MAX(0, MIN(nmax, l)); + // min is the best fit so far and can be at or near max, so v can be inf, nan or out of range for nearest_int + const float v = iscale*(x[i] - min); + const int l = v > 0 ? nearest_int(MIN(v, nmax)) : 0; Laux[i] = l; float w = weights ? weights[i] : x[i]*x[i]; sum_l += w*l;