tests: extend test-quantize-fns to test nrc=2 (i8mm) kernels (#16234)

* Test for nrc=2 as well | i8mm kernels

* Trigger only on supported HW

* Remove trailing whitespace

* Address review comment

* test: properly prepare nrc=2 inputs with independent data per row

* tests : make nrc=2 dot product inputs distinct

Assisted-by: Kiro

* tests : use non-trivial strides in nrc=2 dot product test

* tests : fail nrc=2 dot product test on non-finite errors
This commit is contained in:
Rohanjames1997
2026-09-11 13:19:37 -05:00
committed by GitHub
parent 8172e6577a
commit 982937a333
+56 -13
View File
@@ -5,6 +5,8 @@
#undef NDEBUG
#include <assert.h>
#include <algorithm>
#include <cmath>
#include <math.h>
#include <stdio.h>
#include <string>
@@ -32,9 +34,9 @@ static const char* RESULT_STR[] = {"ok", "FAILED"};
// Generate synthetic data
static void generate_data(float offset, size_t n, float * dst) {
static void generate_data(float offset, size_t n, float * dst, float amplitude = 2.0f) {
for (size_t i = 0; i < n; i++) {
dst[i] = 0.1 + 2*cosf(i + offset);
dst[i] = 0.1 + amplitude*cosf(i + offset);
}
}
@@ -83,23 +85,50 @@ static float dot_product(const float * a1, const float * a2, size_t test_size) {
}
// Total dot product error
static float dot_product_error(const ggml_type_traits * qfns, const ggml_type_traits_cpu * qfns_cpu, size_t test_size, const float * test_data1, const float * test_data2) {
GGML_UNUSED(qfns);
std::vector<uint8_t> tmp_q1(2*test_size);
std::vector<uint8_t> tmp_q2(2*test_size);
static float dot_product_error(const ggml_type_traits_cpu * qfns_cpu, ggml_type src0_type, size_t test_size,
const float * test_data1, const float * test_data2,
const float * test_data3, const float * test_data4,
const int nrc) {
const auto * vdot = ggml_get_type_traits_cpu(qfns_cpu->vec_dot_type);
const size_t pad = 64;
const size_t bx = ggml_row_size(src0_type, test_size) + pad;
const size_t by = ggml_row_size(qfns_cpu->vec_dot_type, test_size) + pad;
std::vector<uint8_t> tmp_q1(bx * nrc);
std::vector<uint8_t> tmp_q2(by * nrc);
qfns_cpu->from_float(test_data1, tmp_q1.data(), test_size);
vdot->from_float(test_data2, tmp_q2.data(), test_size);
float result = INFINITY;
qfns_cpu->vec_dot(test_size, &result, 0, tmp_q1.data(), 0, tmp_q2.data(), 0, 1);
if (nrc == 1) {
float result = INFINITY;
qfns_cpu->vec_dot(test_size, &result, 0, tmp_q1.data(), 0, tmp_q2.data(), 0, 1);
const float dot_ref = dot_product(test_data1, test_data2, test_size);
const float dot_ref = dot_product(test_data1, test_data2, test_size);
return fabsf(result - dot_ref) / test_size;
}
return fabsf(result - dot_ref) / test_size;
// nrc == 2: kernel computes a 2x2 dot product matrix
// Output layout: s[0]=dot(vx0,vy0), s[1]=dot(vx1,vy0), s[bs]=dot(vx0,vy1), s[bs+1]=dot(vx1,vy1)
// row and output strides are padded, same as in the mul_mat path
qfns_cpu->from_float(test_data3, tmp_q1.data() + bx, test_size);
vdot->from_float(test_data4, tmp_q2.data() + by, test_size);
const size_t bs = 16;
std::vector<float> result(bs + 2, INFINITY);
qfns_cpu->vec_dot(test_size, result.data(), bs, tmp_q1.data(), bx, tmp_q2.data(), by, 2);
const float ref00 = dot_product(test_data1, test_data2, test_size);
const float ref10 = dot_product(test_data3, test_data2, test_size);
const float ref01 = dot_product(test_data1, test_data4, test_size);
const float ref11 = dot_product(test_data3, test_data4, test_size);
const auto err = [test_size](float val, float ref) {
const float e = fabsf(val - ref) / test_size;
return std::isfinite(e) ? e : INFINITY;
};
return std::max({err(result[0], ref00), err(result[1], ref10), err(result[bs], ref01), err(result[bs + 1], ref11)});
}
static int test_vec_dot_f32(bool verbose) {
@@ -133,9 +162,13 @@ static int test_vec_dot_q(bool verbose) {
std::vector<float> test_data(test_size);
std::vector<float> test_data2(test_size);
std::vector<float> test_data3(test_size);
std::vector<float> test_data4(test_size);
generate_data(0.0, test_data.size(), test_data.data());
generate_data(1.0, test_data2.size(), test_data2.data());
generate_data(3.0, test_data3.size(), test_data3.data(), 1.0f);
generate_data(4.0, test_data4.size(), test_data4.data(), 1.5f);
for (int i = 0; i < GGML_TYPE_COUNT; i++) {
ggml_type type = (ggml_type) i;
@@ -178,7 +211,7 @@ static int test_vec_dot_q(bool verbose) {
printf("%5s reference implementation error: %s (%f)\n", ggml_type_name(type), RESULT_STR[failed], reference_error);
}
const float vec_dot_error = dot_product_error(qfns, qfns_cpu, test_size, test_data.data(), test_data2.data());
const float vec_dot_error = dot_product_error(qfns_cpu, type, test_size, test_data.data(), test_data2.data(), nullptr, nullptr, 1);
const float max_allowed_error = type == GGML_TYPE_Q2_K || type == GGML_TYPE_IQ2_XS || type == GGML_TYPE_IQ2_XXS ||
type == GGML_TYPE_IQ3_XXS || type == GGML_TYPE_IQ3_S || type == GGML_TYPE_IQ2_S
? MAX_DOT_PRODUCT_ERROR_LOWBIT
@@ -194,6 +227,16 @@ static int test_vec_dot_q(bool verbose) {
if (failed || verbose) {
printf("%5s dot product error: %s (%f)\n", ggml_type_name(type), RESULT_STR[failed], vec_dot_error);
}
// Test nrc=2 path for types that support it
if (qfns_cpu->nrows == 2) {
const float vec_dot_error_nrc2 = dot_product_error(qfns_cpu, type, test_size, test_data.data(), test_data2.data(), test_data3.data(), test_data4.data(), 2);
failed = !(vec_dot_error_nrc2 < max_allowed_error);
num_failed += failed;
if (failed || verbose) {
printf("%5s dot product error (nrc=2): %s (%f)\n", ggml_type_name(type), RESULT_STR[failed], vec_dot_error_nrc2);
}
}
}
}