Add Q2_0 quantization: type definition and CPU backend (#24448)

This commit is contained in:
Pasha Khosravi
2026-07-07 12:05:47 -07:00
committed by GitHub
parent c198af4dc2
commit bec4772f6a
17 changed files with 262 additions and 3 deletions
+76
View File
@@ -71,6 +71,44 @@ void quantize_row_q1_0_ref(const float * GGML_RESTRICT x, block_q1_0 * GGML_REST
}
}
void quantize_row_q2_0_ref(const float * GGML_RESTRICT x, block_q2_0 * GGML_RESTRICT y, int64_t k) {
static const int qk = QK2_0;
assert(k % qk == 0);
const int nb = k / qk;
for (int i = 0; i < nb; i++) {
// Compute scale as max absolute value in the block
float amax = 0.0f;
for (int j = 0; j < qk; j++) {
const float a = fabsf(x[i*qk + j]);
if (a > amax) amax = a;
}
const float d = amax;
const float id = d > 0.0f ? 1.0f / d : 0.0f;
y[i].d = GGML_FP32_TO_FP16(d);
// Clear quant bytes
for (int j = 0; j < qk / 4; ++j) {
y[i].qs[j] = 0;
}
// Encode 2-bit values: round(w/d) clamped to [-1, 2], then add 1
// 00 (-1) = -scale, 01 (0) = 0, 10 (+1) = +scale, 11 (+2) = 2*scale
for (int j = 0; j < qk; ++j) {
const float w = x[i*qk + j];
int q = (int)roundf(w * id) + 1;
if (q < 0) q = 0;
if (q > 3) q = 3;
const int byte_index = j / 4;
const int bit_offset = (j % 4) * 2;
y[i].qs[byte_index] |= ((uint8_t)q << bit_offset);
}
}
}
// reference implementation for deterministic creation of model files
void quantize_row_q4_0_ref(const float * GGML_RESTRICT x, block_q4_0 * GGML_RESTRICT y, int64_t k) {
static const int qk = QK4_0;
@@ -398,6 +436,26 @@ void dequantize_row_q1_0(const block_q1_0 * GGML_RESTRICT x, float * GGML_RESTRI
}
}
void dequantize_row_q2_0(const block_q2_0 * GGML_RESTRICT x, float * GGML_RESTRICT y, int64_t k) {
static const int qk = QK2_0;
assert(k % qk == 0);
const int nb = k / qk;
for (int i = 0; i < nb; i++) {
const float d = GGML_FP16_TO_FP32(x[i].d);
for (int j = 0; j < qk; ++j) {
const int byte_index = j / 4;
const int bit_offset = (j % 4) * 2;
const uint8_t q = (x[i].qs[byte_index] >> bit_offset) & 0x03;
// 00=-1, 01=0, 10=+1, 11=+2
y[i*qk + j] = ((int)q - 1) * d;
}
}
}
void dequantize_row_q4_0(const block_q4_0 * GGML_RESTRICT x, float * GGML_RESTRICT y, int64_t k) {
static const int qk = QK4_0;
@@ -2052,6 +2110,20 @@ size_t quantize_q1_0(const float * GGML_RESTRICT src, void * GGML_RESTRICT dst,
return nrow * row_size;
}
size_t quantize_q2_0(const float * GGML_RESTRICT src, void * GGML_RESTRICT dst, int64_t nrow, int64_t n_per_row, const float * quant_weights) {
if (!quant_weights) {
quantize_row_q2_0_ref(src, dst, (int64_t)nrow*n_per_row);
return nrow * ggml_row_size(GGML_TYPE_Q2_0, n_per_row);
}
size_t row_size = ggml_row_size(GGML_TYPE_Q2_0, n_per_row);
char * qrow = (char *)dst;
for (int64_t row = 0; row < nrow; ++row) {
quantize_row_q2_0_ref(src, (block_q2_0*)qrow, n_per_row);
src += n_per_row;
qrow += row_size;
}
return nrow * row_size;
}
size_t quantize_q4_0(const float * GGML_RESTRICT src, void * GGML_RESTRICT dst, int64_t nrow, int64_t n_per_row, const float * quant_weights) {
if (!quant_weights) {
@@ -5461,6 +5533,10 @@ bool ggml_validate_row_data(enum ggml_type type, const void * data, size_t nbyte
{
VALIDATE_ROW_DATA_D_F16_IMPL(block_q1_0, data, nb);
} break;
case GGML_TYPE_Q2_0:
{
VALIDATE_ROW_DATA_D_F16_IMPL(block_q2_0, data, nb);
} break;
case GGML_TYPE_Q4_0:
{
VALIDATE_ROW_DATA_D_F16_IMPL(block_q4_0, data, nb);