Add Q2_0 quantization: type definition and CPU backend (#24448)

This commit is contained in:
Pasha Khosravi
2026-07-07 12:05:47 -07:00
committed by GitHub
parent c198af4dc2
commit bec4772f6a
17 changed files with 262 additions and 3 deletions
+3 -1
View File
@@ -429,7 +429,8 @@ extern "C" {
GGML_TYPE_MXFP4 = 39, // MXFP4 (1 block)
GGML_TYPE_NVFP4 = 40, // NVFP4 (4 blocks, E4M3 scale)
GGML_TYPE_Q1_0 = 41,
GGML_TYPE_COUNT = 42,
GGML_TYPE_Q2_0 = 42,
GGML_TYPE_COUNT = 43,
};
// precision
@@ -473,6 +474,7 @@ extern "C" {
GGML_FTYPE_MOSTLY_MXFP4 = 25, // except 1d tensors
GGML_FTYPE_MOSTLY_NVFP4 = 26, // except 1d tensors
GGML_FTYPE_MOSTLY_Q1_0 = 27, // except 1d tensors
GGML_FTYPE_MOSTLY_Q2_0 = 28, // except 1d tensors
};
// available tensor operations: