Add Q2_0 quantization: type definition and CPU backend (#24448)

This commit is contained in:
Pasha Khosravi
2026-07-07 12:05:47 -07:00
committed by GitHub
parent c198af4dc2
commit bec4772f6a
17 changed files with 262 additions and 3 deletions
+3
View File
@@ -4533,6 +4533,7 @@ class GGMLQuantizationType(IntEnum):
MXFP4 = 39
NVFP4 = 40
Q1_0 = 41
Q2_0 = 42
class ExpertGatingFuncType(IntEnum):
@@ -4588,6 +4589,7 @@ class LlamaFileType(IntEnum):
MOSTLY_MXFP4_MOE = 38 # except 1d tensors
MOSTLY_NVFP4 = 39 # except 1d tensors
MOSTLY_Q1_0 = 40 # except 1d tensors
MOSTLY_Q2_0 = 41 # except 1d tensors
GUESSED = 1024 # not specified in the model file
@@ -4713,6 +4715,7 @@ GGML_QUANT_SIZES: dict[GGMLQuantizationType, tuple[int, int]] = {
GGMLQuantizationType.MXFP4: (32, 1 + 16),
GGMLQuantizationType.NVFP4: (64, 4 + 32),
GGMLQuantizationType.Q1_0: (128, 2 + 16),
GGMLQuantizationType.Q2_0: (64, 2 + 16),
}