Add Q2_0 quantization: type definition and CPU backend (#24448)
This commit is contained in:
@@ -4533,6 +4533,7 @@ class GGMLQuantizationType(IntEnum):
|
||||
MXFP4 = 39
|
||||
NVFP4 = 40
|
||||
Q1_0 = 41
|
||||
Q2_0 = 42
|
||||
|
||||
|
||||
class ExpertGatingFuncType(IntEnum):
|
||||
@@ -4588,6 +4589,7 @@ class LlamaFileType(IntEnum):
|
||||
MOSTLY_MXFP4_MOE = 38 # except 1d tensors
|
||||
MOSTLY_NVFP4 = 39 # except 1d tensors
|
||||
MOSTLY_Q1_0 = 40 # except 1d tensors
|
||||
MOSTLY_Q2_0 = 41 # except 1d tensors
|
||||
|
||||
GUESSED = 1024 # not specified in the model file
|
||||
|
||||
@@ -4713,6 +4715,7 @@ GGML_QUANT_SIZES: dict[GGMLQuantizationType, tuple[int, int]] = {
|
||||
GGMLQuantizationType.MXFP4: (32, 1 + 16),
|
||||
GGMLQuantizationType.NVFP4: (64, 4 + 32),
|
||||
GGMLQuantizationType.Q1_0: (128, 2 + 16),
|
||||
GGMLQuantizationType.Q2_0: (64, 2 + 16),
|
||||
}
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user