diff --git a/gguf-py/gguf/constants.py b/gguf-py/gguf/constants.py index c9ec92bd8..6b0a26b63 100644 --- a/gguf-py/gguf/constants.py +++ b/gguf-py/gguf/constants.py @@ -11,6 +11,7 @@ GGUF_MAGIC = 0x46554747 # "GGUF" GGUF_VERSION = 3 GGUF_DEFAULT_ALIGNMENT = 32 GGML_QUANT_VERSION = 2 # GGML_QNT_VERSION from ggml.h +GGML_MAX_DIMS = 4 # GGML_MAX_DIMS from ggml.h # # metadata keys diff --git a/gguf-py/gguf/gguf_reader.py b/gguf-py/gguf/gguf_reader.py index 0a1b85f50..ea241ada2 100644 --- a/gguf-py/gguf/gguf_reader.py +++ b/gguf-py/gguf/gguf_reader.py @@ -22,6 +22,7 @@ if __name__ == "__main__": sys.path.insert(0, str(Path(__file__).parent.parent)) from gguf.constants import ( + GGML_MAX_DIMS, GGML_QUANT_SIZES, GGUF_DEFAULT_ALIGNMENT, GGUF_MAGIC, @@ -266,6 +267,8 @@ class GGUFReader: # Get Tensor Dimensions Count n_dims = self._get(offs, np.uint32) offs += int(n_dims.nbytes) + if n_dims[0] > GGML_MAX_DIMS: + raise ValueError(f'Tensor dimensions count {n_dims[0]} exceeds GGML_MAX_DIMS ({GGML_MAX_DIMS})') # Get Tensor Dimension Array dims = self._get(offs, np.uint64, n_dims[0]) @@ -326,7 +329,10 @@ class GGUFReader: raise ValueError(f'Found duplicated tensor with name {tensor_name}') tensor_names.add(tensor_name) ggml_type = GGMLQuantizationType(raw_dtype[0]) - n_elems = int(np.prod(dims)) + # use Python ints: np.prod on uint64 wraps silently on overflow + n_elems = 1 + for dim in dims.tolist(): + n_elems *= int(dim) np_dims = tuple(reversed(dims.tolist())) block_size, type_size = GGML_QUANT_SIZES[ggml_type] n_bytes = n_elems * type_size // block_size diff --git a/gguf-py/tests/test_gguf_reader_validation.py b/gguf-py/tests/test_gguf_reader_validation.py new file mode 100644 index 000000000..98f30a969 --- /dev/null +++ b/gguf-py/tests/test_gguf_reader_validation.py @@ -0,0 +1,37 @@ +import struct +import numpy as np +import pytest + +from gguf.gguf_reader import GGUFReader + + +def _write_gguf(path, n_dims_field, dims): + buf = b'GGUF' + struct.pack('