From: hcl Date: Tue, 4 Aug 2026 09:12:48 +0000 (+0800) Subject: gguf-py: validate n_dims and guard against uint64 overflow in reader (#25401) X-Git-Tag: upstream/0.0.10438~172 X-Git-Url: https://git.djapps.eu/?a=commitdiff_plain;h=5788b510a1e3394fcc2d6b13ba2d9d8fc4a5c139;p=pkg%2Fggml%2Fsources%2Fllama.cpp gguf-py: validate n_dims and guard against uint64 overflow in reader (#25401) The Python GGUF reader lacked two guards the C++ loader has: - n_dims read as uint32 with no GGML_MAX_DIMS bound -> crafted file with huge n_dims triggers oversized memmap read / OOM. - np.prod(dims) on uint64 wraps silently -> a crafted dims triple can overflow to a tiny element count, passing an undersized read through. Add a GGML_MAX_DIMS check and compute the element count with Python ints. Fixes #25378 --- diff --git a/gguf-py/gguf/constants.py b/gguf-py/gguf/constants.py index c9ec92bd8..6b0a26b63 100644 --- a/gguf-py/gguf/constants.py +++ b/gguf-py/gguf/constants.py @@ -11,6 +11,7 @@ GGUF_MAGIC = 0x46554747 # "GGUF" GGUF_VERSION = 3 GGUF_DEFAULT_ALIGNMENT = 32 GGML_QUANT_VERSION = 2 # GGML_QNT_VERSION from ggml.h +GGML_MAX_DIMS = 4 # GGML_MAX_DIMS from ggml.h # # metadata keys diff --git a/gguf-py/gguf/gguf_reader.py b/gguf-py/gguf/gguf_reader.py index 0a1b85f50..ea241ada2 100644 --- a/gguf-py/gguf/gguf_reader.py +++ b/gguf-py/gguf/gguf_reader.py @@ -22,6 +22,7 @@ if __name__ == "__main__": sys.path.insert(0, str(Path(__file__).parent.parent)) from gguf.constants import ( + GGML_MAX_DIMS, GGML_QUANT_SIZES, GGUF_DEFAULT_ALIGNMENT, GGUF_MAGIC, @@ -266,6 +267,8 @@ class GGUFReader: # Get Tensor Dimensions Count n_dims = self._get(offs, np.uint32) offs += int(n_dims.nbytes) + if n_dims[0] > GGML_MAX_DIMS: + raise ValueError(f'Tensor dimensions count {n_dims[0]} exceeds GGML_MAX_DIMS ({GGML_MAX_DIMS})') # Get Tensor Dimension Array dims = self._get(offs, np.uint64, n_dims[0]) @@ -326,7 +329,10 @@ class GGUFReader: raise ValueError(f'Found duplicated tensor with name {tensor_name}') tensor_names.add(tensor_name) ggml_type = GGMLQuantizationType(raw_dtype[0]) - n_elems = int(np.prod(dims)) + # use Python ints: np.prod on uint64 wraps silently on overflow + n_elems = 1 + for dim in dims.tolist(): + n_elems *= int(dim) np_dims = tuple(reversed(dims.tolist())) block_size, type_size = GGML_QUANT_SIZES[ggml_type] n_bytes = n_elems * type_size // block_size diff --git a/gguf-py/tests/test_gguf_reader_validation.py b/gguf-py/tests/test_gguf_reader_validation.py new file mode 100644 index 000000000..98f30a969 --- /dev/null +++ b/gguf-py/tests/test_gguf_reader_validation.py @@ -0,0 +1,37 @@ +import struct +import numpy as np +import pytest + +from gguf.gguf_reader import GGUFReader + + +def _write_gguf(path, n_dims_field, dims): + buf = b'GGUF' + struct.pack('