]> git.djapps.eu Git - pkg/ggml/sources/llama.cpp/commitdiff
convert : minor fixes for numpy 2.x (#23571)
authorSigbjørn Skjæret <redacted>
Sun, 24 May 2026 07:51:31 +0000 (09:51 +0200)
committerGitHub <redacted>
Sun, 24 May 2026 07:51:31 +0000 (09:51 +0200)
examples/convert_legacy_llama.py
gguf-py/gguf/quants.py

index c4ec5c524e9b13063bcfc74b2bf1a7f8d17c8a36..5c9305b1237d3bb3310eff0dcec6be1b108bc748 100755 (executable)
@@ -1308,7 +1308,8 @@ def do_dump_model(model_plus: ModelPlus) -> None:
 
 def main(args_in: list[str] | None = None) -> None:
     output_choices = ["f32", "f16"]
-    if np.uint32(1) == np.uint32(1).newbyteorder("<"):
+    dummy_val = np.uint32(1)
+    if dummy_val == dummy_val.view(dummy_val.dtype.newbyteorder("<")):
         # We currently only support Q8_0 output on little endian systems.
         output_choices.append("q8_0")
     parser = argparse.ArgumentParser(description="Convert a LLaMA model to a GGML compatible file")
index 1d9d9ab7d70e95028bf72ed7ef7d4cab392d470f..80966b6ef1518a45b86745d94eb70d05c3c5490f 100644 (file)
@@ -28,6 +28,7 @@ def quant_shape_from_byte_shape(shape: Sequence[int], quant_type: GGMLQuantizati
 # This is faster than np.vectorize and np.apply_along_axis because it works on more than one row at a time
 def _apply_over_grouped_rows(func: Callable[[np.ndarray], np.ndarray], arr: np.ndarray, otype: DTypeLike, oshape: tuple[int, ...]) -> np.ndarray:
     rows = arr.reshape((-1, arr.shape[-1]))
+    assert len(rows.shape)
     osize = 1
     for dim in oshape:
         osize *= dim