]> git.djapps.eu Git - pkg/ggml/sources/llama.cpp/commitdiff
convert : force f16 or f32 on step3-vl conv weights (#21646)
authorSigbjørn Skjæret <redacted>
Sun, 12 Apr 2026 17:22:29 +0000 (19:22 +0200)
committerGitHub <redacted>
Sun, 12 Apr 2026 17:22:29 +0000 (19:22 +0200)
convert_hf_to_gguf.py

index c96afc78b6eb926df4e0ddb76351df77095d9c45..374a55fb176990640bb952c61776f9e3c3c9f270 100755 (executable)
@@ -4992,6 +4992,8 @@ class Step3VLVisionModel(MmprojModel):
     def tensor_force_quant(self, name, new_name, bid, n_dims):
         if ".position_embd." in new_name:
             return gguf.GGMLQuantizationType.F32
+        if ("mm.0." in new_name or "mm.1." in new_name) and new_name.endswith(".weight"):
+            return gguf.GGMLQuantizationType.F16 if self.ftype == gguf.LlamaFileType.MOSTLY_F16 else gguf.GGMLQuantizationType.F32
         return super().tensor_force_quant(name, new_name, bid, n_dims)
 
     def modify_tensors(self, data_torch: Tensor, name: str, bid: int | None) -> Iterable[tuple[str, Tensor]]: