]> git.djapps.eu Git - pkg/ggml/sources/llama.cpp/commitdiff
llama-quant : exclude i32 ffn_gate_tid2eid routing table from quantization (#25787)
authorYash Raj Pandey <redacted>
Sat, 18 Jul 2026 11:43:18 +0000 (07:43 -0400)
committerGitHub <redacted>
Sat, 18 Jul 2026 11:43:18 +0000 (13:43 +0200)
DeepSeek-V4's ffn_gate_tid2eid tensor is an i32 token-id -> expert-id
index table, not weights. It was never added to the name-based
exclusion list alongside ffn_gate_inp.weight, so llama-quantize tries
to quantize it and fails since i32 cannot convert to a float type.

Fixes ggml-org/llama.cpp#25754

Signed-off-by: Yash Raj Pandey <redacted>
src/llama-quant.cpp

index b66759b277608f6f3d204994b2113242a774102a..b187f5de96a87ecc8c764c07c929b5109d927506 100644 (file)
@@ -306,6 +306,9 @@ static bool tensor_allows_quantization(const llama_model_quantize_params * param
     // NOTE: can't use LLM_TN here because the layer number is not known
     quantize &= name.find("ffn_gate_inp.weight") == std::string::npos;
 
+    // do not quantize the i32 token-id -> expert-id routing table (DeepSeek-V4)
+    quantize &= name.find("ffn_gate_tid2eid.weight") == std::string::npos;
+
     // these are very small (e.g. 4x4)
     quantize &= name.find("altup")  == std::string::npos;
     quantize &= name.find("laurel") == std::string::npos;