mirror of
https://github.com/LostRuins/koboldcpp.git
synced 2026-08-06 15:05:31 +00:00
llama-quant : exclude i32 ffn_gate_tid2eid routing table from quantization (#25787)
DeepSeek-V4's ffn_gate_tid2eid tensor is an i32 token-id -> expert-id index table, not weights. It was never added to the name-based exclusion list alongside ffn_gate_inp.weight, so llama-quantize tries to quantize it and fails since i32 cannot convert to a float type. Fixes ggml-org/llama.cpp#25754 Signed-off-by: Yash Raj Pandey <yashpn62@gmail.com>
This commit is contained in:
parent
86a9c79f86
commit
4937ca83f4
1 changed files with 3 additions and 0 deletions
|
|
@ -306,6 +306,9 @@ static bool tensor_allows_quantization(const llama_model_quantize_params * param
|
|||
// NOTE: can't use LLM_TN here because the layer number is not known
|
||||
quantize &= name.find("ffn_gate_inp.weight") == std::string::npos;
|
||||
|
||||
// do not quantize the i32 token-id -> expert-id routing table (DeepSeek-V4)
|
||||
quantize &= name.find("ffn_gate_tid2eid.weight") == std::string::npos;
|
||||
|
||||
// these are very small (e.g. 4x4)
|
||||
quantize &= name.find("altup") == std::string::npos;
|
||||
quantize &= name.find("laurel") == std::string::npos;
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue