support windows support q4_0 and q5_0 dequant on cpu Add CopyRight from pygguf(It was added before, but disappear after merge). Add some TODO in the code.

2025-09-17 02:29:41 +00:00 · 2024-08-07 12:19:06 +08:00 · 2024-08-07 12:19:06 +08:00 · 0a2fd52cea
commit 0a2fd52cea
parent 442e13bc97
32 changed files with 248 additions and 108 deletions
--- a/ktransformers/optimize/optimize_rules/DeepSeek-V2-Chat.yaml
+++ b/ktransformers/optimize/optimize_rules/DeepSeek-V2-Chat.yaml
@ -26,7 +26,7 @@
      prefill_device: "cuda"
      prefill_mlp_type: "MLPExpertsTorch"
      generate_device: "cpu"
-      generate_mlp_type:  "MLPCPUExperts"
+      generate_mlp_type: "MLPCPUExperts"
      out_device: "cuda"
  recursive: False # don't recursively inject submodules of this module
 - match: