Merge branch 'master' into concedo_experimental

# Conflicts: # Makefile # README.md
2025-09-16 20:09:41 +00:00 · 2023-11-27 14:06:14 +08:00 · 2023-11-27 14:06:14 +08:00 · 8acd7be734
commit 8acd7be734
parent ec1796bec1 f3b269813f
20 changed files with 1274 additions and 136 deletions
--- a/common/common.h
+++ b/common/common.h
@ -130,6 +130,7 @@ struct gpt_params {
    bool numa              = false; // attempt optimizations that help on some NUMA systems
    bool verbose_prompt    = false; // print prompt tokens before generation
    bool infill            = false; // use infill mode
+    bool dump_kv_cache     = false; // dump the KV cache contents for debugging purposes

    // multimodal models (see examples/llava)
    std::string mmproj = ""; // path to multimodal projector
@ -226,3 +227,13 @@ std::string get_sortable_timestamp();
 void dump_non_result_info_yaml(
    FILE * stream, const gpt_params & params, const llama_context * lctx,
    const std::string & timestamp, const std::vector<int> & prompt_tokens, const char * model_desc);
+
+//
+// KV cache utils
+//
+
+// Dump the KV cache view with the number of sequences per cell.
+void dump_kv_cache_view(const llama_kv_cache_view & view, int row_size = 80);
+
+// Dump the KV cache view showing individual sequences in each cell (long output).
+void dump_kv_cache_view_seqs(const llama_kv_cache_view & view, int row_size = 40);