note: also has support for completion tokens count

2025-09-14 10:59:41 +00:00 · 2024-11-01 00:44:14 +08:00 · 2024-11-01 00:44:14 +08:00 · a46f8acd03
commit a46f8acd03
parent aa26a58085 8f275a7c45
31 changed files with 138676 additions and 137399 deletions
--- a/src/llama-vocab.cpp
+++ b/src/llama-vocab.cpp
@ -2226,3 +2226,19 @@ int32_t llama_detokenize_impl(

    return total <= text_len_max ? total : -total;
 }
+
+std::string llama_detokenize(const struct llama_vocab & vocab, const std::vector<llama_token> & tokens, bool special) {
+    std::string text;
+    text.resize(std::max(text.capacity(), tokens.size()));
+    int32_t n_chars = llama_detokenize_impl(vocab, tokens.data(), (int32_t)tokens.size(), &text[0], (int32_t)text.size(), false, special);
+    if (n_chars < 0) {
+        text.resize(-n_chars);
+        n_chars = llama_detokenize_impl(vocab, tokens.data(), (int32_t)tokens.size(), &text[0], (int32_t)text.size(), false, special);
+        GGML_ASSERT(n_chars <= (int32_t)text.size());  // whitespace trimming is performed after per-token detokenization
+    }
+
+    text.resize(n_chars);
+
+    // NOTE: the original tokenizer decodes bytes after collecting the pieces.
+    return text;
+}