mirror of
https://github.com/LostRuins/koboldcpp.git
synced 2026-10-03 03:25:40 +00:00
server : improve startup log messages (#29125)
* server-models : show source per model in log - Show [source] tag (preset/models_dir/cache) per model instead of cryptic * marker - Show HF hub cache path in the 'Loaded cached model presets' log - Add hf_cache::get_cache_dir() public accessor Assisted-by: pi:llama.cpp/Qwen3.8-27B * cont : pad log
This commit is contained in:
parent
eb1e1f495f
commit
1af554f8fc
5 changed files with 39 additions and 33 deletions
|
|
@ -122,6 +122,8 @@ struct common_params_context {
|
|||
|
||||
// parse input arguments from CLI
|
||||
// if one argument has invalid value, it will automatically display usage of the specific argument (and not the full usage message)
|
||||
// TODO: this function can load ggml backend (by calling llama_support_rpc)
|
||||
// this is a side-effect that should be avoided
|
||||
bool common_params_parse(int argc, char ** argv, common_params & params, llama_example ex, void(*print_usage)(int, char **) = nullptr);
|
||||
|
||||
// load all backends and print the list of available (non-CPU) devices to stdout
|
||||
|
|
|
|||
|
|
@ -30,8 +30,8 @@ namespace hf_cache {
|
|||
|
||||
namespace fs = std::filesystem;
|
||||
|
||||
static fs::path get_cache_directory() {
|
||||
static const fs::path cache = []() {
|
||||
std::string get_cache_path() {
|
||||
static const std::string cache = []() {
|
||||
struct {
|
||||
const char * var;
|
||||
fs::path path;
|
||||
|
|
@ -46,14 +46,14 @@ static fs::path get_cache_directory() {
|
|||
for (const auto & entry : entries) {
|
||||
if (auto * p = std::getenv(entry.var); p && *p) {
|
||||
fs::path base(p);
|
||||
return entry.path.empty() ? base : base / entry.path;
|
||||
return (entry.path.empty() ? base : base / entry.path).string();
|
||||
}
|
||||
}
|
||||
#ifndef _WIN32
|
||||
const struct passwd * pw = getpwuid(getuid());
|
||||
|
||||
if (pw && pw->pw_dir && *pw->pw_dir) {
|
||||
return fs::path(pw->pw_dir) / ".cache" / "huggingface" / "hub";
|
||||
return (fs::path(pw->pw_dir) / ".cache" / "huggingface" / "hub").string();
|
||||
}
|
||||
#endif
|
||||
throw std::runtime_error("Failed to determine HF cache directory");
|
||||
|
|
@ -80,7 +80,7 @@ static std::string repo_to_folder_name(const std::string & repo_id) {
|
|||
}
|
||||
|
||||
static fs::path get_repo_path(const std::string & repo_id) {
|
||||
return get_cache_directory() / repo_to_folder_name(repo_id);
|
||||
return fs::path(get_cache_path()) / repo_to_folder_name(repo_id);
|
||||
}
|
||||
|
||||
static bool is_hex_char(const char c) {
|
||||
|
|
@ -393,8 +393,8 @@ static std::string get_cached_ref(const fs::path & repo_path) {
|
|||
}
|
||||
|
||||
hf_files get_cached_files(const std::string & repo_id) {
|
||||
fs::path cache_dir = get_cache_directory();
|
||||
if (!fs::exists(cache_dir)) {
|
||||
const fs::path cache_path = get_cache_path();
|
||||
if (!fs::exists(cache_path)) {
|
||||
return {};
|
||||
}
|
||||
|
||||
|
|
@ -405,7 +405,7 @@ hf_files get_cached_files(const std::string & repo_id) {
|
|||
|
||||
hf_files files;
|
||||
|
||||
for (const auto & repo : fs::directory_iterator(cache_dir)) {
|
||||
for (const auto & repo : fs::directory_iterator(cache_path)) {
|
||||
if (!repo.is_directory()) {
|
||||
continue;
|
||||
}
|
||||
|
|
|
|||
|
|
@ -32,4 +32,7 @@ std::string finalize_file(const hf_file & file);
|
|||
// Remove the entire cached directory for a repo, returns true if removed
|
||||
bool remove_cached_repo(const std::string & repo_id);
|
||||
|
||||
// Returns the HuggingFace hub cache path
|
||||
std::string get_cache_path();
|
||||
|
||||
} // namespace hf_cache
|
||||
|
|
|
|||
|
|
@ -7,6 +7,7 @@
|
|||
#include "build-info.h"
|
||||
#include "preset.h"
|
||||
#include "download.h"
|
||||
#include "hf-cache.h"
|
||||
#include "http.h"
|
||||
#include "subproc.h"
|
||||
|
||||
|
|
@ -677,19 +678,19 @@ void server_models::load_models() {
|
|||
// Phase 1: load presets from all sources - pure I/O, no lock needed
|
||||
// 1. cached models
|
||||
common_presets cached_models = ctx_preset.load_from_cache();
|
||||
SRV_INF("Loaded %zu cached model presets\n", cached_models.size());
|
||||
SRV_TRC("Loaded %zu cached model presets from %s\n", cached_models.size(), hf_cache::get_cache_path().c_str());
|
||||
// 2. local models from --models-dir
|
||||
common_presets local_models;
|
||||
if (!base_params.models_dir.empty()) {
|
||||
local_models = ctx_preset.load_from_models_dir(base_params.models_dir);
|
||||
SRV_INF("Loaded %zu local model presets from %s\n", local_models.size(), base_params.models_dir.c_str());
|
||||
SRV_TRC("Loaded %zu local model presets from %s\n", local_models.size(), base_params.models_dir.c_str());
|
||||
}
|
||||
// 3. custom-path models from presets
|
||||
common_preset global = {};
|
||||
common_presets custom_presets = {};
|
||||
if (!base_params.models_preset.empty()) {
|
||||
custom_presets = ctx_preset.load_from_ini(base_params.models_preset, global);
|
||||
SRV_INF("Loaded %zu custom model presets from %s\n", custom_presets.size(), base_params.models_preset.c_str());
|
||||
SRV_TRC("Loaded %zu custom model presets from %s\n", custom_presets.size(), base_params.models_preset.c_str());
|
||||
}
|
||||
|
||||
// cascade, apply global preset first
|
||||
|
|
@ -762,8 +763,6 @@ void server_models::load_models() {
|
|||
}
|
||||
|
||||
// Helpers that read `mapping` - must be called while holding the lock.
|
||||
std::unordered_set<std::string> custom_names;
|
||||
for (const auto & [name, preset] : custom_presets) custom_names.insert(name);
|
||||
auto join_set = [](const std::set<std::string> & s) {
|
||||
std::string result;
|
||||
for (const auto & v : s) {
|
||||
|
|
@ -773,13 +772,19 @@ void server_models::load_models() {
|
|||
return result;
|
||||
};
|
||||
auto log_available_models = [&]() {
|
||||
SRV_INF("Available models (%zu) (*: custom preset)\n", mapping.size());
|
||||
SRV_INF("Available models (%zu):\n", mapping.size());
|
||||
if (mapping.empty()) {
|
||||
SRV_INF("%s", " no models found on the system (visit https://llama.app/models for suggestions)\n");
|
||||
} else {
|
||||
for (const auto & [name, inst] : mapping) {
|
||||
bool has_custom = custom_names.find(name) != custom_names.end();
|
||||
const std::string source = server_model_source_to_string(inst.meta.source);
|
||||
|
||||
std::string info;
|
||||
if (!inst.meta.aliases.empty()) info += " (aliases: " + join_set(inst.meta.aliases) + ")";
|
||||
if (!inst.meta.tags.empty()) info += " [tags: " + join_set(inst.meta.tags) + "]";
|
||||
SRV_INF(" %c %s%s\n", has_custom ? '*' : ' ', name.c_str(), info.c_str());
|
||||
|
||||
SRV_INF(" [%10s] %s%s\n", source.c_str(), name.c_str(), info.c_str());
|
||||
}
|
||||
}
|
||||
};
|
||||
auto apply_stop_timeout = [&]() {
|
||||
|
|
|
|||
|
|
@ -102,6 +102,8 @@ int llama_server(int argc, char ** argv) {
|
|||
// touch it. lifecycle is symmetric, stop_gc() runs in clean_up() before backend free
|
||||
server_stream_session_manager_start();
|
||||
|
||||
SRV_INF("%s", "initializing ...\n");
|
||||
|
||||
if (!common_params_parse(argc, argv, params, LLAMA_EXAMPLE_SERVER)) {
|
||||
return 1;
|
||||
}
|
||||
|
|
@ -320,11 +322,7 @@ int llama_server(common_params & params, int argc, char ** argv) {
|
|||
};
|
||||
|
||||
if (params.cors_origins == "*" && params.api_keys.empty()) {
|
||||
SRV_WRN("%s", "-----------------\n");
|
||||
SRV_WRN("%s", "CORS is set to allow all origins ('*') and no API key is set\n");
|
||||
SRV_WRN("%s", "this can be a security risk (cross-origin attacks)\n");
|
||||
SRV_WRN("%s", "more info: https://github.com/ggml-org/llama.cpp/pull/25655\n");
|
||||
SRV_WRN("%s", "-----------------\n");
|
||||
SRV_WRN("%s", "security: no API key is set and CORS allows all origins (see https://github.com/ggml-org/llama.cpp/pull/25655)\n");
|
||||
}
|
||||
|
||||
// CORS proxy (EXPERIMENTAL, only used by the Web UI for MCP)
|
||||
|
|
@ -372,14 +370,13 @@ int llama_server(common_params & params, int argc, char ** argv) {
|
|||
ctx_http.post("/tools", ex_wrapper(res_403));
|
||||
}
|
||||
|
||||
if (warn_names.size() > 0) {
|
||||
SRV_WRN("%s", "-----------------\n");
|
||||
SRV_WRN("%s", "the following feature(s) are enabled:\n");
|
||||
if (!warn_names.empty()) {
|
||||
std::string features;
|
||||
for (const auto & name : warn_names) {
|
||||
SRV_WRN(" %s\n", name.c_str());
|
||||
if (!features.empty()) features += ", ";
|
||||
features += name;
|
||||
}
|
||||
SRV_WRN("%s", "do not expose the server to untrusted environments\n");
|
||||
SRV_WRN("%s", "-----------------\n");
|
||||
SRV_WRN("security: %s enabled - do not expose to untrusted environments\n", features.c_str());
|
||||
}
|
||||
|
||||
//
|
||||
|
|
@ -517,8 +514,7 @@ int llama_server(common_params & params, int argc, char ** argv) {
|
|||
// TODO: remove this in the future
|
||||
// check the string to also handle the .sock case
|
||||
if (string_ends_with(ctx_http.listening_address, ":8080")) {
|
||||
SRV_WRN("%s", "NOTICE: server default port will be changed to :9931 in a future release\n");
|
||||
SRV_WRN("%s", " ref: https://github.com/ggml-org/llama.cpp/pull/26508\n");
|
||||
SRV_WRN("%s", "notice: server default port will be changed to :9931 in a future release (ref: https://github.com/ggml-org/llama.cpp/pull/26508)\n");
|
||||
}
|
||||
|
||||
if (is_router_server) {
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue