server : improve startup log messages (#29125)

* server-models : show source per model in log

- Show [source] tag (preset/models_dir/cache) per model instead of cryptic * marker
- Show HF hub cache path in the 'Loaded cached model presets' log
- Add hf_cache::get_cache_dir() public accessor

Assisted-by: pi:llama.cpp/Qwen3.8-27B

* cont : pad log
This commit is contained in:
Georgi Gerganov 2026-09-19 15:37:13 +03:00 • committed by GitHub
parent eb1e1f495f
commit 1af554f8fc
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
5 changed files with 39 additions and 33 deletions

View file

@ -122,6 +122,8 @@ struct common_params_context {
// parse input arguments from CLI
// if one argument has invalid value, it will automatically display usage of the specific argument (and not the full usage message)
// TODO: this function can load ggml backend (by calling llama_support_rpc)
// this is a side-effect that should be avoided
bool common_params_parse(int argc, char ** argv, common_params & params, llama_example ex, void(*print_usage)(int, char **) = nullptr);
// load all backends and print the list of available (non-CPU) devices to stdout

View file

@ -30,8 +30,8 @@ namespace hf_cache {
namespace fs = std::filesystem;
static fs::path get_cache_directory() {
static const fs::path cache = []() {
std::string get_cache_path() {
static const std::string cache = []() {
struct {
const char * var;
fs::path path;
@ -46,14 +46,14 @@ static fs::path get_cache_directory() {
for (const auto & entry : entries) {
if (auto * p = std::getenv(entry.var); p && *p) {
fs::path base(p);
return entry.path.empty() ? base : base / entry.path;
return (entry.path.empty() ? base : base / entry.path).string();
}
}
#ifndef _WIN32
const struct passwd * pw = getpwuid(getuid());
if (pw && pw->pw_dir && *pw->pw_dir) {
return fs::path(pw->pw_dir) / ".cache" / "huggingface" / "hub";
return (fs::path(pw->pw_dir) / ".cache" / "huggingface" / "hub").string();
}
#endif
throw std::runtime_error("Failed to determine HF cache directory");
@ -80,7 +80,7 @@ static std::string repo_to_folder_name(const std::string & repo_id) {
}
static fs::path get_repo_path(const std::string & repo_id) {
return get_cache_directory() / repo_to_folder_name(repo_id);
return fs::path(get_cache_path()) / repo_to_folder_name(repo_id);
}
static bool is_hex_char(const char c) {
@ -393,8 +393,8 @@ static std::string get_cached_ref(const fs::path & repo_path) {
}
hf_files get_cached_files(const std::string & repo_id) {
fs::path cache_dir = get_cache_directory();
if (!fs::exists(cache_dir)) {
const fs::path cache_path = get_cache_path();
if (!fs::exists(cache_path)) {
return {};
}
@ -405,7 +405,7 @@ hf_files get_cached_files(const std::string & repo_id) {
hf_files files;
for (const auto & repo : fs::directory_iterator(cache_dir)) {
for (const auto & repo : fs::directory_iterator(cache_path)) {
if (!repo.is_directory()) {
continue;
}

View file

@ -32,4 +32,7 @@ std::string finalize_file(const hf_file & file);
// Remove the entire cached directory for a repo, returns true if removed
bool remove_cached_repo(const std::string & repo_id);
// Returns the HuggingFace hub cache path
std::string get_cache_path();
} // namespace hf_cache

View file

@ -7,6 +7,7 @@
#include "build-info.h"
#include "preset.h"
#include "download.h"
#include "hf-cache.h"
#include "http.h"
#include "subproc.h"
@ -677,19 +678,19 @@ void server_models::load_models() {
// Phase 1: load presets from all sources - pure I/O, no lock needed
// 1. cached models
common_presets cached_models = ctx_preset.load_from_cache();
SRV_INF("Loaded %zu cached model presets\n", cached_models.size());
SRV_TRC("Loaded %zu cached model presets from %s\n", cached_models.size(), hf_cache::get_cache_path().c_str());
// 2. local models from --models-dir
common_presets local_models;
if (!base_params.models_dir.empty()) {
local_models = ctx_preset.load_from_models_dir(base_params.models_dir);
SRV_INF("Loaded %zu local model presets from %s\n", local_models.size(), base_params.models_dir.c_str());
SRV_TRC("Loaded %zu local model presets from %s\n", local_models.size(), base_params.models_dir.c_str());
}
// 3. custom-path models from presets
common_preset global = {};
common_presets custom_presets = {};
if (!base_params.models_preset.empty()) {
custom_presets = ctx_preset.load_from_ini(base_params.models_preset, global);
SRV_INF("Loaded %zu custom model presets from %s\n", custom_presets.size(), base_params.models_preset.c_str());
SRV_TRC("Loaded %zu custom model presets from %s\n", custom_presets.size(), base_params.models_preset.c_str());
}
// cascade, apply global preset first
@ -762,8 +763,6 @@ void server_models::load_models() {
}
// Helpers that read `mapping` - must be called while holding the lock.
std::unordered_set<std::string> custom_names;
for (const auto & [name, preset] : custom_presets) custom_names.insert(name);
auto join_set = [](const std::set<std::string> & s) {
std::string result;
for (const auto & v : s) {
@ -773,13 +772,19 @@ void server_models::load_models() {
return result;
};
auto log_available_models = [&]() {
SRV_INF("Available models (%zu) (*: custom preset)\n", mapping.size());
SRV_INF("Available models (%zu):\n", mapping.size());
if (mapping.empty()) {
SRV_INF("%s", " no models found on the system (visit https://llama.app/models for suggestions)\n");
} else {
for (const auto & [name, inst] : mapping) {
bool has_custom = custom_names.find(name) != custom_names.end();
const std::string source = server_model_source_to_string(inst.meta.source);
std::string info;
if (!inst.meta.aliases.empty()) info += " (aliases: " + join_set(inst.meta.aliases) + ")";
if (!inst.meta.tags.empty()) info += " [tags: " + join_set(inst.meta.tags) + "]";
SRV_INF(" %c %s%s\n", has_custom ? '*' : ' ', name.c_str(), info.c_str());
SRV_INF(" [%10s] %s%s\n", source.c_str(), name.c_str(), info.c_str());
}
}
};
auto apply_stop_timeout = [&]() {

View file

@ -102,6 +102,8 @@ int llama_server(int argc, char ** argv) {
// touch it. lifecycle is symmetric, stop_gc() runs in clean_up() before backend free
server_stream_session_manager_start();
SRV_INF("%s", "initializing ...\n");
if (!common_params_parse(argc, argv, params, LLAMA_EXAMPLE_SERVER)) {
return 1;
}
@ -320,11 +322,7 @@ int llama_server(common_params & params, int argc, char ** argv) {
};
if (params.cors_origins == "*" && params.api_keys.empty()) {
SRV_WRN("%s", "-----------------\n");
SRV_WRN("%s", "CORS is set to allow all origins ('*') and no API key is set\n");
SRV_WRN("%s", "this can be a security risk (cross-origin attacks)\n");
SRV_WRN("%s", "more info: https://github.com/ggml-org/llama.cpp/pull/25655\n");
SRV_WRN("%s", "-----------------\n");
SRV_WRN("%s", "security: no API key is set and CORS allows all origins (see https://github.com/ggml-org/llama.cpp/pull/25655)\n");
}
// CORS proxy (EXPERIMENTAL, only used by the Web UI for MCP)
@ -372,14 +370,13 @@ int llama_server(common_params & params, int argc, char ** argv) {
ctx_http.post("/tools", ex_wrapper(res_403));
}
if (warn_names.size() > 0) {
SRV_WRN("%s", "-----------------\n");
SRV_WRN("%s", "the following feature(s) are enabled:\n");
if (!warn_names.empty()) {
std::string features;
for (const auto & name : warn_names) {
SRV_WRN(" %s\n", name.c_str());
if (!features.empty()) features += ", ";
features += name;
}
SRV_WRN("%s", "do not expose the server to untrusted environments\n");
SRV_WRN("%s", "-----------------\n");
SRV_WRN("security: %s enabled - do not expose to untrusted environments\n", features.c_str());
}
//
@ -517,8 +514,7 @@ int llama_server(common_params & params, int argc, char ** argv) {
// TODO: remove this in the future
// check the string to also handle the .sock case
if (string_ends_with(ctx_http.listening_address, ":8080")) {
SRV_WRN("%s", "NOTICE: server default port will be changed to :9931 in a future release\n");
SRV_WRN("%s", " ref: https://github.com/ggml-org/llama.cpp/pull/26508\n");
SRV_WRN("%s", "notice: server default port will be changed to :9931 in a future release (ref: https://github.com/ggml-org/llama.cpp/pull/26508)\n");
}
if (is_router_server) {