mirror of
https://github.com/ggml-org/llama.cpp.git
synced 2026-10-11 07:20:33 +02:00
server : report the trained context in the models listing
update_caps already resolves the model file offline to read its modalities, so it now reads the trained context from the same GGUF metadata, and GET /models reports it as context_length when it is known. A router listing then carries the context without any Hub request, which lets the UI sort and filter by it offline. Assisted-by: pi:zai-org/GLM-5.3-Flash
This commit is contained in:
@@ -585,6 +585,11 @@ void server_model_meta::update_caps(const common_params & base) {
|
||||
|
||||
// offline discovery cannot see video; a loaded model reports it
|
||||
architecture = server_model_architecture_json(inp_image, inp_audio, false, output_modalities);
|
||||
|
||||
// the trained context, read from the GGUF metadata
|
||||
if (!params.model.path.empty()) {
|
||||
n_ctx_train = common_get_gguf_n_ctx_train(params.model.path);
|
||||
}
|
||||
}
|
||||
|
||||
//
|
||||
@@ -2125,6 +2130,10 @@ void server_models_routes::init_routes() {
|
||||
// TODO: add other fields, may require reading GGUF metadata
|
||||
};
|
||||
|
||||
if (meta.n_ctx_train > 0) {
|
||||
model_info["context_length"] = meta.n_ctx_train;
|
||||
}
|
||||
|
||||
// merge with loaded_info from the child process if available
|
||||
if (meta.is_running()) {
|
||||
for (auto it = meta.loaded_info.begin(); it != meta.loaded_info.end(); ++it) {
|
||||
|
||||
@@ -86,6 +86,7 @@ struct server_model_meta {
|
||||
int stop_timeout = 0; // seconds to wait before force-killing the model instance during shutdown
|
||||
bool hidden = false; // hidden from GET /models, but still accept if requested
|
||||
json architecture = server_model_architecture_json(false, false, false, {"text"});
|
||||
uint32_t n_ctx_train = 0; // trained context, read from the GGUF metadata; 0 when unknown
|
||||
|
||||
bool is_ready() const {
|
||||
return status == SERVER_MODEL_STATUS_LOADED;
|
||||
|
||||
Reference in New Issue
Block a user