From 6a7da90b519ca6d3a697d8bdd194bbab7b9775f4 Mon Sep 17 00:00:00 2001 From: Aleksander Grygier Date: Fri, 9 Oct 2026 13:07:49 +0200 Subject: [PATCH] server : report the trained context in the models listing update_caps already resolves the model file offline to read its modalities, so it now reads the trained context from the same GGUF metadata, and GET /models reports it as context_length when it is known. A router listing then carries the context without any Hub request, which lets the UI sort and filter by it offline. Assisted-by: pi:zai-org/GLM-5.3-Flash --- tools/server/server-models.cpp | 9 +++++++++ tools/server/server-models.h | 1 + 2 files changed, 10 insertions(+) diff --git a/tools/server/server-models.cpp b/tools/server/server-models.cpp index 35d21c8767..454522f04f 100644 --- a/tools/server/server-models.cpp +++ b/tools/server/server-models.cpp @@ -585,6 +585,11 @@ void server_model_meta::update_caps(const common_params & base) { // offline discovery cannot see video; a loaded model reports it architecture = server_model_architecture_json(inp_image, inp_audio, false, output_modalities); + + // the trained context, read from the GGUF metadata + if (!params.model.path.empty()) { + n_ctx_train = common_get_gguf_n_ctx_train(params.model.path); + } } // @@ -2125,6 +2130,10 @@ void server_models_routes::init_routes() { // TODO: add other fields, may require reading GGUF metadata }; + if (meta.n_ctx_train > 0) { + model_info["context_length"] = meta.n_ctx_train; + } + // merge with loaded_info from the child process if available if (meta.is_running()) { for (auto it = meta.loaded_info.begin(); it != meta.loaded_info.end(); ++it) { diff --git a/tools/server/server-models.h b/tools/server/server-models.h index 238955e7c4..f415b7c5fd 100644 --- a/tools/server/server-models.h +++ b/tools/server/server-models.h @@ -86,6 +86,7 @@ struct server_model_meta { int stop_timeout = 0; // seconds to wait before force-killing the model instance during shutdown bool hidden = false; // hidden from GET /models, but still accept if requested json architecture = server_model_architecture_json(false, false, false, {"text"}); + uint32_t n_ctx_train = 0; // trained context, read from the GGUF metadata; 0 when unknown bool is_ready() const { return status == SERVER_MODEL_STATUS_LOADED;