Commit 4d60b4d08 for llama.cpp

commit 4d60b4d087103cd96fe8feeb557709cd8998cd56
Author: Adrien Gallouët <angt@huggingface.co>
Date:   Mon Oct 5 17:08:29 2026 +0200

    common, server : report model input/output modalities in GET /models (#29987)

    Signed-off-by: Adrien Gallouët <angt@huggingface.co>

diff --git a/common/arg.cpp b/common/arg.cpp
index 9ed1455b4..14d82f695 100644
--- a/common/arg.cpp
+++ b/common/arg.cpp
@@ -682,7 +682,10 @@ void common_models_handler_apply(common_models_handler & handler, common_params
             // if HF repo is a preset repo, we simply run server in router mode with the preset.ini file
             params.models_preset_hf = params.model.hf_repo; // only for showing a warning
             params.models_preset    = hf_cache::finalize_file(plan.preset);
-            params.model = common_params_model{}; // make sure to clear model, so server starts in router mode
+            // clear the model so the server starts in router mode
+            params.model.path.clear();
+            params.model.hf_repo.clear();
+            params.model.docker_repo.clear();
         });
     }

diff --git a/common/common.cpp b/common/common.cpp
index 463137e5a..c301c22ca 100644
--- a/common/common.cpp
+++ b/common/common.cpp
@@ -1,4 +1,5 @@
 #include "ggml.h"
+#include "ggml-cpp.h"
 #include "gguf.h"

 #include "build-info.h"
@@ -1191,6 +1192,41 @@ common_decision_type common_get_decision_type(const struct llama_model * model)
     return common_decision_type_from_string(buf);
 }

+common_decision_type common_get_decision_type(const std::string & fname) {
+    struct gguf_init_params gguf_params = {
+        /* .no_alloc = */ true,
+        /* .ctx      = */ nullptr,
+    };
+
+    gguf_context_ptr gguf_ctx(gguf_init_from_file(fname.c_str(), gguf_params));
+    if (!gguf_ctx) {
+        return COMMON_DECISION_TYPE_UNKNOWN; // missing or unreadable file
+    }
+
+    std::string arch;
+    const int64_t arch_id = gguf_find_key(gguf_ctx.get(), "general.architecture");
+    if (arch_id < 0) {
+        return COMMON_DECISION_TYPE_UNKNOWN; // no architecture in the metadata
+    }
+    if (gguf_get_kv_type(gguf_ctx.get(), arch_id) != GGUF_TYPE_STRING) {
+        return COMMON_DECISION_TYPE_UNKNOWN; // malformed metadata
+    }
+    arch = gguf_get_val_str(gguf_ctx.get(), arch_id);
+    if (arch.empty()) {
+        return COMMON_DECISION_TYPE_UNKNOWN;
+    }
+
+    const std::string key = arch + ".decision.type";
+    const int64_t type_id = gguf_find_key(gguf_ctx.get(), key.c_str());
+    if (type_id < 0) {
+        return COMMON_DECISION_TYPE_NONE;
+    }
+    if (gguf_get_kv_type(gguf_ctx.get(), type_id) != GGUF_TYPE_STRING) {
+        return COMMON_DECISION_TYPE_UNKNOWN; // malformed metadata
+    }
+    return common_decision_type_from_string(gguf_get_val_str(gguf_ctx.get(), type_id));
+}
+
 common_init_result::common_init_result(common_params & params, bool model_only) :
     pimpl(new impl{}) {
     auto mparams = common_model_params_to_llama(params);
diff --git a/common/common.h b/common/common.h
index e1ef70a9e..2f50d90c6 100644
--- a/common/common.h
+++ b/common/common.h
@@ -965,6 +965,10 @@ enum common_decision_type {

 common_decision_type common_get_decision_type(const struct llama_model * model);

+// same as above, but reads a GGUF file; it does not load the model
+// returns COMMON_DECISION_TYPE_UNKNOWN if the file is missing, unreadable, or invalid
+common_decision_type common_get_decision_type(const std::string & fname);
+
 // note: defines the model, context, samplers, ets. lifetimes
 struct common_init_result {
     common_init_result(common_params & params, bool model_only = false);
diff --git a/tools/server/README.md b/tools/server/README.md
index d2d6ab2be..4ab9238b2 100644
--- a/tools/server/README.md
+++ b/tools/server/README.md
@@ -1248,6 +1248,30 @@ Returns information about the loaded model. See [OpenAI Models API documentation

 The returned list always has one single element. The `meta` field can be `null` (for example, while the model is still loading).

+Each object in `data` has an `architecture` object. It has two string arrays:
+
+- `input_modalities` lists what the model can read. It always has `text`, plus each media type that the model supports.
+- `output_modalities` lists what the model can produce.
+
+One output value is special:
+
+| Value | Meaning |
+|---|---|
+| `decisions` | The model is a native decision model. Serve it with [`/v1/systemone`](#post-v1systemone-typesafe-compatible-system-one-api). |
+
+A language model that classifies with prompts does not get `decisions`. Only native decision models do.
+
+Check for membership. Tolerate values that you do not know:
+
+```js
+const useSystemOne =
+    model.architecture?.output_modalities?.includes("decisions") === true;
+```
+
+Without decision metadata, `output_modalities` is `["text"]`. This default is for compatibility only. It does not mean that the model can generate text. Values can change. New combinations such as `["text", "decisions"]` use the same shape.
+
+The router returns the same `architecture` object in [`GET /models`](#get-models-list-available-models). You can find a native decision model without a probe or a model load. This works for unloaded and sleeping models too. Older servers can omit `architecture`. If it is absent, use the legacy behavior of your client.
+
 By default, model `id` field is the path to model file, specified via `-m`. You can set a custom value for model `id` field via `--alias` argument. For example, `--alias gpt-4o-mini`.

 Example:
@@ -1259,6 +1283,10 @@ Example:
         {
             "id": "../models/Meta-Llama-3.1-8B-Instruct-Q4_K_M.gguf",
             "object": "model",
+            "architecture": {
+                "input_modalities": ["text"],
+                "output_modalities": ["text"]
+            },
             "created": 1735142223,
             "owned_by": "llamacpp",
             "meta": {
@@ -1990,6 +2018,37 @@ Note:
     - If a model is not running, it will be added or updated according to the source
 2. When the model is loaded, the info from `/v1/models` is forwarded to router's `/v1/models`. This includes metadata about the model and the runtime instance.

+Each object in `data` has the same `architecture` object as [`GET /v1/models`](#get-v1models-openai-compatible-model-info-api) of a direct server. The server computes both arrays offline. It does not load the model, download files, or run inference. `output_modalities` comes from the GGUF metadata. `input_modalities` comes from the projector file. A native decision model shows `decisions` before its first load, after unload, and while it sleeps:
+
+```json
+{
+  "object": "list",
+  "data": [
+    {
+      "id": "my-decision-model",
+      "object": "model",
+      "tags": ["local"],
+      "architecture": {
+        "input_modalities": ["text"],
+        "output_modalities": ["decisions"]
+      },
+      "status": {
+        "value": "unloaded"
+      }
+    }
+  ]
+}
+```
+
+The values work like this:
+
+- A loaded model reports both arrays. Its values replace the cached values in full.
+- The cache keeps the values across sleep and unload. A known decision model stays advertised.
+- Before the first report, the values come from the offline computation.
+- Offline computation cannot see video. Only a loaded model reports `video` in `input_modalities`.
+- If the metadata or the model file is not available, both arrays are `["text"]`.
+- A source or preset refresh computes both arrays again. A replaced model does not keep old values.
+
 The `status` object can be:

 ```json
diff --git a/tools/server/server-common.cpp b/tools/server/server-common.cpp
index 076a85741..4b73aa908 100644
--- a/tools/server/server-common.cpp
+++ b/tools/server/server-common.cpp
@@ -143,6 +143,47 @@ const char * get_media_marker() {
     return marker.c_str();
 }

+//
+// model output modalities
+//
+
+std::vector<std::string> server_model_output_modalities(common_decision_type decision_type) {
+    switch (decision_type) {
+        case COMMON_DECISION_TYPE_OPENJEV:
+        case COMMON_DECISION_TYPE_LEV:
+        case COMMON_DECISION_TYPE_KEV:
+        case COMMON_DECISION_TYPE_NIMBLE:
+        case COMMON_DECISION_TYPE_LAYA:
+        case COMMON_DECISION_TYPE_CLEF:
+            return {"decisions"};
+        default:
+            // fallback when there is no decision type or the metadata is bad
+            return {"text"};
+    }
+}
+
+json server_model_architecture_json(
+        bool inp_image,
+        bool inp_audio,
+        bool inp_video,
+        const std::vector<std::string> & output_modalities) {
+    std::vector<std::string> input_modalities = {"text"};
+    if (inp_image) {
+        input_modalities.push_back("image");
+    }
+    if (inp_audio) {
+        input_modalities.push_back("audio");
+    }
+    if (inp_video) {
+        input_modalities.push_back("video");
+    }
+
+    return {
+        {"input_modalities",  input_modalities},
+        {"output_modalities", output_modalities},
+    };
+}
+
 //
 // lora utils
 //
diff --git a/tools/server/server-common.h b/tools/server/server-common.h
index 20cdeacf2..6165d871c 100644
--- a/tools/server/server-common.h
+++ b/tools/server/server-common.h
@@ -105,6 +105,20 @@ std::string gen_tool_call_id();
 // get a random marker; note: each time the server restarts, the marker will be different
 const char * get_media_marker();

+//
+// model output modalities
+//
+
+// output modalities for architecture.output_modalities in GET /models
+std::vector<std::string> server_model_output_modalities(common_decision_type decision_type);
+
+// architecture object of GET /models; shared by the direct server and the router
+json server_model_architecture_json(
+        bool inp_image,
+        bool inp_audio,
+        bool inp_video,
+        const std::vector<std::string> & output_modalities);
+
 //
 // lora utils
 //
diff --git a/tools/server/server-context.cpp b/tools/server/server-context.cpp
index 13be63acd..d1af984e6 100644
--- a/tools/server/server-context.cpp
+++ b/tools/server/server-context.cpp
@@ -4537,40 +4537,41 @@ server_context_meta server_context::get_meta() const {
     const char * ftype_name = llama_ftype_name(llama_model_ftype(impl->model_tgt));

     return server_context_meta {
-        /* build_info             */ std::string(llama_build_info()),
-        /* model_name             */ impl->model_name,
-        /* model_aliases          */ impl->model_aliases,
-        /* model_tags             */ impl->model_tags,
-        /* model_path             */ impl->params_base.model.path,
-        /* has_mtmd               */ impl->mctx != nullptr,
-        /* has_inp_image          */ impl->chat_params.allow_image,
-        /* has_inp_audio          */ impl->chat_params.allow_audio,
-        /* has_inp_video          */ impl->chat_params.allow_video,
-        /* json_ui_settings       */ impl->json_ui_settings,
-        /* slot_n_ctx             */ impl->n_ctx_slot(),
-        /* pooling_type           */ llama_pooling_type(impl->ctx_tgt),
-
-        /* chat_params            */ impl->chat_params,
-        /* chat_template_caps     */ common_chat_templates_get_caps(impl->chat_params.tmpls.get()),
-
-        /* bos_token_str          */ bos_token_str,
-        /* eos_token_str          */ eos_token_str,
-        /* fim_pre_token          */ llama_vocab_fim_pre(impl->vocab),
-        /* fim_sub_token          */ llama_vocab_fim_suf(impl->vocab),
-        /* fim_mid_token          */ llama_vocab_fim_mid(impl->vocab),
-        /* fim_pad_token          */ llama_vocab_fim_pad(impl->vocab),
-        /* fim_rep_token          */ llama_vocab_fim_rep(impl->vocab),
-        /* fim_sep_token          */ llama_vocab_fim_sep(impl->vocab),
-
-        /* logit_bias_eog         */ impl->params_base.sampling.logit_bias_eog,
-
-        /* model_vocab_type       */ llama_vocab_type(impl->vocab),
-        /* model_vocab_n_tokens   */ llama_vocab_n_tokens(impl->vocab),
-        /* model_n_ctx_train      */ llama_model_n_ctx_train(impl->model_tgt),
-        /* model_n_embd_inp       */ llama_model_n_embd(impl->model_tgt),
-        /* model_n_params         */ llama_model_n_params(impl->model_tgt),
-        /* model_size             */ llama_model_size(impl->model_tgt),
-        /* model_ftype            */ ftype_name,
+        /* build_info              */ std::string(llama_build_info()),
+        /* model_name              */ impl->model_name,
+        /* model_aliases           */ impl->model_aliases,
+        /* model_tags              */ impl->model_tags,
+        /* model_path              */ impl->params_base.model.path,
+        /* model_output_modalities */ server_model_output_modalities(common_get_decision_type(impl->model_tgt)),
+        /* has_mtmd                */ impl->mctx != nullptr,
+        /* has_inp_image           */ impl->chat_params.allow_image,
+        /* has_inp_audio           */ impl->chat_params.allow_audio,
+        /* has_inp_video           */ impl->chat_params.allow_video,
+        /* json_ui_settings        */ impl->json_ui_settings,
+        /* slot_n_ctx              */ impl->n_ctx_slot(),
+        /* pooling_type            */ llama_pooling_type(impl->ctx_tgt),
+
+        /* chat_params             */ impl->chat_params,
+        /* chat_template_caps      */ common_chat_templates_get_caps(impl->chat_params.tmpls.get()),
+
+        /* bos_token_str           */ bos_token_str,
+        /* eos_token_str           */ eos_token_str,
+        /* fim_pre_token           */ llama_vocab_fim_pre(impl->vocab),
+        /* fim_sub_token           */ llama_vocab_fim_suf(impl->vocab),
+        /* fim_mid_token           */ llama_vocab_fim_mid(impl->vocab),
+        /* fim_pad_token           */ llama_vocab_fim_pad(impl->vocab),
+        /* fim_rep_token           */ llama_vocab_fim_rep(impl->vocab),
+        /* fim_sep_token           */ llama_vocab_fim_sep(impl->vocab),
+
+        /* logit_bias_eog          */ impl->params_base.sampling.logit_bias_eog,
+
+        /* model_vocab_type        */ llama_vocab_type(impl->vocab),
+        /* model_vocab_n_tokens    */ llama_vocab_n_tokens(impl->vocab),
+        /* model_n_ctx_train       */ llama_model_n_ctx_train(impl->model_tgt),
+        /* model_n_embd_inp        */ llama_model_n_embd(impl->model_tgt),
+        /* model_n_params          */ llama_model_n_params(impl->model_tgt),
+        /* model_size              */ llama_model_size(impl->model_tgt),
+        /* model_ftype             */ ftype_name,
     };
 }

@@ -4898,6 +4899,11 @@ static json get_res_model_info(const server_context_meta & meta) {
         {"aliases",  meta.model_aliases},
         {"tags",     meta.model_tags},
         {"object",   "model"},
+        {"architecture", server_model_architecture_json(
+            meta.has_inp_image,
+            meta.has_inp_audio,
+            meta.has_inp_video,
+            meta.model_output_modalities)},
         {"created",  std::time(0)},
         {"owned_by", "llamacpp"},
         {"meta",     {
diff --git a/tools/server/server-context.h b/tools/server/server-context.h
index c554bb95b..7fe003531 100644
--- a/tools/server/server-context.h
+++ b/tools/server/server-context.h
@@ -19,6 +19,7 @@ struct server_context_meta {
     std::set<std::string> model_aliases;
     std::set<std::string> model_tags;
     std::string model_path;
+    std::vector<std::string> model_output_modalities; // output modalities for GET /models
     bool has_mtmd;
     bool has_inp_image;
     bool has_inp_audio;
diff --git a/tools/server/server-models.cpp b/tools/server/server-models.cpp
index d42b523b2..35d21c876 100644
--- a/tools/server/server-models.cpp
+++ b/tools/server/server-models.cpp
@@ -537,9 +537,16 @@ void server_model_meta::update_args(common_preset_context & ctx_preset, std::str
     }
 }

-void server_model_meta::update_caps() {
+void server_model_meta::update_caps(const common_params & base) {
+    // reset to the default so a failed refresh cannot keep old values
+    architecture = server_model_architecture_json(false, false, false, {"text"});
+
+    // resolve the model file offline; do not download
+    common_params params;
+    params.model = base.model;
+    // --no-mmproj applies to child models and blocks auto-attached projectors
+    params.no_mmproj = base.no_mmproj;
     try {
-        common_params params;
         preset.apply_to_params(params, {
             "LLAMA_ARG_MODEL",
             "LLAMA_ARG_MODEL_URL",
@@ -551,16 +558,33 @@ void server_model_meta::update_caps() {
         });
         params.offline = true;
         common_models_handler handler = common_models_handler_init(params, LLAMA_EXAMPLE_SERVER);
-        common_models_handler_apply(handler, params); // note: this won't download the model because offline=true
-        if (params.no_mmproj || params.mmproj.path.empty()) {
-            multimodal = { false, false };
-        } else {
-            multimodal = mtmd_get_cap_from_file(params.mmproj.path.c_str());
+        common_models_handler_apply(handler, params);
+    } catch (const std::exception & e) {
+        LOG_WRN("failed to resolve the model of '%s': %s\n", name.c_str(), e.what());
+        return;
+    }
+
+    // read the output modalities from the GGUF metadata
+    std::vector<std::string> output_modalities = {"text"};
+    if (!params.model.path.empty()) {
+        output_modalities = server_model_output_modalities(common_get_decision_type(params.model.path));
+    }
+
+    bool inp_image = false;
+    bool inp_audio = false;
+    try {
+        if (!params.no_mmproj && !params.mmproj.path.empty()) {
+            mtmd_caps caps = mtmd_get_cap_from_file(params.mmproj.path.c_str());
+            inp_image = caps.inp_vision;
+            inp_audio = caps.inp_audio;
         }
     } catch (const std::exception & e) {
-        LOG_WRN("failed to initialize common_params for multimodal capability detection: %s\n", e.what());
-        multimodal = { false, false };
+        LOG_WRN("failed to read the multimodal capabilities of '%s': %s\n", name.c_str(), e.what());
+        // keep the output modalities from the GGUF metadata
     }
+
+    // offline discovery cannot see video; a loaded model reports it
+    architecture = server_model_architecture_json(inp_image, inp_audio, false, output_modalities);
 }

 //
@@ -651,7 +675,7 @@ void server_models::add_model(server_model_meta && meta) {
     }

     meta.update_args(ctx_preset, bin_path); // render args
-    meta.update_caps();
+    meta.update_caps(base_params);
     std::string name = meta.name;
     mapping[name] = instance_t{
         /* subproc */ std::make_shared<server_subproc>(),
@@ -841,7 +865,6 @@ void server_models::load_models() {
                 /* progress      */ {},
                 /* exit_code     */ 0,
                 /* stop_timeout  */ DEFAULT_STOP_TIMEOUT,
-                /* multimodal    */ mtmd_caps{false, false},
                 // /* need_download */ false,
             };
             add_model(std::move(meta));
@@ -962,7 +985,7 @@ void server_models::load_models() {

             inst.meta.exit_code = 0; // clear failed state so the model can be reloaded
             inst.meta.update_args(ctx_preset, bin_path);
-            inst.meta.update_caps();
+            inst.meta.update_caps(base_params);
         }

         // add models that are new in this reload, load-on-startup is not honored here since a
@@ -983,7 +1006,6 @@ void server_models::load_models() {
                     /* progress      */ {},
                     /* exit_code     */ 0,
                     /* stop_timeout  */ DEFAULT_STOP_TIMEOUT,
-                    /* multimodal    */ mtmd_caps{false, false},
                     // /* need_download */ false,
                 };
                 add_model(std::move(meta));
@@ -1307,6 +1329,27 @@ void server_models::update_status(const std::string & name, const update_status_
         }
         if (!args.loaded_info.is_null()) {
             meta.loaded_info = args.loaded_info;
+            // the child replaces both arrays in full; a bad or missing value changes nothing
+            if (args.loaded_info.contains("architecture") && args.loaded_info.at("architecture").is_object()) {
+                const json & child_arch = args.loaded_info.at("architecture");
+                for (const char * key : { "input_modalities", "output_modalities" }) {
+                    if (!child_arch.contains(key) || !child_arch.at(key).is_array()) {
+                        continue;
+                    }
+                    std::vector<std::string> modalities;
+                    bool valid = true;
+                    for (const auto & m : child_arch.at(key)) {
+                        if (!m.is_string()) {
+                            valid = false;
+                            break;
+                        }
+                        modalities.push_back(m.get<std::string>());
+                    }
+                    if (valid) {
+                        meta.architecture[key] = std::move(modalities);
+                    }
+                }
+            }
         }
         if (!args.progress.is_null()) {
             meta.progress = args.progress;
@@ -2067,19 +2110,6 @@ void server_models_routes::init_routes() {
                 status["failed"]    = true;
             }

-            // pi coding agent multimodal compatibility
-            json input_modalities = json::array({"text"});
-            if (meta.multimodal.inp_vision) {
-                input_modalities.push_back("image");
-            }
-            if (meta.multimodal.inp_audio) {
-                input_modalities.push_back("audio");
-            }
-            json architecture {
-                {"input_modalities",  input_modalities},
-                {"output_modalities", json::array({"text"})},
-            };
-
             json model_info = json {
                 {"id",            meta.name},
                 {"aliases",       meta.aliases},
@@ -2088,7 +2118,7 @@ void server_models_routes::init_routes() {
                 {"owned_by",      "llamacpp"}, // for OAI-compat
                 {"created",       t},          // for OAI-compat
                 {"status",        status},
-                {"architecture",  architecture},
+                {"architecture",  meta.architecture},
                 {"source",        server_model_source_to_string(meta.source)},
                 {"can_remove",    meta.source == SERVER_MODEL_SOURCE_CACHE},
                 // {"need_download", meta.need_download},
diff --git a/tools/server/server-models.h b/tools/server/server-models.h
index 0a6999ee3..238955e7c 100644
--- a/tools/server/server-models.h
+++ b/tools/server/server-models.h
@@ -84,8 +84,8 @@ struct server_model_meta {
     json progress; // reflect load or download progress info, if any
     int exit_code = 0; // exit code of the model instance process (only valid if status == FAILED)
     int stop_timeout = 0; // seconds to wait before force-killing the model instance during shutdown
-    mtmd_caps multimodal; // multimodal capabilities
     bool hidden = false; // hidden from GET /models, but still accept if requested
+    json architecture = server_model_architecture_json(false, false, false, {"text"});

     bool is_ready() const {
         return status == SERVER_MODEL_STATUS_LOADED;
@@ -104,7 +104,7 @@ struct server_model_meta {
     }

     void update_args(common_preset_context & ctx_presets, std::string bin_path);
-    void update_caps();
+    void update_caps(const common_params & base);
 };

 struct server_models_routes;