Commit a657f7e98 for llama.cpp

commit a657f7e981ff8764d2ccce1e74ec3c7e9bf2cfa2
Author: Tarek Dakhran <tarek@liquid.ai>
Date:   Thu Oct 8 02:07:49 2026 +0200

    model : add LiquidAI/d1-omni-600M decision model (#30114)

    * model : add LiquidAI/d1-omni-600M decision model

    Assisted-by: Claude Opus 5.5

    * mtmd : keep conformer GLU sigmoid on CUDA

    Assisted-by: Claude Opus 5.5

    * server : take d1omni audio through images and input_audio, scope memory-less lfm2 to non-causal

    Assisted-by: Claude Opus 5.5

    * common : rename decision type d1omni to lfm2-d1-omni, server : make images an alias of files

    Assisted-by: Claude Opus 5.5

diff --git a/common/common.cpp b/common/common.cpp
index 0e891cea6..28ea8680e 100644
--- a/common/common.cpp
+++ b/common/common.cpp
@@ -1171,6 +1171,7 @@ static const std::map<common_decision_type, std::string> COMMON_DECISION_TYPE_NA
     { COMMON_DECISION_TYPE_CLEF,           "clef"          },
     { COMMON_DECISION_TYPE_PPLX_DECIDER,   "pplx-decider"  },
     { COMMON_DECISION_TYPE_LFM2_D1,        "lfm2-d1"       },
+    { COMMON_DECISION_TYPE_LFM2_D1_OMNI,   "lfm2-d1-omni"  },
 };

 static common_decision_type common_decision_type_from_string(const std::string & str) {
@@ -1284,7 +1285,8 @@ common_init_result::common_init_result(common_params & params, bool model_only)
     // these decision models return a score for each token via the embeddings output
     // TODO: maybe improve this in the future
     const auto decision_type = common_get_decision_type(model);
-    if (decision_type == COMMON_DECISION_TYPE_LAYA || decision_type == COMMON_DECISION_TYPE_KEV || decision_type == COMMON_DECISION_TYPE_CLEF) {
+    if (decision_type == COMMON_DECISION_TYPE_LAYA || decision_type == COMMON_DECISION_TYPE_KEV || decision_type == COMMON_DECISION_TYPE_CLEF ||
+        decision_type == COMMON_DECISION_TYPE_LFM2_D1_OMNI) {
         params.embedding    = true;
         params.pooling_type = LLAMA_POOLING_TYPE_NONE;

diff --git a/common/common.h b/common/common.h
index 6f8acf31d..0a85f11f9 100644
--- a/common/common.h
+++ b/common/common.h
@@ -964,6 +964,7 @@ enum common_decision_type {
     COMMON_DECISION_TYPE_CLEF,    // all questions in one prompt, score of option i read from the embeddings output at row i
     COMMON_DECISION_TYPE_PPLX_DECIDER, // same as openjev, label codes of 1 or 2 letters
     COMMON_DECISION_TYPE_LFM2_D1, // same as openjev, the labels depend on the question type
+    COMMON_DECISION_TYPE_LFM2_D1_OMNI, // same as laya, other prompt layout
     COMMON_DECISION_TYPE_UNKNOWN, // a decision model of a type that is not supported
 };

diff --git a/conversion/__init__.py b/conversion/__init__.py
index 34c56bbee..72960be1a 100644
--- a/conversion/__init__.py
+++ b/conversion/__init__.py
@@ -162,6 +162,7 @@ TEXT_MODEL_MAP: dict[str, str] = {
     "Lfm2BidirectionalModel": "lfm2",
     "Lfm2ForCausalLM": "lfm2",
     "D1Model": "lfm2",
+    "D1OmniModel": "lfm2",
     "Lfm2Model": "lfm2",
     "Lfm2MoeForCausalLM": "lfm2",
     "Llama4ForCausalLM": "llama",
@@ -335,6 +336,7 @@ MMPROJ_MODEL_MAP: dict[str, str] = {
     "KimiK25ForConditionalGeneration": "kimivl",
     "KimiVLForConditionalGeneration": "kimivl",
     "Lfm2AudioForConditionalGeneration": "lfm2",
+    "D1OmniModel": "lfm2",
     "Lfm2VlForConditionalGeneration": "lfm2",
     "LightOnOCRForConditionalGeneration": "lighton_ocr",
     "Llama4ForConditionalGeneration": "llama4",
diff --git a/conversion/lfm2.py b/conversion/lfm2.py
index d4343675e..113c91116 100644
--- a/conversion/lfm2.py
+++ b/conversion/lfm2.py
@@ -161,6 +161,121 @@ class LFM2ColBertModel(LFM2Model):
         yield f"{self.dense_tensor_name}.weight", tensor.clone()


+def _is_d1_omni_checkpoint(dir_model: Path) -> bool:
+    if not (dir_model / "config.json").is_file():
+        return False
+    with open(dir_model / "config.json", encoding="utf-8") as f:
+        return json.load(f).get("model_type") == "d1_omni"
+
+
+@ModelBase.register_hparams_loader(_is_d1_omni_checkpoint)
+def _load_d1_omni_hparams(dir_model: Path) -> dict[str, Any]:
+    logger.info("gguf: detected d1-omni checkpoint")
+    hparams = ModelBase.load_hparams(dir_model, False, guess=False)
+    text = hparams["text_config"]
+    n_layer, n_layer_head = text["num_hidden_layers"], hparams["head_layers"]
+    # the trunk uses the LFM2 FFN sizing, the head blocks are appended with a plain 4x MLP
+    n_ff = int(text["block_ffn_dim_multiplier"] * int(2 * text["intermediate_size"] / 3))
+    n_ff = text["block_multiple_of"] * ((n_ff + text["block_multiple_of"] - 1) // text["block_multiple_of"])
+    text["num_hidden_layers"] = n_layer + n_layer_head
+    text["intermediate_size"] = [n_ff] * n_layer + [4 * text["hidden_size"]] * n_layer_head
+    text["block_auto_adjust_ff_dim"] = False
+    return hparams
+
+
+@ModelBase.register("D1OmniModel")
+@ModelBase.example("LiquidAI/d1-omni-600M")
+class D1OmniModel(LFM2Model):
+    model_arch = gguf.MODEL_ARCH.LFM2
+
+    # the server cuts the text to these lengths, see server-decision.cpp
+    _MAX_LENGTH = 16384
+    _IMAGE_TEXT_LENGTH = 896
+    _AUDIO_TEXT_LENGTH = 15360
+
+    def set_vocab(self):
+        super().set_vocab()
+        # the systemone template writes the BOS, after the media
+        self.gguf_writer.remove_key(gguf.Keys.Tokenizer.ADD_BOS)
+        self.gguf_writer.add_add_bos_token(False)
+        self.gguf_writer.add_token_type_count(3)  # choice, score, noul
+        self.gguf_writer.add_chat_template([{"name": "systemone", "template": self._systemone_template()}])
+
+    @staticmethod
+    def _systemone_template() -> str:
+        # follows prompt.py of the model repo, the server cuts each marked piece to its token budget
+        # the media (images, or an audio clip if audio is true) come first
+        description = jinja_str_or_json("o.description")
+        has_description = "o.description is not none and o.description != ''"
+        yes_no = "{{ 'yes' if o.key == 'true' else 'no' }}"
+        option_code = "{% if loop.index0 < 10 %}00{% elif loop.index0 < 100 %}0{% endif %}{{ loop.index0 }}"
+        option = (
+            "{% if type == 'choice' and audio %}option_" + option_code + ": "
+            "{% if " + has_description + " %}" + description + "{% else %}{{ o.key }}{% endif %}"
+            "{% elif type == 'choice' %}{{ o.key }}{% if " + has_description + " %}: " + description + "{% endif %}"
+            "{% elif type == 'score' %}level {{ o.key }}: " + description
+            + "{% elif audio %}{{ o.key }}: " + yes_no
+            + "{% else %}{{ o.key }}: {% if " + has_description + " %}" + description
+            + "{% elif images and not ns.criteria %}" + yes_no
+            + "{% elif o.key == 'true' %}yes, the statement holds"
+            "{% else %}no, the statement does not hold{% endif %}{% endif %}"
+        )
+        state = "{% if state is string %}{{ state }}{% elif state is not none %}{{ state | tojson }}{% elif audio %}{}{% endif %}"
+        return (
+            "{% set ns = namespace(criteria=false) %}"
+            "{% for o in options %}{% if o.description is not none %}{% set ns.criteria = true %}{% endif %}{% endfor %}"
+            "{% for image in images %}{{ image }}{% endfor %}{{ sep }}"
+            "<|startoftext|><|reserved_7|>{{ sep }}{{ mark_state }}" + state
+            + "{{ sep }}{{ mark_question }}<|reserved_8|>" + jinja_str_or_json("instructions")
+            + "{% for o in options %}{{ sep }}<|reserved_9|><|mask|>{{ sep }}{{ mark_option }} " + option
+            + "{{ sep }}<|reserved_10|>{% endfor %}{{ sep }}<|reserved_11|>"
+        )
+
+    def set_gguf_parameters(self):
+        lengths = (self.hparams["max_length"], self.hparams["image_text_length"], self.hparams["audio_text_length"])
+        if lengths != (self._MAX_LENGTH, self._IMAGE_TEXT_LENGTH, self._AUDIO_TEXT_LENGTH):
+            raise ValueError(f"unexpected text lengths: {lengths}")
+        n_head, n_layer_head = self.hparams["num_attention_heads"], self.hparams["head_layers"]
+        self.hparams["num_key_value_heads"] = [
+            self.hparams["num_key_value_heads"] if t != "conv" else 0 for t in self.hparams["layer_types"]
+        ] + [n_head] * n_layer_head
+
+        # the head needs per-layer sizes, LFM2Model writes a single feed forward length
+        TextModel.set_gguf_parameters(self)
+        self.gguf_writer.add_vocab_size(self.hparams["vocab_size"])
+        self.gguf_writer.add_shortconv_l_cache(self.hparams["conv_L_cache"])
+        self.gguf_writer.add_layer_norm_eps(1e-5)  # nn.LayerNorm of the head
+        self.gguf_writer.add_causal_attention(False)
+
+        self.gguf_writer.add_decision_type(gguf.DecisionType.LFM2_D1_OMNI)
+        self.gguf_writer.add_decision_block_count(n_layer_head)
+        # "choice:3-5" -> "choice.3_5", "choice:11+" -> "choice.11"
+        for name, value in self.hparams["temperatures"].items():
+            self.gguf_writer.add_decision_temperature(name.replace(":", ".").replace("-", "_").rstrip("+"), value)
+
+    @classmethod
+    def filter_tensors(cls, item: tuple[str, Callable[[], Tensor]]) -> tuple[str, Callable[[], Tensor]] | None:
+        name, gen = item
+
+        if name.startswith(("vision.", "audio.")):
+            return None
+
+        name = name.replace("encoder.", "model.", 1) if name.startswith("encoder.") else name
+        name = name.replace("head.head.layers.", "head.layers.").replace("in_proj_", "in_proj.")
+        name = name.removeprefix("head.") if name.startswith(("head.type_emb", "head.scorer")) else name
+
+        return super().filter_tensors((name, gen))
+
+    def modify_tensors(self, data_torch: Tensor, name: str, bid: int | None) -> Iterable[tuple[str, Tensor]]:
+        if name.startswith("head.layers.") and bid is not None:
+            # the head blocks come after the trunk blocks
+            suffix = name.split(".", 3)[3]
+            bid += self.block_count - self.hparams["head_layers"]
+            name = f"head.layers.{bid}.{suffix}"
+
+        yield from super().modify_tensors(data_torch, name, bid)
+
+
 @ModelBase.register("Lfm2MoeForCausalLM")
 @ModelBase.example("LiquidAI/LFM2-8B-A1B")
 class LFM2MoeModel(TextModel):
@@ -276,6 +391,58 @@ class LFM2VLModel(MmprojModel):
         yield from super().modify_tensors(data_torch, name, bid)


+@ModelBase.register("D1OmniModel")
+@ModelBase.example("LiquidAI/d1-omni-600M")
+class D1OmniMmprojModel(ConformerAudioModel):
+    has_vision_encoder = True
+    has_audio_encoder  = True
+
+    def __init__(self, *args, **kwargs):
+        super().__init__(*args, **kwargs)
+        assert self.hparams_vision is not None and self.hparams_audio is not None
+        # dynamic resolution, as LFM2VLModel
+        self.hparams_vision["image_size"] = 256
+        # the images are normalized to [-1, 1] (vision.py of the model repo)
+        self.preprocessor_config = {**self.preprocessor_config, "image_mean": [0.5] * 3, "image_std": [0.5] * 3}
+        self.hparams_audio["hidden_size"] = self.hparams_audio["d_model"]
+        self.hparams_audio["intermediate_size"] = self.hparams_audio["d_model"] * self.hparams_audio["ff_expansion_factor"]
+        self.hparams_audio["num_attention_heads"] = self.hparams_audio["n_heads"]
+
+    def set_gguf_parameters(self):
+        super().set_gguf_parameters()
+        self.gguf_writer.add_clip_vision_projector_type(gguf.VisionProjectorType.D1OMNI_V)
+        self.gguf_writer.add_vision_attention_layernorm_eps(self.find_vparam(["layer_norm_eps"]))
+        self.gguf_writer.add_vision_projector_scale_factor(self.global_config.get("downsample_factor", 2))
+        self.gguf_writer.add_vision_use_gelu(True)
+
+        assert self.hparams_audio is not None
+        self.gguf_writer.add_clip_audio_projector_type(gguf.VisionProjectorType.D1OMNI_A)
+        self.gguf_writer.add_audio_num_mel_bins(self.hparams_audio["feat_in"])
+        self.gguf_writer.add_audio_attention_layernorm_eps(1e-5)
+
+    @classmethod
+    def filter_tensors(cls, item: tuple[str, Callable[[], Tensor]]) -> tuple[str, Callable[[], Tensor]] | None:
+        name, gen = item
+
+        if name.startswith(("encoder.", "head.")):
+            return None
+
+        name = name.replace("vision.tower.", "vision_tower.").replace("vision.projector.", "multi_modal_projector.")
+        name = name.replace("audio.encoder.", "conformer.")
+        # the residual block continues the adapter: norm, linear, gelu, linear, then norm, down, up
+        for old, new in (("adapter.norm", 0), ("adapter.linear_1", 1), ("adapter.linear_2", 3),
+                         ("residual.ln", 4), ("residual.down", 5), ("residual.up", 6)):
+            name = name.replace(f"audio.{old}.", f"audio_adapter.model.{new}.")
+
+        return super().filter_tensors((name, gen))
+
+    def modify_tensors(self, data_torch: Tensor, name: str, bid: int | None) -> Iterable[tuple[str, Tensor]]:
+        if "patch_embedding.weight" in name:
+            data_torch = data_torch.view(data_torch.shape[0], 16, 16, 3).permute(0, 3, 1, 2)
+
+        yield from super().modify_tensors(data_torch, name, bid)
+
+
 @ModelBase.register("Lfm2AudioForConditionalGeneration")
 @ModelBase.example("LiquidAI/LFM2.5-Audio-1.5B", "LiquidAI/LFM2-Audio-1.5B")
 class LFM2AudioModel(ConformerAudioModel):
diff --git a/gguf-py/gguf/constants.py b/gguf-py/gguf/constants.py
index c181cb44a..fcf325fe9 100644
--- a/gguf-py/gguf/constants.py
+++ b/gguf-py/gguf/constants.py
@@ -5172,6 +5172,10 @@ MODEL_TENSORS: dict[MODEL_ARCH, list[MODEL_TENSOR]] = {
         MODEL_TENSOR.ATTN_OUT,
         MODEL_TENSOR.OUTPUT,
         MODEL_TENSOR.DENSE_2_OUT, # LFM2-ColBert-350M
+        MODEL_TENSOR.TOKEN_TYPES, # decision head
+        MODEL_TENSOR.CLS,
+        MODEL_TENSOR.CLS_NORM,
+        MODEL_TENSOR.CLS_OUT,
     ],
     MODEL_ARCH.LFM2MOE: [
         MODEL_TENSOR.TOKEN_EMBD,
@@ -6072,6 +6076,7 @@ class DecisionType:
     CLEF    = "clef"     # joint head over all questions, one score per option
     PPLX_DECIDER = "pplx-decider"  # same as openjev, label codes of 1 or 2 letters
     LFM2_D1 = "lfm2-d1"  # same as openjev, the labels depend on the question type
+    LFM2_D1_OMNI = "lfm2-d1-omni"  # same head as laya on a bidirectional LFM2 trunk, other prompt layout


 class VisionProjectorType:
@@ -6133,6 +6138,8 @@ class VisionProjectorType:
     GRANITE4_VISION = "granite4_vision"
     MUSE_GLIMMER   = "muse-glimmer"
     COHERE2V       = "cohere2v"
+    D1OMNI_V       = "d1omni_v"  # lfm2 vision, without separator tokens
+    D1OMNI_A       = "d1omni_a"  # lfm2a audio, with a residual block after the projector


 # Items here are (block size, type size)
diff --git a/src/llama-model.cpp b/src/llama-model.cpp
index 9f05a66e4..bf04945bf 100644
--- a/src/llama-model.cpp
+++ b/src/llama-model.cpp
@@ -2368,6 +2368,11 @@ ggml_tensor * llama_model::get_rope_factors(const llama_cparams & cparams, int i
 llama_memory_i * llama_model::create_memory(const llama_memory_params & params, const llama_cparams & cparams) const {
     llama_memory_i * res;

+    // the non-causal LFM2 decision graph reads the whole prompt in one batch, nothing is kept
+    if (arch == LLM_ARCH_LFM2 && !hparams.causal_attn && hparams.n_layer_decision > 0) {
+        return nullptr;
+    }
+
     switch (arch) {
         // Models that need specific instantiation should be handled in the
         // switch statement
diff --git a/src/models/lfm2.cpp b/src/models/lfm2.cpp
index 07b71ccd3..2488c0746 100644
--- a/src/models/lfm2.cpp
+++ b/src/models/lfm2.cpp
@@ -4,6 +4,9 @@

 #include <algorithm>

+// question types of a decision model: choice, score, noul
+static const uint32_t N_DECISION_TYPES = 3;
+
 void llama_model_lfm2::load_arch_hparams(llama_model_loader & ml) {
     ml.get_key(LLM_KV_SHORTCONV_L_CACHE,           hparams.n_shortconv_l_cache);
     ml.get_key(LLM_KV_ATTENTION_LAYERNORM_RMS_EPS, hparams.f_norm_rms_eps);
@@ -23,6 +26,15 @@ void llama_model_lfm2::load_arch_hparams(llama_model_loader & ml) {
         default:    type = LLM_TYPE_UNKNOWN;
     }

+    ml.get_key(LLM_KV_DECISION_BLOCK_COUNT, hparams.n_layer_decision, false);
+    if (hparams.n_layer_decision > 0) {
+        if (hparams.n_layer_decision >= hparams.n_layer() || hparams.causal_attn) {
+            throw std::runtime_error("invalid decision head");
+        }
+        ml.get_key(LLM_KV_ATTENTION_LAYERNORM_EPS, hparams.f_norm_eps);
+        hparams.n_embd_out_impl = N_DECISION_TYPES;
+    }
+
     if (const auto is_swa = ml.get_key(LLM_KV_ATTENTION_SLIDING_WINDOW, hparams.n_swa, false); is_swa && hparams.n_swa > 0) {
         hparams.swa_type = LLAMA_SWA_TYPE_STANDARD;
         for (uint32_t il = 0; il < hparams.n_layer(); ++il) {
@@ -37,13 +49,49 @@ void llama_model_lfm2::load_arch_tensors(llama_model_loader &) {
     tok_embd = create_tensor(tn(LLM_TENSOR_TOKEN_EMBD, "weight"), {n_embd, n_vocab}, 0);

     output_norm = create_tensor(tn(LLM_TENSOR_OUTPUT_NORM_LFM2, "weight"), {n_embd}, 0);
-    output      = create_tensor(tn(LLM_TENSOR_OUTPUT,           "weight"), {n_embd, n_vocab}, TENSOR_NOT_REQUIRED);

-    if (output == NULL) {
-        output = create_tensor(tn(LLM_TENSOR_TOKEN_EMBD, "weight"), {n_embd, n_vocab}, TENSOR_DUPLICATED);
+    if (hparams.n_layer_decision > 0) {
+        // decision head: plain pre-norm blocks with biases
+        for (int i = n_layer - (int) hparams.n_layer_decision; i < n_layer; ++i) {
+            auto & layer = layers[i];
+            const int64_t n_ff_head = hparams.n_ff(i);
+
+            layer.attn_norm   = create_tensor(tn(LLM_TENSOR_ATTN_NORM, "weight", i), {n_embd}, 0);
+            layer.attn_norm_b = create_tensor(tn(LLM_TENSOR_ATTN_NORM, "bias",   i), {n_embd}, 0);
+
+            layer.wqkv   = create_tensor(tn(LLM_TENSOR_ATTN_QKV, "weight", i), {n_embd, 3 * n_embd}, 0);
+            layer.wqkv_b = create_tensor(tn(LLM_TENSOR_ATTN_QKV, "bias",   i), {3 * n_embd}, 0);
+            layer.wo     = create_tensor(tn(LLM_TENSOR_ATTN_OUT, "weight", i), {n_embd, n_embd}, 0);
+            layer.wo_b   = create_tensor(tn(LLM_TENSOR_ATTN_OUT, "bias",   i), {n_embd}, 0);
+
+            layer.ffn_norm   = create_tensor(tn(LLM_TENSOR_FFN_NORM, "weight", i), {n_embd}, 0);
+            layer.ffn_norm_b = create_tensor(tn(LLM_TENSOR_FFN_NORM, "bias",   i), {n_embd}, 0);
+            layer.ffn_up     = create_tensor(tn(LLM_TENSOR_FFN_UP,   "weight", i), {n_embd, n_ff_head}, 0);
+            layer.ffn_up_b   = create_tensor(tn(LLM_TENSOR_FFN_UP,   "bias",   i), {n_ff_head}, 0);
+            layer.ffn_down   = create_tensor(tn(LLM_TENSOR_FFN_DOWN, "weight", i), {n_ff_head, n_embd}, 0);
+            layer.ffn_down_b = create_tensor(tn(LLM_TENSOR_FFN_DOWN, "bias",   i), {n_embd}, 0);
+        }
+
+        if (n_token_types != N_DECISION_TYPES) {
+            throw std::runtime_error("decision model must have one token type per question type");
+        }
+        type_embd = create_tensor(tn(LLM_TENSOR_TOKEN_TYPES, "weight"), {n_embd, n_token_types}, 0);
+
+        cls_norm   = create_tensor(tn(LLM_TENSOR_CLS_NORM, "weight"), {n_embd},    0);
+        cls_norm_b = create_tensor(tn(LLM_TENSOR_CLS_NORM, "bias"),   {n_embd},    0);
+        cls        = create_tensor(tn(LLM_TENSOR_CLS,      "weight"), {n_embd, n_embd}, 0);
+        cls_b      = create_tensor(tn(LLM_TENSOR_CLS,      "bias"),   {n_embd},    0);
+        cls_out    = create_tensor(tn(LLM_TENSOR_CLS_OUT,  "weight"), {n_embd, 1}, 0);
+        cls_out_b  = create_tensor(tn(LLM_TENSOR_CLS_OUT,  "bias"),   {1},         0);
+    } else {
+        output = create_tensor(tn(LLM_TENSOR_OUTPUT, "weight"), {n_embd, n_vocab}, TENSOR_NOT_REQUIRED);
+
+        if (output == NULL) {
+            output = create_tensor(tn(LLM_TENSOR_TOKEN_EMBD, "weight"), {n_embd, n_vocab}, TENSOR_DUPLICATED);
+        }
     }

-    for (int i = 0; i < n_layer; ++i) {
+    for (int i = 0; i < n_layer - (int) hparams.n_layer_decision; ++i) {
         auto & layer = layers[i];

         const bool is_moe_layer = i >= static_cast<int>(hparams.n_layer_dense_lead);
@@ -87,6 +135,9 @@ void llama_model_lfm2::load_arch_tensors(llama_model_loader &) {
 }

 std::unique_ptr<llm_graph_context> llama_model_lfm2::build_arch_graph(const llm_graph_params & params) const {
+    if (hparams.n_layer_decision > 0) {
+        return std::make_unique<graph_decision>(*this, params);
+    }
     if (hparams.swa_type == LLAMA_SWA_TYPE_STANDARD) {
         return std::make_unique<graph<true>>(*this, params);
     } else {
@@ -294,6 +345,242 @@ llama_model_lfm2::graph<iswa>::graph(const llama_model & model, const llm_graph_
     ggml_build_forward_expand(gf, cur);
 }

+// media entries (an image or audio prefix) are embeddings, text entries are tokens
+static bool lfm2_is_media(const llama_ubatch & ubatch, int64_t i) {
+    return ubatch.is_mixed() ? ubatch.type[i] != 0 : ubatch.token == nullptr;
+}
+
+// non-causal within a sequence, the media never reads the text, so it is a function of the media alone
+// in the head, the text and the media only read their own kind
+class llm_graph_input_attn_media : public llm_graph_input_attn_no_cache {
+public:
+    llm_graph_input_attn_media(const llama_hparams & hparams, const llama_cparams & cparams, bool is_head) :
+        llm_graph_input_attn_no_cache(hparams, cparams), is_head(is_head) {}
+
+    void set_input(const llama_ubatch * ubatch) override {
+        const int64_t n_tokens = ubatch->n_tokens;
+
+        std::vector<bool> is_media(n_tokens);
+        for (int64_t i = 0; i < n_tokens; ++i) {
+            is_media[i] = lfm2_is_media(*ubatch, i);
+        }
+
+        const auto fill_mask = [&](auto * data, auto zero, auto ninf) {
+            for (int64_t i1 = 0; i1 < n_tokens; ++i1) {
+                for (int64_t i0 = 0; i0 < n_tokens; ++i0) {
+                    bool visible = ubatch->seq_id[i0][0] == ubatch->seq_id[i1][0];
+                    if (is_head) {
+                        visible = visible && is_media[i0] == is_media[i1];
+                    } else {
+                        visible = visible && !(is_media[i1] && !is_media[i0]);
+                    }
+                    data[i1 * n_tokens + i0] = visible ? zero : ninf;
+                }
+            }
+        };
+
+        GGML_ASSERT(ggml_backend_buffer_is_host(self_kq_mask->buffer));
+        if (self_kq_mask->type == GGML_TYPE_F16) {
+            fill_mask((ggml_fp16_t *) self_kq_mask->data, ggml_fp32_to_fp16(0.0f), ggml_fp32_to_fp16(-INFINITY));
+        } else {
+            fill_mask((float *) self_kq_mask->data, 0.0f, -INFINITY);
+        }
+    }
+
+    const bool is_head;
+};
+
+// 1 where the previous (next) token is the left (right) neighbor in the same sequence
+// the last media entry does not read the text on its right
+class llm_graph_input_conv_mask : public llm_graph_input_i {
+public:
+    void set_input(const llama_ubatch * ubatch) override {
+        const int64_t n_tokens = ubatch->n_tokens;
+
+        std::vector<float> data_left(n_tokens, 0.0f);
+        std::vector<float> data_right(n_tokens, 0.0f);
+        for (int64_t i = 0; i + 1 < n_tokens; ++i) {
+            const bool is_next = ubatch->seq_id[i][0] == ubatch->seq_id[i + 1][0] && ubatch->pos[i] + 1 == ubatch->pos[i + 1];
+            data_right[i]    = is_next && !(lfm2_is_media(*ubatch, i) && !lfm2_is_media(*ubatch, i + 1));
+            data_left[i + 1] = is_next;
+        }
+        ggml_backend_tensor_set(left,  data_left.data(),  0, ggml_nbytes(left));
+        ggml_backend_tensor_set(right, data_right.data(), 0, ggml_nbytes(right));
+    }
+
+    ggml_tensor * left  = nullptr; // F32 [1, n_tokens]
+    ggml_tensor * right = nullptr; // F32 [1, n_tokens]
+};
+
+llama_model_lfm2::graph_decision::graph_decision(const llama_model & model, const llm_graph_params & params) :
+    llm_graph_context(params) {
+    const int64_t n_embd_head = hparams.n_embd_head_v();
+    const int     n_layer_enc = n_layer - hparams.n_layer_decision;
+
+    ggml_tensor * cur = build_inp_embd(model.tok_embd);
+    cb(cur, "model.embed_tokens", -1);
+
+    ggml_tensor * inp_pos     = build_inp_pos();
+    ggml_tensor * inp_out_ids = build_inp_out_ids();
+
+    const auto type_mask = cparams.flash_attn ? GGML_TYPE_F16 : GGML_TYPE_F32;
+
+    llm_graph_input_attn_no_cache * inp_attn[2];
+    for (bool is_head : {false, true}) {
+        auto inp = std::make_unique<llm_graph_input_attn_media>(hparams, cparams, is_head);
+        inp->self_kq_mask = ggml_new_tensor_4d(ctx0, type_mask, n_tokens, n_tokens, 1, 1);
+        ggml_set_input(inp->self_kq_mask);
+        inp->self_kq_mask_cnv = inp->self_kq_mask;
+        inp_attn[is_head] = (llm_graph_input_attn_no_cache *) res->add_input(std::move(inp));
+    }
+
+    auto inp_conv = std::make_unique<llm_graph_input_conv_mask>();
+    inp_conv->left  = ggml_new_tensor_2d(ctx0, GGML_TYPE_F32, 1, n_tokens);
+    inp_conv->right = ggml_new_tensor_2d(ctx0, GGML_TYPE_F32, 1, n_tokens);
+    ggml_set_input(inp_conv->left);
+    ggml_set_input(inp_conv->right);
+    ggml_tensor * conv_left  = inp_conv->left;
+    ggml_tensor * conv_right = inp_conv->right;
+    res->add_input(std::move(inp_conv));
+
+    for (int il = 0; il < n_layer_enc; ++il) {
+        const auto & layer = model.layers[il];
+
+        ggml_tensor * inpL = cur;
+        cur = build_norm(cur, layer.attn_norm, NULL, LLM_NORM_RMS, il);
+        cb(cur, "model.layers.{}.operator_norm", il);
+
+        if (hparams.is_recr(il)) {
+            ggml_tensor * bcx = build_lora_mm(layer.shortconv.in_proj, cur);
+            cb(bcx, "model.layers.{}.conv.in_proj", il);
+
+            ggml_tensor * b = ggml_view_2d(ctx0, bcx, n_embd, n_tokens, bcx->nb[1], 0 * n_embd * ggml_element_size(bcx));
+            ggml_tensor * c = ggml_view_2d(ctx0, bcx, n_embd, n_tokens, bcx->nb[1], 1 * n_embd * ggml_element_size(bcx));
+            ggml_tensor * x = ggml_view_2d(ctx0, bcx, n_embd, n_tokens, bcx->nb[1], 2 * n_embd * ggml_element_size(bcx));
+
+            // centred 3-tap conv, a tap outside the sequence reads 0
+            ggml_tensor * bx  = ggml_mul(ctx0, b, x);
+            ggml_tensor * bxp = ggml_pad_ext(ctx0, bx, 0, 0, 1, 1, 0, 0, 0, 0);
+            ggml_tensor * prv = ggml_view_2d(ctx0, bxp, n_embd, n_tokens, bxp->nb[1], 0);
+            ggml_tensor * nxt = ggml_view_2d(ctx0, bxp, n_embd, n_tokens, bxp->nb[1], 2 * bxp->nb[1]);
+
+            GGML_ASSERT(hparams.n_shortconv_l_cache == 3);
+            ggml_tensor * taps = ggml_cont(ctx0, ggml_transpose(ctx0, layer.shortconv.conv));
+            ggml_tensor * tap0 = ggml_view_1d(ctx0, taps, n_embd, 0 * taps->nb[1]);
+            ggml_tensor * tap1 = ggml_view_1d(ctx0, taps, n_embd, 1 * taps->nb[1]);
+            ggml_tensor * tap2 = ggml_view_1d(ctx0, taps, n_embd, 2 * taps->nb[1]);
+
+            ggml_tensor * y = ggml_mul(ctx0, bx, tap1);
+            y = ggml_add(ctx0, y, ggml_mul(ctx0, ggml_mul(ctx0, prv, tap0), conv_left));
+            y = ggml_add(ctx0, y, ggml_mul(ctx0, ggml_mul(ctx0, nxt, tap2), conv_right));
+            cb(y, "model.layers.{}.conv.conv", il);
+
+            cur = build_lora_mm(layer.shortconv.out_proj, ggml_mul(ctx0, c, y));
+            cb(cur, "model.layers.{}.conv.out_proj", il);
+        } else {
+            auto [q, k, v] = build_qkv(layer, cur, n_embd_head, n_head, hparams.n_head_kv(il), il);
+
+            q = build_norm(q, layer.attn_q_norm, NULL, LLM_NORM_RMS, il);
+            k = build_norm(k, layer.attn_k_norm, NULL, LLM_NORM_RMS, il);
+
+            q = ggml_rope_ext(ctx0, q, inp_pos, nullptr, n_rot, rope_type, n_ctx_orig, freq_base, freq_scale, ext_factor,
+                              attn_factor, beta_fast, beta_slow);
+            k = ggml_rope_ext(ctx0, k, inp_pos, nullptr, n_rot, rope_type, n_ctx_orig, freq_base, freq_scale, ext_factor,
+                              attn_factor, beta_fast, beta_slow);
+
+            cur = build_attn(inp_attn[0],
+                    layer.wo, NULL, layer.wo_s,
+                    q, k, v, nullptr, nullptr, nullptr, 1.0f / sqrtf(float(n_embd_head)), il);
+            cb(cur, "model.layers.{}.self_attn.out_proj", il);
+        }
+
+        cur = ggml_add(ctx0, cur, inpL);
+
+        ggml_tensor * ffn_out = build_norm(cur, layer.ffn_norm, NULL, LLM_NORM_RMS, il);
+        ffn_out = build_ffn(ffn_out,
+                layer.ffn_up,   NULL, NULL,
+                layer.ffn_gate, NULL, NULL,
+                layer.ffn_down, NULL, NULL,
+                NULL, LLM_FFN_SILU, LLM_FFN_PAR, il);
+
+        cur = ggml_add(ctx0, cur, ffn_out);
+        cb(cur, "l_out", il);
+    }
+
+    cur = build_norm(cur, model.output_norm, NULL, LLM_NORM_RMS, -1);
+    cb(cur, "result_norm", -1);
+
+    cur = build_decision_head(model, cur, inp_attn[1], inp_out_ids);
+
+    res->t_embd = cur;
+    ggml_build_forward_expand(gf, cur);
+}
+
+// same as llama_model_modern_bert::graph::build_decision_head(), with the head counts of the head layers
+ggml_tensor * llama_model_lfm2::graph_decision::build_decision_head(
+        const llama_model & model,
+        ggml_tensor * inp,
+        llm_graph_input_attn_no_cache * inp_attn,
+        ggml_tensor * inp_out_ids) {
+    const int64_t n_embd_head = hparams.n_embd_head_v();
+    const int     n_layer_enc = n_layer - hparams.n_layer_decision;
+
+    ggml_tensor * scores = nullptr;
+
+    // the question type is not a graph input, so the head is evaluated for each of them
+    for (uint32_t it = 0; it < N_DECISION_TYPES; ++it) {
+        ggml_tensor * type_row = ggml_view_1d(ctx0, model.type_embd, n_embd, it * model.type_embd->nb[1]);
+        ggml_tensor * inpL = ggml_add(ctx0, inp, type_row);
+
+        for (int il = n_layer_enc; il < n_layer; ++il) {
+            const auto & layer = model.layers[il];
+
+            ggml_tensor * cur = build_norm(inpL, layer.attn_norm, layer.attn_norm_b, LLM_NORM, il);
+            cb(cur, "attn_norm", il);
+
+            // no positional encoding in the head
+            auto [Qcur, Kcur, Vcur] = build_qkv(layer, cur, n_embd_head, hparams.n_head(il), hparams.n_head_kv(il), il);
+
+            cur = build_attn(inp_attn,
+                        layer.wo, layer.wo_b, layer.wo_s,
+                        Qcur, Kcur, Vcur, nullptr, nullptr, nullptr, 1.0f/sqrtf(float(n_embd_head)), il);
+            cb(cur, "kqv_out", il);
+
+            if (il == n_layer - 1 && inp_out_ids) {
+                cur  = ggml_get_rows(ctx0,  cur, inp_out_ids);
+                inpL = ggml_get_rows(ctx0, inpL, inp_out_ids);
+            }
+
+            ggml_tensor * ffn_inp = ggml_add(ctx0, cur, inpL);
+            cb(ffn_inp, "ffn_inp", il);
+
+            cur = build_norm(ffn_inp, layer.ffn_norm, layer.ffn_norm_b, LLM_NORM, il);
+            cb(cur, "ffn_norm", il);
+
+            cur = build_ffn(cur,
+                    layer.ffn_up,   layer.ffn_up_b,   NULL,
+                    NULL,           NULL,             NULL,
+                    layer.ffn_down, layer.ffn_down_b, NULL,
+                    NULL,
+                    LLM_FFN_RELU,
+                    LLM_FFN_SEQ, il);
+
+            inpL = ggml_add(ctx0, cur, ffn_inp);
+        }
+
+        // scorer
+        ggml_tensor * cur = build_norm(inpL, model.cls_norm, model.cls_norm_b, LLM_NORM, -1);
+        cur = ggml_add(ctx0, build_lora_mm(model.cls, cur), model.cls_b);
+        cur = ggml_gelu_erf(ctx0, cur);
+        cur = ggml_add(ctx0, build_lora_mm(model.cls_out, cur), model.cls_out_b);
+
+        scores = scores ? ggml_concat(ctx0, scores, cur, 0) : cur;
+    }
+    cb(scores, "decision_scores", -1);
+
+    return scores;
+}
+
 // Explicit template instantiations
 template struct llama_model_lfm2::graph<true>;
 template struct llama_model_lfm2::graph<false>;
diff --git a/src/models/models.h b/src/models/models.h
index 1ef0c5156..d6ddf9d16 100644
--- a/src/models/models.h
+++ b/src/models/models.h
@@ -2186,6 +2186,17 @@ struct llama_model_lfm2 : public llama_model_base {
         graph(const llama_model & model, const llm_graph_params & params);
     };

+    // non-causal trunk without memory, then the decision head
+    struct graph_decision : public llm_graph_context {
+        graph_decision(const llama_model & model, const llm_graph_params & params);
+
+        ggml_tensor * build_decision_head(
+                const llama_model & model,
+                ggml_tensor * inp,
+                llm_graph_input_attn_no_cache * inp_attn,
+                ggml_tensor * inp_out_ids);
+    };
+
     std::unique_ptr<llm_graph_context> build_arch_graph(const llm_graph_params & params) const override;
 };

diff --git a/tools/mtmd/clip-impl.h b/tools/mtmd/clip-impl.h
index abf75d998..de171b7a5 100644
--- a/tools/mtmd/clip-impl.h
+++ b/tools/mtmd/clip-impl.h
@@ -476,6 +476,7 @@ enum projector_type {
     PROJECTOR_TYPE_MERALION,
     PROJECTOR_TYPE_MUSIC_FLAMINGO,
     PROJECTOR_TYPE_LFM2,
+    PROJECTOR_TYPE_D1OMNI_V,
     PROJECTOR_TYPE_KIMIVL,
     PROJECTOR_TYPE_PADDLEOCR,
     PROJECTOR_TYPE_LIGHTONOCR,
@@ -488,6 +489,7 @@ enum projector_type {
     PROJECTOR_TYPE_DEEPSEEKOCR2,
     PROJECTOR_TYPE_DEEPSEEK4V,
     PROJECTOR_TYPE_LFM2A,
+    PROJECTOR_TYPE_D1OMNI_A,
     PROJECTOR_TYPE_GLM4V,
     PROJECTOR_TYPE_GLM5V,
     PROJECTOR_TYPE_YOUTUVL,
@@ -544,6 +546,7 @@ static std::map<projector_type, std::string> PROJECTOR_TYPE_NAMES = {
     { PROJECTOR_TYPE_MERALION,          "meralion"},
     { PROJECTOR_TYPE_MUSIC_FLAMINGO,    "musicflamingo"},
     { PROJECTOR_TYPE_LFM2,              "lfm2"},
+    { PROJECTOR_TYPE_D1OMNI_V,          "d1omni_v"},
     { PROJECTOR_TYPE_KIMIVL,            "kimivl"},
     { PROJECTOR_TYPE_PADDLEOCR,         "paddleocr"},
     { PROJECTOR_TYPE_LIGHTONOCR,        "lightonocr"},
@@ -556,6 +559,7 @@ static std::map<projector_type, std::string> PROJECTOR_TYPE_NAMES = {
     { PROJECTOR_TYPE_DEEPSEEKOCR2,      "deepseekocr2"},
     { PROJECTOR_TYPE_DEEPSEEK4V,        "deepseek4v"},
     { PROJECTOR_TYPE_LFM2A,             "lfm2a"},
+    { PROJECTOR_TYPE_D1OMNI_A,          "d1omni_a"},
     { PROJECTOR_TYPE_GLM4V,             "glm4v"},
     { PROJECTOR_TYPE_GLM5V,             "glm5v"},
     { PROJECTOR_TYPE_YOUTUVL,           "youtuvl"},
diff --git a/tools/mtmd/clip-model.h b/tools/mtmd/clip-model.h
index 33f679fc0..a5ff38822 100644
--- a/tools/mtmd/clip-model.h
+++ b/tools/mtmd/clip-model.h
@@ -623,6 +623,10 @@ struct clip_model {
     ggml_tensor * mm_3_b = nullptr;
     ggml_tensor * mm_4_w = nullptr;
     ggml_tensor * mm_4_b = nullptr;
+    ggml_tensor * mm_5_w = nullptr;
+    ggml_tensor * mm_5_b = nullptr;
+    ggml_tensor * mm_6_w = nullptr;
+    ggml_tensor * mm_6_b = nullptr;

     // GLMV-Edge projection
     ggml_tensor * mm_model_adapter_conv_w = nullptr;
diff --git a/tools/mtmd/clip.cpp b/tools/mtmd/clip.cpp
index b7302e4bf..ca38d76df 100644
--- a/tools/mtmd/clip.cpp
+++ b/tools/mtmd/clip.cpp
@@ -939,6 +939,7 @@ static std::unique_ptr<clip_graph> clip_get_graph_builder(clip_ctx * ctx, const
         case PROJECTOR_TYPE_IDEFICS3:
         case PROJECTOR_TYPE_COHERE2V:
         case PROJECTOR_TYPE_LFM2:
+        case PROJECTOR_TYPE_D1OMNI_V:
         case PROJECTOR_TYPE_JANUS_PRO:
         case PROJECTOR_TYPE_PHI4:
             {
@@ -1073,6 +1074,7 @@ static std::unique_ptr<clip_graph> clip_get_graph_builder(clip_ctx * ctx, const
                 builder = std::make_unique<clip_graph_deepseekocr2>(ctx, img);
             } break;
         case PROJECTOR_TYPE_LFM2A:
+        case PROJECTOR_TYPE_D1OMNI_A:
             {
                 builder = std::make_unique<clip_graph_conformer>(ctx, img);
             } break;
@@ -1525,6 +1527,7 @@ struct clip_model_loader {
                         }
                     } break;
                 case PROJECTOR_TYPE_LFM2:
+                case PROJECTOR_TYPE_D1OMNI_V:
                     {
                         // default for older GGUFs
                         std::string resize_algo = "bilinear";
@@ -1540,6 +1543,10 @@ struct clip_model_loader {
                         }
                         hparams.image_resize_algo_rf = hparams.image_resize_algo;
                         hparams.image_resize_algo_ov = hparams.image_resize_algo;
+                        // the tiles stretch the image to the grid (d1-omni vision.py)
+                        if (model.proj_type == PROJECTOR_TYPE_D1OMNI_V) {
+                            hparams.image_pad_rf = PAD_NONE;
+                        }
                         get_u32(KEY_PROJ_SCALE_FACTOR, hparams.n_merge, false);
                         // ref: https://huggingface.co/LiquidAI/LFM2.5-VL-1.6B/blob/main/processor_config.json
                         hparams.set_limit_image_tokens(64, 256);
@@ -1980,6 +1987,7 @@ struct clip_model_loader {
                         hparams.set_warmup_n_tokens(32*32);
                     } break;
                 case PROJECTOR_TYPE_LFM2A:
+                case PROJECTOR_TYPE_D1OMNI_A:
                     {
                         // audio preprocessing params
                         hparams.audio_chunk_len        = 1; // in seconds
@@ -2799,6 +2807,7 @@ struct clip_model_loader {
                     model.mm_2_b = get_tensor(string_format(TN_LLAVA_PROJ, 2, "bias"));
                 } break;
             case PROJECTOR_TYPE_LFM2:
+            case PROJECTOR_TYPE_D1OMNI_V:
                 {
                     model.mm_input_norm_w = get_tensor(TN_MM_INP_NORM, false);
                     model.mm_input_norm_b = get_tensor(TN_MM_INP_NORM_B, false);
@@ -3411,6 +3420,7 @@ struct clip_model_loader {
                     model.mm_input_proj_w = get_tensor(string_format(TN_A_MM_INP_PROJ, "weight"));
                 } break;
             case PROJECTOR_TYPE_LFM2A:
+            case PROJECTOR_TYPE_D1OMNI_A:
                 {
                     for (int i : {0, 2, 3, 5, 6}) {
                         model.pre_encode_conv_X_w[i] = get_tensor(string_format(TN_CONV1D, i, "weight"));
@@ -3426,6 +3436,16 @@ struct clip_model_loader {
                     model.mm_3_w = get_tensor(string_format(TN_MM_AUDIO_MLP, 3, "weight"));
                     model.mm_3_b = get_tensor(string_format(TN_MM_AUDIO_MLP, 3, "bias"));

+                    // residual block after the projector: norm, down, up
+                    if (model.proj_type == PROJECTOR_TYPE_D1OMNI_A) {
+                        model.mm_4_w = get_tensor(string_format(TN_MM_AUDIO_MLP, 4, "weight"));
+                        model.mm_4_b = get_tensor(string_format(TN_MM_AUDIO_MLP, 4, "bias"));
+                        model.mm_5_w = get_tensor(string_format(TN_MM_AUDIO_MLP, 5, "weight"));
+                        model.mm_5_b = get_tensor(string_format(TN_MM_AUDIO_MLP, 5, "bias"));
+                        model.mm_6_w = get_tensor(string_format(TN_MM_AUDIO_MLP, 6, "weight"));
+                        model.mm_6_b = get_tensor(string_format(TN_MM_AUDIO_MLP, 6, "bias"));
+                    }
+
                     for (int il = 0; il < hparams.n_layer; ++il) {
                         auto & layer = model.layers[il];

@@ -4277,6 +4297,7 @@ int clip_n_output_tokens(const clip_ctx * ctx, const clip_image_f32 * img) {
                 n_patches = ctx->model.hparams.image_size / ctx->model.hparams.patch_size;
             } break;
         case PROJECTOR_TYPE_LFM2:
+        case PROJECTOR_TYPE_D1OMNI_V:
         case PROJECTOR_TYPE_KIMIVL:
         case PROJECTOR_TYPE_KIMIK25:
             {
@@ -4400,6 +4421,7 @@ int clip_n_output_tokens(const clip_ctx * ctx, const clip_image_f32 * img) {
             }
         } break;
         case PROJECTOR_TYPE_LFM2A:
+        case PROJECTOR_TYPE_D1OMNI_A:
             {
                 n_patches = ((((img->nx() + 1) / 2) + 1) / 2 + 1) / 2;
             } break;
@@ -5358,6 +5380,7 @@ bool clip_encode(struct clip_ctx * ctx, struct clip_encode_params * params) {
         case PROJECTOR_TYPE_GLMA:
         case PROJECTOR_TYPE_ULTRAVOX:
         case PROJECTOR_TYPE_LFM2:
+        case PROJECTOR_TYPE_D1OMNI_V:
         case PROJECTOR_TYPE_VOXTRAL:
         case PROJECTOR_TYPE_MERALION:
         case PROJECTOR_TYPE_MUSIC_FLAMINGO:
@@ -5617,6 +5640,7 @@ bool clip_encode(struct clip_ctx * ctx, struct clip_encode_params * params) {
                 }
             } break;
         case PROJECTOR_TYPE_LFM2A:
+        case PROJECTOR_TYPE_D1OMNI_A:
             {
                 GGML_ASSERT(imgs.entries.size() == 1);
                 const auto n_frames = clip_n_output_tokens(ctx, &imgs.entries.front());
@@ -6080,6 +6104,7 @@ int clip_n_mmproj_embd(const struct clip_ctx * ctx) {
             return ctx->model.mm_2_w->ne[1];
         case PROJECTOR_TYPE_GLMA:
         case PROJECTOR_TYPE_LFM2:
+        case PROJECTOR_TYPE_D1OMNI_V:
         case PROJECTOR_TYPE_KIMIVL:
         case PROJECTOR_TYPE_PADDLEOCR:
         case PROJECTOR_TYPE_KIMIK25:
@@ -6096,6 +6121,8 @@ int clip_n_mmproj_embd(const struct clip_ctx * ctx) {
             return ctx->model.mm_fc_w->ne[1];
         case PROJECTOR_TYPE_LFM2A:
             return ctx->model.position_embeddings->ne[0];
+        case PROJECTOR_TYPE_D1OMNI_A:
+            return ctx->model.mm_3_w->ne[1];
         case PROJECTOR_TYPE_GRANITE_SPEECH:
             return ctx->model.qf_proj_blocks[0].qf_proj_linear_w->ne[1];
         case PROJECTOR_TYPE_GRANITE4_VISION:
diff --git a/tools/mtmd/models/conformer.cpp b/tools/mtmd/models/conformer.cpp
index 18c3d27bc..9463f9f1c 100644
--- a/tools/mtmd/models/conformer.cpp
+++ b/tools/mtmd/models/conformer.cpp
@@ -4,7 +4,7 @@ ggml_cgraph * clip_graph_conformer::build() {
     const int n_frames   = img.nx();
     const int n_pos      = n_frames / 2;
     const int n_pos_embd = (((((n_frames + 1) / 2) + 1) / 2 + 1) / 2) * 2 - 1;
-    GGML_ASSERT(model.position_embeddings->ne[1] >= n_pos);
+    GGML_ASSERT(!model.position_embeddings || model.position_embeddings->ne[1] >= n_pos);

     ggml_tensor * pos_emb = ggml_new_tensor_2d(ctx0, GGML_TYPE_F32, 512, n_pos_embd);
     ggml_set_name(pos_emb, "pos_emb");
@@ -164,7 +164,7 @@ ggml_cgraph * clip_graph_conformer::build() {
             // TODO @ngxson : support this ops in ggml
             {
                 int64_t       d    = x->ne[0] / 2;
-                ggml_tensor * gate = ggml_sigmoid(ctx0, ggml_view_2d(ctx0, x, d, x->ne[1], x->nb[1], d * x->nb[0]));
+                ggml_tensor * gate = ggml_sigmoid(ctx0, ggml_cont(ctx0, ggml_view_2d(ctx0, x, d, x->ne[1], x->nb[1], d * x->nb[0])));
                 x                  = ggml_mul(ctx0, ggml_view_2d(ctx0, x, d, x->ne[1], x->nb[1], 0), gate);
                 x                  = ggml_cont(ctx0, ggml_transpose(ctx0, x));
             }
@@ -207,6 +207,13 @@ ggml_cgraph * clip_graph_conformer::build() {
     cb(cur, "audio_adapter.model.{}", 0);
     cur = build_ffn(cur, model.mm_1_w, model.mm_1_b, nullptr, nullptr, model.mm_3_w, model.mm_3_b, FFN_GELU_ERF, -1);

+    // d1omni_a: residual block after the projector
+    if (model.mm_4_w) {
+        ggml_tensor * x = build_norm(cur, model.mm_4_w, model.mm_4_b, NORM_TYPE_NORMAL, 1e-5, -1);
+        x   = build_ffn(x, model.mm_5_w, model.mm_5_b, nullptr, nullptr, model.mm_6_w, model.mm_6_b, FFN_GELU_ERF, -1);
+        cur = ggml_add(ctx0, cur, x);
+    }
+
     cb(cur, "projected", -1);

     ggml_build_forward_expand(gf, cur);
diff --git a/tools/mtmd/models/siglip.cpp b/tools/mtmd/models/siglip.cpp
index 96d21ceee..8cc901a28 100644
--- a/tools/mtmd/models/siglip.cpp
+++ b/tools/mtmd/models/siglip.cpp
@@ -4,7 +4,7 @@ ggml_cgraph * clip_graph_siglip::build() {
     ggml_tensor * inp = build_inp();

     ggml_tensor * learned_pos_embd = model.position_embeddings;
-    if (proj_type == PROJECTOR_TYPE_LFM2 || proj_type == PROJECTOR_TYPE_PHI4) {
+    if (proj_type == PROJECTOR_TYPE_LFM2 || proj_type == PROJECTOR_TYPE_D1OMNI_V || proj_type == PROJECTOR_TYPE_PHI4) {
         learned_pos_embd = resize_position_embeddings();
     }

@@ -55,7 +55,7 @@ ggml_cgraph * clip_graph_siglip::build() {
         cur = build_mm(model.mm_2_w, cur);
         cur = ggml_add(ctx0, cur, model.mm_2_b);

-    } else if (proj_type == PROJECTOR_TYPE_LFM2) {
+    } else if (proj_type == PROJECTOR_TYPE_LFM2 || proj_type == PROJECTOR_TYPE_D1OMNI_V) {
         // pixel unshuffle block
         const int scale_factor = model.hparams.n_merge;
         cur = build_patch_merge_permute(cur, scale_factor);
@@ -70,11 +70,12 @@ ggml_cgraph * clip_graph_siglip::build() {
             cur = ggml_add(ctx0, cur, model.mm_input_norm_b);
         }

+        // d1-omni uses the exact gelu in the projector
         cur = build_ffn(cur,
             model.mm_1_w, model.mm_1_b,
             nullptr, nullptr,
             model.mm_2_w, model.mm_2_b,
-            FFN_GELU,
+            proj_type == PROJECTOR_TYPE_D1OMNI_V ? FFN_GELU_ERF : FFN_GELU,
             -1);

     } else if (proj_type == PROJECTOR_TYPE_JANUS_PRO) {
diff --git a/tools/mtmd/mtmd-audio.cpp b/tools/mtmd/mtmd-audio.cpp
index 6c8b19441..1b4ea60bf 100644
--- a/tools/mtmd/mtmd-audio.cpp
+++ b/tools/mtmd/mtmd-audio.cpp
@@ -994,6 +994,42 @@ bool mtmd_audio_preprocessor_conformer::preprocess(const float *
     return true;
 }

+//
+// mtmd_audio_preprocessor_d1omni
+//
+
+bool mtmd_audio_preprocessor_d1omni::preprocess(const float *                 samples,
+                                                size_t                        n_samples,
+                                                std::vector<mtmd_audio_mel> & output) const {
+    if (n_samples == 0) {
+        return false;
+    }
+    const size_t n_max = 30 * hparams.audio_sample_rate;
+    const size_t n_min = hparams.audio_sample_rate / 2;
+
+    std::vector<float> buf(samples, samples + std::min(n_samples, n_max));
+    buf.resize(std::max(buf.size(), n_min), 0.0f);
+
+    if (!mtmd_audio_preprocessor_conformer::preprocess(buf.data(), buf.size(), output)) {
+        return false;
+    }
+
+    // the encoder reads one frame per hop, not the extra frame of the centre padding (NeMo: seq_len)
+    const int64_t n_frames = buf.size() / hparams.audio_hop_len;
+    for (auto & mel : output) {
+        if (mel.n_len <= n_frames) {
+            continue;
+        }
+        std::vector<float> data((size_t) mel.n_mel * n_frames);
+        for (int64_t j = 0; j < mel.n_mel; ++j) {
+            std::copy_n(mel.data.begin() + (size_t) j * mel.n_len, n_frames, data.begin() + (size_t) j * n_frames);
+        }
+        mel.n_len = n_frames;
+        mel.data  = std::move(data);
+    }
+    return true;
+}
+
 //
 // mtmd_audio_preprocessor_granite_speech
 //
diff --git a/tools/mtmd/mtmd-audio.h b/tools/mtmd/mtmd-audio.h
index 36db50526..4ca6cbd55 100644
--- a/tools/mtmd/mtmd-audio.h
+++ b/tools/mtmd/mtmd-audio.h
@@ -80,6 +80,12 @@ struct mtmd_audio_preprocessor_conformer : mtmd_audio_preprocessor {
     mtmd_audio_cache cache;
 };

+// same as conformer, the audio is cut to 30 s and padded to 0.5 s (d1-omni audio.py)
+struct mtmd_audio_preprocessor_d1omni : mtmd_audio_preprocessor_conformer {
+    using mtmd_audio_preprocessor_conformer::mtmd_audio_preprocessor_conformer;
+    bool preprocess(const float * samples, size_t n_samples, std::vector<mtmd_audio_mel> & output) const override;
+};
+
 struct mtmd_audio_preprocessor_granite_speech : mtmd_audio_preprocessor {
     mtmd_audio_preprocessor_granite_speech(const clip_ctx * ctx) : mtmd_audio_preprocessor(ctx) {}
     void initialize() override;
diff --git a/tools/mtmd/mtmd.cpp b/tools/mtmd/mtmd.cpp
index b2319b901..b6e331bc7 100644
--- a/tools/mtmd/mtmd.cpp
+++ b/tools/mtmd/mtmd.cpp
@@ -871,6 +871,11 @@ struct mtmd_context {
                     ov_img_first       = false;
                     image_preproc = std::make_unique<mtmd_image_preprocessor_lfm2>(ctx_v);
                 } break;
+            case PROJECTOR_TYPE_D1OMNI_V:
+                {
+                    // same tiles and thumbnail as lfm2, without separator tokens
+                    image_preproc = std::make_unique<mtmd_image_preprocessor_lfm2>(ctx_v);
+                } break;
             case PROJECTOR_TYPE_GLM4V:
                 {
                     // <|begin_of_image|> ... (image embeddings) ... <|end_of_image|>
@@ -982,6 +987,10 @@ struct mtmd_context {
                 {
                     audio_preproc = std::make_unique<mtmd_audio_preprocessor_conformer>(ctx_a);
                 } break;
+            case PROJECTOR_TYPE_D1OMNI_A:
+                {
+                    audio_preproc = std::make_unique<mtmd_audio_preprocessor_d1omni>(ctx_a);
+                } break;
             case PROJECTOR_TYPE_GRANITE_SPEECH:
                 {
                     audio_preproc = std::make_unique<mtmd_audio_preprocessor_granite_speech>(ctx_a);
diff --git a/tools/server/README.md b/tools/server/README.md
index 8b271395b..9231f06e8 100644
--- a/tools/server/README.md
+++ b/tools/server/README.md
@@ -1703,9 +1703,11 @@ Follows the [TypeSafe API](https://docs.typesafe.ai/api), streaming is not suppo

 *Options:*

-`state`: The content to evaluate. Can be a string, an object or an array. A value that is not a string is given to the model as JSON text. For lfm2-d1, it can be `null`, for example to ask about images only.
+`state`: The content to evaluate. Can be a string, an object or an array. A value that is not a string is given to the model as JSON text. For lfm2-d1 and lfm2-d1-omni, it can be `null`, for example to ask about images only.

-`images`: Optional. An array of images, the maximum number may be limited depending on the model. Each one is a data URL (`data:image/...;base64,...`). See the image input section below.
+`files`: Optional. An array of input files, the maximum number may be limited depending on the model. Each one is a data URL (`data:image/...;base64,...`). For audio-capable models, it can be audio clips (`data:audio/...;base64,...`). See the image input section below.
+
+`images`: Optional. An alias of `files`.

 `questions`: An object that maps a question id to a question. Each question has these fields:

@@ -1718,20 +1720,20 @@ Follows the [TypeSafe API](https://docs.typesafe.ai/api), streaming is not suppo

 The questions of a request are answered independently, an answer does not depend on the other questions. The exception is clef: it reads all the questions in one prompt and decides them jointly.

-The number of options of a `choice` question is limited by the model, for example: 52 for openjev, 255 for laya, clef, pplx-decider and lfm2-d1. For laya, long questions and options are truncated to the token budget the model was trained with.
+The number of options of a `choice` question is limited by the model, for example: 52 for openjev, 255 for laya, clef, pplx-decider, lfm2-d1 and lfm2-d1-omni. For laya, long questions and options are truncated to the token budget the model was trained with.

-For laya and clef, the whole prompt is evaluated in one batch: it must fit in `--ubatch-size`. A server that runs clef only serves this endpoint, text generation is not available.
+For laya, clef and lfm2-d1-omni, the whole prompt is evaluated in one batch: it must fit in `--ubatch-size`. An lfm2-d1-omni prompt is cut to 16384 tokens. A server that runs clef only serves this endpoint, text generation is not available.

 *Image input:*

-Image input needs a model that supports it (for example: openjev, clef, pplx-decider, lfm2-d1) and its multimodal projector, see `--mmproj`.
+Image input needs a model that supports it (for example: openjev, clef, pplx-decider, lfm2-d1, lfm2-d1-omni) and its multimodal projector, see `--mmproj`.

 Images can be given in two ways, and both can be used in the same request:

-- The `images` field.
-- A `state` made of chat messages, either an array of messages or an object with a `messages` array. An `image_url` part in the `content` of a message is taken as an image, in the same format as chat completions. Only data URLs are accepted.
+- The `files` field, or its alias `images`.
+- A `state` made of chat messages, either an array of messages or an object with a `messages` array. An `image_url` part in the `content` of a message is taken as an image, in the same format as chat completions. Only data URLs are accepted. For lfm2-d1-omni, an `input_audio` part is taken as an audio clip, as base64 data.

-All the images are placed before the state in the prompt, the ones from `images` first. The image parts are removed from the state.
+All the images are placed before the state in the prompt, the ones from `files` and `images` first. The image parts are removed from the state.

 *Response:*

diff --git a/tools/server/server-common.cpp b/tools/server/server-common.cpp
index f56595231..d364ba898 100644
--- a/tools/server/server-common.cpp
+++ b/tools/server/server-common.cpp
@@ -157,6 +157,7 @@ std::vector<std::string> server_model_output_modalities(common_decision_type dec
         case COMMON_DECISION_TYPE_CLEF:
         case COMMON_DECISION_TYPE_PPLX_DECIDER:
         case COMMON_DECISION_TYPE_LFM2_D1:
+        case COMMON_DECISION_TYPE_LFM2_D1_OMNI:
             return {"decisions"};
         default:
             // fallback when there is no decision type or the metadata is bad
diff --git a/tools/server/server-context.cpp b/tools/server/server-context.cpp
index 6f3519215..e3270747b 100644
--- a/tools/server/server-context.cpp
+++ b/tools/server/server-context.cpp
@@ -5660,7 +5660,7 @@ void server_routes::init_routes() {
                 scores.push_back(result->scores);
                 n_tokens += result->n_tokens;
             }
-            answers[question.id] = decision.format_answer(question, scores);
+            answers[question.id] = decision.format_answer(question, scores, !files.empty());
         }

         res->ok(json{
diff --git a/tools/server/server-decision.cpp b/tools/server/server-decision.cpp
index 71657a359..11ad66a19 100644
--- a/tools/server/server-decision.cpp
+++ b/tools/server/server-decision.cpp
@@ -126,6 +126,12 @@ void server_decision_context::init(const llama_model * model) {
     } else if (model_type == COMMON_DECISION_TYPE_LFM2_D1) {
         n_options_max   = 255;
         noul_true_first = true;
+    } else if (model_type == COMMON_DECISION_TYPE_LFM2_D1_OMNI) {
+        token_marker = llama_vocab_mask(vocab);
+        if (token_marker == LLAMA_TOKEN_NULL) {
+            throw std::runtime_error("decision model has no mask token");
+        }
+        n_options_max = 255;
     } else {
         throw std::runtime_error("unsupported decision model type: " + type_name);
     }
@@ -139,8 +145,8 @@ void server_decision_context::init(const llama_model * model) {
 //

 std::vector<server_decision_question> server_decision_context::parse_questions(const json & body) const {
-    // d1 accepts a null state (images only)
-    if (!body.contains("state") || (body.at("state").is_null() && type != COMMON_DECISION_TYPE_LFM2_D1)) {
+    // lfm2-d1 and lfm2-d1-omni accept a null state, for example to ask about images only
+    if (!body.contains("state") || (body.at("state").is_null() && type != COMMON_DECISION_TYPE_LFM2_D1 && type != COMMON_DECISION_TYPE_LFM2_D1_OMNI)) {
         throw std::invalid_argument("\"state\" must be provided");
     }
     if (!body.contains("questions") || !body.at("questions").is_object() || body.at("questions").empty()) {
@@ -193,7 +199,15 @@ std::vector<server_decision_question> server_decision_context::parse_questions(c
                 throw err("\"criteria\" must be an object");
             }
             for (const char * key : {"false", "true"}) {
-                question.options.push_back({key, criteria.is_object() && criteria.contains(key) ? criteria.at(key) : json()});
+                json description;
+                if (criteria.is_object() && criteria.contains(key)) {
+                    description = criteria.at(key);
+                } else if (criteria.is_object() && type == COMMON_DECISION_TYPE_LFM2_D1_OMNI) {
+                    // lfm2-d1-omni also reads the descriptions under "no" and "yes"
+                    const char * alias = std::string(key) == "true" ? "yes" : "no";
+                    description = criteria.contains(alias) ? criteria.at(alias) : json();
+                }
+                question.options.push_back({key, description});
             }
             if (noul_true_first) {
                 std::swap(question.options[0], question.options[1]);
@@ -217,9 +231,10 @@ std::vector<server_decision_question> server_decision_context::parse_questions(c

 static const size_t DECISION_MAX_IMAGES = 8;

+// any media, mtmd tells an audio clip from an image by its content
 static void decision_load_image(const json & url, std::vector<raw_buffer> & files) {
-    if (!url.is_string() || !string_starts_with(url.get<std::string>(), "data:image/")) {
-        throw std::invalid_argument("images must be data URLs (data:image/...;base64,...)");
+    if (!url.is_string() || !string_starts_with(url.get<std::string>(), "data:")) {
+        throw std::invalid_argument("images must be data URLs (data:image/...;base64,... or data:audio/...;base64,...)");
     }
     if (files.size() >= DECISION_MAX_IMAGES) {
         throw std::invalid_argument(string_format("too many images, the maximum is %zu", DECISION_MAX_IMAGES));
@@ -227,15 +242,30 @@ static void decision_load_image(const json & url, std::vector<raw_buffer> & file
     handle_media(files, url.get<std::string>(), "");
 }

+// audio is base64 data, as a data URL or not (OpenAI input_audio)
+static void decision_load_audio(const json & data, std::vector<raw_buffer> & files) {
+    if (!data.is_string() || string_starts_with(data.get<std::string>(), "http") || string_starts_with(data.get<std::string>(), "file://")) {
+        throw std::invalid_argument("audio must be base64 data");
+    }
+    if (files.size() >= DECISION_MAX_IMAGES) {
+        throw std::invalid_argument(string_format("too many media files, the maximum is %zu", DECISION_MAX_IMAGES));
+    }
+    handle_media(files, data.get<std::string>(), "");
+}
+
 json server_decision_context::parse_state(const json & body, std::vector<raw_buffer> & files) const {
     if (body.contains("videos") && !body.at("videos").is_null() && !body.at("videos").empty()) {
         throw std::invalid_argument("\"videos\" is not supported");
     }
-    if (body.contains("images") && !body.at("images").is_null()) {
-        if (!body.at("images").is_array()) {
-            throw std::invalid_argument("\"images\" must be an array");
+    // "images" is an alias of "files"
+    for (const char * key : {"files", "images"}) {
+        if (!body.contains(key) || body.at(key).is_null()) {
+            continue;
+        }
+        if (!body.at(key).is_array()) {
+            throw std::invalid_argument(string_format("\"%s\" must be an array", key));
         }
-        for (const auto & url : body.at("images")) {
+        for (const auto & url : body.at(key)) {
             decision_load_image(url, files);
         }
     }
@@ -247,7 +277,7 @@ json server_decision_context::parse_state(const json & body, std::vector<raw_buf
         return state;
     }

-    // chat messages: take the image parts out of the content
+    // chat messages: take the image and audio parts out of the content
     json messages_out = json::array();
     for (const auto & msg : messages) {
         if (!msg.is_object() || !msg.contains("content") || !msg.at("content").is_array()) {
@@ -259,6 +289,9 @@ json server_decision_context::parse_state(const json & body, std::vector<raw_buf
             if (part.is_object() && json_value(part, "type", std::string()) == "image_url" && part.contains("image_url")) {
                 const json & image_url = part.at("image_url");
                 decision_load_image(image_url.is_object() && image_url.contains("url") ? image_url.at("url") : image_url, files);
+            } else if (part.is_object() && json_value(part, "type", std::string()) == "input_audio" && part.contains("input_audio")) {
+                const json input_audio = json_value(part, "input_audio", json::object());
+                decision_load_audio(input_audio.contains("data") ? input_audio.at("data") : json_value(input_audio, "url", json()), files);
             } else {
                 content.push_back(part);
             }
@@ -364,6 +397,41 @@ static std::string decision_kev_text(const json & val) {
     return std::regex_replace(decision_kev_render(val), re_special, "<\xC2\xA6$1\xC2\xA6>");
 }

+// lfm2-d1-omni: special tokens written in the input must not be parsed as such, in keys too (d1-omni prompt.py: escape)
+static json decision_d1omni_escape(const json & val) {
+    static const std::regex re_special("<\\|([A-Za-z0-9_]+)\\|>");
+    if (val.is_string()) {
+        return std::regex_replace(val.get<std::string>(), re_special, "<\xC2\xA6$1\xC2\xA6>");
+    }
+    if (val.is_array()) {
+        json out = json::array();
+        for (const auto & item : val) {
+            out.push_back(decision_d1omni_escape(item));
+        }
+        return out;
+    }
+    if (val.is_object()) {
+        json out = json::object();
+        for (const auto & [key, item] : val.items()) {
+            out[std::regex_replace(key, re_special, "<\xC2\xA6$1\xC2\xA6>")] = decision_d1omni_escape(item);
+        }
+        return out;
+    }
+    return val;
+}
+
+// given to the lfm2-d1-omni template: text between the pieces of the prompt, and at the start of the pieces that are cut to a token budget
+static const std::string D1OMNI_MARKER        = "<<d1omni:";
+static const std::string D1OMNI_SEP           = "<<d1omni:sep>>";
+static const std::string D1OMNI_MARK_STATE    = "<<d1omni:state>>";
+static const std::string D1OMNI_MARK_QUESTION = "<<d1omni:question>>";
+static const std::string D1OMNI_MARK_OPTION   = "<<d1omni:option>>";
+
+// max_length, image_text_length and audio_text_length of the model config, the converter checks them
+static const size_t D1OMNI_MAX_TOKENS       = 16384;
+static const size_t D1OMNI_MAX_TOKENS_IMAGE = 896;
+static const size_t D1OMNI_MAX_TOKENS_AUDIO = 15360;
+
 size_t server_decision_context::n_variants(const server_decision_question & question) const {
     // lev shows the options of a choice in 2 orders, to cancel the preference for the first label
     if (type == COMMON_DECISION_TYPE_LEV && question.type == SERVER_DECISION_QUESTION_CHOICE && question.options.size() > 1) {
@@ -515,7 +583,8 @@ std::string server_decision_context::render(
         const std::vector<server_decision_question> & questions,
         const server_decision_question & question,
         size_t variant,
-        size_t n_images) const {
+        size_t n_images,
+        bool is_audio) const {
     // the template is given raw JSON values, it serializes the ones that are not strings
     json inp = json{
         {"id",           question.id},
@@ -554,6 +623,15 @@ std::string server_decision_context::render(
         inp = decision_replace_text(inp, text_marker, " ");
     }

+    if (type == COMMON_DECISION_TYPE_LFM2_D1_OMNI) {
+        inp = decision_replace_text(decision_d1omni_escape(inp), D1OMNI_MARKER, "<<d1omni ");
+        inp["audio"]         = is_audio;
+        inp["sep"]           = D1OMNI_SEP;
+        inp["mark_state"]    = D1OMNI_MARK_STATE;
+        inp["mark_question"] = D1OMNI_MARK_QUESTION;
+        inp["mark_option"]   = D1OMNI_MARK_OPTION;
+    }
+
     // the template puts one media marker per image
     json images = json::array();
     if (n_images > 0) {
@@ -580,6 +658,11 @@ void server_decision_context::fill_task(
         mtmd_context * mctx,
         const mtmd_helper_init_opt & init_opt,
         server_task & task) const {
+    if (type == COMMON_DECISION_TYPE_LFM2_D1_OMNI) {
+        fill_task_d1omni(state, questions, question, variant, files, mctx, init_opt, task);
+        return;
+    }
+
     const std::string prompt = render(state, questions, question, variant, files.size());

     if (type == COMMON_DECISION_TYPE_OPENJEV || type == COMMON_DECISION_TYPE_LEV || type == COMMON_DECISION_TYPE_NIMBLE || type == COMMON_DECISION_TYPE_PPLX_DECIDER) {
@@ -678,6 +761,129 @@ void server_decision_context::fill_task_laya(llama_tokens & tokens, const server
     task.decision.column = question.type;
 }

+// each piece is cut to its budget as the model was trained (d1-omni prompt.py: encode), the state gets the room that is left
+// the media come first, the text after them has a budget of its own
+void server_decision_context::fill_task_d1omni(
+        const json & state,
+        const std::vector<server_decision_question> & questions,
+        const server_decision_question & question,
+        size_t variant,
+        const std::vector<raw_buffer> & files,
+        mtmd_context * mctx,
+        const mtmd_helper_init_opt & init_opt,
+        server_task & task) const {
+    const auto invalid = std::runtime_error("unexpected layout of the decision prompt");
+
+    // the prompt depends on the kind of media, mtmd tells an audio clip from an image by its content
+    std::string markers;
+    server_tokens media(llama_tokens(), false);
+    bool is_audio = false;
+    if (!files.empty()) {
+        for (size_t i = 0; i < files.size(); i++) {
+            markers += get_media_marker();
+        }
+        media = process_mtmd_prompt(mctx, markers, files, init_opt);
+        for (size_t i = 0; i < media.size(); i++) {
+            if (media[i] == LLAMA_TOKEN_NULL) {
+                const auto & chunk = media.find_chunk(i);
+                is_audio = is_audio || mtmd_input_chunk_get_type(chunk.get()) == MTMD_INPUT_CHUNK_TYPE_AUDIO;
+                i += mtmd_input_chunk_get_n_tokens(chunk.get()) - 1;
+            }
+        }
+    }
+    if (is_audio && files.size() > 1) {
+        throw std::invalid_argument("a request has images or one audio clip, not both");
+    }
+    const size_t n_media = media.size();
+
+    const std::string prompt = render(state, questions, question, variant, files.size(), is_audio);
+    std::vector<std::string> pieces = string_split(prompt, D1OMNI_SEP);
+    if (pieces.empty() || pieces[0] != markers) {
+        throw invalid;
+    }
+
+    size_t n_max = D1OMNI_MAX_TOKENS;
+    if (!files.empty()) {
+        n_max = std::min(is_audio ? D1OMNI_MAX_TOKENS_AUDIO : D1OMNI_MAX_TOKENS_IMAGE, D1OMNI_MAX_TOKENS - std::min(D1OMNI_MAX_TOKENS, n_media));
+        if (n_max < 64) {
+            throw std::invalid_argument(string_format("the media take %zu of the %zu positions, send fewer images", n_media, D1OMNI_MAX_TOKENS));
+        }
+    }
+
+    // the options get max(96, min(24 n + 32, max / 2)) tokens, shared evenly
+    const int64_t n_options      = question.options.size();
+    const int64_t n_budget       = std::max<int64_t>(96, std::min<int64_t>(24 * n_options + 32, n_max / 2));
+    const int64_t n_option_max   = std::max<int64_t>(2, (n_budget - 3 * n_options) / n_options);
+    const int64_t n_question_max = std::max<int64_t>(16, n_budget);
+
+    llama_tokens head;  // before the state
+    llama_tokens body;  // the state
+    llama_tokens tail;  // after the state
+    bool has_state = false;
+    for (size_t i_piece = 1; i_piece < pieces.size(); i_piece++) {
+        std::string piece = pieces[i_piece];
+        int64_t n_piece_max = -1;
+        bool is_state = false;
+        if (string_starts_with(piece, D1OMNI_MARK_STATE)) {
+            piece    = piece.substr(D1OMNI_MARK_STATE.size());
+            is_state = true;
+        } else if (string_starts_with(piece, D1OMNI_MARK_QUESTION)) {
+            piece = piece.substr(D1OMNI_MARK_QUESTION.size());
+            n_piece_max = n_question_max;
+        } else if (string_starts_with(piece, D1OMNI_MARK_OPTION)) {
+            piece = piece.substr(D1OMNI_MARK_OPTION.size());
+            n_piece_max = n_option_max;
+        }
+
+        llama_tokens tokens = common_tokenize(vocab, piece, false, true);
+        if (n_piece_max >= 0 && (int64_t) tokens.size() > n_piece_max) {
+            tokens.resize(n_piece_max);
+        }
+
+        if (is_state) {
+            if (has_state) {
+                throw invalid;
+            }
+            body      = std::move(tokens);
+            has_state = true;
+        } else {
+            llama_tokens & dst = has_state ? tail : head;
+            dst.insert(dst.end(), tokens.begin(), tokens.end());
+        }
+    }
+    if (!has_state) {
+        throw invalid;
+    }
+
+    const size_t n_room = n_max - std::min(n_max, head.size() + tail.size());
+    body.resize(std::min(body.size(), n_room));
+
+    llama_tokens tokens = std::move(head);
+    tokens.insert(tokens.end(), body.begin(), body.end());
+    tokens.insert(tokens.end(), tail.begin(), tail.end());
+    tokens.resize(std::min(tokens.size(), n_max));
+
+    for (size_t i = 0; i < tokens.size(); i++) {
+        if (tokens[i] == token_marker) {
+            task.decision.markers.push_back(n_media + i);
+        }
+    }
+    if ((int64_t) task.decision.markers.size() != n_options) {
+        throw std::invalid_argument("the options do not fit in the context");
+    }
+
+    // the output has one score per question type
+    task.decision.column = question.type;
+    if (files.empty()) {
+        task.tokens = server_tokens(tokens, false);
+    } else {
+        task.tokens = std::move(media);
+        for (const llama_token token : tokens) {
+            task.tokens.push_back(token);
+        }
+    }
+}
+
 //
 // joint prompt (clef)
 //
@@ -850,14 +1056,15 @@ static double decision_confidence_score(const std::vector<double> & probs) {
     return std::max(0.0, 1.0 - dist / dist_uniform);
 }

-json server_decision_context::format_answer(const server_decision_question & question, const std::vector<std::vector<float>> & scores) const {
+json server_decision_context::format_answer(const server_decision_question & question, const std::vector<std::vector<float>> & scores, bool has_media) const {
     const size_t n = n_outputs(question);
     if (scores.size() != n_variants(question)) {
         throw std::runtime_error("decision result does not match the number of variants");
     }

     // softmax over the outputs of each variant, then the average of the variants
-    const float temperature = get_temperature(question);
+    // lfm2-d1-omni: image and audio answers are not calibrated
+    const float temperature = has_media && type == COMMON_DECISION_TYPE_LFM2_D1_OMNI ? 1.0f : get_temperature(question);
     std::vector<double> probs(n, 0.0);
     for (size_t v = 0; v < scores.size(); v++) {
         const auto & s = scores[v];
diff --git a/tools/server/server-decision.h b/tools/server/server-decision.h
index 8f357f374..5c7e44f37 100644
--- a/tools/server/server-decision.h
+++ b/tools/server/server-decision.h
@@ -62,6 +62,7 @@ struct server_decision_context {
             case COMMON_DECISION_TYPE_CLEF:
             case COMMON_DECISION_TYPE_PPLX_DECIDER:
             case COMMON_DECISION_TYPE_LFM2_D1:
+            case COMMON_DECISION_TYPE_LFM2_D1_OMNI:
                 return true;
             default:
                 return false;
@@ -72,7 +73,8 @@ struct server_decision_context {
     std::vector<server_decision_question> parse_questions(const json & body) const;

     // returns the state without its images, they are appended to files in order
-    // images come from "images" and from the image_url parts of a state made of chat messages
+    // images come from "files" (alias "images") and from the image_url and input_audio parts of a state made of chat messages
+    // an image can be an audio clip if the model supports it
     json parse_state(const json & body, std::vector<raw_buffer> & files) const;

     // number of prompts that are evaluated to answer this question, each one shows the options in a different order
@@ -101,7 +103,7 @@ struct server_decision_context {
             server_task & task) const;

     // scores: the raw model outputs of each variant
-    json format_answer(const server_decision_question & question, const std::vector<std::vector<float>> & scores) const;
+    json format_answer(const server_decision_question & question, const std::vector<std::vector<float>> & scores, bool has_media = false) const;

 private:
     const llama_vocab * vocab = nullptr;
@@ -128,12 +130,22 @@ private:
             const std::vector<server_decision_question> & questions,
             const server_decision_question & question,
             size_t variant,
-            size_t n_images) const;
+            size_t n_images,
+            bool is_audio = false) const;
     json render_options(const server_decision_question & question, size_t variant) const;
     size_t n_outputs(const server_decision_question & question) const;
     // LFM2_D1: label text and tokens of each option
     void d1_labels(const server_decision_question & question, std::vector<std::string> & texts, std::vector<llama_tokens> & groups) const;
     void fill_task_laya(llama_tokens & tokens, const server_decision_question & question, server_task & task) const;
+    void fill_task_d1omni(
+            const json & state,
+            const std::vector<server_decision_question> & questions,
+            const server_decision_question & question,
+            size_t variant,
+            const std::vector<raw_buffer> & files,
+            mtmd_context * mctx,
+            const mtmd_helper_init_opt & init_opt,
+            server_task & task) const;

     float get_temperature(const server_decision_question & question) const;
 };