Commit 69f201a20 for llama.cpp

commit 69f201a2051ea9b9e9b50c3cc56afd9e2ae64414
Author: tc-mb <157115220+tc-mb@users.noreply.github.com>
Date:   Sun Oct 11 02:41:59 2026 +0800

    model : support MiniCPM-V 4.7 (#29416)

    * mtmd : add MiniCPM-V 4.7 support

    Signed-off-by: tc-mb <tianchi_cai@icloud.com>

    * model : allow mrope time from an extra position slot

    Signed-off-by: tc-mb <tianchi_cai@icloud.com>

    * Update conversion/minicpm.py

    Co-authored-by: Sigbjørn Skjæret <sigbjorn.skjaeret@huggingface.co>

    * Slim down comments

    Signed-off-by: tc-mb <tianchi_cai@icloud.com>

    * fix for "do not hand-wrap comments"

    Signed-off-by: tc-mb <tianchi_cai@icloud.com>

    * fix ci

    Signed-off-by: tc-mb <tianchi_cai@icloud.com>

    * rm 3d repo for pr one

    Signed-off-by: tc-mb <tianchi_cai@icloud.com>

    * gguf: add rope.section_order metadata

    * fix comments

    * handle grid layout

    * allow compat

    ---------

    Signed-off-by: tc-mb <tianchi_cai@icloud.com>
    Co-authored-by: Sigbjørn Skjæret <sigbjorn.skjaeret@huggingface.co>
    Co-authored-by: Xuan Son Nguyen <son@huggingface.co>

diff --git a/conversion/__init__.py b/conversion/__init__.py
index 5bf472889..2a62dbac9 100644
--- a/conversion/__init__.py
+++ b/conversion/__init__.py
@@ -189,6 +189,7 @@ TEXT_MODEL_MAP: dict[str, str] = {
     "MiniCPM3ForCausalLM": "minicpm",
     "MiniCPMForCausalLM": "minicpm",
     "MiniCPMV4_6ForConditionalGeneration": "minicpm",
+    "MiniCPMV4_7ForConditionalGeneration": "minicpm",
     "MiniMaxText01ForCausalLM": "minimax",
     "MiniMaxM1ForCausalLM": "minimax",
     "MiniMaxM2ForCausalLM": "minimax",
@@ -346,6 +347,7 @@ MMPROJ_MODEL_MAP: dict[str, str] = {
     "MiMoV2ForCausalLM": "mimo",
     "MiniMaxM3SparseForConditionalGeneration": "minimax",
     "MiniCPMV4_6ForConditionalGeneration": "minicpm",
+    "MiniCPMV4_7ForConditionalGeneration": "minicpm",
     "Mistral3ForConditionalGeneration": "llava",
     "NemotronH_Nano_VL_V2": "nemotron",
     "MuseGlimmerForConditionalGeneration": "muse_glimmer",
diff --git a/conversion/base.py b/conversion/base.py
index 21caf4a17..d1dc6fb77 100644
--- a/conversion/base.py
+++ b/conversion/base.py
@@ -1389,14 +1389,17 @@ class TextModel(ModelBase):
         name, gen = item

         # Skip multimodal tensors
-        if name.startswith(("mlp", "vit.", "vpm.", "siglip2.", "conformer.", "merger.", "resampler.", "sound_encoder.", "sound_projection.", "speech_embeddings.")) \
+        # strip the "model." wrapper so the prefixes below match (name is not returned)
+        if name.startswith("model."):
+            name = name[len("model."):]
+        if name.startswith(("mlp", "vit.", "vpm.", "siglip2.", "conformer.", "connector.", "merger.", "resampler.", "sound_encoder.", "sound_projection.", "speech_embeddings.")) \
                 or "visual." in name or "vision." in name or "audio." in name or "talker." in name \
                 or "vision_" in name or "audio_" in name \
                 or "token2wav." in name or "code2wav." in name \
                 or "projector." in name or "pre_mm_projector_norm" in name \
                 or "image_newline" in name or "view_seperator" in name \
                 or "patch_embed" in name or "patch_embedding" in name \
-                or "patch_merger." in name or "patch_merge_mlp." in name or "model.connector." in name:
+                or "patch_merger." in name or "patch_merge_mlp." in name:
             return None

         return super().filter_tensors(item)
diff --git a/conversion/minicpm.py b/conversion/minicpm.py
index 678d7bec1..937ac033a 100644
--- a/conversion/minicpm.py
+++ b/conversion/minicpm.py
@@ -139,9 +139,16 @@ class MiniCPMV4_6TextModel(Qwen3_5TextModel):
 @ModelBase.register("MiniCPMV4_6ForConditionalGeneration")
 @ModelBase.example("openbmb/MiniCPM-V-4_6")
 class MiniCPMV4_6VisionModel(MmprojModel):
+    projector_type = gguf.VisionProjectorType.MINICPMV4_6
+    # fallback for checkpoints whose preprocessor config omits `scale_resolution`
+    default_scale_resolution: int | None = None
+
+    def get_downsample_mode(self) -> str:
+        return self.preprocessor_config.get("downsample_mode", "16x")
+
     def __init__(self, *args, **kwargs):
         super().__init__(*args, **kwargs)
-        self.downsample_mode = self.preprocessor_config.get("downsample_mode", "16x")
+        self.downsample_mode = self.get_downsample_mode()
         if self.downsample_mode not in {"4x", "16x"}:
             raise ValueError(f"Unsupported downsample mode: {self.downsample_mode}")
         if self.downsample_mode == "4x":
@@ -157,7 +164,8 @@ class MiniCPMV4_6VisionModel(MmprojModel):
             # The CLIP loader in tools/mtmd/clip.cpp consumes `clip.vision.image_size`
             # as the slice size and warmup resolution, so report `scale_resolution` there
             # to match the upstream MiniCPMV4_6ImageProcessorPil slicing rules.
-            scale_resolution = self.preprocessor_config.get("scale_resolution")
+            scale_resolution = self.preprocessor_config.get(
+                "scale_resolution", self.default_scale_resolution)
             if scale_resolution is not None:
                 self.hparams_vision["image_size"] = int(scale_resolution)

@@ -166,12 +174,15 @@ class MiniCPMV4_6VisionModel(MmprojModel):
         assert self.hparams_vision is not None

         # projector type string is consumed by clip_projector_type_from_string() in clip.cpp
-        # (mapped to PROJECTOR_TYPE_MINICPMV4_6).
-        self.gguf_writer.add_clip_projector_type(gguf.VisionProjectorType.MINICPMV4_6)
+        self.gguf_writer.add_clip_projector_type(self.projector_type)

         self.gguf_writer.add_vision_projector_scale_factor(
             2 if self.downsample_mode == "4x" else 4)

+        max_slice_nums = self.preprocessor_config.get("max_slice_nums")
+        if max_slice_nums is not None:
+            self.gguf_writer.add_vision_max_slice_nums(int(max_slice_nums))
+
         # borrow wa_layer_indexes for vit_merger insertion point
         insert_layer_id = int(self.global_config.get(
             "insert_layer_id", self.hparams_vision.get("insert_layer_id", 6)))
@@ -191,3 +202,73 @@ class MiniCPMV4_6VisionModel(MmprojModel):
             return None

         return super().filter_tensors(item)
+
+
+# MiniCPM-V 4.7 shares the v4.6 stack: the same Qwen3.5 text tower (MoE variant when the checkpoint says so) and the same SigLIP + vit_merger + merger vision tower.
+
+@ModelBase.register("MiniCPMV4_7ForConditionalGeneration")
+@ModelBase.example("openbmb/MiniCPM-V-4.7")
+class MiniCPMV4_7TextModel(Qwen3_5TextModel):
+    model_arch = gguf.MODEL_ARCH.QWEN35
+
+    def set_gguf_parameters(self):
+        super().set_gguf_parameters()
+        # mtmd puts the time of the image canvas in slot z, slot t stays the KV cache position
+        self.gguf_writer.add_rope_section_order(gguf.RopeSectionOrder.ZYXT)
+
+    def __init__(self, dir_model, ftype, fname_out, *, hparams: dict | None = None, **kwargs):
+        if hparams is None:
+            hparams = ModelBase.load_hparams(dir_model, is_mistral_format=False)
+        text_config = hparams.get("text_config", {})
+        if text_config.get("model_type") == "qwen3_5_moe_text":
+            self.model_arch = gguf.MODEL_ARCH.QWEN35MOE
+        else:
+            self.model_arch = gguf.MODEL_ARCH.QWEN35
+        super().__init__(dir_model, ftype, fname_out, hparams=hparams, **kwargs)
+
+    @classmethod
+    def filter_tensors(cls, item: tuple[str, Callable[[], Tensor]]) -> tuple[str, Callable[[], Tensor]] | None:
+        name, gen = item
+
+        # MTP tensors are not used yet
+        if name.startswith("mtp"):
+            return None
+
+        return super().filter_tensors(item)
+
+
+@ModelBase.register("MiniCPMV4_7ForConditionalGeneration")
+@ModelBase.example("openbmb/MiniCPM-V-4.7")
+class MiniCPMV4_7VisionModel(MiniCPMV4_6VisionModel):
+    projector_type = gguf.VisionProjectorType.MINICPMV4_7
+    # MiniCPMV4_7ImageProcessorPil default
+    default_scale_resolution = 448
+    # rows of v.tok_embd_sep, the order must match clip_suffix_rows() in clip-impl.h
+    tok_embd_sep = ["</image>", "<slice>", "</slice>", "\n"]
+
+    def get_downsample_mode(self) -> str:
+        # 4.7 moved downsample_mode to the model config; preprocessor value takes priority
+        return self.preprocessor_config.get(
+            "downsample_mode", self.global_config.get("downsample_mode", "16x"))
+
+    @classmethod
+    def filter_tensors(cls, item: tuple[str, Callable[[], Tensor]]) -> tuple[str, Callable[[], Tensor]] | None:
+        # keep the text tok_embd, the separator rows are taken from it in modify_tensors
+        if item[0] == "model.language_model.embed_tokens.weight":
+            return item
+        return super().filter_tensors(item)
+
+    def modify_tensors(self, data_torch: Tensor, name: str, bid: int | None) -> Iterable[tuple[str, Tensor]]:
+        if name == "model.language_model.embed_tokens.weight":
+            # the tile separators are text tokens; clip appends their embeddings so that one chunk holds the whole image
+            from transformers import AutoTokenizer
+            tokenizer = AutoTokenizer.from_pretrained(self.dir_model)
+            ids = []
+            for text in self.tok_embd_sep:
+                tok = tokenizer.encode(text, add_special_tokens=False)
+                if len(tok) != 1:
+                    raise ValueError(f"separator {text!r} must be a single token, got {tok}")
+                ids.append(tok[0])
+            yield self.format_tensor_name(gguf.MODEL_TENSOR.V_TOK_EMBD_SEP, suffix=""), data_torch[ids]
+            return
+        yield from super().modify_tensors(data_torch, name, bid)
diff --git a/docs/multimodal/minicpmv4.7.md b/docs/multimodal/minicpmv4.7.md
new file mode 100644
index 000000000..97bec1307
--- /dev/null
+++ b/docs/multimodal/minicpmv4.7.md
@@ -0,0 +1,55 @@
+## MiniCPM-V 4.7
+
+### Prepare models and code
+
+Download [MiniCPM-V-4.7](https://huggingface.co/openbmb/MiniCPM-V-4.7) PyTorch model from huggingface to "MiniCPM-V-4.7" folder.
+
+The model must be the standard `transformers` checkpoint (no `trust_remote_code` for the text and vision graph used here); the architecture in `config.json` is `MiniCPMV4_7ForConditionalGeneration` with a `qwen3_5_text` (or `qwen3_5_moe_text`) text model and a SigLIP-based vision tower plus a window-attention `vit_merger`, same as MiniCPM-V 4.6.
+
+If the checkpoint ships no MTP weights, pass `--no-mtp` to skip the nextn layers.
+
+### Build llama.cpp
+
+If there are differences in usage, please refer to the official build [documentation](https://github.com/ggml-org/llama.cpp/blob/master/docs/build.md)
+
+Clone llama.cpp:
+```bash
+git clone https://github.com/ggml-org/llama.cpp
+cd llama.cpp
+```
+
+Build llama.cpp using `CMake`:
+```bash
+cmake -B build
+cmake --build build --config Release
+```
+
+
+### Usage of MiniCPM-V 4.7
+
+MiniCPM-V 4.7 is converted directly through `convert_hf_to_gguf.py`. The same script is invoked twice on the original Hugging Face directory: once to produce the language-model GGUF and once with `--mmproj` to produce the multimodal projector GGUF.
+
+```bash
+# language model
+python ./convert_hf_to_gguf.py ../MiniCPM-V-4.7 --outfile ../MiniCPM-V-4.7/ggml-model-f16.gguf --no-mtp
+
+# multimodal projector (vision tower + window-attention vit_merger + DownsampleMLP merger)
+python ./convert_hf_to_gguf.py ../MiniCPM-V-4.7 --mmproj --outfile ../MiniCPM-V-4.7/mmproj-model-f16.gguf
+
+# optional: quantize to Q4_K_M
+./build/bin/llama-quantize ../MiniCPM-V-4.7/ggml-model-f16.gguf ../MiniCPM-V-4.7/ggml-model-Q4_K_M.gguf Q4_K_M
+```
+
+The default projector merges 16x (4x4 patches into one token). To keep 4x more visual tokens, copy the model dir and set `"downsample_mode": "4x"` in the copy's `preprocessor_config.json` before running the `--mmproj` conversion; the loader reads `clip.vision.projector.scale_factor` to pick the graph.
+
+
+Inference on Linux or Mac
+```bash
+# run in single-turn mode
+./build/bin/llama-mtmd-cli -m ../MiniCPM-V-4.7/ggml-model-f16.gguf --mmproj ../MiniCPM-V-4.7/mmproj-model-f16.gguf -c 4096 --jinja --image xx.jpg -p "What is in the image?"
+
+# run in conversation mode
+./build/bin/llama-mtmd-cli -m ../MiniCPM-V-4.7/ggml-model-Q4_K_M.gguf --mmproj ../MiniCPM-V-4.7/mmproj-model-f16.gguf --jinja
+```
+
+The chat template enables thinking by default. Pass `--chat-template-kwargs '{"enable_thinking": false}'` to `llama-server` to turn it off.
diff --git a/gguf-py/gguf/constants.py b/gguf-py/gguf/constants.py
index c003ed2b6..9ed43c7b4 100644
--- a/gguf-py/gguf/constants.py
+++ b/gguf-py/gguf/constants.py
@@ -259,6 +259,7 @@ class Keys:
         DIMENSION_COUNT           = "{arch}.rope.dimension_count"
         DIMENSION_COUNT_SWA       = "{arch}.rope.dimension_count_swa"
         DIMENSION_SECTIONS        = "{arch}.rope.dimension_sections"
+        SECTION_ORDER             = "{arch}.rope.section_order"
         FREQ_BASE                 = "{arch}.rope.freq_base"
         FREQ_BASE_SWA             = "{arch}.rope.freq_base_swa"
         SCALING_TYPE              = "{arch}.rope.scaling.type"
@@ -409,6 +410,7 @@ class Keys:
         BLOCK_COUNT           = "clip.vision.block_count"
         IMAGE_MEAN            = "clip.vision.image_mean"
         IMAGE_STD             = "clip.vision.image_std"
+        MAX_SLICE_NUMS        = "clip.vision.max_slice_nums"
         IMAGE_RESIZE_ALGO     = "clip.vision.image_resize_algo"
         SPATIAL_MERGE_SIZE    = "clip.vision.spatial_merge_size"
         SWIGLU_CLAMP          = "clip.vision.swiglu_clamp"
@@ -1051,6 +1053,7 @@ class MODEL_TENSOR(IntEnum):
     V_SAM_NET_3          = auto() # Deepseek-OCR
     V_ENC_EMBD_IMGNL     = auto() # Deepseek-OCR
     V_ENC_EMBD_VSEP      = auto() # Deepseek-OCR
+    V_TOK_EMBD_SEP       = auto() # MiniCPM-V 4.7
     V_RESMPL_QUERY_768   = auto() # Deepseek-OCR-2
     V_RESMPL_QUERY_1024  = auto() # Deepseek-OCR-2

@@ -1832,6 +1835,7 @@ TENSOR_NAMES: dict[MODEL_TENSOR, str] = {
     MODEL_TENSOR.V_SAM_NET_3:               "v.sam.net_3",
     MODEL_TENSOR.V_ENC_EMBD_IMGNL:          "v.image_newline", # Deepseek-OCR, Granite4Vision
     MODEL_TENSOR.V_ENC_EMBD_VSEP:           "v.view_seperator", # Deepseek-OCR
+    MODEL_TENSOR.V_TOK_EMBD_SEP:            "v.tok_embd_sep", # MiniCPM-V 4.7
     MODEL_TENSOR.V_RESMPL_QUERY_768:        "v.resample_query_768", # Deepseek-OCR-2 qwen2
     MODEL_TENSOR.V_RESMPL_QUERY_1024:       "v.resample_query_1024", # Deepseek-OCR-2 qwen2
     # Granite4 Vision
@@ -2079,6 +2083,7 @@ MODEL_TENSORS: dict[MODEL_ARCH, list[MODEL_TENSOR]] = {
         MODEL_TENSOR.V_ENC_EMBD_POS,
         MODEL_TENSOR.V_ENC_EMBD_IMGNL,
         MODEL_TENSOR.V_ENC_EMBD_VSEP,
+        MODEL_TENSOR.V_TOK_EMBD_SEP,
         MODEL_TENSOR.V_ENC_INPUT_NORM,
         MODEL_TENSOR.V_ENC_ATTN_QKV,
         MODEL_TENSOR.V_ENC_ATTN_Q,
@@ -5927,6 +5932,12 @@ class RopeScalingType(Enum):
     LONGROPE = 'longrope'


+# M-RoPE: input position slot (t, y, x, z) that feeds each RoPE section, in section order
+class RopeSectionOrder(Enum):
+    TYXZ = 'tyxz' # default
+    ZYXT = 'zyxt'
+
+
 class PoolingType(IntEnum):
     NONE = 0
     MEAN = 1
@@ -6133,6 +6144,7 @@ class VisionProjectorType:
     PARAKEET       = "parakeet"  # audio
     MINIMAXM3      = "minimax_m3"
     MINICPMV4_6    = "minicpmv4_6"
+    MINICPMV4_7    = "minicpmv4_7"
     GRANITE_SPEECH = "granite_speech"  # audio
     MIMOVL         = "mimovl"
     MIMO_AUDIO     = "mimo_audio"
diff --git a/gguf-py/gguf/gguf_writer.py b/gguf-py/gguf/gguf_writer.py
index 3d063bb3a..d14d3fade 100644
--- a/gguf-py/gguf/gguf_writer.py
+++ b/gguf-py/gguf/gguf_writer.py
@@ -24,6 +24,7 @@ from .constants import (
     GGUFEndian,
     GGUFValueType,
     Keys,
+    RopeSectionOrder,
     RopeScalingType,
     PoolingType,
     TokenType,
@@ -1154,6 +1155,9 @@ class GGUFWriter:
     def add_rope_dimension_sections(self, dims: Sequence[int]) -> None:
         self.add_array(Keys.Rope.DIMENSION_SECTIONS.format(arch=self.arch), dims)

+    def add_rope_section_order(self, value: RopeSectionOrder) -> None:
+        self.add_string(Keys.Rope.SECTION_ORDER.format(arch=self.arch), value.value)
+
     def add_rope_freq_base(self, value: float) -> None:
         self.add_float32(Keys.Rope.FREQ_BASE.format(arch=self.arch), value)

@@ -1462,6 +1466,9 @@ class GGUFWriter:
     def add_vision_projector_scale_factor(self, value: int) -> None:
         self.add_uint32(Keys.ClipVision.Projector.SCALE_FACTOR, value)

+    def add_vision_max_slice_nums(self, value: int) -> None:
+        self.add_uint32(Keys.ClipVision.MAX_SLICE_NUMS, value)
+
     def add_vision_n_wa_pattern(self, value: int) -> None:
         """Add window attention pattern interval for vision models.

diff --git a/include/llama.h b/include/llama.h
index cfc69cee2..0d476a299 100644
--- a/include/llama.h
+++ b/include/llama.h
@@ -1078,7 +1078,7 @@ extern "C" {

     // Set custom position for the token at index idx in the batch
     // For M-RoPE models:
-    //     - Embedding tokens must have multiple positions per token
+    //     - Embedding tokens must have n_pos_per_embd positions per token, in order [t, y, x, z]; t is also the KV cache position
     //     - Text token only requires one single position per token
     LLAMA_API bool llama_batch_ext_set_pos(
                                 struct llama_batch_ext * batch,
diff --git a/src/llama-arch.cpp b/src/llama-arch.cpp
index 5774867d9..5d9ec1333 100644
--- a/src/llama-arch.cpp
+++ b/src/llama-arch.cpp
@@ -330,6 +330,7 @@ static const std::map<llm_kv, const char *> LLM_KV_NAMES = {
     { LLM_KV_ROPE_DIMENSION_COUNT,           "%s.rope.dimension_count"                 },
     { LLM_KV_ROPE_DIMENSION_COUNT_SWA,       "%s.rope.dimension_count_swa"             },
     { LLM_KV_ROPE_DIMENSION_SECTIONS,        "%s.rope.dimension_sections"              },
+    { LLM_KV_ROPE_SECTION_ORDER,             "%s.rope.section_order"                   },
     { LLM_KV_ROPE_FREQ_BASE,                 "%s.rope.freq_base"                       },
     { LLM_KV_ROPE_FREQ_BASE_SWA,             "%s.rope.freq_base_swa"                   },
     { LLM_KV_ROPE_SCALE_LINEAR,              "%s.rope.scale_linear"                    },
diff --git a/src/llama-arch.h b/src/llama-arch.h
index 1bf4744ff..ba06be8c6 100644
--- a/src/llama-arch.h
+++ b/src/llama-arch.h
@@ -335,6 +335,7 @@ enum llm_kv {
     LLM_KV_ROPE_DIMENSION_COUNT,
     LLM_KV_ROPE_DIMENSION_COUNT_SWA,
     LLM_KV_ROPE_DIMENSION_SECTIONS,
+    LLM_KV_ROPE_SECTION_ORDER,
     LLM_KV_ROPE_FREQ_BASE,
     LLM_KV_ROPE_FREQ_BASE_SWA,
     LLM_KV_ROPE_SCALE_LINEAR,
diff --git a/src/llama-graph.cpp b/src/llama-graph.cpp
index b22b95e6d..1440601d6 100644
--- a/src/llama-graph.cpp
+++ b/src/llama-graph.cpp
@@ -172,7 +172,33 @@ void llm_graph_input_pos::set_input(const llama_ubatch * ubatch) {
     if (ubatch->pos && pos) {
         const int64_t n_tokens = ubatch->n_tokens;

-        ggml_backend_tensor_set(pos, ubatch->pos, 0, n_tokens*n_pos_per_embd*ggml_element_size(pos));
+        const bool has_embd = ubatch->is_mixed() || ubatch->token == nullptr;
+        if (rope_section_order == LLAMA_ROPE_SECTION_ORDER_TYXZ || !has_embd) {
+            ggml_backend_tensor_set(pos, ubatch->pos, 0, n_tokens*n_pos_per_embd*ggml_element_size(pos));
+            return;
+        }
+
+        // input is always [t, y, x, z]
+        // token entries are expanded by the batch to [p, p, p, 0]
+        GGML_ASSERT(n_pos_per_embd == 4);
+
+        // slot index per section, slots are t = 0, y = 1, x = 2, z = 3
+        std::array<int64_t, 4> slot_of_section = { 0, 1, 2, 3 };
+        switch (rope_section_order) {
+            case LLAMA_ROPE_SECTION_ORDER_TYXZ: break;
+            case LLAMA_ROPE_SECTION_ORDER_ZYXT: slot_of_section = { 3, 1, 2, 0 }; break;
+            default: GGML_ABORT("unsupported rope section order");
+        }
+
+        std::vector<llama_pos> pos_data(n_tokens*n_pos_per_embd);
+        for (int64_t i = 0; i < n_tokens; ++i) {
+            const bool is_embd = ubatch->is_mixed() ? ubatch->type[i] != 0 : true;
+            for (int64_t s = 0; s < 4; ++s) {
+                const int64_t slot = is_embd ? slot_of_section[s] : s;
+                pos_data[s*n_tokens + i] = ubatch->pos[slot*n_tokens + i];
+            }
+        }
+        ggml_backend_tensor_set(pos, pos_data.data(), 0, pos_data.size()*ggml_element_size(pos));
     }
 }

@@ -2620,7 +2646,7 @@ ggml_tensor * llm_graph_context::build_inp_embd(ggml_tensor * tok_embd, float to
 }

 ggml_tensor * llm_graph_context::build_inp_pos() const {
-    auto inp = std::make_unique<llm_graph_input_pos>(hparams.n_pos_per_embd());
+    auto inp = std::make_unique<llm_graph_input_pos>(hparams.n_pos_per_embd(), hparams.rope_section_order);

     auto & cur = inp->pos;

diff --git a/src/llama-graph.h b/src/llama-graph.h
index 06bb8c472..2ff75d9a0 100644
--- a/src/llama-graph.h
+++ b/src/llama-graph.h
@@ -169,7 +169,8 @@ public:

 class llm_graph_input_pos : public llm_graph_input_i {
 public:
-    llm_graph_input_pos(uint32_t n_pos_per_embd) : n_pos_per_embd(n_pos_per_embd) {}
+    llm_graph_input_pos(uint32_t n_pos_per_embd, llama_rope_section_order rope_section_order = LLAMA_ROPE_SECTION_ORDER_TYXZ)
+        : n_pos_per_embd(n_pos_per_embd), rope_section_order(rope_section_order) {}
     virtual ~llm_graph_input_pos() = default;

     void set_input(const llama_ubatch * ubatch) override;
@@ -179,6 +180,7 @@ public:
     ggml_tensor * pos = nullptr; // I32 [n_batch]

     const uint32_t n_pos_per_embd = 1;
+    const llama_rope_section_order rope_section_order = LLAMA_ROPE_SECTION_ORDER_TYXZ;
 };

 // temperature tuning, used by llama4
diff --git a/src/llama-hparams.h b/src/llama-hparams.h
index 98afe8a62..064de68f3 100644
--- a/src/llama-hparams.h
+++ b/src/llama-hparams.h
@@ -36,6 +36,13 @@ enum llama_non_causal_type {
     LLAMA_NON_CAUSAL_TYPE_SWA_FULL = 2, // all layers non-causal, SWA not applied between tokens of the current ubatch (deepseek 4)
 };

+// M-RoPE: which input position slot feeds each RoPE section
+enum llama_rope_section_order {
+    LLAMA_ROPE_SECTION_ORDER_UNSPECIFIED = -1,
+    LLAMA_ROPE_SECTION_ORDER_TYXZ        = 0, // default, slot i feeds section i
+    LLAMA_ROPE_SECTION_ORDER_ZYXT        = 1, // MiniCPM-V 4.7: time last
+};
+
 // forward declaration; full definition in llama-graph.h
 enum llm_ffn_op_type : int;

@@ -166,6 +173,8 @@ struct llama_hparams {

     std::array<int, 4> rope_sections;

+    enum llama_rope_section_order rope_section_order = LLAMA_ROPE_SECTION_ORDER_TYXZ;
+
     // Per-layer RoPE enable flags (1 = use RoPE, 0 = NoPE)
     // by default, all layers use RoPE (controlled by rope_finetuned)
     std::array<uint32_t, LLAMA_MAX_LAYERS> rope_pattern;
diff --git a/src/llama-model-saver.cpp b/src/llama-model-saver.cpp
index 8d9728319..1c7cbd900 100644
--- a/src/llama-model-saver.cpp
+++ b/src/llama-model-saver.cpp
@@ -365,6 +365,7 @@ void llama_model_saver::add_kv_from_model() {
     add_kv(LLM_KV_ROPE_DIMENSION_COUNT,              hparams.n_rot_full);
     add_kv(LLM_KV_ROPE_DIMENSION_COUNT_SWA,          hparams.n_rot_swa);
     add_kv(LLM_KV_ROPE_DIMENSION_SECTIONS,           hparams.rope_sections);
+    add_kv(LLM_KV_ROPE_SECTION_ORDER,                llama_rope_section_order_name(hparams.rope_section_order));
     add_kv(LLM_KV_ROPE_FREQ_BASE,                    hparams.rope_freq_base_train);
     add_kv(LLM_KV_ROPE_FREQ_BASE_SWA,                hparams.rope_freq_base_train_swa);
     // add_kv(LLM_KV_ROPE_SCALE_LINEAR,                 rope_scaling_factor); // old name
diff --git a/src/llama-model.cpp b/src/llama-model.cpp
index b3a0cf5d5..6fece1019 100644
--- a/src/llama-model.cpp
+++ b/src/llama-model.cpp
@@ -1064,6 +1064,25 @@ static llama_rope_scaling_type llama_rope_scaling_type_from_string(const std::st
     return LLAMA_ROPE_SCALING_TYPE_UNSPECIFIED;
 }

+static const std::map<llama_rope_section_order, const char *> LLAMA_ROPE_SECTION_ORDERS = {
+    { LLAMA_ROPE_SECTION_ORDER_TYXZ, "tyxz" },
+    { LLAMA_ROPE_SECTION_ORDER_ZYXT, "zyxt" },
+};
+
+std::string llama_rope_section_order_name(llama_rope_section_order rope_section_order) {
+    return LLAMA_ROPE_SECTION_ORDERS.at(rope_section_order);
+}
+
+static llama_rope_section_order llama_rope_section_order_from_string(const std::string & name) {
+    for (const auto & kv : LLAMA_ROPE_SECTION_ORDERS) {
+        if (kv.second == name) {
+            return kv.first;
+        }
+    }
+
+    return LLAMA_ROPE_SECTION_ORDER_UNSPECIFIED;
+}
+
 // Maps GGUF activation names to the FFN op type used by the graph builders.
 static const std::map<std::string, llm_ffn_op_type> LLM_FFN_OP_TYPES_FROM_STRING = {
     { "gelu",              LLM_FFN_GEGLU_ERF },
@@ -1447,6 +1466,13 @@ void llama_model_base::load_hparams(llama_model_loader & ml) {
     hparams.rope_scaling_type_train = llama_rope_scaling_type_from_string(rope_scaling);
     GGML_ASSERT(hparams.rope_scaling_type_train != LLAMA_ROPE_SCALING_TYPE_UNSPECIFIED);

+    std::string rope_section_order("tyxz");
+    ml.get_key(LLM_KV_ROPE_SECTION_ORDER, rope_section_order, false);
+    hparams.rope_section_order = llama_rope_section_order_from_string(rope_section_order);
+    if (hparams.rope_section_order == LLAMA_ROPE_SECTION_ORDER_UNSPECIFIED) {
+        throw std::runtime_error("unknown rope section order: " + rope_section_order);
+    }
+
     // TODO: Handle SWA metadata similarly when models start implementing it
     // rope_freq_scale (inverse of the kv) is optional
     float ropescale = 0.0f;
@@ -1517,6 +1543,10 @@ void llama_model_base::load_hparams(llama_model_loader & ml) {
     }

     hparams.rope_type = llama_model_rope_type(this);
+
+    if (hparams.rope_section_order != LLAMA_ROPE_SECTION_ORDER_TYXZ && hparams.n_pos_per_embd() != 4) {
+        throw std::runtime_error("rope section order " + llama_rope_section_order_name(hparams.rope_section_order) + " requires M-RoPE");
+    }
 }

 void llama_model_base::load_vocab(llama_model_loader & ml) {
@@ -2158,6 +2188,9 @@ void llama_model::print_info() const {
         if (const auto & s = hparams.rope_sections; s[0] || s[1] || s[2] || s[3]) {
             LLAMA_LOG_INFO("%s: mrope sections        = [%d, %d, %d, %d]\n", __func__, s[0], s[1], s[2], s[3]);
         }
+        if (hparams.rope_section_order != LLAMA_ROPE_SECTION_ORDER_TYXZ) {
+            LLAMA_LOG_INFO("%s: rope section order    = %s\n", __func__, llama_rope_section_order_name(hparams.rope_section_order).c_str());
+        }
         if (!classifier_labels.empty()) {
             LLAMA_LOG_INFO("%s: n_cls_out             = %u\n", __func__, hparams.n_cls_out);

diff --git a/src/llama-model.h b/src/llama-model.h
index 82124471f..3aa5823a5 100644
--- a/src/llama-model.h
+++ b/src/llama-model.h
@@ -160,6 +160,7 @@ enum llm_type {
 };

 std::string llama_rope_scaling_type_name(llama_rope_scaling_type rope_scaling_type);
+std::string llama_rope_section_order_name(llama_rope_section_order rope_section_order);

 // Map a GGUF activation-name string to llm_ffn_op_type. Returns `fallback` if
 // the string is empty or not recognized.
diff --git a/tools/mtmd/clip-graph.h b/tools/mtmd/clip-graph.h
index 06eee4697..33e546b7d 100644
--- a/tools/mtmd/clip-graph.h
+++ b/tools/mtmd/clip-graph.h
@@ -166,4 +166,7 @@ struct clip_graph {
     // Generic function to stack frames for audio processing
     // Abstracts out the StackAudioFrames logic used by ultravox
     ggml_tensor * build_stack(ggml_tensor * cur, int32_t stack_factor, int32_t n_embed);
+
+    // append the separators of img.suffix_type after the image tokens
+    ggml_tensor * build_suffix(ggml_tensor * cur);
 };
diff --git a/tools/mtmd/clip-impl.h b/tools/mtmd/clip-impl.h
index de171b7a5..30cb30789 100644
--- a/tools/mtmd/clip-impl.h
+++ b/tools/mtmd/clip-impl.h
@@ -68,6 +68,7 @@

 #define KEY_MM_PATCH_MERGE_TYPE    "clip.vision.mm_patch_merge_type"
 #define KEY_IMAGE_GRID_PINPOINTS   "clip.vision.image_grid_pinpoints"
+#define KEY_MAX_SLICE_NUMS         "clip.vision.max_slice_nums"
 #define KEY_WIN_ATTN_PATTERN       "clip.vision.n_wa_pattern"
 #define KEY_WIN_ATTN_LAYER_INDEXES "clip.vision.wa_layer_indexes"
 #define KEY_WA_PATTERN_MODE        "clip.vision.wa_pattern_mode"
@@ -146,6 +147,7 @@
 #define TN_MVLM_PROJ_PEG   "mm.model.peg.%d.%s"
 #define TN_IMAGE_NEWLINE   "v.image_newline"
 #define TN_IMAGE_SEPERATOR "v.view_seperator"
+#define TN_TOK_EMBD_SEP    "v.tok_embd_sep"
 #define TN_MM_INP_NORM     "mm.input_norm.weight"
 #define TN_MM_INP_NORM_B   "mm.input_norm.bias"
 #define TN_MM_INP_PROJ     "mm.input_projection.weight" // gemma3
@@ -500,6 +502,7 @@ enum projector_type {
     PROJECTOR_TYPE_PARAKEET,
     PROJECTOR_TYPE_EXAONE4_5,
     PROJECTOR_TYPE_MINICPMV4_6,
+    PROJECTOR_TYPE_MINICPMV4_7,
     PROJECTOR_TYPE_GRANITE_SPEECH,
     PROJECTOR_TYPE_MIMOVL,
     PROJECTOR_TYPE_MINIMAX_M3,
@@ -569,6 +572,7 @@ static std::map<projector_type, std::string> PROJECTOR_TYPE_NAMES = {
     { PROJECTOR_TYPE_EXAONE4_5,         "exaone4_5"},
     { PROJECTOR_TYPE_HUNYUANVL,         "hunyuanvl"},
     { PROJECTOR_TYPE_MINICPMV4_6,       "minicpmv4_6"},
+    { PROJECTOR_TYPE_MINICPMV4_7,       "minicpmv4_7"},
     { PROJECTOR_TYPE_GRANITE_SPEECH,    "granite_speech"},
     { PROJECTOR_TYPE_MIMOVL,            "mimovl"},
     { PROJECTOR_TYPE_MINIMAX_M3,        "minimax_m3"},
@@ -660,6 +664,38 @@ struct clip_image_u8 {

 struct mtmd_serialization; // forward declaration

+// separators appended after the image tokens of one entry, as rows of v.tok_embd_sep
+enum clip_suffix_type : int32_t {
+    CLIP_SUFFIX_NONE = 0,
+    // MiniCPM-V 4.7 tiles
+    CLIP_SUFFIX_MINICPMV_OV,       // </image>
+    CLIP_SUFFIX_MINICPMV_OV_SLICE, // </image><slice>
+    CLIP_SUFFIX_MINICPMV_SLICE,    // </slice><slice>
+    CLIP_SUFFIX_MINICPMV_ROW_END,  // </slice>\n<slice>
+    CLIP_SUFFIX_MINICPMV_LAST,     // </slice>
+    CLIP_SUFFIX_COUNT,
+};
+
+// rows of v.tok_embd_sep for each suffix type
+// MiniCPM-V 4.7 rows (set by the converter): 0 = </image>, 1 = <slice>, 2 = </slice>, 3 = \n
+static inline const std::vector<int> & clip_suffix_rows(clip_suffix_type type) {
+    static const std::vector<int> none;
+    static const std::vector<int> minicpmv_ov       = { 0 };
+    static const std::vector<int> minicpmv_ov_slice = { 0, 1 };
+    static const std::vector<int> minicpmv_slice    = { 2, 1 };
+    static const std::vector<int> minicpmv_row_end  = { 2, 3, 1 };
+    static const std::vector<int> minicpmv_last     = { 2 };
+    switch (type) {
+        case CLIP_SUFFIX_NONE:              return none;
+        case CLIP_SUFFIX_MINICPMV_OV:       return minicpmv_ov;
+        case CLIP_SUFFIX_MINICPMV_OV_SLICE: return minicpmv_ov_slice;
+        case CLIP_SUFFIX_MINICPMV_SLICE:    return minicpmv_slice;
+        case CLIP_SUFFIX_MINICPMV_ROW_END:  return minicpmv_row_end;
+        case CLIP_SUFFIX_MINICPMV_LAST:     return minicpmv_last;
+        default: GGML_ABORT("invalid suffix type");
+    }
+}
+
 // For images, buf.size() == nx*ny*3
 //     Memory layout: RGBRGBRGB...
 // For seq, buf.size() == nx*ny*3*nt
@@ -675,10 +711,12 @@ struct clip_image_f32 {
     // deepseek4v: number of leading IMAGE_PAD embeddings, aligns IMAGE_START to the LLM compressor ratio
     // depends on the chunk position, set at tokenize time (see mtmd_tokenizer::add_media)
     int32_t lead_pad = 0;
+    // separators appended after the image tokens
+    clip_suffix_type suffix_type = CLIP_SUFFIX_NONE;

-    // llava-next "anyres" tiling, used by Granite4 Vision
-    // the whole grid is encoded and assembled in a single graph
-    // NOTE: excluded from serialized: a deserialized image is always a placeholder, which is never encoded
+    // tile grid of the image group this entry belongs to
+    // llava-next "anyres" (Granite4 Vision): the whole grid is encoded and assembled in a single graph
+    // MiniCPM-V 4.7: set on the overview entry, the decoder positions of all tiles are derived from it
     struct anyres_info {
         int grid_x = 0; // tiles per row, 0 means the image is not tiled
         int grid_y = 0; // tiles per column
diff --git a/tools/mtmd/clip-model.h b/tools/mtmd/clip-model.h
index a5ff38822..657671ded 100644
--- a/tools/mtmd/clip-model.h
+++ b/tools/mtmd/clip-model.h
@@ -71,6 +71,7 @@ struct clip_hparams {
     std::vector<clip_image_size> image_res_candidates;
     int32_t preproc_min_tiles = 0;
     int32_t preproc_max_tiles = 0;
+    int32_t max_slice_nums = 9; // llava-uhd slice cap; per-model, carried in the GGUF
     int32_t preproc_tile_size = 0; // local tile size (deepseek-ocr)
     resize_algo image_resize_algo_rf = RESIZE_ALGO_BICUBIC;
     resize_algo image_resize_algo_ov = RESIZE_ALGO_BICUBIC;
@@ -614,6 +615,7 @@ struct clip_model {

     ggml_tensor * image_newline = nullptr;
     ggml_tensor * view_seperator = nullptr;
+    ggml_tensor * tok_embd_sep = nullptr; // [n_embd_text, n_sep] rows of the text model tok_embd (MiniCPM-V 4.7)


     // Yi type models with mlp+normalization projection
diff --git a/tools/mtmd/clip.cpp b/tools/mtmd/clip.cpp
index db6abf633..fccb00c3a 100644
--- a/tools/mtmd/clip.cpp
+++ b/tools/mtmd/clip.cpp
@@ -909,6 +909,16 @@ ggml_tensor * clip_graph::build_stack(ggml_tensor * cur, int32_t stack_factor, i

 // aka pixel_shuffle / pixel_unshuffle / patch_merger (Kimi-VL)
 // support dynamic resolution
+ggml_tensor * clip_graph::build_suffix(ggml_tensor * cur) {
+    for (int idx : clip_suffix_rows(img.suffix_type)) {
+        GGML_ASSERT(model.tok_embd_sep && idx < model.tok_embd_sep->ne[1]);
+        ggml_tensor * row = ggml_view_2d(ctx0, model.tok_embd_sep, model.tok_embd_sep->ne[0], 1,
+                                         model.tok_embd_sep->nb[1], idx * model.tok_embd_sep->nb[1]);
+        cur = ggml_concat(ctx0, cur, ggml_cast(ctx0, row, cur->type), 1);
+    }
+    return cur;
+}
+
 ggml_tensor * clip_graph::build_patch_merge_permute(ggml_tensor * cur, int scale_factor) {
     GGML_ASSERT(scale_factor > 1);

@@ -1020,6 +1030,7 @@ static std::unique_ptr<clip_graph> clip_get_graph_builder(clip_ctx * ctx, const
                 builder = std::make_unique<clip_graph_minicpmv>(ctx, img);
             } break;
         case PROJECTOR_TYPE_MINICPMV4_6:
+        case PROJECTOR_TYPE_MINICPMV4_7:
             {
                 builder = std::make_unique<clip_graph_minicpmv4_6>(ctx, img);
             } break;
@@ -1332,6 +1343,7 @@ struct clip_model_loader {
             if (is_vision) {
                 get_u32(KEY_IMAGE_SIZE, hparams.image_size);
                 get_u32(KEY_PATCH_SIZE, hparams.patch_size);
+                get_u32(KEY_MAX_SLICE_NUMS, hparams.max_slice_nums, false);
                 get_i32(KEY_MINICPMV_VERSION, hparams.minicpmv_version, false); // legacy
                 get_u32(KEY_MINICPMV_QUERY_NUM, hparams.minicpmv_query_num, false);
                 if (hparams.minicpmv_query_num == 0) {
@@ -1471,13 +1483,18 @@ struct clip_model_loader {
                         }
                     } break;
                 case PROJECTOR_TYPE_MINICPMV4_6:
+                case PROJECTOR_TYPE_MINICPMV4_7:
                     {
-                        // MiniCPM-V 4.6 unified merger projector
+                        // MiniCPM-V 4.6/4.7 unified merger projector
                         // ViT merger 2x2 + final merger 2x2 = 4x spatial merge per dimension
                         hparams.n_merge = 4;
                         get_u32(KEY_PROJ_SCALE_FACTOR, hparams.n_merge, false);
                         GGML_ASSERT(hparams.n_merge == 2 || hparams.n_merge == 4);

+                        // no padding: the reference stretches the refined image to the target size
+                        hparams.image_pad_ov = PAD_NONE;
+                        hparams.image_pad_rf = PAD_NONE;
+
                         // borrow wa_layer_indexes for vit_merger insertion point
                         std::vector<int> wa_layer_indexes_vec;
                         get_arr_int(KEY_WIN_ATTN_LAYER_INDEXES, wa_layer_indexes_vec, false);
@@ -2393,6 +2410,7 @@ struct clip_model_loader {
                     || model.proj_type == PROJECTOR_TYPE_IDEFICS3
                     || model.proj_type == PROJECTOR_TYPE_MINICPMV
                     || model.proj_type == PROJECTOR_TYPE_MINICPMV4_6
+                    || model.proj_type == PROJECTOR_TYPE_MINICPMV4_7
                 ) && layer.ff_up_w && layer.ff_down_w && layer.ff_down_w->ne[0] == hparams.n_embd;
             if (is_ffn_swapped) {
                 // swap up and down weights
@@ -2495,6 +2513,7 @@ struct clip_model_loader {
                     model.mm_model_ln_post_b = get_tensor(string_format(TN_MINICPMV_LN, "post", "bias"));
                 } break;
             case PROJECTOR_TYPE_MINICPMV4_6:
+            case PROJECTOR_TYPE_MINICPMV4_7:
                 {
                     const bool merger_required = hparams.n_merge == 4;
                     auto get_merger_tensor = [&](const std::string & name, bool required = true) {
@@ -2526,6 +2545,7 @@ struct clip_model_loader {
                     model.mm_ffn_up_b     = get_tensor(string_format(TN_MM_UP,   "bias"), false);
                     model.mm_ffn_down_w   = get_tensor(string_format(TN_MM_DOWN, "weight"));
                     model.mm_ffn_down_b   = get_tensor(string_format(TN_MM_DOWN, "bias"), false);
+                    model.tok_embd_sep    = get_tensor(TN_TOK_EMBD_SEP, model.proj_type == PROJECTOR_TYPE_MINICPMV4_7);
                 } break;
             case PROJECTOR_TYPE_GLM_EDGE:
                 {
@@ -4169,6 +4189,8 @@ int clip_n_output_tokens_x(const clip_ctx * ctx, const clip_image_f32 * img) {
         case PROJECTOR_TYPE_MUSE_GLIMMER:
             return (img->nx() / params.patch_size) / 2;
         case PROJECTOR_TYPE_STEP3VL:
+        case PROJECTOR_TYPE_MINICPMV4_6:
+        case PROJECTOR_TYPE_MINICPMV4_7:
             return img->nx() / (params.patch_size * params.n_merge);
         case PROJECTOR_TYPE_DEEPSEEKOCR:
         case PROJECTOR_TYPE_DEEPSEEKOCR2:
@@ -4197,6 +4219,8 @@ int clip_n_output_tokens_y(const clip_ctx * ctx, const clip_image_f32 * img) {
         case PROJECTOR_TYPE_MUSE_GLIMMER:
             return (img->ny() / params.patch_size) / 2;
         case PROJECTOR_TYPE_STEP3VL:
+        case PROJECTOR_TYPE_MINICPMV4_6:
+        case PROJECTOR_TYPE_MINICPMV4_7:
             return img->ny() / (params.patch_size * params.n_merge);
         default:
             break;
@@ -4262,6 +4286,7 @@ int clip_n_output_tokens(const clip_ctx * ctx, const clip_image_f32 * img) {
                 }
             } break;
         case PROJECTOR_TYPE_MINICPMV4_6:
+        case PROJECTOR_TYPE_MINICPMV4_7:
             {
                 n_patches /= params.n_merge * params.n_merge;
             } break;
@@ -4513,6 +4538,8 @@ int clip_n_output_tokens(const clip_ctx * ctx, const clip_image_f32 * img) {
             GGML_ABORT("unsupported projector type");
     }

+    n_patches += (int) clip_suffix_rows(img->suffix_type).size();
+
     return n_patches;
 }

@@ -4836,6 +4863,7 @@ bool clip_encode(struct clip_ctx * ctx, struct clip_encode_params * params) {
                 set_input_f32("omega", omega);
             } break;
         case PROJECTOR_TYPE_MINICPMV4_6:
+        case PROJECTOR_TYPE_MINICPMV4_7:
             {
                 const bool is_4x = hparams.n_merge == 2;

@@ -6080,6 +6108,7 @@ int clip_n_mmproj_embd(const struct clip_ctx * ctx) {
         case PROJECTOR_TYPE_MINICPMV:
             return ctx->model.mm_model_proj->ne[0];
         case PROJECTOR_TYPE_MINICPMV4_6:
+        case PROJECTOR_TYPE_MINICPMV4_7:
             return ctx->model.mm_ffn_down_w->ne[1];
         case PROJECTOR_TYPE_GLM_EDGE:
             return ctx->model.mm_model_mlp_3_w->ne[1];
diff --git a/tools/mtmd/models/minicpmv.cpp b/tools/mtmd/models/minicpmv.cpp
index 16514a337..0e0fc0381 100644
--- a/tools/mtmd/models/minicpmv.cpp
+++ b/tools/mtmd/models/minicpmv.cpp
@@ -350,6 +350,8 @@ ggml_cgraph * clip_graph_minicpmv4_6::build() {
         inpL = cur;
     }

+    inpL = build_suffix(inpL);
+
     ggml_build_forward_expand(gf, inpL);
     return gf;
 }
diff --git a/tools/mtmd/mtmd-image.cpp b/tools/mtmd/mtmd-image.cpp
index 5ce0fa059..a34fd4548 100644
--- a/tools/mtmd/mtmd-image.cpp
+++ b/tools/mtmd/mtmd-image.cpp
@@ -507,9 +507,7 @@ mtmd_image_preproc_out mtmd_image_preprocessor_llava_uhd::preprocess(const clip_

 mtmd_image_preprocessor_llava_uhd::slice_instructions mtmd_image_preprocessor_llava_uhd::get_slice_instructions(const clip_image_size & original_size) const {
     mtmd_image_preprocessor_llava_uhd::slice_instructions res;
-    // align slices by patch_size * n_merge so an integer number of merger output tokens fits per slice
-    const int n_merge         = hparams.n_merge;
-    const int patch_size      = hparams.patch_size * n_merge;
+    const int patch_size      = get_slice_align();
     const int slice_size      = hparams.image_size;
     const int original_width  = original_size.width;
     const int original_height = original_size.height;
@@ -568,7 +566,7 @@ mtmd_image_preprocessor_llava_uhd::slice_instructions mtmd_image_preprocessor_ll
     res.overview_size = best_size;

     {
-        const int max_slice_nums = 9; // TODO: this is only used by minicpmv, maybe remove it
+        const int max_slice_nums = hparams.max_slice_nums > 0 ? hparams.max_slice_nums : 9;
         const float log_ratio = log((float)original_width / original_height);
         const float ratio = (float)original_width * original_height / (slice_size * slice_size);
         const int multiple = fmin(ceil(ratio), max_slice_nums);
@@ -691,7 +689,7 @@ clip_image_size mtmd_image_preprocessor_llava_uhd::select_best_resolution(const
 }

 int mtmd_image_preprocessor_llava_uhd::ensure_divide(int length, int patch_size) const {
-    return std::max(static_cast<int>(std::round(static_cast<float>(length) / patch_size) * patch_size), patch_size);
+    return std::max(align_round(static_cast<double>(length) / patch_size) * patch_size, patch_size);
 }

 clip_image_size mtmd_image_preprocessor_llava_uhd::get_refine_size(const clip_image_size & original_size, const clip_image_size & grid, int scale_resolution, int patch_size, bool allow_upscale) const {
@@ -893,17 +891,15 @@ mtmd_image_preproc_out mtmd_image_preprocessor_longest_edge::preprocess(const cl
 //

 mtmd_image_preprocessor_llava_uhd::slice_instructions mtmd_image_preprocessor_minicpmv::get_slice_instructions(const clip_image_size & original_size) const {
-    if (hparams.n_merge == 2) {
-        const int   slice_size = hparams.image_size;
-        const float ratio      = (float)original_size.width * original_size.height / (slice_size * slice_size);
-        if (ratio <= 1.0f) {
-            mtmd_image_preprocessor_llava_uhd::slice_instructions inst;
-            const int patch_size = hparams.patch_size * hparams.n_merge;
-            inst.overview_size = get_best_resize(original_size, slice_size, patch_size, true);
-            inst.refined_size  = clip_image_size{0, 0};
-            inst.grid_size     = clip_image_size{0, 0};
-            return inst;
-        }
+    // overview only for small images, unlike generic llava-uhd which slices once one side exceeds scale resolution
+    const int   slice_size = hparams.image_size;
+    const float ratio      = (float) original_size.width * original_size.height / (slice_size * slice_size);
+    if (ratio <= 1.0f) {
+        mtmd_image_preprocessor_llava_uhd::slice_instructions inst;
+        inst.overview_size = get_best_resize(original_size, slice_size, get_slice_align(), true);
+        inst.refined_size  = clip_image_size{0, 0};
+        inst.grid_size     = clip_image_size{0, 0};
+        return inst;
     }
     return mtmd_image_preprocessor_llava_uhd::get_slice_instructions(original_size);
 }
diff --git a/tools/mtmd/mtmd-image.h b/tools/mtmd/mtmd-image.h
index 2dff2b47d..b318b2d42 100644
--- a/tools/mtmd/mtmd-image.h
+++ b/tools/mtmd/mtmd-image.h
@@ -83,6 +83,17 @@ struct mtmd_image_preprocessor_llava_uhd : mtmd_image_preprocessor {
     slice_output slice_image(const clip_image_u8 & img, const slice_instructions & inst) const;

 protected:
+    // align slices to a multiple of the merger factor (integer merger tokens per slice)
+    virtual int get_slice_align() const {
+        const int merge = hparams.n_merge > 0 ? hparams.n_merge : 1;
+        return hparams.patch_size * merge;
+    }
+
+    // rounding for snapping a length to a multiple of the align size
+    virtual int align_round(double v) const {
+        return static_cast<int>(std::round(v));
+    }
+
     clip_image_size get_best_resize(const clip_image_size & original_size, int scale_resolution, int patch_size, bool allow_upscale = false) const;

     /**
@@ -155,6 +166,26 @@ private:
 struct mtmd_image_preprocessor_minicpmv : mtmd_image_preprocessor_llava_uhd {
     using mtmd_image_preprocessor_llava_uhd::mtmd_image_preprocessor_llava_uhd;
     slice_instructions get_slice_instructions(const clip_image_size & original_size) const override;
+
+protected:
+    // always patch_size * 4, even in 4x mode (the 2x2 vit_merger slot stays)
+    int get_slice_align() const override {
+        return hparams.patch_size * 4;
+    }
+
+    // Python's round() breaks ties to even, unlike std::round
+    int align_round(double v) const override {
+        const double fl = std::floor(v);
+        const double diff = v - fl;
+        if (diff > 0.5) {
+            return static_cast<int>(fl) + 1;
+        }
+        if (diff < 0.5) {
+            return static_cast<int>(fl);
+        }
+        const int lo = static_cast<int>(fl);
+        return (lo % 2 == 0) ? lo : lo + 1;
+    }
 };

 // custom llava-uhd slicing logic for LFM2
diff --git a/tools/mtmd/mtmd.cpp b/tools/mtmd/mtmd.cpp
index b6e331bc7..2c2b06936 100644
--- a/tools/mtmd/mtmd.cpp
+++ b/tools/mtmd/mtmd.cpp
@@ -19,6 +19,7 @@

 #include <algorithm>
 #include <cerrno>
+#include <cmath>
 #include <cstdio>
 #include <cstdlib>
 #include <cstring>
@@ -27,7 +28,10 @@
 #include <vector>

 // remember to bump this if the serialization format changes
-#define MTMD_SERIALIZATION_VERSION 2
+#define MTMD_SERIALIZATION_VERSION 3
+
+// oldest compat version that can be loaded
+#define MTMD_SERIALIZATION_VERSION_MIN 2

 struct mtmd_serialization {
     // note: using 64-bit here for future-proofing
@@ -45,7 +49,7 @@ struct mtmd_serialization {
         // copy buf to data
         data.assign(buf, buf + len);
         uint64_t ver_in = read<uint64_t>();
-        if (ver_in != version) {
+        if (ver_in < MTMD_SERIALIZATION_VERSION_MIN || ver_in > version) {
             throw std::runtime_error("version mismatch");
         }
         this->version = ver_in;
@@ -106,6 +110,11 @@ void clip_image_f32::serialize(mtmd_serialization & ser) const {
     ser.write(add_viewsep);
     ser.write(add_newline);
     ser.write(lead_pad);
+    ser.write((int32_t)suffix_type);
+    ser.write((int32_t)anyres.grid_x);
+    ser.write((int32_t)anyres.grid_y);
+    ser.write((int32_t)anyres.orig_nx);
+    ser.write((int32_t)anyres.orig_ny);
     ser.write((int32_t)nx_);
     ser.write((int32_t)ny_);
 }
@@ -113,6 +122,17 @@ void clip_image_f32::deserialize(mtmd_serialization & ser) {
     add_viewsep = ser.read<bool>();
     add_newline = ser.read<bool>();
     lead_pad = ser.read<int32_t>();
+    if (ser.version >= 3) {
+        const int32_t suffix_raw = ser.read<int32_t>();
+        if (suffix_raw < 0 || suffix_raw >= CLIP_SUFFIX_COUNT) {
+            throw std::runtime_error("invalid suffix type");
+        }
+        suffix_type = (clip_suffix_type)suffix_raw;
+        anyres.grid_x = ser.read<int32_t>();
+        anyres.grid_y = ser.read<int32_t>();
+        anyres.orig_nx = ser.read<int32_t>();
+        anyres.orig_ny = ser.read<int32_t>();
+    }
     nx_ = ser.read<int32_t>();
     ny_ = ser.read<int32_t>();
     buf.clear(); // always a placeholder after loading
@@ -204,9 +224,11 @@ enum mtmd_pos_type {
     MTMD_POS_TYPE_NORMAL,    // number of positions equals to number of tokens
     MTMD_POS_TYPE_MROPE,     // qwen-vl mrope style, each image takes max(t,h,w) position indexes
     MTMD_POS_TYPE_HUNYUANVL, // HunyuanVL mrope + BOI/EOI/newline layout with XD-RoPE dim-3
+    MTMD_POS_TYPE_CANVAS,    // MiniCPM-V 4.7: overview + slices in one chunk, sharing one 2D canvas (see mtmd_image_tokens::canvas_tile_grid)
     MTMD_POS_TYPE_COUNT,     // for validation
 };

+
 struct mtmd_image_tokens {
     uint32_t nx = 0; // number of tokens in x direction
     uint32_t ny = 0; // number of tokens in y direction
@@ -218,6 +240,14 @@ struct mtmd_image_tokens {
             // [BOI] [row0 tokens + newline] ... [row(ny-1) tokens + newline] [EOI]
             return (nx + 1) * ny + 2;
         }
+        if (pos == MTMD_POS_TYPE_CANVAS) {
+            uint32_t n = 0;
+            for (size_t k = 0; k < batch_f32.entries.size(); ++k) {
+                const auto [gw, gh] = canvas_tile_grid(k);
+                n += gw * gh + (uint32_t) clip_suffix_rows(batch_f32.entries[k].suffix_type).size();
+            }
+            return n;
+        }
         uint32_t nz = batch_f32.entries.size();
         if (n_temporal_merge > 1) {
             // [QWEN_VIDEO] this logic is quite ugly, it's mostly to make qwen-vl temporal merge work, can be improved in the future
@@ -243,8 +273,17 @@ struct mtmd_image_tokens {
         return false;
     }

+    // MTMD_POS_TYPE_CANVAS: entries are [overview, slices row by row], nx/ny is the token grid of the last entry
+    // returns the token grid (w, h) of entry k, scaled from its pixel size
+    std::pair<uint32_t, uint32_t> canvas_tile_grid(size_t k) const {
+        const auto & ref = batch_f32.entries.back();
+        const auto & e   = batch_f32.entries[k];
+        return { (uint32_t) e.nx() * nx / ref.nx(), (uint32_t) e.ny() * ny / ref.ny() };
+    }
+
     bool can_batch_with(const mtmd_image_tokens & other) {
-        return nx == other.nx && ny == other.ny && pos == other.pos;
+        // a canvas chunk holds a whole image group, its layout is not given by nx/ny alone
+        return nx == other.nx && ny == other.ny && pos == other.pos && pos != MTMD_POS_TYPE_CANVAS;
     }

     mtmd_image_tokens clone() {
@@ -516,6 +555,9 @@ struct mtmd_context {
     bool tok_row_end_trail = false;
     bool ov_img_first      = false;

+    // MiniCPM-V 4.6/4.7 prepends an <image_id>N</image_id> tag before <image>
+    bool use_image_id = false;
+
     // string template for slice image delimiters with row/col (idefics3)
     std::string sli_img_start_tmpl;

@@ -680,6 +722,7 @@ struct mtmd_context {
                     image_preproc = std::make_unique<mtmd_image_preprocessor_llava_uhd>(ctx_v);
                 } break;
             case PROJECTOR_TYPE_MINICPMV4_6:
+            case PROJECTOR_TYPE_MINICPMV4_7:
                 {
                     slice_tmpl        = MTMD_SLICE_TMPL_MINICPMV_2_6;
                     tok_ov_img_start  = {lookup_token("<image>")};
@@ -689,6 +732,7 @@ struct mtmd_context {
                     tok_row_end       = {lookup_token("\n")};
                     tok_row_end_trail = false; // no trailing end-of-row token
                     ov_img_first      = true;
+                    use_image_id      = true;
                     image_preproc     = std::make_unique<mtmd_image_preprocessor_minicpmv>(ctx_v);
                 } break;
             case PROJECTOR_TYPE_QWEN2VL:
@@ -1429,7 +1473,15 @@ struct mtmd_tokenizer {
             const bool has_tiling_grid = (preproc_out.grid_x > 0 && preproc_out.grid_y > 0)
                 || preproc_out.has_overview();

-            if (has_tiling_grid) {
+            if (has_tiling_grid && ctx->proj_type_v() == PROJECTOR_TYPE_MINICPMV4_7) {
+                GGML_ASSERT(bitmaps.size() == 1);
+                if (ctx->use_image_id) {
+                    add_text("<image_id>" + std::to_string(n_images_added) + "</image_id>", true);
+                }
+                add_text(ctx->tok_ov_img_start);
+                // the separators after <image> are appended by clip, see add_canvas_chunk()
+                add_canvas_chunk(std::move(preproc_out), bitmaps[0]->id);
+            } else if (has_tiling_grid) {
                 // [QWEN_VIDEO] we do not support "frame merging" for llama-uhd style, so no batching for now
                 GGML_ASSERT(bitmaps.size() == 1);

@@ -1448,6 +1500,9 @@ struct mtmd_tokenizer {

                 // add overview image (first)
                 if (ctx->ov_img_first) {
+                    if (ctx->use_image_id) {
+                        add_text("<image_id>" + std::to_string(n_images_added) + "</image_id>", true);
+                    }
                     add_text(ctx->tok_ov_img_start);
                     cur.entries.emplace_back(std::move(ov_chunk));
                     add_text(ctx->tok_ov_img_end);
@@ -1673,6 +1728,62 @@ struct mtmd_tokenizer {
         return 0;
     }

+    // MiniCPM-V 4.7: the overview and all slices go in one chunk, clip appends the separators after each tile:
+    //   [ov] </image><slice> [S00] </slice><slice> [S01] </slice>\n<slice> [S10] </slice><slice> [S11] </slice>
+    void add_canvas_chunk(mtmd_image_preproc_out && preproc_out, const std::string & id) {
+        const int n_col = preproc_out.grid_x;
+        const int n_row = preproc_out.grid_y;
+        auto & slices = preproc_out.entries;
+        GGML_ASSERT(preproc_out.has_overview());
+        GGML_ASSERT((int) slices.size() == n_col * n_row);
+
+        auto & ov = preproc_out.overview;
+        ov.suffix_type = CLIP_SUFFIX_MINICPMV_OV;
+        if (!slices.empty()) {
+            ov.suffix_type = CLIP_SUFFIX_MINICPMV_OV_SLICE;
+            ov.anyres.grid_x = n_col;
+            ov.anyres.grid_y = n_row;
+        }
+        for (int y = 0; y < n_row; y++) {
+            for (int x = 0; x < n_col; x++) {
+                auto & suffix = slices[y * n_col + x].suffix_type;
+                if (y == n_row - 1 && x == n_col - 1) {
+                    suffix = CLIP_SUFFIX_MINICPMV_LAST;
+                } else if (x == n_col - 1) {
+                    suffix = CLIP_SUFFIX_MINICPMV_ROW_END;
+                } else {
+                    suffix = CLIP_SUFFIX_MINICPMV_SLICE;
+                }
+            }
+        }
+
+        mtmd_image_tokens_ptr image_tokens(new mtmd_image_tokens);
+        image_tokens->pos = MTMD_POS_TYPE_CANVAS;
+        image_tokens->id  = id;
+        auto & entries = image_tokens->batch_f32.entries;
+        entries.push_back(std::move(ov));
+        for (auto & slice : slices) {
+            entries.push_back(std::move(slice));
+        }
+        // token grid of the last entry, the grids of the other entries are scaled from it
+        image_tokens->nx = clip_n_output_tokens_x(ctx->ctx_v, &entries.back());
+        image_tokens->ny = clip_n_output_tokens_y(ctx->ctx_v, &entries.back());
+
+        size_t n_tokens = 0;
+        for (const auto & entry : entries) {
+            n_tokens += clip_n_output_tokens(ctx->ctx_v, &entry);
+        }
+        GGML_ASSERT(n_tokens == image_tokens->n_tokens());
+
+        mtmd_input_chunk chunk{
+            MTMD_INPUT_CHUNK_TYPE_IMAGE,
+            {}, // text tokens
+            std::move(image_tokens),
+            nullptr, // audio tokens
+        };
+        cur.entries.emplace_back(std::move(chunk));
+    }
+
     std::vector<mtmd_input_chunk> split_batch_to_chunk(mtmd_image_preproc_out && preproc_out, const std::string & id) {
         std::vector<mtmd_input_chunk> chunks;

@@ -1814,6 +1925,24 @@ static int32_t mtmd_encode_impl(mtmd_context * ctx, const mtmd_image_tokens * im
         return 1;
     }

+    if (image_tokens->pos == MTMD_POS_TYPE_CANVAS) {
+        // the tiles differ in size, encode them one by one
+        size_t offset = 0;
+        for (const auto & entry : image_tokens->batch_f32.entries) {
+            clip_image_f32_batch one;
+            one.entries.push_back(entry);
+            std::vector<float> embd((size_t) n_embd_out * clip_n_output_tokens(ctx_clip, &entry));
+            if (!clip_image_batch_encode(ctx_clip, ctx->n_threads, &one, embd)) {
+                return 1;
+            }
+            GGML_ASSERT(offset + embd.size() <= out_embd.size());
+            std::copy(embd.begin(), embd.end(), out_embd.begin() + offset);
+            offset += embd.size();
+        }
+        GGML_ASSERT(offset == out_embd.size());
+        return 0;
+    }
+
     bool ok = clip_image_batch_encode(
         ctx_clip,
         ctx->n_threads,
@@ -2494,6 +2623,67 @@ size_t mtmd_image_tokens_get_ny(const mtmd_image_tokens * image_tokens) {
     return image_tokens->ny;
 }

+// map a tile coordinate onto the canvas like the reference: round(linspace(0, canvas - 1, grid)), round() breaks ties to even
+static uint32_t mtmd_canvas_scale(uint32_t coord, uint32_t grid, uint32_t canvas) {
+    if (grid <= 1 || canvas <= 1) {
+        return 0;
+    }
+    const double v = (double) coord * (double) (canvas - 1) / (double) (grid - 1);
+    return std::min((uint32_t) std::nearbyint(v), canvas - 1);
+}
+
+// MTMD_POS_TYPE_CANVAS: every tile shares the <image> token before the chunk as origin
+// the overview is stretched over the whole canvas, each slice fills its own cell; the time component is the origin, in slot z
+// a tile takes one position in slot t (the KV cache position), the separators after it take one position each
+static mtmd_decoder_pos mtmd_canvas_decoder_pos(const mtmd_image_tokens * image_tokens, llama_pos pos_0, size_t i) {
+    const auto & entries = image_tokens->batch_f32.entries;
+    const auto & grid    = entries[0].anyres;
+    const uint32_t nx = image_tokens->nx;
+    const uint32_t ny = image_tokens->ny;
+    const uint32_t canvas_w = grid.is_tiled() ? grid.grid_x * nx : nx;
+    const uint32_t canvas_h = grid.is_tiled() ? grid.grid_y * ny : ny;
+    const uint32_t base = pos_0 - 1;
+
+    mtmd_decoder_pos pos;
+    uint32_t t = pos_0;
+    for (size_t k = 0; k < entries.size(); ++k) {
+        const auto [gw, gh] = image_tokens->canvas_tile_grid(k);
+        if (i < gw * gh) {
+            const uint32_t row = i / gw;
+            const uint32_t col = i % gw;
+            uint32_t h;
+            uint32_t w;
+            if (k == 0) {
+                h = mtmd_canvas_scale(row, gh, canvas_h);
+                w = mtmd_canvas_scale(col, gw, canvas_w);
+            } else {
+                const uint32_t s = k - 1;
+                h = (s / grid.grid_x) * ny + row;
+                w = (s % grid.grid_x) * nx + col;
+            }
+            pos.t = t;
+            pos.x = base + w;
+            pos.y = base + h;
+            pos.z = base;
+            return pos;
+        }
+        i -= gw * gh;
+
+        const size_t n_sep = clip_suffix_rows(entries[k].suffix_type).size();
+        if (i < n_sep) {
+            const uint32_t p = t + 1 + i;
+            pos.t = p;
+            pos.x = p;
+            pos.y = p;
+            pos.z = p;
+            return pos;
+        }
+        i -= n_sep;
+        t += 1 + n_sep;
+    }
+    GGML_ABORT("token index out of range");
+}
+
 mtmd_decoder_pos mtmd_image_tokens_get_decoder_pos(const mtmd_image_tokens * image_tokens, llama_pos pos_0, size_t i) {
     mtmd_decoder_pos pos;
     switch (image_tokens->pos) {
@@ -2543,6 +2733,10 @@ mtmd_decoder_pos mtmd_image_tokens_get_decoder_pos(const mtmd_image_tokens * ima
                     pos.z = image_tokens->image_idx;
                 }
             } break;
+        case MTMD_POS_TYPE_CANVAS:
+            {
+                pos = mtmd_canvas_decoder_pos(image_tokens, pos_0, i);
+            } break;
         default:
             GGML_ABORT("invalid position type");
     }
@@ -2563,6 +2757,15 @@ llama_pos mtmd_image_tokens_get_n_pos(const mtmd_image_tokens * image_tokens) {
             // HunyuanVL: the sequential (dim-0) position advances by the full token count
             // (includes BOI/EOI and row newline tokens), not by max(nx, ny)
             return image_tokens->n_tokens();
+        case MTMD_POS_TYPE_CANVAS:
+            {
+                // one position per tile, plus one per separator
+                llama_pos n_pos = 0;
+                for (const auto & entry : image_tokens->batch_f32.entries) {
+                    n_pos += 1 + (llama_pos) clip_suffix_rows(entry.suffix_type).size();
+                }
+                return n_pos;
+            }
         default:
             GGML_ABORT("invalid position type");
     }