model : support MiniCPM-V 4.7 (#29416)

* mtmd : add MiniCPM-V 4.7 support

Signed-off-by: tc-mb <tianchi_cai@icloud.com>

* model : allow mrope time from an extra position slot

Signed-off-by: tc-mb <tianchi_cai@icloud.com>

* Update conversion/minicpm.py

Co-authored-by: Sigbjørn Skjæret <sigbjorn.skjaeret@huggingface.co>

* Slim down comments

Signed-off-by: tc-mb <tianchi_cai@icloud.com>

* fix for "do not hand-wrap comments"

Signed-off-by: tc-mb <tianchi_cai@icloud.com>

* fix ci

Signed-off-by: tc-mb <tianchi_cai@icloud.com>

* rm 3d repo for pr one

Signed-off-by: tc-mb <tianchi_cai@icloud.com>

* gguf: add rope.section_order metadata

* fix comments

* handle grid layout

* allow compat

---------

Signed-off-by: tc-mb <tianchi_cai@icloud.com>
Co-authored-by: Sigbjørn Skjæret <sigbjorn.skjaeret@huggingface.co>
Co-authored-by: Xuan Son Nguyen <son@huggingface.co>
This commit is contained in:
tc-mb
2026-10-10 20:41:59 +02:00
committed by GitHub
co-authored by Sigbjørn Skjæret Xuan Son Nguyen
parent abee0c8476
commit 69f201a205
23 changed files with 572 additions and 34 deletions
+2
View File
@@ -189,6 +189,7 @@ TEXT_MODEL_MAP: dict[str, str] = {
"MiniCPM3ForCausalLM": "minicpm",
"MiniCPMForCausalLM": "minicpm",
"MiniCPMV4_6ForConditionalGeneration": "minicpm",
"MiniCPMV4_7ForConditionalGeneration": "minicpm",
"MiniMaxText01ForCausalLM": "minimax",
"MiniMaxM1ForCausalLM": "minimax",
"MiniMaxM2ForCausalLM": "minimax",
@@ -346,6 +347,7 @@ MMPROJ_MODEL_MAP: dict[str, str] = {
"MiMoV2ForCausalLM": "mimo",
"MiniMaxM3SparseForConditionalGeneration": "minimax",
"MiniCPMV4_6ForConditionalGeneration": "minicpm",
"MiniCPMV4_7ForConditionalGeneration": "minicpm",
"Mistral3ForConditionalGeneration": "llava",
"NemotronH_Nano_VL_V2": "nemotron",
"MuseGlimmerForConditionalGeneration": "muse_glimmer",
+5 -2
View File
@@ -1389,14 +1389,17 @@ class TextModel(ModelBase):
name, gen = item
# Skip multimodal tensors
if name.startswith(("mlp", "vit.", "vpm.", "siglip2.", "conformer.", "merger.", "resampler.", "sound_encoder.", "sound_projection.", "speech_embeddings.")) \
# strip the "model." wrapper so the prefixes below match (name is not returned)
if name.startswith("model."):
name = name[len("model."):]
if name.startswith(("mlp", "vit.", "vpm.", "siglip2.", "conformer.", "connector.", "merger.", "resampler.", "sound_encoder.", "sound_projection.", "speech_embeddings.")) \
or "visual." in name or "vision." in name or "audio." in name or "talker." in name \
or "vision_" in name or "audio_" in name \
or "token2wav." in name or "code2wav." in name \
or "projector." in name or "pre_mm_projector_norm" in name \
or "image_newline" in name or "view_seperator" in name \
or "patch_embed" in name or "patch_embedding" in name \
or "patch_merger." in name or "patch_merge_mlp." in name or "model.connector." in name:
or "patch_merger." in name or "patch_merge_mlp." in name:
return None
return super().filter_tensors(item)
+85 -4
View File
@@ -139,9 +139,16 @@ class MiniCPMV4_6TextModel(Qwen3_5TextModel):
@ModelBase.register("MiniCPMV4_6ForConditionalGeneration")
@ModelBase.example("openbmb/MiniCPM-V-4_6")
class MiniCPMV4_6VisionModel(MmprojModel):
projector_type = gguf.VisionProjectorType.MINICPMV4_6
# fallback for checkpoints whose preprocessor config omits `scale_resolution`
default_scale_resolution: int | None = None
def get_downsample_mode(self) -> str:
return self.preprocessor_config.get("downsample_mode", "16x")
def __init__(self, *args, **kwargs):
super().__init__(*args, **kwargs)
self.downsample_mode = self.preprocessor_config.get("downsample_mode", "16x")
self.downsample_mode = self.get_downsample_mode()
if self.downsample_mode not in {"4x", "16x"}:
raise ValueError(f"Unsupported downsample mode: {self.downsample_mode}")
if self.downsample_mode == "4x":
@@ -157,7 +164,8 @@ class MiniCPMV4_6VisionModel(MmprojModel):
# The CLIP loader in tools/mtmd/clip.cpp consumes `clip.vision.image_size`
# as the slice size and warmup resolution, so report `scale_resolution` there
# to match the upstream MiniCPMV4_6ImageProcessorPil slicing rules.
scale_resolution = self.preprocessor_config.get("scale_resolution")
scale_resolution = self.preprocessor_config.get(
"scale_resolution", self.default_scale_resolution)
if scale_resolution is not None:
self.hparams_vision["image_size"] = int(scale_resolution)
@@ -166,12 +174,15 @@ class MiniCPMV4_6VisionModel(MmprojModel):
assert self.hparams_vision is not None
# projector type string is consumed by clip_projector_type_from_string() in clip.cpp
# (mapped to PROJECTOR_TYPE_MINICPMV4_6).
self.gguf_writer.add_clip_projector_type(gguf.VisionProjectorType.MINICPMV4_6)
self.gguf_writer.add_clip_projector_type(self.projector_type)
self.gguf_writer.add_vision_projector_scale_factor(
2 if self.downsample_mode == "4x" else 4)
max_slice_nums = self.preprocessor_config.get("max_slice_nums")
if max_slice_nums is not None:
self.gguf_writer.add_vision_max_slice_nums(int(max_slice_nums))
# borrow wa_layer_indexes for vit_merger insertion point
insert_layer_id = int(self.global_config.get(
"insert_layer_id", self.hparams_vision.get("insert_layer_id", 6)))
@@ -191,3 +202,73 @@ class MiniCPMV4_6VisionModel(MmprojModel):
return None
return super().filter_tensors(item)
# MiniCPM-V 4.7 shares the v4.6 stack: the same Qwen3.5 text tower (MoE variant when the checkpoint says so) and the same SigLIP + vit_merger + merger vision tower.
@ModelBase.register("MiniCPMV4_7ForConditionalGeneration")
@ModelBase.example("openbmb/MiniCPM-V-4.7")
class MiniCPMV4_7TextModel(Qwen3_5TextModel):
model_arch = gguf.MODEL_ARCH.QWEN35
def set_gguf_parameters(self):
super().set_gguf_parameters()
# mtmd puts the time of the image canvas in slot z, slot t stays the KV cache position
self.gguf_writer.add_rope_section_order(gguf.RopeSectionOrder.ZYXT)
def __init__(self, dir_model, ftype, fname_out, *, hparams: dict | None = None, **kwargs):
if hparams is None:
hparams = ModelBase.load_hparams(dir_model, is_mistral_format=False)
text_config = hparams.get("text_config", {})
if text_config.get("model_type") == "qwen3_5_moe_text":
self.model_arch = gguf.MODEL_ARCH.QWEN35MOE
else:
self.model_arch = gguf.MODEL_ARCH.QWEN35
super().__init__(dir_model, ftype, fname_out, hparams=hparams, **kwargs)
@classmethod
def filter_tensors(cls, item: tuple[str, Callable[[], Tensor]]) -> tuple[str, Callable[[], Tensor]] | None:
name, gen = item
# MTP tensors are not used yet
if name.startswith("mtp"):
return None
return super().filter_tensors(item)
@ModelBase.register("MiniCPMV4_7ForConditionalGeneration")
@ModelBase.example("openbmb/MiniCPM-V-4.7")
class MiniCPMV4_7VisionModel(MiniCPMV4_6VisionModel):
projector_type = gguf.VisionProjectorType.MINICPMV4_7
# MiniCPMV4_7ImageProcessorPil default
default_scale_resolution = 448
# rows of v.tok_embd_sep, the order must match clip_suffix_rows() in clip-impl.h
tok_embd_sep = ["</image>", "<slice>", "</slice>", "\n"]
def get_downsample_mode(self) -> str:
# 4.7 moved downsample_mode to the model config; preprocessor value takes priority
return self.preprocessor_config.get(
"downsample_mode", self.global_config.get("downsample_mode", "16x"))
@classmethod
def filter_tensors(cls, item: tuple[str, Callable[[], Tensor]]) -> tuple[str, Callable[[], Tensor]] | None:
# keep the text tok_embd, the separator rows are taken from it in modify_tensors
if item[0] == "model.language_model.embed_tokens.weight":
return item
return super().filter_tensors(item)
def modify_tensors(self, data_torch: Tensor, name: str, bid: int | None) -> Iterable[tuple[str, Tensor]]:
if name == "model.language_model.embed_tokens.weight":
# the tile separators are text tokens; clip appends their embeddings so that one chunk holds the whole image
from transformers import AutoTokenizer
tokenizer = AutoTokenizer.from_pretrained(self.dir_model)
ids = []
for text in self.tok_embd_sep:
tok = tokenizer.encode(text, add_special_tokens=False)
if len(tok) != 1:
raise ValueError(f"separator {text!r} must be a single token, got {tok}")
ids.append(tok[0])
yield self.format_tensor_name(gguf.MODEL_TENSOR.V_TOK_EMBD_SEP, suffix=""), data_torch[ids]
return
yield from super().modify_tensors(data_torch, name, bid)
+55
View File
@@ -0,0 +1,55 @@
## MiniCPM-V 4.7
### Prepare models and code
Download [MiniCPM-V-4.7](https://huggingface.co/openbmb/MiniCPM-V-4.7) PyTorch model from huggingface to "MiniCPM-V-4.7" folder.
The model must be the standard `transformers` checkpoint (no `trust_remote_code` for the text and vision graph used here); the architecture in `config.json` is `MiniCPMV4_7ForConditionalGeneration` with a `qwen3_5_text` (or `qwen3_5_moe_text`) text model and a SigLIP-based vision tower plus a window-attention `vit_merger`, same as MiniCPM-V 4.6.
If the checkpoint ships no MTP weights, pass `--no-mtp` to skip the nextn layers.
### Build llama.cpp
If there are differences in usage, please refer to the official build [documentation](https://github.com/ggml-org/llama.cpp/blob/master/docs/build.md)
Clone llama.cpp:
```bash
git clone https://github.com/ggml-org/llama.cpp
cd llama.cpp
```
Build llama.cpp using `CMake`:
```bash
cmake -B build
cmake --build build --config Release
```
### Usage of MiniCPM-V 4.7
MiniCPM-V 4.7 is converted directly through `convert_hf_to_gguf.py`. The same script is invoked twice on the original Hugging Face directory: once to produce the language-model GGUF and once with `--mmproj` to produce the multimodal projector GGUF.
```bash
# language model
python ./convert_hf_to_gguf.py ../MiniCPM-V-4.7 --outfile ../MiniCPM-V-4.7/ggml-model-f16.gguf --no-mtp
# multimodal projector (vision tower + window-attention vit_merger + DownsampleMLP merger)
python ./convert_hf_to_gguf.py ../MiniCPM-V-4.7 --mmproj --outfile ../MiniCPM-V-4.7/mmproj-model-f16.gguf
# optional: quantize to Q4_K_M
./build/bin/llama-quantize ../MiniCPM-V-4.7/ggml-model-f16.gguf ../MiniCPM-V-4.7/ggml-model-Q4_K_M.gguf Q4_K_M
```
The default projector merges 16x (4x4 patches into one token). To keep 4x more visual tokens, copy the model dir and set `"downsample_mode": "4x"` in the copy's `preprocessor_config.json` before running the `--mmproj` conversion; the loader reads `clip.vision.projector.scale_factor` to pick the graph.
Inference on Linux or Mac
```bash
# run in single-turn mode
./build/bin/llama-mtmd-cli -m ../MiniCPM-V-4.7/ggml-model-f16.gguf --mmproj ../MiniCPM-V-4.7/mmproj-model-f16.gguf -c 4096 --jinja --image xx.jpg -p "What is in the image?"
# run in conversation mode
./build/bin/llama-mtmd-cli -m ../MiniCPM-V-4.7/ggml-model-Q4_K_M.gguf --mmproj ../MiniCPM-V-4.7/mmproj-model-f16.gguf --jinja
```
The chat template enables thinking by default. Pass `--chat-template-kwargs '{"enable_thinking": false}'` to `llama-server` to turn it off.
+12
View File
@@ -259,6 +259,7 @@ class Keys:
DIMENSION_COUNT = "{arch}.rope.dimension_count"
DIMENSION_COUNT_SWA = "{arch}.rope.dimension_count_swa"
DIMENSION_SECTIONS = "{arch}.rope.dimension_sections"
SECTION_ORDER = "{arch}.rope.section_order"
FREQ_BASE = "{arch}.rope.freq_base"
FREQ_BASE_SWA = "{arch}.rope.freq_base_swa"
SCALING_TYPE = "{arch}.rope.scaling.type"
@@ -409,6 +410,7 @@ class Keys:
BLOCK_COUNT = "clip.vision.block_count"
IMAGE_MEAN = "clip.vision.image_mean"
IMAGE_STD = "clip.vision.image_std"
MAX_SLICE_NUMS = "clip.vision.max_slice_nums"
IMAGE_RESIZE_ALGO = "clip.vision.image_resize_algo"
SPATIAL_MERGE_SIZE = "clip.vision.spatial_merge_size"
SWIGLU_CLAMP = "clip.vision.swiglu_clamp"
@@ -1051,6 +1053,7 @@ class MODEL_TENSOR(IntEnum):
V_SAM_NET_3 = auto() # Deepseek-OCR
V_ENC_EMBD_IMGNL = auto() # Deepseek-OCR
V_ENC_EMBD_VSEP = auto() # Deepseek-OCR
V_TOK_EMBD_SEP = auto() # MiniCPM-V 4.7
V_RESMPL_QUERY_768 = auto() # Deepseek-OCR-2
V_RESMPL_QUERY_1024 = auto() # Deepseek-OCR-2
@@ -1832,6 +1835,7 @@ TENSOR_NAMES: dict[MODEL_TENSOR, str] = {
MODEL_TENSOR.V_SAM_NET_3: "v.sam.net_3",
MODEL_TENSOR.V_ENC_EMBD_IMGNL: "v.image_newline", # Deepseek-OCR, Granite4Vision
MODEL_TENSOR.V_ENC_EMBD_VSEP: "v.view_seperator", # Deepseek-OCR
MODEL_TENSOR.V_TOK_EMBD_SEP: "v.tok_embd_sep", # MiniCPM-V 4.7
MODEL_TENSOR.V_RESMPL_QUERY_768: "v.resample_query_768", # Deepseek-OCR-2 qwen2
MODEL_TENSOR.V_RESMPL_QUERY_1024: "v.resample_query_1024", # Deepseek-OCR-2 qwen2
# Granite4 Vision
@@ -2079,6 +2083,7 @@ MODEL_TENSORS: dict[MODEL_ARCH, list[MODEL_TENSOR]] = {
MODEL_TENSOR.V_ENC_EMBD_POS,
MODEL_TENSOR.V_ENC_EMBD_IMGNL,
MODEL_TENSOR.V_ENC_EMBD_VSEP,
MODEL_TENSOR.V_TOK_EMBD_SEP,
MODEL_TENSOR.V_ENC_INPUT_NORM,
MODEL_TENSOR.V_ENC_ATTN_QKV,
MODEL_TENSOR.V_ENC_ATTN_Q,
@@ -5927,6 +5932,12 @@ class RopeScalingType(Enum):
LONGROPE = 'longrope'
# M-RoPE: input position slot (t, y, x, z) that feeds each RoPE section, in section order
class RopeSectionOrder(Enum):
TYXZ = 'tyxz' # default
ZYXT = 'zyxt'
class PoolingType(IntEnum):
NONE = 0
MEAN = 1
@@ -6133,6 +6144,7 @@ class VisionProjectorType:
PARAKEET = "parakeet" # audio
MINIMAXM3 = "minimax_m3"
MINICPMV4_6 = "minicpmv4_6"
MINICPMV4_7 = "minicpmv4_7"
GRANITE_SPEECH = "granite_speech" # audio
MIMOVL = "mimovl"
MIMO_AUDIO = "mimo_audio"
+7
View File
@@ -24,6 +24,7 @@ from .constants import (
GGUFEndian,
GGUFValueType,
Keys,
RopeSectionOrder,
RopeScalingType,
PoolingType,
TokenType,
@@ -1154,6 +1155,9 @@ class GGUFWriter:
def add_rope_dimension_sections(self, dims: Sequence[int]) -> None:
self.add_array(Keys.Rope.DIMENSION_SECTIONS.format(arch=self.arch), dims)
def add_rope_section_order(self, value: RopeSectionOrder) -> None:
self.add_string(Keys.Rope.SECTION_ORDER.format(arch=self.arch), value.value)
def add_rope_freq_base(self, value: float) -> None:
self.add_float32(Keys.Rope.FREQ_BASE.format(arch=self.arch), value)
@@ -1462,6 +1466,9 @@ class GGUFWriter:
def add_vision_projector_scale_factor(self, value: int) -> None:
self.add_uint32(Keys.ClipVision.Projector.SCALE_FACTOR, value)
def add_vision_max_slice_nums(self, value: int) -> None:
self.add_uint32(Keys.ClipVision.MAX_SLICE_NUMS, value)
def add_vision_n_wa_pattern(self, value: int) -> None:
"""Add window attention pattern interval for vision models.
+1 -1
View File
@@ -1078,7 +1078,7 @@ extern "C" {
// Set custom position for the token at index idx in the batch
// For M-RoPE models:
// - Embedding tokens must have multiple positions per token
// - Embedding tokens must have n_pos_per_embd positions per token, in order [t, y, x, z]; t is also the KV cache position
// - Text token only requires one single position per token
LLAMA_API bool llama_batch_ext_set_pos(
struct llama_batch_ext * batch,
+1
View File
@@ -330,6 +330,7 @@ static const std::map<llm_kv, const char *> LLM_KV_NAMES = {
{ LLM_KV_ROPE_DIMENSION_COUNT, "%s.rope.dimension_count" },
{ LLM_KV_ROPE_DIMENSION_COUNT_SWA, "%s.rope.dimension_count_swa" },
{ LLM_KV_ROPE_DIMENSION_SECTIONS, "%s.rope.dimension_sections" },
{ LLM_KV_ROPE_SECTION_ORDER, "%s.rope.section_order" },
{ LLM_KV_ROPE_FREQ_BASE, "%s.rope.freq_base" },
{ LLM_KV_ROPE_FREQ_BASE_SWA, "%s.rope.freq_base_swa" },
{ LLM_KV_ROPE_SCALE_LINEAR, "%s.rope.scale_linear" },
+1
View File
@@ -335,6 +335,7 @@ enum llm_kv {
LLM_KV_ROPE_DIMENSION_COUNT,
LLM_KV_ROPE_DIMENSION_COUNT_SWA,
LLM_KV_ROPE_DIMENSION_SECTIONS,
LLM_KV_ROPE_SECTION_ORDER,
LLM_KV_ROPE_FREQ_BASE,
LLM_KV_ROPE_FREQ_BASE_SWA,
LLM_KV_ROPE_SCALE_LINEAR,
+28 -2
View File
@@ -172,7 +172,33 @@ void llm_graph_input_pos::set_input(const llama_ubatch * ubatch) {
if (ubatch->pos && pos) {
const int64_t n_tokens = ubatch->n_tokens;
ggml_backend_tensor_set(pos, ubatch->pos, 0, n_tokens*n_pos_per_embd*ggml_element_size(pos));
const bool has_embd = ubatch->is_mixed() || ubatch->token == nullptr;
if (rope_section_order == LLAMA_ROPE_SECTION_ORDER_TYXZ || !has_embd) {
ggml_backend_tensor_set(pos, ubatch->pos, 0, n_tokens*n_pos_per_embd*ggml_element_size(pos));
return;
}
// input is always [t, y, x, z]
// token entries are expanded by the batch to [p, p, p, 0]
GGML_ASSERT(n_pos_per_embd == 4);
// slot index per section, slots are t = 0, y = 1, x = 2, z = 3
std::array<int64_t, 4> slot_of_section = { 0, 1, 2, 3 };
switch (rope_section_order) {
case LLAMA_ROPE_SECTION_ORDER_TYXZ: break;
case LLAMA_ROPE_SECTION_ORDER_ZYXT: slot_of_section = { 3, 1, 2, 0 }; break;
default: GGML_ABORT("unsupported rope section order");
}
std::vector<llama_pos> pos_data(n_tokens*n_pos_per_embd);
for (int64_t i = 0; i < n_tokens; ++i) {
const bool is_embd = ubatch->is_mixed() ? ubatch->type[i] != 0 : true;
for (int64_t s = 0; s < 4; ++s) {
const int64_t slot = is_embd ? slot_of_section[s] : s;
pos_data[s*n_tokens + i] = ubatch->pos[slot*n_tokens + i];
}
}
ggml_backend_tensor_set(pos, pos_data.data(), 0, pos_data.size()*ggml_element_size(pos));
}
}
@@ -2620,7 +2646,7 @@ ggml_tensor * llm_graph_context::build_inp_embd(ggml_tensor * tok_embd, float to
}
ggml_tensor * llm_graph_context::build_inp_pos() const {
auto inp = std::make_unique<llm_graph_input_pos>(hparams.n_pos_per_embd());
auto inp = std::make_unique<llm_graph_input_pos>(hparams.n_pos_per_embd(), hparams.rope_section_order);
auto & cur = inp->pos;
+3 -1
View File
@@ -169,7 +169,8 @@ public:
class llm_graph_input_pos : public llm_graph_input_i {
public:
llm_graph_input_pos(uint32_t n_pos_per_embd) : n_pos_per_embd(n_pos_per_embd) {}
llm_graph_input_pos(uint32_t n_pos_per_embd, llama_rope_section_order rope_section_order = LLAMA_ROPE_SECTION_ORDER_TYXZ)
: n_pos_per_embd(n_pos_per_embd), rope_section_order(rope_section_order) {}
virtual ~llm_graph_input_pos() = default;
void set_input(const llama_ubatch * ubatch) override;
@@ -179,6 +180,7 @@ public:
ggml_tensor * pos = nullptr; // I32 [n_batch]
const uint32_t n_pos_per_embd = 1;
const llama_rope_section_order rope_section_order = LLAMA_ROPE_SECTION_ORDER_TYXZ;
};
// temperature tuning, used by llama4
+9
View File
@@ -36,6 +36,13 @@ enum llama_non_causal_type {
LLAMA_NON_CAUSAL_TYPE_SWA_FULL = 2, // all layers non-causal, SWA not applied between tokens of the current ubatch (deepseek 4)
};
// M-RoPE: which input position slot feeds each RoPE section
enum llama_rope_section_order {
LLAMA_ROPE_SECTION_ORDER_UNSPECIFIED = -1,
LLAMA_ROPE_SECTION_ORDER_TYXZ = 0, // default, slot i feeds section i
LLAMA_ROPE_SECTION_ORDER_ZYXT = 1, // MiniCPM-V 4.7: time last
};
// forward declaration; full definition in llama-graph.h
enum llm_ffn_op_type : int;
@@ -166,6 +173,8 @@ struct llama_hparams {
std::array<int, 4> rope_sections;
enum llama_rope_section_order rope_section_order = LLAMA_ROPE_SECTION_ORDER_TYXZ;
// Per-layer RoPE enable flags (1 = use RoPE, 0 = NoPE)
// by default, all layers use RoPE (controlled by rope_finetuned)
std::array<uint32_t, LLAMA_MAX_LAYERS> rope_pattern;
+1
View File
@@ -365,6 +365,7 @@ void llama_model_saver::add_kv_from_model() {
add_kv(LLM_KV_ROPE_DIMENSION_COUNT, hparams.n_rot_full);
add_kv(LLM_KV_ROPE_DIMENSION_COUNT_SWA, hparams.n_rot_swa);
add_kv(LLM_KV_ROPE_DIMENSION_SECTIONS, hparams.rope_sections);
add_kv(LLM_KV_ROPE_SECTION_ORDER, llama_rope_section_order_name(hparams.rope_section_order));
add_kv(LLM_KV_ROPE_FREQ_BASE, hparams.rope_freq_base_train);
add_kv(LLM_KV_ROPE_FREQ_BASE_SWA, hparams.rope_freq_base_train_swa);
// add_kv(LLM_KV_ROPE_SCALE_LINEAR, rope_scaling_factor); // old name
+33
View File
@@ -1064,6 +1064,25 @@ static llama_rope_scaling_type llama_rope_scaling_type_from_string(const std::st
return LLAMA_ROPE_SCALING_TYPE_UNSPECIFIED;
}
static const std::map<llama_rope_section_order, const char *> LLAMA_ROPE_SECTION_ORDERS = {
{ LLAMA_ROPE_SECTION_ORDER_TYXZ, "tyxz" },
{ LLAMA_ROPE_SECTION_ORDER_ZYXT, "zyxt" },
};
std::string llama_rope_section_order_name(llama_rope_section_order rope_section_order) {
return LLAMA_ROPE_SECTION_ORDERS.at(rope_section_order);
}
static llama_rope_section_order llama_rope_section_order_from_string(const std::string & name) {
for (const auto & kv : LLAMA_ROPE_SECTION_ORDERS) {
if (kv.second == name) {
return kv.first;
}
}
return LLAMA_ROPE_SECTION_ORDER_UNSPECIFIED;
}
// Maps GGUF activation names to the FFN op type used by the graph builders.
static const std::map<std::string, llm_ffn_op_type> LLM_FFN_OP_TYPES_FROM_STRING = {
{ "gelu", LLM_FFN_GEGLU_ERF },
@@ -1447,6 +1466,13 @@ void llama_model_base::load_hparams(llama_model_loader & ml) {
hparams.rope_scaling_type_train = llama_rope_scaling_type_from_string(rope_scaling);
GGML_ASSERT(hparams.rope_scaling_type_train != LLAMA_ROPE_SCALING_TYPE_UNSPECIFIED);
std::string rope_section_order("tyxz");
ml.get_key(LLM_KV_ROPE_SECTION_ORDER, rope_section_order, false);
hparams.rope_section_order = llama_rope_section_order_from_string(rope_section_order);
if (hparams.rope_section_order == LLAMA_ROPE_SECTION_ORDER_UNSPECIFIED) {
throw std::runtime_error("unknown rope section order: " + rope_section_order);
}
// TODO: Handle SWA metadata similarly when models start implementing it
// rope_freq_scale (inverse of the kv) is optional
float ropescale = 0.0f;
@@ -1517,6 +1543,10 @@ void llama_model_base::load_hparams(llama_model_loader & ml) {
}
hparams.rope_type = llama_model_rope_type(this);
if (hparams.rope_section_order != LLAMA_ROPE_SECTION_ORDER_TYXZ && hparams.n_pos_per_embd() != 4) {
throw std::runtime_error("rope section order " + llama_rope_section_order_name(hparams.rope_section_order) + " requires M-RoPE");
}
}
void llama_model_base::load_vocab(llama_model_loader & ml) {
@@ -2158,6 +2188,9 @@ void llama_model::print_info() const {
if (const auto & s = hparams.rope_sections; s[0] || s[1] || s[2] || s[3]) {
LLAMA_LOG_INFO("%s: mrope sections = [%d, %d, %d, %d]\n", __func__, s[0], s[1], s[2], s[3]);
}
if (hparams.rope_section_order != LLAMA_ROPE_SECTION_ORDER_TYXZ) {
LLAMA_LOG_INFO("%s: rope section order = %s\n", __func__, llama_rope_section_order_name(hparams.rope_section_order).c_str());
}
if (!classifier_labels.empty()) {
LLAMA_LOG_INFO("%s: n_cls_out = %u\n", __func__, hparams.n_cls_out);
+1
View File
@@ -160,6 +160,7 @@ enum llm_type {
};
std::string llama_rope_scaling_type_name(llama_rope_scaling_type rope_scaling_type);
std::string llama_rope_section_order_name(llama_rope_section_order rope_section_order);
// Map a GGUF activation-name string to llm_ffn_op_type. Returns `fallback` if
// the string is empty or not recognized.
+3
View File
@@ -166,4 +166,7 @@ struct clip_graph {
// Generic function to stack frames for audio processing
// Abstracts out the StackAudioFrames logic used by ultravox
ggml_tensor * build_stack(ggml_tensor * cur, int32_t stack_factor, int32_t n_embed);
// append the separators of img.suffix_type after the image tokens
ggml_tensor * build_suffix(ggml_tensor * cur);
};
+41 -3
View File
@@ -68,6 +68,7 @@
#define KEY_MM_PATCH_MERGE_TYPE "clip.vision.mm_patch_merge_type"
#define KEY_IMAGE_GRID_PINPOINTS "clip.vision.image_grid_pinpoints"
#define KEY_MAX_SLICE_NUMS "clip.vision.max_slice_nums"
#define KEY_WIN_ATTN_PATTERN "clip.vision.n_wa_pattern"
#define KEY_WIN_ATTN_LAYER_INDEXES "clip.vision.wa_layer_indexes"
#define KEY_WA_PATTERN_MODE "clip.vision.wa_pattern_mode"
@@ -146,6 +147,7 @@
#define TN_MVLM_PROJ_PEG "mm.model.peg.%d.%s"
#define TN_IMAGE_NEWLINE "v.image_newline"
#define TN_IMAGE_SEPERATOR "v.view_seperator"
#define TN_TOK_EMBD_SEP "v.tok_embd_sep"
#define TN_MM_INP_NORM "mm.input_norm.weight"
#define TN_MM_INP_NORM_B "mm.input_norm.bias"
#define TN_MM_INP_PROJ "mm.input_projection.weight" // gemma3
@@ -500,6 +502,7 @@ enum projector_type {
PROJECTOR_TYPE_PARAKEET,
PROJECTOR_TYPE_EXAONE4_5,
PROJECTOR_TYPE_MINICPMV4_6,
PROJECTOR_TYPE_MINICPMV4_7,
PROJECTOR_TYPE_GRANITE_SPEECH,
PROJECTOR_TYPE_MIMOVL,
PROJECTOR_TYPE_MINIMAX_M3,
@@ -569,6 +572,7 @@ static std::map<projector_type, std::string> PROJECTOR_TYPE_NAMES = {
{ PROJECTOR_TYPE_EXAONE4_5, "exaone4_5"},
{ PROJECTOR_TYPE_HUNYUANVL, "hunyuanvl"},
{ PROJECTOR_TYPE_MINICPMV4_6, "minicpmv4_6"},
{ PROJECTOR_TYPE_MINICPMV4_7, "minicpmv4_7"},
{ PROJECTOR_TYPE_GRANITE_SPEECH, "granite_speech"},
{ PROJECTOR_TYPE_MIMOVL, "mimovl"},
{ PROJECTOR_TYPE_MINIMAX_M3, "minimax_m3"},
@@ -660,6 +664,38 @@ struct clip_image_u8 {
struct mtmd_serialization; // forward declaration
// separators appended after the image tokens of one entry, as rows of v.tok_embd_sep
enum clip_suffix_type : int32_t {
CLIP_SUFFIX_NONE = 0,
// MiniCPM-V 4.7 tiles
CLIP_SUFFIX_MINICPMV_OV, // </image>
CLIP_SUFFIX_MINICPMV_OV_SLICE, // </image><slice>
CLIP_SUFFIX_MINICPMV_SLICE, // </slice><slice>
CLIP_SUFFIX_MINICPMV_ROW_END, // </slice>\n<slice>
CLIP_SUFFIX_MINICPMV_LAST, // </slice>
CLIP_SUFFIX_COUNT,
};
// rows of v.tok_embd_sep for each suffix type
// MiniCPM-V 4.7 rows (set by the converter): 0 = </image>, 1 = <slice>, 2 = </slice>, 3 = \n
static inline const std::vector<int> & clip_suffix_rows(clip_suffix_type type) {
static const std::vector<int> none;
static const std::vector<int> minicpmv_ov = { 0 };
static const std::vector<int> minicpmv_ov_slice = { 0, 1 };
static const std::vector<int> minicpmv_slice = { 2, 1 };
static const std::vector<int> minicpmv_row_end = { 2, 3, 1 };
static const std::vector<int> minicpmv_last = { 2 };
switch (type) {
case CLIP_SUFFIX_NONE: return none;
case CLIP_SUFFIX_MINICPMV_OV: return minicpmv_ov;
case CLIP_SUFFIX_MINICPMV_OV_SLICE: return minicpmv_ov_slice;
case CLIP_SUFFIX_MINICPMV_SLICE: return minicpmv_slice;
case CLIP_SUFFIX_MINICPMV_ROW_END: return minicpmv_row_end;
case CLIP_SUFFIX_MINICPMV_LAST: return minicpmv_last;
default: GGML_ABORT("invalid suffix type");
}
}
// For images, buf.size() == nx*ny*3
// Memory layout: RGBRGBRGB...
// For seq, buf.size() == nx*ny*3*nt
@@ -675,10 +711,12 @@ struct clip_image_f32 {
// deepseek4v: number of leading IMAGE_PAD embeddings, aligns IMAGE_START to the LLM compressor ratio
// depends on the chunk position, set at tokenize time (see mtmd_tokenizer::add_media)
int32_t lead_pad = 0;
// separators appended after the image tokens
clip_suffix_type suffix_type = CLIP_SUFFIX_NONE;
// llava-next "anyres" tiling, used by Granite4 Vision
// the whole grid is encoded and assembled in a single graph
// NOTE: excluded from serialized: a deserialized image is always a placeholder, which is never encoded
// tile grid of the image group this entry belongs to
// llava-next "anyres" (Granite4 Vision): the whole grid is encoded and assembled in a single graph
// MiniCPM-V 4.7: set on the overview entry, the decoder positions of all tiles are derived from it
struct anyres_info {
int grid_x = 0; // tiles per row, 0 means the image is not tiled
int grid_y = 0; // tiles per column
+2
View File
@@ -71,6 +71,7 @@ struct clip_hparams {
std::vector<clip_image_size> image_res_candidates;
int32_t preproc_min_tiles = 0;
int32_t preproc_max_tiles = 0;
int32_t max_slice_nums = 9; // llava-uhd slice cap; per-model, carried in the GGUF
int32_t preproc_tile_size = 0; // local tile size (deepseek-ocr)
resize_algo image_resize_algo_rf = RESIZE_ALGO_BICUBIC;
resize_algo image_resize_algo_ov = RESIZE_ALGO_BICUBIC;
@@ -614,6 +615,7 @@ struct clip_model {
ggml_tensor * image_newline = nullptr;
ggml_tensor * view_seperator = nullptr;
ggml_tensor * tok_embd_sep = nullptr; // [n_embd_text, n_sep] rows of the text model tok_embd (MiniCPM-V 4.7)
// Yi type models with mlp+normalization projection
+30 -1
View File
@@ -909,6 +909,16 @@ ggml_tensor * clip_graph::build_stack(ggml_tensor * cur, int32_t stack_factor, i
// aka pixel_shuffle / pixel_unshuffle / patch_merger (Kimi-VL)
// support dynamic resolution
ggml_tensor * clip_graph::build_suffix(ggml_tensor * cur) {
for (int idx : clip_suffix_rows(img.suffix_type)) {
GGML_ASSERT(model.tok_embd_sep && idx < model.tok_embd_sep->ne[1]);
ggml_tensor * row = ggml_view_2d(ctx0, model.tok_embd_sep, model.tok_embd_sep->ne[0], 1,
model.tok_embd_sep->nb[1], idx * model.tok_embd_sep->nb[1]);
cur = ggml_concat(ctx0, cur, ggml_cast(ctx0, row, cur->type), 1);
}
return cur;
}
ggml_tensor * clip_graph::build_patch_merge_permute(ggml_tensor * cur, int scale_factor) {
GGML_ASSERT(scale_factor > 1);
@@ -1020,6 +1030,7 @@ static std::unique_ptr<clip_graph> clip_get_graph_builder(clip_ctx * ctx, const
builder = std::make_unique<clip_graph_minicpmv>(ctx, img);
} break;
case PROJECTOR_TYPE_MINICPMV4_6:
case PROJECTOR_TYPE_MINICPMV4_7:
{
builder = std::make_unique<clip_graph_minicpmv4_6>(ctx, img);
} break;
@@ -1332,6 +1343,7 @@ struct clip_model_loader {
if (is_vision) {
get_u32(KEY_IMAGE_SIZE, hparams.image_size);
get_u32(KEY_PATCH_SIZE, hparams.patch_size);
get_u32(KEY_MAX_SLICE_NUMS, hparams.max_slice_nums, false);
get_i32(KEY_MINICPMV_VERSION, hparams.minicpmv_version, false); // legacy
get_u32(KEY_MINICPMV_QUERY_NUM, hparams.minicpmv_query_num, false);
if (hparams.minicpmv_query_num == 0) {
@@ -1471,13 +1483,18 @@ struct clip_model_loader {
}
} break;
case PROJECTOR_TYPE_MINICPMV4_6:
case PROJECTOR_TYPE_MINICPMV4_7:
{
// MiniCPM-V 4.6 unified merger projector
// MiniCPM-V 4.6/4.7 unified merger projector
// ViT merger 2x2 + final merger 2x2 = 4x spatial merge per dimension
hparams.n_merge = 4;
get_u32(KEY_PROJ_SCALE_FACTOR, hparams.n_merge, false);
GGML_ASSERT(hparams.n_merge == 2 || hparams.n_merge == 4);
// no padding: the reference stretches the refined image to the target size
hparams.image_pad_ov = PAD_NONE;
hparams.image_pad_rf = PAD_NONE;
// borrow wa_layer_indexes for vit_merger insertion point
std::vector<int> wa_layer_indexes_vec;
get_arr_int(KEY_WIN_ATTN_LAYER_INDEXES, wa_layer_indexes_vec, false);
@@ -2393,6 +2410,7 @@ struct clip_model_loader {
|| model.proj_type == PROJECTOR_TYPE_IDEFICS3
|| model.proj_type == PROJECTOR_TYPE_MINICPMV
|| model.proj_type == PROJECTOR_TYPE_MINICPMV4_6
|| model.proj_type == PROJECTOR_TYPE_MINICPMV4_7
) && layer.ff_up_w && layer.ff_down_w && layer.ff_down_w->ne[0] == hparams.n_embd;
if (is_ffn_swapped) {
// swap up and down weights
@@ -2495,6 +2513,7 @@ struct clip_model_loader {
model.mm_model_ln_post_b = get_tensor(string_format(TN_MINICPMV_LN, "post", "bias"));
} break;
case PROJECTOR_TYPE_MINICPMV4_6:
case PROJECTOR_TYPE_MINICPMV4_7:
{
const bool merger_required = hparams.n_merge == 4;
auto get_merger_tensor = [&](const std::string & name, bool required = true) {
@@ -2526,6 +2545,7 @@ struct clip_model_loader {
model.mm_ffn_up_b = get_tensor(string_format(TN_MM_UP, "bias"), false);
model.mm_ffn_down_w = get_tensor(string_format(TN_MM_DOWN, "weight"));
model.mm_ffn_down_b = get_tensor(string_format(TN_MM_DOWN, "bias"), false);
model.tok_embd_sep = get_tensor(TN_TOK_EMBD_SEP, model.proj_type == PROJECTOR_TYPE_MINICPMV4_7);
} break;
case PROJECTOR_TYPE_GLM_EDGE:
{
@@ -4169,6 +4189,8 @@ int clip_n_output_tokens_x(const clip_ctx * ctx, const clip_image_f32 * img) {
case PROJECTOR_TYPE_MUSE_GLIMMER:
return (img->nx() / params.patch_size) / 2;
case PROJECTOR_TYPE_STEP3VL:
case PROJECTOR_TYPE_MINICPMV4_6:
case PROJECTOR_TYPE_MINICPMV4_7:
return img->nx() / (params.patch_size * params.n_merge);
case PROJECTOR_TYPE_DEEPSEEKOCR:
case PROJECTOR_TYPE_DEEPSEEKOCR2:
@@ -4197,6 +4219,8 @@ int clip_n_output_tokens_y(const clip_ctx * ctx, const clip_image_f32 * img) {
case PROJECTOR_TYPE_MUSE_GLIMMER:
return (img->ny() / params.patch_size) / 2;
case PROJECTOR_TYPE_STEP3VL:
case PROJECTOR_TYPE_MINICPMV4_6:
case PROJECTOR_TYPE_MINICPMV4_7:
return img->ny() / (params.patch_size * params.n_merge);
default:
break;
@@ -4262,6 +4286,7 @@ int clip_n_output_tokens(const clip_ctx * ctx, const clip_image_f32 * img) {
}
} break;
case PROJECTOR_TYPE_MINICPMV4_6:
case PROJECTOR_TYPE_MINICPMV4_7:
{
n_patches /= params.n_merge * params.n_merge;
} break;
@@ -4513,6 +4538,8 @@ int clip_n_output_tokens(const clip_ctx * ctx, const clip_image_f32 * img) {
GGML_ABORT("unsupported projector type");
}
n_patches += (int) clip_suffix_rows(img->suffix_type).size();
return n_patches;
}
@@ -4836,6 +4863,7 @@ bool clip_encode(struct clip_ctx * ctx, struct clip_encode_params * params) {
set_input_f32("omega", omega);
} break;
case PROJECTOR_TYPE_MINICPMV4_6:
case PROJECTOR_TYPE_MINICPMV4_7:
{
const bool is_4x = hparams.n_merge == 2;
@@ -6080,6 +6108,7 @@ int clip_n_mmproj_embd(const struct clip_ctx * ctx) {
case PROJECTOR_TYPE_MINICPMV:
return ctx->model.mm_model_proj->ne[0];
case PROJECTOR_TYPE_MINICPMV4_6:
case PROJECTOR_TYPE_MINICPMV4_7:
return ctx->model.mm_ffn_down_w->ne[1];
case PROJECTOR_TYPE_GLM_EDGE:
return ctx->model.mm_model_mlp_3_w->ne[1];
+2
View File
@@ -350,6 +350,8 @@ ggml_cgraph * clip_graph_minicpmv4_6::build() {
inpL = cur;
}
inpL = build_suffix(inpL);
ggml_build_forward_expand(gf, inpL);
return gf;
}
+12 -16
View File
@@ -507,9 +507,7 @@ mtmd_image_preproc_out mtmd_image_preprocessor_llava_uhd::preprocess(const clip_
mtmd_image_preprocessor_llava_uhd::slice_instructions mtmd_image_preprocessor_llava_uhd::get_slice_instructions(const clip_image_size & original_size) const {
mtmd_image_preprocessor_llava_uhd::slice_instructions res;
// align slices by patch_size * n_merge so an integer number of merger output tokens fits per slice
const int n_merge = hparams.n_merge;
const int patch_size = hparams.patch_size * n_merge;
const int patch_size = get_slice_align();
const int slice_size = hparams.image_size;
const int original_width = original_size.width;
const int original_height = original_size.height;
@@ -568,7 +566,7 @@ mtmd_image_preprocessor_llava_uhd::slice_instructions mtmd_image_preprocessor_ll
res.overview_size = best_size;
{
const int max_slice_nums = 9; // TODO: this is only used by minicpmv, maybe remove it
const int max_slice_nums = hparams.max_slice_nums > 0 ? hparams.max_slice_nums : 9;
const float log_ratio = log((float)original_width / original_height);
const float ratio = (float)original_width * original_height / (slice_size * slice_size);
const int multiple = fmin(ceil(ratio), max_slice_nums);
@@ -691,7 +689,7 @@ clip_image_size mtmd_image_preprocessor_llava_uhd::select_best_resolution(const
}
int mtmd_image_preprocessor_llava_uhd::ensure_divide(int length, int patch_size) const {
return std::max(static_cast<int>(std::round(static_cast<float>(length) / patch_size) * patch_size), patch_size);
return std::max(align_round(static_cast<double>(length) / patch_size) * patch_size, patch_size);
}
clip_image_size mtmd_image_preprocessor_llava_uhd::get_refine_size(const clip_image_size & original_size, const clip_image_size & grid, int scale_resolution, int patch_size, bool allow_upscale) const {
@@ -893,17 +891,15 @@ mtmd_image_preproc_out mtmd_image_preprocessor_longest_edge::preprocess(const cl
//
mtmd_image_preprocessor_llava_uhd::slice_instructions mtmd_image_preprocessor_minicpmv::get_slice_instructions(const clip_image_size & original_size) const {
if (hparams.n_merge == 2) {
const int slice_size = hparams.image_size;
const float ratio = (float)original_size.width * original_size.height / (slice_size * slice_size);
if (ratio <= 1.0f) {
mtmd_image_preprocessor_llava_uhd::slice_instructions inst;
const int patch_size = hparams.patch_size * hparams.n_merge;
inst.overview_size = get_best_resize(original_size, slice_size, patch_size, true);
inst.refined_size = clip_image_size{0, 0};
inst.grid_size = clip_image_size{0, 0};
return inst;
}
// overview only for small images, unlike generic llava-uhd which slices once one side exceeds scale resolution
const int slice_size = hparams.image_size;
const float ratio = (float) original_size.width * original_size.height / (slice_size * slice_size);
if (ratio <= 1.0f) {
mtmd_image_preprocessor_llava_uhd::slice_instructions inst;
inst.overview_size = get_best_resize(original_size, slice_size, get_slice_align(), true);
inst.refined_size = clip_image_size{0, 0};
inst.grid_size = clip_image_size{0, 0};
return inst;
}
return mtmd_image_preprocessor_llava_uhd::get_slice_instructions(original_size);
}
+31
View File
@@ -83,6 +83,17 @@ struct mtmd_image_preprocessor_llava_uhd : mtmd_image_preprocessor {
slice_output slice_image(const clip_image_u8 & img, const slice_instructions & inst) const;
protected:
// align slices to a multiple of the merger factor (integer merger tokens per slice)
virtual int get_slice_align() const {
const int merge = hparams.n_merge > 0 ? hparams.n_merge : 1;
return hparams.patch_size * merge;
}
// rounding for snapping a length to a multiple of the align size
virtual int align_round(double v) const {
return static_cast<int>(std::round(v));
}
clip_image_size get_best_resize(const clip_image_size & original_size, int scale_resolution, int patch_size, bool allow_upscale = false) const;
/**
@@ -155,6 +166,26 @@ private:
struct mtmd_image_preprocessor_minicpmv : mtmd_image_preprocessor_llava_uhd {
using mtmd_image_preprocessor_llava_uhd::mtmd_image_preprocessor_llava_uhd;
slice_instructions get_slice_instructions(const clip_image_size & original_size) const override;
protected:
// always patch_size * 4, even in 4x mode (the 2x2 vit_merger slot stays)
int get_slice_align() const override {
return hparams.patch_size * 4;
}
// Python's round() breaks ties to even, unlike std::round
int align_round(double v) const override {
const double fl = std::floor(v);
const double diff = v - fl;
if (diff > 0.5) {
return static_cast<int>(fl) + 1;
}
if (diff < 0.5) {
return static_cast<int>(fl);
}
const int lo = static_cast<int>(fl);
return (lo % 2 == 0) ? lo : lo + 1;
}
};
// custom llava-uhd slicing logic for LFM2
+207 -4
View File
@@ -19,6 +19,7 @@
#include <algorithm>
#include <cerrno>
#include <cmath>
#include <cstdio>
#include <cstdlib>
#include <cstring>
@@ -27,7 +28,10 @@
#include <vector>
// remember to bump this if the serialization format changes
#define MTMD_SERIALIZATION_VERSION 2
#define MTMD_SERIALIZATION_VERSION 3
// oldest compat version that can be loaded
#define MTMD_SERIALIZATION_VERSION_MIN 2
struct mtmd_serialization {
// note: using 64-bit here for future-proofing
@@ -45,7 +49,7 @@ struct mtmd_serialization {
// copy buf to data
data.assign(buf, buf + len);
uint64_t ver_in = read<uint64_t>();
if (ver_in != version) {
if (ver_in < MTMD_SERIALIZATION_VERSION_MIN || ver_in > version) {
throw std::runtime_error("version mismatch");
}
this->version = ver_in;
@@ -106,6 +110,11 @@ void clip_image_f32::serialize(mtmd_serialization & ser) const {
ser.write(add_viewsep);
ser.write(add_newline);
ser.write(lead_pad);
ser.write((int32_t)suffix_type);
ser.write((int32_t)anyres.grid_x);
ser.write((int32_t)anyres.grid_y);
ser.write((int32_t)anyres.orig_nx);
ser.write((int32_t)anyres.orig_ny);
ser.write((int32_t)nx_);
ser.write((int32_t)ny_);
}
@@ -113,6 +122,17 @@ void clip_image_f32::deserialize(mtmd_serialization & ser) {
add_viewsep = ser.read<bool>();
add_newline = ser.read<bool>();
lead_pad = ser.read<int32_t>();
if (ser.version >= 3) {
const int32_t suffix_raw = ser.read<int32_t>();
if (suffix_raw < 0 || suffix_raw >= CLIP_SUFFIX_COUNT) {
throw std::runtime_error("invalid suffix type");
}
suffix_type = (clip_suffix_type)suffix_raw;
anyres.grid_x = ser.read<int32_t>();
anyres.grid_y = ser.read<int32_t>();
anyres.orig_nx = ser.read<int32_t>();
anyres.orig_ny = ser.read<int32_t>();
}
nx_ = ser.read<int32_t>();
ny_ = ser.read<int32_t>();
buf.clear(); // always a placeholder after loading
@@ -204,9 +224,11 @@ enum mtmd_pos_type {
MTMD_POS_TYPE_NORMAL, // number of positions equals to number of tokens
MTMD_POS_TYPE_MROPE, // qwen-vl mrope style, each image takes max(t,h,w) position indexes
MTMD_POS_TYPE_HUNYUANVL, // HunyuanVL mrope + BOI/EOI/newline layout with XD-RoPE dim-3
MTMD_POS_TYPE_CANVAS, // MiniCPM-V 4.7: overview + slices in one chunk, sharing one 2D canvas (see mtmd_image_tokens::canvas_tile_grid)
MTMD_POS_TYPE_COUNT, // for validation
};
struct mtmd_image_tokens {
uint32_t nx = 0; // number of tokens in x direction
uint32_t ny = 0; // number of tokens in y direction
@@ -218,6 +240,14 @@ struct mtmd_image_tokens {
// [BOI] [row0 tokens + newline] ... [row(ny-1) tokens + newline] [EOI]
return (nx + 1) * ny + 2;
}
if (pos == MTMD_POS_TYPE_CANVAS) {
uint32_t n = 0;
for (size_t k = 0; k < batch_f32.entries.size(); ++k) {
const auto [gw, gh] = canvas_tile_grid(k);
n += gw * gh + (uint32_t) clip_suffix_rows(batch_f32.entries[k].suffix_type).size();
}
return n;
}
uint32_t nz = batch_f32.entries.size();
if (n_temporal_merge > 1) {
// [QWEN_VIDEO] this logic is quite ugly, it's mostly to make qwen-vl temporal merge work, can be improved in the future
@@ -243,8 +273,17 @@ struct mtmd_image_tokens {
return false;
}
// MTMD_POS_TYPE_CANVAS: entries are [overview, slices row by row], nx/ny is the token grid of the last entry
// returns the token grid (w, h) of entry k, scaled from its pixel size
std::pair<uint32_t, uint32_t> canvas_tile_grid(size_t k) const {
const auto & ref = batch_f32.entries.back();
const auto & e = batch_f32.entries[k];
return { (uint32_t) e.nx() * nx / ref.nx(), (uint32_t) e.ny() * ny / ref.ny() };
}
bool can_batch_with(const mtmd_image_tokens & other) {
return nx == other.nx && ny == other.ny && pos == other.pos;
// a canvas chunk holds a whole image group, its layout is not given by nx/ny alone
return nx == other.nx && ny == other.ny && pos == other.pos && pos != MTMD_POS_TYPE_CANVAS;
}
mtmd_image_tokens clone() {
@@ -516,6 +555,9 @@ struct mtmd_context {
bool tok_row_end_trail = false;
bool ov_img_first = false;
// MiniCPM-V 4.6/4.7 prepends an <image_id>N</image_id> tag before <image>
bool use_image_id = false;
// string template for slice image delimiters with row/col (idefics3)
std::string sli_img_start_tmpl;
@@ -680,6 +722,7 @@ struct mtmd_context {
image_preproc = std::make_unique<mtmd_image_preprocessor_llava_uhd>(ctx_v);
} break;
case PROJECTOR_TYPE_MINICPMV4_6:
case PROJECTOR_TYPE_MINICPMV4_7:
{
slice_tmpl = MTMD_SLICE_TMPL_MINICPMV_2_6;
tok_ov_img_start = {lookup_token("<image>")};
@@ -689,6 +732,7 @@ struct mtmd_context {
tok_row_end = {lookup_token("\n")};
tok_row_end_trail = false; // no trailing end-of-row token
ov_img_first = true;
use_image_id = true;
image_preproc = std::make_unique<mtmd_image_preprocessor_minicpmv>(ctx_v);
} break;
case PROJECTOR_TYPE_QWEN2VL:
@@ -1429,7 +1473,15 @@ struct mtmd_tokenizer {
const bool has_tiling_grid = (preproc_out.grid_x > 0 && preproc_out.grid_y > 0)
|| preproc_out.has_overview();
if (has_tiling_grid) {
if (has_tiling_grid && ctx->proj_type_v() == PROJECTOR_TYPE_MINICPMV4_7) {
GGML_ASSERT(bitmaps.size() == 1);
if (ctx->use_image_id) {
add_text("<image_id>" + std::to_string(n_images_added) + "</image_id>", true);
}
add_text(ctx->tok_ov_img_start);
// the separators after <image> are appended by clip, see add_canvas_chunk()
add_canvas_chunk(std::move(preproc_out), bitmaps[0]->id);
} else if (has_tiling_grid) {
// [QWEN_VIDEO] we do not support "frame merging" for llama-uhd style, so no batching for now
GGML_ASSERT(bitmaps.size() == 1);
@@ -1448,6 +1500,9 @@ struct mtmd_tokenizer {
// add overview image (first)
if (ctx->ov_img_first) {
if (ctx->use_image_id) {
add_text("<image_id>" + std::to_string(n_images_added) + "</image_id>", true);
}
add_text(ctx->tok_ov_img_start);
cur.entries.emplace_back(std::move(ov_chunk));
add_text(ctx->tok_ov_img_end);
@@ -1673,6 +1728,62 @@ struct mtmd_tokenizer {
return 0;
}
// MiniCPM-V 4.7: the overview and all slices go in one chunk, clip appends the separators after each tile:
// [ov] </image><slice> [S00] </slice><slice> [S01] </slice>\n<slice> [S10] </slice><slice> [S11] </slice>
void add_canvas_chunk(mtmd_image_preproc_out && preproc_out, const std::string & id) {
const int n_col = preproc_out.grid_x;
const int n_row = preproc_out.grid_y;
auto & slices = preproc_out.entries;
GGML_ASSERT(preproc_out.has_overview());
GGML_ASSERT((int) slices.size() == n_col * n_row);
auto & ov = preproc_out.overview;
ov.suffix_type = CLIP_SUFFIX_MINICPMV_OV;
if (!slices.empty()) {
ov.suffix_type = CLIP_SUFFIX_MINICPMV_OV_SLICE;
ov.anyres.grid_x = n_col;
ov.anyres.grid_y = n_row;
}
for (int y = 0; y < n_row; y++) {
for (int x = 0; x < n_col; x++) {
auto & suffix = slices[y * n_col + x].suffix_type;
if (y == n_row - 1 && x == n_col - 1) {
suffix = CLIP_SUFFIX_MINICPMV_LAST;
} else if (x == n_col - 1) {
suffix = CLIP_SUFFIX_MINICPMV_ROW_END;
} else {
suffix = CLIP_SUFFIX_MINICPMV_SLICE;
}
}
}
mtmd_image_tokens_ptr image_tokens(new mtmd_image_tokens);
image_tokens->pos = MTMD_POS_TYPE_CANVAS;
image_tokens->id = id;
auto & entries = image_tokens->batch_f32.entries;
entries.push_back(std::move(ov));
for (auto & slice : slices) {
entries.push_back(std::move(slice));
}
// token grid of the last entry, the grids of the other entries are scaled from it
image_tokens->nx = clip_n_output_tokens_x(ctx->ctx_v, &entries.back());
image_tokens->ny = clip_n_output_tokens_y(ctx->ctx_v, &entries.back());
size_t n_tokens = 0;
for (const auto & entry : entries) {
n_tokens += clip_n_output_tokens(ctx->ctx_v, &entry);
}
GGML_ASSERT(n_tokens == image_tokens->n_tokens());
mtmd_input_chunk chunk{
MTMD_INPUT_CHUNK_TYPE_IMAGE,
{}, // text tokens
std::move(image_tokens),
nullptr, // audio tokens
};
cur.entries.emplace_back(std::move(chunk));
}
std::vector<mtmd_input_chunk> split_batch_to_chunk(mtmd_image_preproc_out && preproc_out, const std::string & id) {
std::vector<mtmd_input_chunk> chunks;
@@ -1814,6 +1925,24 @@ static int32_t mtmd_encode_impl(mtmd_context * ctx, const mtmd_image_tokens * im
return 1;
}
if (image_tokens->pos == MTMD_POS_TYPE_CANVAS) {
// the tiles differ in size, encode them one by one
size_t offset = 0;
for (const auto & entry : image_tokens->batch_f32.entries) {
clip_image_f32_batch one;
one.entries.push_back(entry);
std::vector<float> embd((size_t) n_embd_out * clip_n_output_tokens(ctx_clip, &entry));
if (!clip_image_batch_encode(ctx_clip, ctx->n_threads, &one, embd)) {
return 1;
}
GGML_ASSERT(offset + embd.size() <= out_embd.size());
std::copy(embd.begin(), embd.end(), out_embd.begin() + offset);
offset += embd.size();
}
GGML_ASSERT(offset == out_embd.size());
return 0;
}
bool ok = clip_image_batch_encode(
ctx_clip,
ctx->n_threads,
@@ -2494,6 +2623,67 @@ size_t mtmd_image_tokens_get_ny(const mtmd_image_tokens * image_tokens) {
return image_tokens->ny;
}
// map a tile coordinate onto the canvas like the reference: round(linspace(0, canvas - 1, grid)), round() breaks ties to even
static uint32_t mtmd_canvas_scale(uint32_t coord, uint32_t grid, uint32_t canvas) {
if (grid <= 1 || canvas <= 1) {
return 0;
}
const double v = (double) coord * (double) (canvas - 1) / (double) (grid - 1);
return std::min((uint32_t) std::nearbyint(v), canvas - 1);
}
// MTMD_POS_TYPE_CANVAS: every tile shares the <image> token before the chunk as origin
// the overview is stretched over the whole canvas, each slice fills its own cell; the time component is the origin, in slot z
// a tile takes one position in slot t (the KV cache position), the separators after it take one position each
static mtmd_decoder_pos mtmd_canvas_decoder_pos(const mtmd_image_tokens * image_tokens, llama_pos pos_0, size_t i) {
const auto & entries = image_tokens->batch_f32.entries;
const auto & grid = entries[0].anyres;
const uint32_t nx = image_tokens->nx;
const uint32_t ny = image_tokens->ny;
const uint32_t canvas_w = grid.is_tiled() ? grid.grid_x * nx : nx;
const uint32_t canvas_h = grid.is_tiled() ? grid.grid_y * ny : ny;
const uint32_t base = pos_0 - 1;
mtmd_decoder_pos pos;
uint32_t t = pos_0;
for (size_t k = 0; k < entries.size(); ++k) {
const auto [gw, gh] = image_tokens->canvas_tile_grid(k);
if (i < gw * gh) {
const uint32_t row = i / gw;
const uint32_t col = i % gw;
uint32_t h;
uint32_t w;
if (k == 0) {
h = mtmd_canvas_scale(row, gh, canvas_h);
w = mtmd_canvas_scale(col, gw, canvas_w);
} else {
const uint32_t s = k - 1;
h = (s / grid.grid_x) * ny + row;
w = (s % grid.grid_x) * nx + col;
}
pos.t = t;
pos.x = base + w;
pos.y = base + h;
pos.z = base;
return pos;
}
i -= gw * gh;
const size_t n_sep = clip_suffix_rows(entries[k].suffix_type).size();
if (i < n_sep) {
const uint32_t p = t + 1 + i;
pos.t = p;
pos.x = p;
pos.y = p;
pos.z = p;
return pos;
}
i -= n_sep;
t += 1 + n_sep;
}
GGML_ABORT("token index out of range");
}
mtmd_decoder_pos mtmd_image_tokens_get_decoder_pos(const mtmd_image_tokens * image_tokens, llama_pos pos_0, size_t i) {
mtmd_decoder_pos pos;
switch (image_tokens->pos) {
@@ -2543,6 +2733,10 @@ mtmd_decoder_pos mtmd_image_tokens_get_decoder_pos(const mtmd_image_tokens * ima
pos.z = image_tokens->image_idx;
}
} break;
case MTMD_POS_TYPE_CANVAS:
{
pos = mtmd_canvas_decoder_pos(image_tokens, pos_0, i);
} break;
default:
GGML_ABORT("invalid position type");
}
@@ -2563,6 +2757,15 @@ llama_pos mtmd_image_tokens_get_n_pos(const mtmd_image_tokens * image_tokens) {
// HunyuanVL: the sequential (dim-0) position advances by the full token count
// (includes BOI/EOI and row newline tokens), not by max(nx, ny)
return image_tokens->n_tokens();
case MTMD_POS_TYPE_CANVAS:
{
// one position per tile, plus one per separator
llama_pos n_pos = 0;
for (const auto & entry : image_tokens->batch_f32.entries) {
n_pos += 1 + (llama_pos) clip_suffix_rows(entry.suffix_type).size();
}
return n_pos;
}
default:
GGML_ABORT("invalid position type");
}