mirror of
https://github.com/ggml-org/llama.cpp.git
synced 2026-10-10 23:10:48 +02:00
model : support MiniCPM-V 4.7 (#29416)
* mtmd : add MiniCPM-V 4.7 support Signed-off-by: tc-mb <tianchi_cai@icloud.com> * model : allow mrope time from an extra position slot Signed-off-by: tc-mb <tianchi_cai@icloud.com> * Update conversion/minicpm.py Co-authored-by: Sigbjørn Skjæret <sigbjorn.skjaeret@huggingface.co> * Slim down comments Signed-off-by: tc-mb <tianchi_cai@icloud.com> * fix for "do not hand-wrap comments" Signed-off-by: tc-mb <tianchi_cai@icloud.com> * fix ci Signed-off-by: tc-mb <tianchi_cai@icloud.com> * rm 3d repo for pr one Signed-off-by: tc-mb <tianchi_cai@icloud.com> * gguf: add rope.section_order metadata * fix comments * handle grid layout * allow compat --------- Signed-off-by: tc-mb <tianchi_cai@icloud.com> Co-authored-by: Sigbjørn Skjæret <sigbjorn.skjaeret@huggingface.co> Co-authored-by: Xuan Son Nguyen <son@huggingface.co>
This commit is contained in:
co-authored by
Sigbjørn Skjæret
Xuan Son Nguyen
parent
abee0c8476
commit
69f201a205
@@ -189,6 +189,7 @@ TEXT_MODEL_MAP: dict[str, str] = {
|
||||
"MiniCPM3ForCausalLM": "minicpm",
|
||||
"MiniCPMForCausalLM": "minicpm",
|
||||
"MiniCPMV4_6ForConditionalGeneration": "minicpm",
|
||||
"MiniCPMV4_7ForConditionalGeneration": "minicpm",
|
||||
"MiniMaxText01ForCausalLM": "minimax",
|
||||
"MiniMaxM1ForCausalLM": "minimax",
|
||||
"MiniMaxM2ForCausalLM": "minimax",
|
||||
@@ -346,6 +347,7 @@ MMPROJ_MODEL_MAP: dict[str, str] = {
|
||||
"MiMoV2ForCausalLM": "mimo",
|
||||
"MiniMaxM3SparseForConditionalGeneration": "minimax",
|
||||
"MiniCPMV4_6ForConditionalGeneration": "minicpm",
|
||||
"MiniCPMV4_7ForConditionalGeneration": "minicpm",
|
||||
"Mistral3ForConditionalGeneration": "llava",
|
||||
"NemotronH_Nano_VL_V2": "nemotron",
|
||||
"MuseGlimmerForConditionalGeneration": "muse_glimmer",
|
||||
|
||||
+5
-2
@@ -1389,14 +1389,17 @@ class TextModel(ModelBase):
|
||||
name, gen = item
|
||||
|
||||
# Skip multimodal tensors
|
||||
if name.startswith(("mlp", "vit.", "vpm.", "siglip2.", "conformer.", "merger.", "resampler.", "sound_encoder.", "sound_projection.", "speech_embeddings.")) \
|
||||
# strip the "model." wrapper so the prefixes below match (name is not returned)
|
||||
if name.startswith("model."):
|
||||
name = name[len("model."):]
|
||||
if name.startswith(("mlp", "vit.", "vpm.", "siglip2.", "conformer.", "connector.", "merger.", "resampler.", "sound_encoder.", "sound_projection.", "speech_embeddings.")) \
|
||||
or "visual." in name or "vision." in name or "audio." in name or "talker." in name \
|
||||
or "vision_" in name or "audio_" in name \
|
||||
or "token2wav." in name or "code2wav." in name \
|
||||
or "projector." in name or "pre_mm_projector_norm" in name \
|
||||
or "image_newline" in name or "view_seperator" in name \
|
||||
or "patch_embed" in name or "patch_embedding" in name \
|
||||
or "patch_merger." in name or "patch_merge_mlp." in name or "model.connector." in name:
|
||||
or "patch_merger." in name or "patch_merge_mlp." in name:
|
||||
return None
|
||||
|
||||
return super().filter_tensors(item)
|
||||
|
||||
+85
-4
@@ -139,9 +139,16 @@ class MiniCPMV4_6TextModel(Qwen3_5TextModel):
|
||||
@ModelBase.register("MiniCPMV4_6ForConditionalGeneration")
|
||||
@ModelBase.example("openbmb/MiniCPM-V-4_6")
|
||||
class MiniCPMV4_6VisionModel(MmprojModel):
|
||||
projector_type = gguf.VisionProjectorType.MINICPMV4_6
|
||||
# fallback for checkpoints whose preprocessor config omits `scale_resolution`
|
||||
default_scale_resolution: int | None = None
|
||||
|
||||
def get_downsample_mode(self) -> str:
|
||||
return self.preprocessor_config.get("downsample_mode", "16x")
|
||||
|
||||
def __init__(self, *args, **kwargs):
|
||||
super().__init__(*args, **kwargs)
|
||||
self.downsample_mode = self.preprocessor_config.get("downsample_mode", "16x")
|
||||
self.downsample_mode = self.get_downsample_mode()
|
||||
if self.downsample_mode not in {"4x", "16x"}:
|
||||
raise ValueError(f"Unsupported downsample mode: {self.downsample_mode}")
|
||||
if self.downsample_mode == "4x":
|
||||
@@ -157,7 +164,8 @@ class MiniCPMV4_6VisionModel(MmprojModel):
|
||||
# The CLIP loader in tools/mtmd/clip.cpp consumes `clip.vision.image_size`
|
||||
# as the slice size and warmup resolution, so report `scale_resolution` there
|
||||
# to match the upstream MiniCPMV4_6ImageProcessorPil slicing rules.
|
||||
scale_resolution = self.preprocessor_config.get("scale_resolution")
|
||||
scale_resolution = self.preprocessor_config.get(
|
||||
"scale_resolution", self.default_scale_resolution)
|
||||
if scale_resolution is not None:
|
||||
self.hparams_vision["image_size"] = int(scale_resolution)
|
||||
|
||||
@@ -166,12 +174,15 @@ class MiniCPMV4_6VisionModel(MmprojModel):
|
||||
assert self.hparams_vision is not None
|
||||
|
||||
# projector type string is consumed by clip_projector_type_from_string() in clip.cpp
|
||||
# (mapped to PROJECTOR_TYPE_MINICPMV4_6).
|
||||
self.gguf_writer.add_clip_projector_type(gguf.VisionProjectorType.MINICPMV4_6)
|
||||
self.gguf_writer.add_clip_projector_type(self.projector_type)
|
||||
|
||||
self.gguf_writer.add_vision_projector_scale_factor(
|
||||
2 if self.downsample_mode == "4x" else 4)
|
||||
|
||||
max_slice_nums = self.preprocessor_config.get("max_slice_nums")
|
||||
if max_slice_nums is not None:
|
||||
self.gguf_writer.add_vision_max_slice_nums(int(max_slice_nums))
|
||||
|
||||
# borrow wa_layer_indexes for vit_merger insertion point
|
||||
insert_layer_id = int(self.global_config.get(
|
||||
"insert_layer_id", self.hparams_vision.get("insert_layer_id", 6)))
|
||||
@@ -191,3 +202,73 @@ class MiniCPMV4_6VisionModel(MmprojModel):
|
||||
return None
|
||||
|
||||
return super().filter_tensors(item)
|
||||
|
||||
|
||||
# MiniCPM-V 4.7 shares the v4.6 stack: the same Qwen3.5 text tower (MoE variant when the checkpoint says so) and the same SigLIP + vit_merger + merger vision tower.
|
||||
|
||||
@ModelBase.register("MiniCPMV4_7ForConditionalGeneration")
|
||||
@ModelBase.example("openbmb/MiniCPM-V-4.7")
|
||||
class MiniCPMV4_7TextModel(Qwen3_5TextModel):
|
||||
model_arch = gguf.MODEL_ARCH.QWEN35
|
||||
|
||||
def set_gguf_parameters(self):
|
||||
super().set_gguf_parameters()
|
||||
# mtmd puts the time of the image canvas in slot z, slot t stays the KV cache position
|
||||
self.gguf_writer.add_rope_section_order(gguf.RopeSectionOrder.ZYXT)
|
||||
|
||||
def __init__(self, dir_model, ftype, fname_out, *, hparams: dict | None = None, **kwargs):
|
||||
if hparams is None:
|
||||
hparams = ModelBase.load_hparams(dir_model, is_mistral_format=False)
|
||||
text_config = hparams.get("text_config", {})
|
||||
if text_config.get("model_type") == "qwen3_5_moe_text":
|
||||
self.model_arch = gguf.MODEL_ARCH.QWEN35MOE
|
||||
else:
|
||||
self.model_arch = gguf.MODEL_ARCH.QWEN35
|
||||
super().__init__(dir_model, ftype, fname_out, hparams=hparams, **kwargs)
|
||||
|
||||
@classmethod
|
||||
def filter_tensors(cls, item: tuple[str, Callable[[], Tensor]]) -> tuple[str, Callable[[], Tensor]] | None:
|
||||
name, gen = item
|
||||
|
||||
# MTP tensors are not used yet
|
||||
if name.startswith("mtp"):
|
||||
return None
|
||||
|
||||
return super().filter_tensors(item)
|
||||
|
||||
|
||||
@ModelBase.register("MiniCPMV4_7ForConditionalGeneration")
|
||||
@ModelBase.example("openbmb/MiniCPM-V-4.7")
|
||||
class MiniCPMV4_7VisionModel(MiniCPMV4_6VisionModel):
|
||||
projector_type = gguf.VisionProjectorType.MINICPMV4_7
|
||||
# MiniCPMV4_7ImageProcessorPil default
|
||||
default_scale_resolution = 448
|
||||
# rows of v.tok_embd_sep, the order must match clip_suffix_rows() in clip-impl.h
|
||||
tok_embd_sep = ["</image>", "<slice>", "</slice>", "\n"]
|
||||
|
||||
def get_downsample_mode(self) -> str:
|
||||
# 4.7 moved downsample_mode to the model config; preprocessor value takes priority
|
||||
return self.preprocessor_config.get(
|
||||
"downsample_mode", self.global_config.get("downsample_mode", "16x"))
|
||||
|
||||
@classmethod
|
||||
def filter_tensors(cls, item: tuple[str, Callable[[], Tensor]]) -> tuple[str, Callable[[], Tensor]] | None:
|
||||
# keep the text tok_embd, the separator rows are taken from it in modify_tensors
|
||||
if item[0] == "model.language_model.embed_tokens.weight":
|
||||
return item
|
||||
return super().filter_tensors(item)
|
||||
|
||||
def modify_tensors(self, data_torch: Tensor, name: str, bid: int | None) -> Iterable[tuple[str, Tensor]]:
|
||||
if name == "model.language_model.embed_tokens.weight":
|
||||
# the tile separators are text tokens; clip appends their embeddings so that one chunk holds the whole image
|
||||
from transformers import AutoTokenizer
|
||||
tokenizer = AutoTokenizer.from_pretrained(self.dir_model)
|
||||
ids = []
|
||||
for text in self.tok_embd_sep:
|
||||
tok = tokenizer.encode(text, add_special_tokens=False)
|
||||
if len(tok) != 1:
|
||||
raise ValueError(f"separator {text!r} must be a single token, got {tok}")
|
||||
ids.append(tok[0])
|
||||
yield self.format_tensor_name(gguf.MODEL_TENSOR.V_TOK_EMBD_SEP, suffix=""), data_torch[ids]
|
||||
return
|
||||
yield from super().modify_tensors(data_torch, name, bid)
|
||||
|
||||
@@ -0,0 +1,55 @@
|
||||
## MiniCPM-V 4.7
|
||||
|
||||
### Prepare models and code
|
||||
|
||||
Download [MiniCPM-V-4.7](https://huggingface.co/openbmb/MiniCPM-V-4.7) PyTorch model from huggingface to "MiniCPM-V-4.7" folder.
|
||||
|
||||
The model must be the standard `transformers` checkpoint (no `trust_remote_code` for the text and vision graph used here); the architecture in `config.json` is `MiniCPMV4_7ForConditionalGeneration` with a `qwen3_5_text` (or `qwen3_5_moe_text`) text model and a SigLIP-based vision tower plus a window-attention `vit_merger`, same as MiniCPM-V 4.6.
|
||||
|
||||
If the checkpoint ships no MTP weights, pass `--no-mtp` to skip the nextn layers.
|
||||
|
||||
### Build llama.cpp
|
||||
|
||||
If there are differences in usage, please refer to the official build [documentation](https://github.com/ggml-org/llama.cpp/blob/master/docs/build.md)
|
||||
|
||||
Clone llama.cpp:
|
||||
```bash
|
||||
git clone https://github.com/ggml-org/llama.cpp
|
||||
cd llama.cpp
|
||||
```
|
||||
|
||||
Build llama.cpp using `CMake`:
|
||||
```bash
|
||||
cmake -B build
|
||||
cmake --build build --config Release
|
||||
```
|
||||
|
||||
|
||||
### Usage of MiniCPM-V 4.7
|
||||
|
||||
MiniCPM-V 4.7 is converted directly through `convert_hf_to_gguf.py`. The same script is invoked twice on the original Hugging Face directory: once to produce the language-model GGUF and once with `--mmproj` to produce the multimodal projector GGUF.
|
||||
|
||||
```bash
|
||||
# language model
|
||||
python ./convert_hf_to_gguf.py ../MiniCPM-V-4.7 --outfile ../MiniCPM-V-4.7/ggml-model-f16.gguf --no-mtp
|
||||
|
||||
# multimodal projector (vision tower + window-attention vit_merger + DownsampleMLP merger)
|
||||
python ./convert_hf_to_gguf.py ../MiniCPM-V-4.7 --mmproj --outfile ../MiniCPM-V-4.7/mmproj-model-f16.gguf
|
||||
|
||||
# optional: quantize to Q4_K_M
|
||||
./build/bin/llama-quantize ../MiniCPM-V-4.7/ggml-model-f16.gguf ../MiniCPM-V-4.7/ggml-model-Q4_K_M.gguf Q4_K_M
|
||||
```
|
||||
|
||||
The default projector merges 16x (4x4 patches into one token). To keep 4x more visual tokens, copy the model dir and set `"downsample_mode": "4x"` in the copy's `preprocessor_config.json` before running the `--mmproj` conversion; the loader reads `clip.vision.projector.scale_factor` to pick the graph.
|
||||
|
||||
|
||||
Inference on Linux or Mac
|
||||
```bash
|
||||
# run in single-turn mode
|
||||
./build/bin/llama-mtmd-cli -m ../MiniCPM-V-4.7/ggml-model-f16.gguf --mmproj ../MiniCPM-V-4.7/mmproj-model-f16.gguf -c 4096 --jinja --image xx.jpg -p "What is in the image?"
|
||||
|
||||
# run in conversation mode
|
||||
./build/bin/llama-mtmd-cli -m ../MiniCPM-V-4.7/ggml-model-Q4_K_M.gguf --mmproj ../MiniCPM-V-4.7/mmproj-model-f16.gguf --jinja
|
||||
```
|
||||
|
||||
The chat template enables thinking by default. Pass `--chat-template-kwargs '{"enable_thinking": false}'` to `llama-server` to turn it off.
|
||||
@@ -259,6 +259,7 @@ class Keys:
|
||||
DIMENSION_COUNT = "{arch}.rope.dimension_count"
|
||||
DIMENSION_COUNT_SWA = "{arch}.rope.dimension_count_swa"
|
||||
DIMENSION_SECTIONS = "{arch}.rope.dimension_sections"
|
||||
SECTION_ORDER = "{arch}.rope.section_order"
|
||||
FREQ_BASE = "{arch}.rope.freq_base"
|
||||
FREQ_BASE_SWA = "{arch}.rope.freq_base_swa"
|
||||
SCALING_TYPE = "{arch}.rope.scaling.type"
|
||||
@@ -409,6 +410,7 @@ class Keys:
|
||||
BLOCK_COUNT = "clip.vision.block_count"
|
||||
IMAGE_MEAN = "clip.vision.image_mean"
|
||||
IMAGE_STD = "clip.vision.image_std"
|
||||
MAX_SLICE_NUMS = "clip.vision.max_slice_nums"
|
||||
IMAGE_RESIZE_ALGO = "clip.vision.image_resize_algo"
|
||||
SPATIAL_MERGE_SIZE = "clip.vision.spatial_merge_size"
|
||||
SWIGLU_CLAMP = "clip.vision.swiglu_clamp"
|
||||
@@ -1051,6 +1053,7 @@ class MODEL_TENSOR(IntEnum):
|
||||
V_SAM_NET_3 = auto() # Deepseek-OCR
|
||||
V_ENC_EMBD_IMGNL = auto() # Deepseek-OCR
|
||||
V_ENC_EMBD_VSEP = auto() # Deepseek-OCR
|
||||
V_TOK_EMBD_SEP = auto() # MiniCPM-V 4.7
|
||||
V_RESMPL_QUERY_768 = auto() # Deepseek-OCR-2
|
||||
V_RESMPL_QUERY_1024 = auto() # Deepseek-OCR-2
|
||||
|
||||
@@ -1832,6 +1835,7 @@ TENSOR_NAMES: dict[MODEL_TENSOR, str] = {
|
||||
MODEL_TENSOR.V_SAM_NET_3: "v.sam.net_3",
|
||||
MODEL_TENSOR.V_ENC_EMBD_IMGNL: "v.image_newline", # Deepseek-OCR, Granite4Vision
|
||||
MODEL_TENSOR.V_ENC_EMBD_VSEP: "v.view_seperator", # Deepseek-OCR
|
||||
MODEL_TENSOR.V_TOK_EMBD_SEP: "v.tok_embd_sep", # MiniCPM-V 4.7
|
||||
MODEL_TENSOR.V_RESMPL_QUERY_768: "v.resample_query_768", # Deepseek-OCR-2 qwen2
|
||||
MODEL_TENSOR.V_RESMPL_QUERY_1024: "v.resample_query_1024", # Deepseek-OCR-2 qwen2
|
||||
# Granite4 Vision
|
||||
@@ -2079,6 +2083,7 @@ MODEL_TENSORS: dict[MODEL_ARCH, list[MODEL_TENSOR]] = {
|
||||
MODEL_TENSOR.V_ENC_EMBD_POS,
|
||||
MODEL_TENSOR.V_ENC_EMBD_IMGNL,
|
||||
MODEL_TENSOR.V_ENC_EMBD_VSEP,
|
||||
MODEL_TENSOR.V_TOK_EMBD_SEP,
|
||||
MODEL_TENSOR.V_ENC_INPUT_NORM,
|
||||
MODEL_TENSOR.V_ENC_ATTN_QKV,
|
||||
MODEL_TENSOR.V_ENC_ATTN_Q,
|
||||
@@ -5927,6 +5932,12 @@ class RopeScalingType(Enum):
|
||||
LONGROPE = 'longrope'
|
||||
|
||||
|
||||
# M-RoPE: input position slot (t, y, x, z) that feeds each RoPE section, in section order
|
||||
class RopeSectionOrder(Enum):
|
||||
TYXZ = 'tyxz' # default
|
||||
ZYXT = 'zyxt'
|
||||
|
||||
|
||||
class PoolingType(IntEnum):
|
||||
NONE = 0
|
||||
MEAN = 1
|
||||
@@ -6133,6 +6144,7 @@ class VisionProjectorType:
|
||||
PARAKEET = "parakeet" # audio
|
||||
MINIMAXM3 = "minimax_m3"
|
||||
MINICPMV4_6 = "minicpmv4_6"
|
||||
MINICPMV4_7 = "minicpmv4_7"
|
||||
GRANITE_SPEECH = "granite_speech" # audio
|
||||
MIMOVL = "mimovl"
|
||||
MIMO_AUDIO = "mimo_audio"
|
||||
|
||||
@@ -24,6 +24,7 @@ from .constants import (
|
||||
GGUFEndian,
|
||||
GGUFValueType,
|
||||
Keys,
|
||||
RopeSectionOrder,
|
||||
RopeScalingType,
|
||||
PoolingType,
|
||||
TokenType,
|
||||
@@ -1154,6 +1155,9 @@ class GGUFWriter:
|
||||
def add_rope_dimension_sections(self, dims: Sequence[int]) -> None:
|
||||
self.add_array(Keys.Rope.DIMENSION_SECTIONS.format(arch=self.arch), dims)
|
||||
|
||||
def add_rope_section_order(self, value: RopeSectionOrder) -> None:
|
||||
self.add_string(Keys.Rope.SECTION_ORDER.format(arch=self.arch), value.value)
|
||||
|
||||
def add_rope_freq_base(self, value: float) -> None:
|
||||
self.add_float32(Keys.Rope.FREQ_BASE.format(arch=self.arch), value)
|
||||
|
||||
@@ -1462,6 +1466,9 @@ class GGUFWriter:
|
||||
def add_vision_projector_scale_factor(self, value: int) -> None:
|
||||
self.add_uint32(Keys.ClipVision.Projector.SCALE_FACTOR, value)
|
||||
|
||||
def add_vision_max_slice_nums(self, value: int) -> None:
|
||||
self.add_uint32(Keys.ClipVision.MAX_SLICE_NUMS, value)
|
||||
|
||||
def add_vision_n_wa_pattern(self, value: int) -> None:
|
||||
"""Add window attention pattern interval for vision models.
|
||||
|
||||
|
||||
+1
-1
@@ -1078,7 +1078,7 @@ extern "C" {
|
||||
|
||||
// Set custom position for the token at index idx in the batch
|
||||
// For M-RoPE models:
|
||||
// - Embedding tokens must have multiple positions per token
|
||||
// - Embedding tokens must have n_pos_per_embd positions per token, in order [t, y, x, z]; t is also the KV cache position
|
||||
// - Text token only requires one single position per token
|
||||
LLAMA_API bool llama_batch_ext_set_pos(
|
||||
struct llama_batch_ext * batch,
|
||||
|
||||
@@ -330,6 +330,7 @@ static const std::map<llm_kv, const char *> LLM_KV_NAMES = {
|
||||
{ LLM_KV_ROPE_DIMENSION_COUNT, "%s.rope.dimension_count" },
|
||||
{ LLM_KV_ROPE_DIMENSION_COUNT_SWA, "%s.rope.dimension_count_swa" },
|
||||
{ LLM_KV_ROPE_DIMENSION_SECTIONS, "%s.rope.dimension_sections" },
|
||||
{ LLM_KV_ROPE_SECTION_ORDER, "%s.rope.section_order" },
|
||||
{ LLM_KV_ROPE_FREQ_BASE, "%s.rope.freq_base" },
|
||||
{ LLM_KV_ROPE_FREQ_BASE_SWA, "%s.rope.freq_base_swa" },
|
||||
{ LLM_KV_ROPE_SCALE_LINEAR, "%s.rope.scale_linear" },
|
||||
|
||||
@@ -335,6 +335,7 @@ enum llm_kv {
|
||||
LLM_KV_ROPE_DIMENSION_COUNT,
|
||||
LLM_KV_ROPE_DIMENSION_COUNT_SWA,
|
||||
LLM_KV_ROPE_DIMENSION_SECTIONS,
|
||||
LLM_KV_ROPE_SECTION_ORDER,
|
||||
LLM_KV_ROPE_FREQ_BASE,
|
||||
LLM_KV_ROPE_FREQ_BASE_SWA,
|
||||
LLM_KV_ROPE_SCALE_LINEAR,
|
||||
|
||||
+28
-2
@@ -172,7 +172,33 @@ void llm_graph_input_pos::set_input(const llama_ubatch * ubatch) {
|
||||
if (ubatch->pos && pos) {
|
||||
const int64_t n_tokens = ubatch->n_tokens;
|
||||
|
||||
ggml_backend_tensor_set(pos, ubatch->pos, 0, n_tokens*n_pos_per_embd*ggml_element_size(pos));
|
||||
const bool has_embd = ubatch->is_mixed() || ubatch->token == nullptr;
|
||||
if (rope_section_order == LLAMA_ROPE_SECTION_ORDER_TYXZ || !has_embd) {
|
||||
ggml_backend_tensor_set(pos, ubatch->pos, 0, n_tokens*n_pos_per_embd*ggml_element_size(pos));
|
||||
return;
|
||||
}
|
||||
|
||||
// input is always [t, y, x, z]
|
||||
// token entries are expanded by the batch to [p, p, p, 0]
|
||||
GGML_ASSERT(n_pos_per_embd == 4);
|
||||
|
||||
// slot index per section, slots are t = 0, y = 1, x = 2, z = 3
|
||||
std::array<int64_t, 4> slot_of_section = { 0, 1, 2, 3 };
|
||||
switch (rope_section_order) {
|
||||
case LLAMA_ROPE_SECTION_ORDER_TYXZ: break;
|
||||
case LLAMA_ROPE_SECTION_ORDER_ZYXT: slot_of_section = { 3, 1, 2, 0 }; break;
|
||||
default: GGML_ABORT("unsupported rope section order");
|
||||
}
|
||||
|
||||
std::vector<llama_pos> pos_data(n_tokens*n_pos_per_embd);
|
||||
for (int64_t i = 0; i < n_tokens; ++i) {
|
||||
const bool is_embd = ubatch->is_mixed() ? ubatch->type[i] != 0 : true;
|
||||
for (int64_t s = 0; s < 4; ++s) {
|
||||
const int64_t slot = is_embd ? slot_of_section[s] : s;
|
||||
pos_data[s*n_tokens + i] = ubatch->pos[slot*n_tokens + i];
|
||||
}
|
||||
}
|
||||
ggml_backend_tensor_set(pos, pos_data.data(), 0, pos_data.size()*ggml_element_size(pos));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2620,7 +2646,7 @@ ggml_tensor * llm_graph_context::build_inp_embd(ggml_tensor * tok_embd, float to
|
||||
}
|
||||
|
||||
ggml_tensor * llm_graph_context::build_inp_pos() const {
|
||||
auto inp = std::make_unique<llm_graph_input_pos>(hparams.n_pos_per_embd());
|
||||
auto inp = std::make_unique<llm_graph_input_pos>(hparams.n_pos_per_embd(), hparams.rope_section_order);
|
||||
|
||||
auto & cur = inp->pos;
|
||||
|
||||
|
||||
+3
-1
@@ -169,7 +169,8 @@ public:
|
||||
|
||||
class llm_graph_input_pos : public llm_graph_input_i {
|
||||
public:
|
||||
llm_graph_input_pos(uint32_t n_pos_per_embd) : n_pos_per_embd(n_pos_per_embd) {}
|
||||
llm_graph_input_pos(uint32_t n_pos_per_embd, llama_rope_section_order rope_section_order = LLAMA_ROPE_SECTION_ORDER_TYXZ)
|
||||
: n_pos_per_embd(n_pos_per_embd), rope_section_order(rope_section_order) {}
|
||||
virtual ~llm_graph_input_pos() = default;
|
||||
|
||||
void set_input(const llama_ubatch * ubatch) override;
|
||||
@@ -179,6 +180,7 @@ public:
|
||||
ggml_tensor * pos = nullptr; // I32 [n_batch]
|
||||
|
||||
const uint32_t n_pos_per_embd = 1;
|
||||
const llama_rope_section_order rope_section_order = LLAMA_ROPE_SECTION_ORDER_TYXZ;
|
||||
};
|
||||
|
||||
// temperature tuning, used by llama4
|
||||
|
||||
@@ -36,6 +36,13 @@ enum llama_non_causal_type {
|
||||
LLAMA_NON_CAUSAL_TYPE_SWA_FULL = 2, // all layers non-causal, SWA not applied between tokens of the current ubatch (deepseek 4)
|
||||
};
|
||||
|
||||
// M-RoPE: which input position slot feeds each RoPE section
|
||||
enum llama_rope_section_order {
|
||||
LLAMA_ROPE_SECTION_ORDER_UNSPECIFIED = -1,
|
||||
LLAMA_ROPE_SECTION_ORDER_TYXZ = 0, // default, slot i feeds section i
|
||||
LLAMA_ROPE_SECTION_ORDER_ZYXT = 1, // MiniCPM-V 4.7: time last
|
||||
};
|
||||
|
||||
// forward declaration; full definition in llama-graph.h
|
||||
enum llm_ffn_op_type : int;
|
||||
|
||||
@@ -166,6 +173,8 @@ struct llama_hparams {
|
||||
|
||||
std::array<int, 4> rope_sections;
|
||||
|
||||
enum llama_rope_section_order rope_section_order = LLAMA_ROPE_SECTION_ORDER_TYXZ;
|
||||
|
||||
// Per-layer RoPE enable flags (1 = use RoPE, 0 = NoPE)
|
||||
// by default, all layers use RoPE (controlled by rope_finetuned)
|
||||
std::array<uint32_t, LLAMA_MAX_LAYERS> rope_pattern;
|
||||
|
||||
@@ -365,6 +365,7 @@ void llama_model_saver::add_kv_from_model() {
|
||||
add_kv(LLM_KV_ROPE_DIMENSION_COUNT, hparams.n_rot_full);
|
||||
add_kv(LLM_KV_ROPE_DIMENSION_COUNT_SWA, hparams.n_rot_swa);
|
||||
add_kv(LLM_KV_ROPE_DIMENSION_SECTIONS, hparams.rope_sections);
|
||||
add_kv(LLM_KV_ROPE_SECTION_ORDER, llama_rope_section_order_name(hparams.rope_section_order));
|
||||
add_kv(LLM_KV_ROPE_FREQ_BASE, hparams.rope_freq_base_train);
|
||||
add_kv(LLM_KV_ROPE_FREQ_BASE_SWA, hparams.rope_freq_base_train_swa);
|
||||
// add_kv(LLM_KV_ROPE_SCALE_LINEAR, rope_scaling_factor); // old name
|
||||
|
||||
@@ -1064,6 +1064,25 @@ static llama_rope_scaling_type llama_rope_scaling_type_from_string(const std::st
|
||||
return LLAMA_ROPE_SCALING_TYPE_UNSPECIFIED;
|
||||
}
|
||||
|
||||
static const std::map<llama_rope_section_order, const char *> LLAMA_ROPE_SECTION_ORDERS = {
|
||||
{ LLAMA_ROPE_SECTION_ORDER_TYXZ, "tyxz" },
|
||||
{ LLAMA_ROPE_SECTION_ORDER_ZYXT, "zyxt" },
|
||||
};
|
||||
|
||||
std::string llama_rope_section_order_name(llama_rope_section_order rope_section_order) {
|
||||
return LLAMA_ROPE_SECTION_ORDERS.at(rope_section_order);
|
||||
}
|
||||
|
||||
static llama_rope_section_order llama_rope_section_order_from_string(const std::string & name) {
|
||||
for (const auto & kv : LLAMA_ROPE_SECTION_ORDERS) {
|
||||
if (kv.second == name) {
|
||||
return kv.first;
|
||||
}
|
||||
}
|
||||
|
||||
return LLAMA_ROPE_SECTION_ORDER_UNSPECIFIED;
|
||||
}
|
||||
|
||||
// Maps GGUF activation names to the FFN op type used by the graph builders.
|
||||
static const std::map<std::string, llm_ffn_op_type> LLM_FFN_OP_TYPES_FROM_STRING = {
|
||||
{ "gelu", LLM_FFN_GEGLU_ERF },
|
||||
@@ -1447,6 +1466,13 @@ void llama_model_base::load_hparams(llama_model_loader & ml) {
|
||||
hparams.rope_scaling_type_train = llama_rope_scaling_type_from_string(rope_scaling);
|
||||
GGML_ASSERT(hparams.rope_scaling_type_train != LLAMA_ROPE_SCALING_TYPE_UNSPECIFIED);
|
||||
|
||||
std::string rope_section_order("tyxz");
|
||||
ml.get_key(LLM_KV_ROPE_SECTION_ORDER, rope_section_order, false);
|
||||
hparams.rope_section_order = llama_rope_section_order_from_string(rope_section_order);
|
||||
if (hparams.rope_section_order == LLAMA_ROPE_SECTION_ORDER_UNSPECIFIED) {
|
||||
throw std::runtime_error("unknown rope section order: " + rope_section_order);
|
||||
}
|
||||
|
||||
// TODO: Handle SWA metadata similarly when models start implementing it
|
||||
// rope_freq_scale (inverse of the kv) is optional
|
||||
float ropescale = 0.0f;
|
||||
@@ -1517,6 +1543,10 @@ void llama_model_base::load_hparams(llama_model_loader & ml) {
|
||||
}
|
||||
|
||||
hparams.rope_type = llama_model_rope_type(this);
|
||||
|
||||
if (hparams.rope_section_order != LLAMA_ROPE_SECTION_ORDER_TYXZ && hparams.n_pos_per_embd() != 4) {
|
||||
throw std::runtime_error("rope section order " + llama_rope_section_order_name(hparams.rope_section_order) + " requires M-RoPE");
|
||||
}
|
||||
}
|
||||
|
||||
void llama_model_base::load_vocab(llama_model_loader & ml) {
|
||||
@@ -2158,6 +2188,9 @@ void llama_model::print_info() const {
|
||||
if (const auto & s = hparams.rope_sections; s[0] || s[1] || s[2] || s[3]) {
|
||||
LLAMA_LOG_INFO("%s: mrope sections = [%d, %d, %d, %d]\n", __func__, s[0], s[1], s[2], s[3]);
|
||||
}
|
||||
if (hparams.rope_section_order != LLAMA_ROPE_SECTION_ORDER_TYXZ) {
|
||||
LLAMA_LOG_INFO("%s: rope section order = %s\n", __func__, llama_rope_section_order_name(hparams.rope_section_order).c_str());
|
||||
}
|
||||
if (!classifier_labels.empty()) {
|
||||
LLAMA_LOG_INFO("%s: n_cls_out = %u\n", __func__, hparams.n_cls_out);
|
||||
|
||||
|
||||
@@ -160,6 +160,7 @@ enum llm_type {
|
||||
};
|
||||
|
||||
std::string llama_rope_scaling_type_name(llama_rope_scaling_type rope_scaling_type);
|
||||
std::string llama_rope_section_order_name(llama_rope_section_order rope_section_order);
|
||||
|
||||
// Map a GGUF activation-name string to llm_ffn_op_type. Returns `fallback` if
|
||||
// the string is empty or not recognized.
|
||||
|
||||
@@ -166,4 +166,7 @@ struct clip_graph {
|
||||
// Generic function to stack frames for audio processing
|
||||
// Abstracts out the StackAudioFrames logic used by ultravox
|
||||
ggml_tensor * build_stack(ggml_tensor * cur, int32_t stack_factor, int32_t n_embed);
|
||||
|
||||
// append the separators of img.suffix_type after the image tokens
|
||||
ggml_tensor * build_suffix(ggml_tensor * cur);
|
||||
};
|
||||
|
||||
+41
-3
@@ -68,6 +68,7 @@
|
||||
|
||||
#define KEY_MM_PATCH_MERGE_TYPE "clip.vision.mm_patch_merge_type"
|
||||
#define KEY_IMAGE_GRID_PINPOINTS "clip.vision.image_grid_pinpoints"
|
||||
#define KEY_MAX_SLICE_NUMS "clip.vision.max_slice_nums"
|
||||
#define KEY_WIN_ATTN_PATTERN "clip.vision.n_wa_pattern"
|
||||
#define KEY_WIN_ATTN_LAYER_INDEXES "clip.vision.wa_layer_indexes"
|
||||
#define KEY_WA_PATTERN_MODE "clip.vision.wa_pattern_mode"
|
||||
@@ -146,6 +147,7 @@
|
||||
#define TN_MVLM_PROJ_PEG "mm.model.peg.%d.%s"
|
||||
#define TN_IMAGE_NEWLINE "v.image_newline"
|
||||
#define TN_IMAGE_SEPERATOR "v.view_seperator"
|
||||
#define TN_TOK_EMBD_SEP "v.tok_embd_sep"
|
||||
#define TN_MM_INP_NORM "mm.input_norm.weight"
|
||||
#define TN_MM_INP_NORM_B "mm.input_norm.bias"
|
||||
#define TN_MM_INP_PROJ "mm.input_projection.weight" // gemma3
|
||||
@@ -500,6 +502,7 @@ enum projector_type {
|
||||
PROJECTOR_TYPE_PARAKEET,
|
||||
PROJECTOR_TYPE_EXAONE4_5,
|
||||
PROJECTOR_TYPE_MINICPMV4_6,
|
||||
PROJECTOR_TYPE_MINICPMV4_7,
|
||||
PROJECTOR_TYPE_GRANITE_SPEECH,
|
||||
PROJECTOR_TYPE_MIMOVL,
|
||||
PROJECTOR_TYPE_MINIMAX_M3,
|
||||
@@ -569,6 +572,7 @@ static std::map<projector_type, std::string> PROJECTOR_TYPE_NAMES = {
|
||||
{ PROJECTOR_TYPE_EXAONE4_5, "exaone4_5"},
|
||||
{ PROJECTOR_TYPE_HUNYUANVL, "hunyuanvl"},
|
||||
{ PROJECTOR_TYPE_MINICPMV4_6, "minicpmv4_6"},
|
||||
{ PROJECTOR_TYPE_MINICPMV4_7, "minicpmv4_7"},
|
||||
{ PROJECTOR_TYPE_GRANITE_SPEECH, "granite_speech"},
|
||||
{ PROJECTOR_TYPE_MIMOVL, "mimovl"},
|
||||
{ PROJECTOR_TYPE_MINIMAX_M3, "minimax_m3"},
|
||||
@@ -660,6 +664,38 @@ struct clip_image_u8 {
|
||||
|
||||
struct mtmd_serialization; // forward declaration
|
||||
|
||||
// separators appended after the image tokens of one entry, as rows of v.tok_embd_sep
|
||||
enum clip_suffix_type : int32_t {
|
||||
CLIP_SUFFIX_NONE = 0,
|
||||
// MiniCPM-V 4.7 tiles
|
||||
CLIP_SUFFIX_MINICPMV_OV, // </image>
|
||||
CLIP_SUFFIX_MINICPMV_OV_SLICE, // </image><slice>
|
||||
CLIP_SUFFIX_MINICPMV_SLICE, // </slice><slice>
|
||||
CLIP_SUFFIX_MINICPMV_ROW_END, // </slice>\n<slice>
|
||||
CLIP_SUFFIX_MINICPMV_LAST, // </slice>
|
||||
CLIP_SUFFIX_COUNT,
|
||||
};
|
||||
|
||||
// rows of v.tok_embd_sep for each suffix type
|
||||
// MiniCPM-V 4.7 rows (set by the converter): 0 = </image>, 1 = <slice>, 2 = </slice>, 3 = \n
|
||||
static inline const std::vector<int> & clip_suffix_rows(clip_suffix_type type) {
|
||||
static const std::vector<int> none;
|
||||
static const std::vector<int> minicpmv_ov = { 0 };
|
||||
static const std::vector<int> minicpmv_ov_slice = { 0, 1 };
|
||||
static const std::vector<int> minicpmv_slice = { 2, 1 };
|
||||
static const std::vector<int> minicpmv_row_end = { 2, 3, 1 };
|
||||
static const std::vector<int> minicpmv_last = { 2 };
|
||||
switch (type) {
|
||||
case CLIP_SUFFIX_NONE: return none;
|
||||
case CLIP_SUFFIX_MINICPMV_OV: return minicpmv_ov;
|
||||
case CLIP_SUFFIX_MINICPMV_OV_SLICE: return minicpmv_ov_slice;
|
||||
case CLIP_SUFFIX_MINICPMV_SLICE: return minicpmv_slice;
|
||||
case CLIP_SUFFIX_MINICPMV_ROW_END: return minicpmv_row_end;
|
||||
case CLIP_SUFFIX_MINICPMV_LAST: return minicpmv_last;
|
||||
default: GGML_ABORT("invalid suffix type");
|
||||
}
|
||||
}
|
||||
|
||||
// For images, buf.size() == nx*ny*3
|
||||
// Memory layout: RGBRGBRGB...
|
||||
// For seq, buf.size() == nx*ny*3*nt
|
||||
@@ -675,10 +711,12 @@ struct clip_image_f32 {
|
||||
// deepseek4v: number of leading IMAGE_PAD embeddings, aligns IMAGE_START to the LLM compressor ratio
|
||||
// depends on the chunk position, set at tokenize time (see mtmd_tokenizer::add_media)
|
||||
int32_t lead_pad = 0;
|
||||
// separators appended after the image tokens
|
||||
clip_suffix_type suffix_type = CLIP_SUFFIX_NONE;
|
||||
|
||||
// llava-next "anyres" tiling, used by Granite4 Vision
|
||||
// the whole grid is encoded and assembled in a single graph
|
||||
// NOTE: excluded from serialized: a deserialized image is always a placeholder, which is never encoded
|
||||
// tile grid of the image group this entry belongs to
|
||||
// llava-next "anyres" (Granite4 Vision): the whole grid is encoded and assembled in a single graph
|
||||
// MiniCPM-V 4.7: set on the overview entry, the decoder positions of all tiles are derived from it
|
||||
struct anyres_info {
|
||||
int grid_x = 0; // tiles per row, 0 means the image is not tiled
|
||||
int grid_y = 0; // tiles per column
|
||||
|
||||
@@ -71,6 +71,7 @@ struct clip_hparams {
|
||||
std::vector<clip_image_size> image_res_candidates;
|
||||
int32_t preproc_min_tiles = 0;
|
||||
int32_t preproc_max_tiles = 0;
|
||||
int32_t max_slice_nums = 9; // llava-uhd slice cap; per-model, carried in the GGUF
|
||||
int32_t preproc_tile_size = 0; // local tile size (deepseek-ocr)
|
||||
resize_algo image_resize_algo_rf = RESIZE_ALGO_BICUBIC;
|
||||
resize_algo image_resize_algo_ov = RESIZE_ALGO_BICUBIC;
|
||||
@@ -614,6 +615,7 @@ struct clip_model {
|
||||
|
||||
ggml_tensor * image_newline = nullptr;
|
||||
ggml_tensor * view_seperator = nullptr;
|
||||
ggml_tensor * tok_embd_sep = nullptr; // [n_embd_text, n_sep] rows of the text model tok_embd (MiniCPM-V 4.7)
|
||||
|
||||
|
||||
// Yi type models with mlp+normalization projection
|
||||
|
||||
+30
-1
@@ -909,6 +909,16 @@ ggml_tensor * clip_graph::build_stack(ggml_tensor * cur, int32_t stack_factor, i
|
||||
|
||||
// aka pixel_shuffle / pixel_unshuffle / patch_merger (Kimi-VL)
|
||||
// support dynamic resolution
|
||||
ggml_tensor * clip_graph::build_suffix(ggml_tensor * cur) {
|
||||
for (int idx : clip_suffix_rows(img.suffix_type)) {
|
||||
GGML_ASSERT(model.tok_embd_sep && idx < model.tok_embd_sep->ne[1]);
|
||||
ggml_tensor * row = ggml_view_2d(ctx0, model.tok_embd_sep, model.tok_embd_sep->ne[0], 1,
|
||||
model.tok_embd_sep->nb[1], idx * model.tok_embd_sep->nb[1]);
|
||||
cur = ggml_concat(ctx0, cur, ggml_cast(ctx0, row, cur->type), 1);
|
||||
}
|
||||
return cur;
|
||||
}
|
||||
|
||||
ggml_tensor * clip_graph::build_patch_merge_permute(ggml_tensor * cur, int scale_factor) {
|
||||
GGML_ASSERT(scale_factor > 1);
|
||||
|
||||
@@ -1020,6 +1030,7 @@ static std::unique_ptr<clip_graph> clip_get_graph_builder(clip_ctx * ctx, const
|
||||
builder = std::make_unique<clip_graph_minicpmv>(ctx, img);
|
||||
} break;
|
||||
case PROJECTOR_TYPE_MINICPMV4_6:
|
||||
case PROJECTOR_TYPE_MINICPMV4_7:
|
||||
{
|
||||
builder = std::make_unique<clip_graph_minicpmv4_6>(ctx, img);
|
||||
} break;
|
||||
@@ -1332,6 +1343,7 @@ struct clip_model_loader {
|
||||
if (is_vision) {
|
||||
get_u32(KEY_IMAGE_SIZE, hparams.image_size);
|
||||
get_u32(KEY_PATCH_SIZE, hparams.patch_size);
|
||||
get_u32(KEY_MAX_SLICE_NUMS, hparams.max_slice_nums, false);
|
||||
get_i32(KEY_MINICPMV_VERSION, hparams.minicpmv_version, false); // legacy
|
||||
get_u32(KEY_MINICPMV_QUERY_NUM, hparams.minicpmv_query_num, false);
|
||||
if (hparams.minicpmv_query_num == 0) {
|
||||
@@ -1471,13 +1483,18 @@ struct clip_model_loader {
|
||||
}
|
||||
} break;
|
||||
case PROJECTOR_TYPE_MINICPMV4_6:
|
||||
case PROJECTOR_TYPE_MINICPMV4_7:
|
||||
{
|
||||
// MiniCPM-V 4.6 unified merger projector
|
||||
// MiniCPM-V 4.6/4.7 unified merger projector
|
||||
// ViT merger 2x2 + final merger 2x2 = 4x spatial merge per dimension
|
||||
hparams.n_merge = 4;
|
||||
get_u32(KEY_PROJ_SCALE_FACTOR, hparams.n_merge, false);
|
||||
GGML_ASSERT(hparams.n_merge == 2 || hparams.n_merge == 4);
|
||||
|
||||
// no padding: the reference stretches the refined image to the target size
|
||||
hparams.image_pad_ov = PAD_NONE;
|
||||
hparams.image_pad_rf = PAD_NONE;
|
||||
|
||||
// borrow wa_layer_indexes for vit_merger insertion point
|
||||
std::vector<int> wa_layer_indexes_vec;
|
||||
get_arr_int(KEY_WIN_ATTN_LAYER_INDEXES, wa_layer_indexes_vec, false);
|
||||
@@ -2393,6 +2410,7 @@ struct clip_model_loader {
|
||||
|| model.proj_type == PROJECTOR_TYPE_IDEFICS3
|
||||
|| model.proj_type == PROJECTOR_TYPE_MINICPMV
|
||||
|| model.proj_type == PROJECTOR_TYPE_MINICPMV4_6
|
||||
|| model.proj_type == PROJECTOR_TYPE_MINICPMV4_7
|
||||
) && layer.ff_up_w && layer.ff_down_w && layer.ff_down_w->ne[0] == hparams.n_embd;
|
||||
if (is_ffn_swapped) {
|
||||
// swap up and down weights
|
||||
@@ -2495,6 +2513,7 @@ struct clip_model_loader {
|
||||
model.mm_model_ln_post_b = get_tensor(string_format(TN_MINICPMV_LN, "post", "bias"));
|
||||
} break;
|
||||
case PROJECTOR_TYPE_MINICPMV4_6:
|
||||
case PROJECTOR_TYPE_MINICPMV4_7:
|
||||
{
|
||||
const bool merger_required = hparams.n_merge == 4;
|
||||
auto get_merger_tensor = [&](const std::string & name, bool required = true) {
|
||||
@@ -2526,6 +2545,7 @@ struct clip_model_loader {
|
||||
model.mm_ffn_up_b = get_tensor(string_format(TN_MM_UP, "bias"), false);
|
||||
model.mm_ffn_down_w = get_tensor(string_format(TN_MM_DOWN, "weight"));
|
||||
model.mm_ffn_down_b = get_tensor(string_format(TN_MM_DOWN, "bias"), false);
|
||||
model.tok_embd_sep = get_tensor(TN_TOK_EMBD_SEP, model.proj_type == PROJECTOR_TYPE_MINICPMV4_7);
|
||||
} break;
|
||||
case PROJECTOR_TYPE_GLM_EDGE:
|
||||
{
|
||||
@@ -4169,6 +4189,8 @@ int clip_n_output_tokens_x(const clip_ctx * ctx, const clip_image_f32 * img) {
|
||||
case PROJECTOR_TYPE_MUSE_GLIMMER:
|
||||
return (img->nx() / params.patch_size) / 2;
|
||||
case PROJECTOR_TYPE_STEP3VL:
|
||||
case PROJECTOR_TYPE_MINICPMV4_6:
|
||||
case PROJECTOR_TYPE_MINICPMV4_7:
|
||||
return img->nx() / (params.patch_size * params.n_merge);
|
||||
case PROJECTOR_TYPE_DEEPSEEKOCR:
|
||||
case PROJECTOR_TYPE_DEEPSEEKOCR2:
|
||||
@@ -4197,6 +4219,8 @@ int clip_n_output_tokens_y(const clip_ctx * ctx, const clip_image_f32 * img) {
|
||||
case PROJECTOR_TYPE_MUSE_GLIMMER:
|
||||
return (img->ny() / params.patch_size) / 2;
|
||||
case PROJECTOR_TYPE_STEP3VL:
|
||||
case PROJECTOR_TYPE_MINICPMV4_6:
|
||||
case PROJECTOR_TYPE_MINICPMV4_7:
|
||||
return img->ny() / (params.patch_size * params.n_merge);
|
||||
default:
|
||||
break;
|
||||
@@ -4262,6 +4286,7 @@ int clip_n_output_tokens(const clip_ctx * ctx, const clip_image_f32 * img) {
|
||||
}
|
||||
} break;
|
||||
case PROJECTOR_TYPE_MINICPMV4_6:
|
||||
case PROJECTOR_TYPE_MINICPMV4_7:
|
||||
{
|
||||
n_patches /= params.n_merge * params.n_merge;
|
||||
} break;
|
||||
@@ -4513,6 +4538,8 @@ int clip_n_output_tokens(const clip_ctx * ctx, const clip_image_f32 * img) {
|
||||
GGML_ABORT("unsupported projector type");
|
||||
}
|
||||
|
||||
n_patches += (int) clip_suffix_rows(img->suffix_type).size();
|
||||
|
||||
return n_patches;
|
||||
}
|
||||
|
||||
@@ -4836,6 +4863,7 @@ bool clip_encode(struct clip_ctx * ctx, struct clip_encode_params * params) {
|
||||
set_input_f32("omega", omega);
|
||||
} break;
|
||||
case PROJECTOR_TYPE_MINICPMV4_6:
|
||||
case PROJECTOR_TYPE_MINICPMV4_7:
|
||||
{
|
||||
const bool is_4x = hparams.n_merge == 2;
|
||||
|
||||
@@ -6080,6 +6108,7 @@ int clip_n_mmproj_embd(const struct clip_ctx * ctx) {
|
||||
case PROJECTOR_TYPE_MINICPMV:
|
||||
return ctx->model.mm_model_proj->ne[0];
|
||||
case PROJECTOR_TYPE_MINICPMV4_6:
|
||||
case PROJECTOR_TYPE_MINICPMV4_7:
|
||||
return ctx->model.mm_ffn_down_w->ne[1];
|
||||
case PROJECTOR_TYPE_GLM_EDGE:
|
||||
return ctx->model.mm_model_mlp_3_w->ne[1];
|
||||
|
||||
@@ -350,6 +350,8 @@ ggml_cgraph * clip_graph_minicpmv4_6::build() {
|
||||
inpL = cur;
|
||||
}
|
||||
|
||||
inpL = build_suffix(inpL);
|
||||
|
||||
ggml_build_forward_expand(gf, inpL);
|
||||
return gf;
|
||||
}
|
||||
|
||||
+12
-16
@@ -507,9 +507,7 @@ mtmd_image_preproc_out mtmd_image_preprocessor_llava_uhd::preprocess(const clip_
|
||||
|
||||
mtmd_image_preprocessor_llava_uhd::slice_instructions mtmd_image_preprocessor_llava_uhd::get_slice_instructions(const clip_image_size & original_size) const {
|
||||
mtmd_image_preprocessor_llava_uhd::slice_instructions res;
|
||||
// align slices by patch_size * n_merge so an integer number of merger output tokens fits per slice
|
||||
const int n_merge = hparams.n_merge;
|
||||
const int patch_size = hparams.patch_size * n_merge;
|
||||
const int patch_size = get_slice_align();
|
||||
const int slice_size = hparams.image_size;
|
||||
const int original_width = original_size.width;
|
||||
const int original_height = original_size.height;
|
||||
@@ -568,7 +566,7 @@ mtmd_image_preprocessor_llava_uhd::slice_instructions mtmd_image_preprocessor_ll
|
||||
res.overview_size = best_size;
|
||||
|
||||
{
|
||||
const int max_slice_nums = 9; // TODO: this is only used by minicpmv, maybe remove it
|
||||
const int max_slice_nums = hparams.max_slice_nums > 0 ? hparams.max_slice_nums : 9;
|
||||
const float log_ratio = log((float)original_width / original_height);
|
||||
const float ratio = (float)original_width * original_height / (slice_size * slice_size);
|
||||
const int multiple = fmin(ceil(ratio), max_slice_nums);
|
||||
@@ -691,7 +689,7 @@ clip_image_size mtmd_image_preprocessor_llava_uhd::select_best_resolution(const
|
||||
}
|
||||
|
||||
int mtmd_image_preprocessor_llava_uhd::ensure_divide(int length, int patch_size) const {
|
||||
return std::max(static_cast<int>(std::round(static_cast<float>(length) / patch_size) * patch_size), patch_size);
|
||||
return std::max(align_round(static_cast<double>(length) / patch_size) * patch_size, patch_size);
|
||||
}
|
||||
|
||||
clip_image_size mtmd_image_preprocessor_llava_uhd::get_refine_size(const clip_image_size & original_size, const clip_image_size & grid, int scale_resolution, int patch_size, bool allow_upscale) const {
|
||||
@@ -893,17 +891,15 @@ mtmd_image_preproc_out mtmd_image_preprocessor_longest_edge::preprocess(const cl
|
||||
//
|
||||
|
||||
mtmd_image_preprocessor_llava_uhd::slice_instructions mtmd_image_preprocessor_minicpmv::get_slice_instructions(const clip_image_size & original_size) const {
|
||||
if (hparams.n_merge == 2) {
|
||||
const int slice_size = hparams.image_size;
|
||||
const float ratio = (float)original_size.width * original_size.height / (slice_size * slice_size);
|
||||
if (ratio <= 1.0f) {
|
||||
mtmd_image_preprocessor_llava_uhd::slice_instructions inst;
|
||||
const int patch_size = hparams.patch_size * hparams.n_merge;
|
||||
inst.overview_size = get_best_resize(original_size, slice_size, patch_size, true);
|
||||
inst.refined_size = clip_image_size{0, 0};
|
||||
inst.grid_size = clip_image_size{0, 0};
|
||||
return inst;
|
||||
}
|
||||
// overview only for small images, unlike generic llava-uhd which slices once one side exceeds scale resolution
|
||||
const int slice_size = hparams.image_size;
|
||||
const float ratio = (float) original_size.width * original_size.height / (slice_size * slice_size);
|
||||
if (ratio <= 1.0f) {
|
||||
mtmd_image_preprocessor_llava_uhd::slice_instructions inst;
|
||||
inst.overview_size = get_best_resize(original_size, slice_size, get_slice_align(), true);
|
||||
inst.refined_size = clip_image_size{0, 0};
|
||||
inst.grid_size = clip_image_size{0, 0};
|
||||
return inst;
|
||||
}
|
||||
return mtmd_image_preprocessor_llava_uhd::get_slice_instructions(original_size);
|
||||
}
|
||||
|
||||
@@ -83,6 +83,17 @@ struct mtmd_image_preprocessor_llava_uhd : mtmd_image_preprocessor {
|
||||
slice_output slice_image(const clip_image_u8 & img, const slice_instructions & inst) const;
|
||||
|
||||
protected:
|
||||
// align slices to a multiple of the merger factor (integer merger tokens per slice)
|
||||
virtual int get_slice_align() const {
|
||||
const int merge = hparams.n_merge > 0 ? hparams.n_merge : 1;
|
||||
return hparams.patch_size * merge;
|
||||
}
|
||||
|
||||
// rounding for snapping a length to a multiple of the align size
|
||||
virtual int align_round(double v) const {
|
||||
return static_cast<int>(std::round(v));
|
||||
}
|
||||
|
||||
clip_image_size get_best_resize(const clip_image_size & original_size, int scale_resolution, int patch_size, bool allow_upscale = false) const;
|
||||
|
||||
/**
|
||||
@@ -155,6 +166,26 @@ private:
|
||||
struct mtmd_image_preprocessor_minicpmv : mtmd_image_preprocessor_llava_uhd {
|
||||
using mtmd_image_preprocessor_llava_uhd::mtmd_image_preprocessor_llava_uhd;
|
||||
slice_instructions get_slice_instructions(const clip_image_size & original_size) const override;
|
||||
|
||||
protected:
|
||||
// always patch_size * 4, even in 4x mode (the 2x2 vit_merger slot stays)
|
||||
int get_slice_align() const override {
|
||||
return hparams.patch_size * 4;
|
||||
}
|
||||
|
||||
// Python's round() breaks ties to even, unlike std::round
|
||||
int align_round(double v) const override {
|
||||
const double fl = std::floor(v);
|
||||
const double diff = v - fl;
|
||||
if (diff > 0.5) {
|
||||
return static_cast<int>(fl) + 1;
|
||||
}
|
||||
if (diff < 0.5) {
|
||||
return static_cast<int>(fl);
|
||||
}
|
||||
const int lo = static_cast<int>(fl);
|
||||
return (lo % 2 == 0) ? lo : lo + 1;
|
||||
}
|
||||
};
|
||||
|
||||
// custom llava-uhd slicing logic for LFM2
|
||||
|
||||
+207
-4
@@ -19,6 +19,7 @@
|
||||
|
||||
#include <algorithm>
|
||||
#include <cerrno>
|
||||
#include <cmath>
|
||||
#include <cstdio>
|
||||
#include <cstdlib>
|
||||
#include <cstring>
|
||||
@@ -27,7 +28,10 @@
|
||||
#include <vector>
|
||||
|
||||
// remember to bump this if the serialization format changes
|
||||
#define MTMD_SERIALIZATION_VERSION 2
|
||||
#define MTMD_SERIALIZATION_VERSION 3
|
||||
|
||||
// oldest compat version that can be loaded
|
||||
#define MTMD_SERIALIZATION_VERSION_MIN 2
|
||||
|
||||
struct mtmd_serialization {
|
||||
// note: using 64-bit here for future-proofing
|
||||
@@ -45,7 +49,7 @@ struct mtmd_serialization {
|
||||
// copy buf to data
|
||||
data.assign(buf, buf + len);
|
||||
uint64_t ver_in = read<uint64_t>();
|
||||
if (ver_in != version) {
|
||||
if (ver_in < MTMD_SERIALIZATION_VERSION_MIN || ver_in > version) {
|
||||
throw std::runtime_error("version mismatch");
|
||||
}
|
||||
this->version = ver_in;
|
||||
@@ -106,6 +110,11 @@ void clip_image_f32::serialize(mtmd_serialization & ser) const {
|
||||
ser.write(add_viewsep);
|
||||
ser.write(add_newline);
|
||||
ser.write(lead_pad);
|
||||
ser.write((int32_t)suffix_type);
|
||||
ser.write((int32_t)anyres.grid_x);
|
||||
ser.write((int32_t)anyres.grid_y);
|
||||
ser.write((int32_t)anyres.orig_nx);
|
||||
ser.write((int32_t)anyres.orig_ny);
|
||||
ser.write((int32_t)nx_);
|
||||
ser.write((int32_t)ny_);
|
||||
}
|
||||
@@ -113,6 +122,17 @@ void clip_image_f32::deserialize(mtmd_serialization & ser) {
|
||||
add_viewsep = ser.read<bool>();
|
||||
add_newline = ser.read<bool>();
|
||||
lead_pad = ser.read<int32_t>();
|
||||
if (ser.version >= 3) {
|
||||
const int32_t suffix_raw = ser.read<int32_t>();
|
||||
if (suffix_raw < 0 || suffix_raw >= CLIP_SUFFIX_COUNT) {
|
||||
throw std::runtime_error("invalid suffix type");
|
||||
}
|
||||
suffix_type = (clip_suffix_type)suffix_raw;
|
||||
anyres.grid_x = ser.read<int32_t>();
|
||||
anyres.grid_y = ser.read<int32_t>();
|
||||
anyres.orig_nx = ser.read<int32_t>();
|
||||
anyres.orig_ny = ser.read<int32_t>();
|
||||
}
|
||||
nx_ = ser.read<int32_t>();
|
||||
ny_ = ser.read<int32_t>();
|
||||
buf.clear(); // always a placeholder after loading
|
||||
@@ -204,9 +224,11 @@ enum mtmd_pos_type {
|
||||
MTMD_POS_TYPE_NORMAL, // number of positions equals to number of tokens
|
||||
MTMD_POS_TYPE_MROPE, // qwen-vl mrope style, each image takes max(t,h,w) position indexes
|
||||
MTMD_POS_TYPE_HUNYUANVL, // HunyuanVL mrope + BOI/EOI/newline layout with XD-RoPE dim-3
|
||||
MTMD_POS_TYPE_CANVAS, // MiniCPM-V 4.7: overview + slices in one chunk, sharing one 2D canvas (see mtmd_image_tokens::canvas_tile_grid)
|
||||
MTMD_POS_TYPE_COUNT, // for validation
|
||||
};
|
||||
|
||||
|
||||
struct mtmd_image_tokens {
|
||||
uint32_t nx = 0; // number of tokens in x direction
|
||||
uint32_t ny = 0; // number of tokens in y direction
|
||||
@@ -218,6 +240,14 @@ struct mtmd_image_tokens {
|
||||
// [BOI] [row0 tokens + newline] ... [row(ny-1) tokens + newline] [EOI]
|
||||
return (nx + 1) * ny + 2;
|
||||
}
|
||||
if (pos == MTMD_POS_TYPE_CANVAS) {
|
||||
uint32_t n = 0;
|
||||
for (size_t k = 0; k < batch_f32.entries.size(); ++k) {
|
||||
const auto [gw, gh] = canvas_tile_grid(k);
|
||||
n += gw * gh + (uint32_t) clip_suffix_rows(batch_f32.entries[k].suffix_type).size();
|
||||
}
|
||||
return n;
|
||||
}
|
||||
uint32_t nz = batch_f32.entries.size();
|
||||
if (n_temporal_merge > 1) {
|
||||
// [QWEN_VIDEO] this logic is quite ugly, it's mostly to make qwen-vl temporal merge work, can be improved in the future
|
||||
@@ -243,8 +273,17 @@ struct mtmd_image_tokens {
|
||||
return false;
|
||||
}
|
||||
|
||||
// MTMD_POS_TYPE_CANVAS: entries are [overview, slices row by row], nx/ny is the token grid of the last entry
|
||||
// returns the token grid (w, h) of entry k, scaled from its pixel size
|
||||
std::pair<uint32_t, uint32_t> canvas_tile_grid(size_t k) const {
|
||||
const auto & ref = batch_f32.entries.back();
|
||||
const auto & e = batch_f32.entries[k];
|
||||
return { (uint32_t) e.nx() * nx / ref.nx(), (uint32_t) e.ny() * ny / ref.ny() };
|
||||
}
|
||||
|
||||
bool can_batch_with(const mtmd_image_tokens & other) {
|
||||
return nx == other.nx && ny == other.ny && pos == other.pos;
|
||||
// a canvas chunk holds a whole image group, its layout is not given by nx/ny alone
|
||||
return nx == other.nx && ny == other.ny && pos == other.pos && pos != MTMD_POS_TYPE_CANVAS;
|
||||
}
|
||||
|
||||
mtmd_image_tokens clone() {
|
||||
@@ -516,6 +555,9 @@ struct mtmd_context {
|
||||
bool tok_row_end_trail = false;
|
||||
bool ov_img_first = false;
|
||||
|
||||
// MiniCPM-V 4.6/4.7 prepends an <image_id>N</image_id> tag before <image>
|
||||
bool use_image_id = false;
|
||||
|
||||
// string template for slice image delimiters with row/col (idefics3)
|
||||
std::string sli_img_start_tmpl;
|
||||
|
||||
@@ -680,6 +722,7 @@ struct mtmd_context {
|
||||
image_preproc = std::make_unique<mtmd_image_preprocessor_llava_uhd>(ctx_v);
|
||||
} break;
|
||||
case PROJECTOR_TYPE_MINICPMV4_6:
|
||||
case PROJECTOR_TYPE_MINICPMV4_7:
|
||||
{
|
||||
slice_tmpl = MTMD_SLICE_TMPL_MINICPMV_2_6;
|
||||
tok_ov_img_start = {lookup_token("<image>")};
|
||||
@@ -689,6 +732,7 @@ struct mtmd_context {
|
||||
tok_row_end = {lookup_token("\n")};
|
||||
tok_row_end_trail = false; // no trailing end-of-row token
|
||||
ov_img_first = true;
|
||||
use_image_id = true;
|
||||
image_preproc = std::make_unique<mtmd_image_preprocessor_minicpmv>(ctx_v);
|
||||
} break;
|
||||
case PROJECTOR_TYPE_QWEN2VL:
|
||||
@@ -1429,7 +1473,15 @@ struct mtmd_tokenizer {
|
||||
const bool has_tiling_grid = (preproc_out.grid_x > 0 && preproc_out.grid_y > 0)
|
||||
|| preproc_out.has_overview();
|
||||
|
||||
if (has_tiling_grid) {
|
||||
if (has_tiling_grid && ctx->proj_type_v() == PROJECTOR_TYPE_MINICPMV4_7) {
|
||||
GGML_ASSERT(bitmaps.size() == 1);
|
||||
if (ctx->use_image_id) {
|
||||
add_text("<image_id>" + std::to_string(n_images_added) + "</image_id>", true);
|
||||
}
|
||||
add_text(ctx->tok_ov_img_start);
|
||||
// the separators after <image> are appended by clip, see add_canvas_chunk()
|
||||
add_canvas_chunk(std::move(preproc_out), bitmaps[0]->id);
|
||||
} else if (has_tiling_grid) {
|
||||
// [QWEN_VIDEO] we do not support "frame merging" for llama-uhd style, so no batching for now
|
||||
GGML_ASSERT(bitmaps.size() == 1);
|
||||
|
||||
@@ -1448,6 +1500,9 @@ struct mtmd_tokenizer {
|
||||
|
||||
// add overview image (first)
|
||||
if (ctx->ov_img_first) {
|
||||
if (ctx->use_image_id) {
|
||||
add_text("<image_id>" + std::to_string(n_images_added) + "</image_id>", true);
|
||||
}
|
||||
add_text(ctx->tok_ov_img_start);
|
||||
cur.entries.emplace_back(std::move(ov_chunk));
|
||||
add_text(ctx->tok_ov_img_end);
|
||||
@@ -1673,6 +1728,62 @@ struct mtmd_tokenizer {
|
||||
return 0;
|
||||
}
|
||||
|
||||
// MiniCPM-V 4.7: the overview and all slices go in one chunk, clip appends the separators after each tile:
|
||||
// [ov] </image><slice> [S00] </slice><slice> [S01] </slice>\n<slice> [S10] </slice><slice> [S11] </slice>
|
||||
void add_canvas_chunk(mtmd_image_preproc_out && preproc_out, const std::string & id) {
|
||||
const int n_col = preproc_out.grid_x;
|
||||
const int n_row = preproc_out.grid_y;
|
||||
auto & slices = preproc_out.entries;
|
||||
GGML_ASSERT(preproc_out.has_overview());
|
||||
GGML_ASSERT((int) slices.size() == n_col * n_row);
|
||||
|
||||
auto & ov = preproc_out.overview;
|
||||
ov.suffix_type = CLIP_SUFFIX_MINICPMV_OV;
|
||||
if (!slices.empty()) {
|
||||
ov.suffix_type = CLIP_SUFFIX_MINICPMV_OV_SLICE;
|
||||
ov.anyres.grid_x = n_col;
|
||||
ov.anyres.grid_y = n_row;
|
||||
}
|
||||
for (int y = 0; y < n_row; y++) {
|
||||
for (int x = 0; x < n_col; x++) {
|
||||
auto & suffix = slices[y * n_col + x].suffix_type;
|
||||
if (y == n_row - 1 && x == n_col - 1) {
|
||||
suffix = CLIP_SUFFIX_MINICPMV_LAST;
|
||||
} else if (x == n_col - 1) {
|
||||
suffix = CLIP_SUFFIX_MINICPMV_ROW_END;
|
||||
} else {
|
||||
suffix = CLIP_SUFFIX_MINICPMV_SLICE;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
mtmd_image_tokens_ptr image_tokens(new mtmd_image_tokens);
|
||||
image_tokens->pos = MTMD_POS_TYPE_CANVAS;
|
||||
image_tokens->id = id;
|
||||
auto & entries = image_tokens->batch_f32.entries;
|
||||
entries.push_back(std::move(ov));
|
||||
for (auto & slice : slices) {
|
||||
entries.push_back(std::move(slice));
|
||||
}
|
||||
// token grid of the last entry, the grids of the other entries are scaled from it
|
||||
image_tokens->nx = clip_n_output_tokens_x(ctx->ctx_v, &entries.back());
|
||||
image_tokens->ny = clip_n_output_tokens_y(ctx->ctx_v, &entries.back());
|
||||
|
||||
size_t n_tokens = 0;
|
||||
for (const auto & entry : entries) {
|
||||
n_tokens += clip_n_output_tokens(ctx->ctx_v, &entry);
|
||||
}
|
||||
GGML_ASSERT(n_tokens == image_tokens->n_tokens());
|
||||
|
||||
mtmd_input_chunk chunk{
|
||||
MTMD_INPUT_CHUNK_TYPE_IMAGE,
|
||||
{}, // text tokens
|
||||
std::move(image_tokens),
|
||||
nullptr, // audio tokens
|
||||
};
|
||||
cur.entries.emplace_back(std::move(chunk));
|
||||
}
|
||||
|
||||
std::vector<mtmd_input_chunk> split_batch_to_chunk(mtmd_image_preproc_out && preproc_out, const std::string & id) {
|
||||
std::vector<mtmd_input_chunk> chunks;
|
||||
|
||||
@@ -1814,6 +1925,24 @@ static int32_t mtmd_encode_impl(mtmd_context * ctx, const mtmd_image_tokens * im
|
||||
return 1;
|
||||
}
|
||||
|
||||
if (image_tokens->pos == MTMD_POS_TYPE_CANVAS) {
|
||||
// the tiles differ in size, encode them one by one
|
||||
size_t offset = 0;
|
||||
for (const auto & entry : image_tokens->batch_f32.entries) {
|
||||
clip_image_f32_batch one;
|
||||
one.entries.push_back(entry);
|
||||
std::vector<float> embd((size_t) n_embd_out * clip_n_output_tokens(ctx_clip, &entry));
|
||||
if (!clip_image_batch_encode(ctx_clip, ctx->n_threads, &one, embd)) {
|
||||
return 1;
|
||||
}
|
||||
GGML_ASSERT(offset + embd.size() <= out_embd.size());
|
||||
std::copy(embd.begin(), embd.end(), out_embd.begin() + offset);
|
||||
offset += embd.size();
|
||||
}
|
||||
GGML_ASSERT(offset == out_embd.size());
|
||||
return 0;
|
||||
}
|
||||
|
||||
bool ok = clip_image_batch_encode(
|
||||
ctx_clip,
|
||||
ctx->n_threads,
|
||||
@@ -2494,6 +2623,67 @@ size_t mtmd_image_tokens_get_ny(const mtmd_image_tokens * image_tokens) {
|
||||
return image_tokens->ny;
|
||||
}
|
||||
|
||||
// map a tile coordinate onto the canvas like the reference: round(linspace(0, canvas - 1, grid)), round() breaks ties to even
|
||||
static uint32_t mtmd_canvas_scale(uint32_t coord, uint32_t grid, uint32_t canvas) {
|
||||
if (grid <= 1 || canvas <= 1) {
|
||||
return 0;
|
||||
}
|
||||
const double v = (double) coord * (double) (canvas - 1) / (double) (grid - 1);
|
||||
return std::min((uint32_t) std::nearbyint(v), canvas - 1);
|
||||
}
|
||||
|
||||
// MTMD_POS_TYPE_CANVAS: every tile shares the <image> token before the chunk as origin
|
||||
// the overview is stretched over the whole canvas, each slice fills its own cell; the time component is the origin, in slot z
|
||||
// a tile takes one position in slot t (the KV cache position), the separators after it take one position each
|
||||
static mtmd_decoder_pos mtmd_canvas_decoder_pos(const mtmd_image_tokens * image_tokens, llama_pos pos_0, size_t i) {
|
||||
const auto & entries = image_tokens->batch_f32.entries;
|
||||
const auto & grid = entries[0].anyres;
|
||||
const uint32_t nx = image_tokens->nx;
|
||||
const uint32_t ny = image_tokens->ny;
|
||||
const uint32_t canvas_w = grid.is_tiled() ? grid.grid_x * nx : nx;
|
||||
const uint32_t canvas_h = grid.is_tiled() ? grid.grid_y * ny : ny;
|
||||
const uint32_t base = pos_0 - 1;
|
||||
|
||||
mtmd_decoder_pos pos;
|
||||
uint32_t t = pos_0;
|
||||
for (size_t k = 0; k < entries.size(); ++k) {
|
||||
const auto [gw, gh] = image_tokens->canvas_tile_grid(k);
|
||||
if (i < gw * gh) {
|
||||
const uint32_t row = i / gw;
|
||||
const uint32_t col = i % gw;
|
||||
uint32_t h;
|
||||
uint32_t w;
|
||||
if (k == 0) {
|
||||
h = mtmd_canvas_scale(row, gh, canvas_h);
|
||||
w = mtmd_canvas_scale(col, gw, canvas_w);
|
||||
} else {
|
||||
const uint32_t s = k - 1;
|
||||
h = (s / grid.grid_x) * ny + row;
|
||||
w = (s % grid.grid_x) * nx + col;
|
||||
}
|
||||
pos.t = t;
|
||||
pos.x = base + w;
|
||||
pos.y = base + h;
|
||||
pos.z = base;
|
||||
return pos;
|
||||
}
|
||||
i -= gw * gh;
|
||||
|
||||
const size_t n_sep = clip_suffix_rows(entries[k].suffix_type).size();
|
||||
if (i < n_sep) {
|
||||
const uint32_t p = t + 1 + i;
|
||||
pos.t = p;
|
||||
pos.x = p;
|
||||
pos.y = p;
|
||||
pos.z = p;
|
||||
return pos;
|
||||
}
|
||||
i -= n_sep;
|
||||
t += 1 + n_sep;
|
||||
}
|
||||
GGML_ABORT("token index out of range");
|
||||
}
|
||||
|
||||
mtmd_decoder_pos mtmd_image_tokens_get_decoder_pos(const mtmd_image_tokens * image_tokens, llama_pos pos_0, size_t i) {
|
||||
mtmd_decoder_pos pos;
|
||||
switch (image_tokens->pos) {
|
||||
@@ -2543,6 +2733,10 @@ mtmd_decoder_pos mtmd_image_tokens_get_decoder_pos(const mtmd_image_tokens * ima
|
||||
pos.z = image_tokens->image_idx;
|
||||
}
|
||||
} break;
|
||||
case MTMD_POS_TYPE_CANVAS:
|
||||
{
|
||||
pos = mtmd_canvas_decoder_pos(image_tokens, pos_0, i);
|
||||
} break;
|
||||
default:
|
||||
GGML_ABORT("invalid position type");
|
||||
}
|
||||
@@ -2563,6 +2757,15 @@ llama_pos mtmd_image_tokens_get_n_pos(const mtmd_image_tokens * image_tokens) {
|
||||
// HunyuanVL: the sequential (dim-0) position advances by the full token count
|
||||
// (includes BOI/EOI and row newline tokens), not by max(nx, ny)
|
||||
return image_tokens->n_tokens();
|
||||
case MTMD_POS_TYPE_CANVAS:
|
||||
{
|
||||
// one position per tile, plus one per separator
|
||||
llama_pos n_pos = 0;
|
||||
for (const auto & entry : image_tokens->batch_f32.entries) {
|
||||
n_pos += 1 + (llama_pos) clip_suffix_rows(entry.suffix_type).size();
|
||||
}
|
||||
return n_pos;
|
||||
}
|
||||
default:
|
||||
GGML_ABORT("invalid position type");
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user