mtmd: deepseek-ocr v1 multi-tile (#24717)

* mtmd: deepseek-ocr v1 multi-tile dynamic resolution + unified image-preprocessors for both versions (ds-ocr v1 and v2)

* remove hacky API

* fuse row into a long image

* almost working

* adapt to new preprocessor api

* rm debugging printf

* improve

* mtmd: dsocr-tiles fixes (#25481)

* ds-ocr img-preproc fuse_row tile-drop fix for multi rows and columns images

* mtmd drop the duplicate redundant img_end

* deepseekocr graph simplify CLS broadcast cleanup

* test-deepseek-ocr: relax v1 single-view tolerance; drop trailing prompt space; make DRY opt-in and n_predict model-specific (#25486)

---------

Co-authored-by: Saba Fallah <10401143+sfallah@users.noreply.github.com>
Co-authored-by: Saba Fallah <sabafallah@gmail.com>
This commit is contained in:
Xuan-Son Nguyen
2026-07-10 16:05:49 +02:00
committed by GitHub
co-authored by Saba Fallah Saba Fallah
parent 07d9378286
commit 3e706dd55f
10 changed files with 239 additions and 119 deletions
+29 -5
View File
@@ -1024,6 +1024,8 @@ static std::unique_ptr<clip_graph> clip_get_graph_builder(clip_ctx * ctx, const
GGML_ABORT("missing cgraph builder");
}
builder->img_batch = &imgs;
// TODO [QWEN_VIDEO]: improve this in the future
builder->n_batch = imgs.entries.size();
@@ -1580,7 +1582,16 @@ struct clip_model_loader {
get_u32(KEY_SAM_N_HEAD, hparams.sam_n_head, true);
get_u32(KEY_SAM_N_EMBD, hparams.sam_n_embd, true);
get_u32(KEY_ATTN_WINDOW_SIZE, hparams.attn_window_size, true);
hparams.preproc_min_tiles = 2;
if (model.proj_type == PROJECTOR_TYPE_DEEPSEEKOCR) {
hparams.preproc_max_tiles = 9;
hparams.preproc_tile_size = 640;
// the CLIP/ViT body runs its layernorms at 1e-5 (the SAM stage uses 1e-6)
hparams.eps = 1e-5f;
}
if (model.proj_type == PROJECTOR_TYPE_DEEPSEEKOCR2) {
hparams.preproc_max_tiles = 6;
hparams.preproc_tile_size = 768;
// qwen2 encoder is GQA, requires KEY_N_HEAD_KV
get_u32(string_format(KEY_N_HEAD_KV, "vision"), hparams.n_head_kv);
}
@@ -3251,6 +3262,9 @@ int clip_n_output_tokens_x(const clip_ctx * ctx, const clip_image_f32 * img) {
return (img->nx() / params.patch_size) / 2;
case PROJECTOR_TYPE_STEP3VL:
return img->nx() / (params.patch_size * params.n_merge);
case PROJECTOR_TYPE_DEEPSEEKOCR:
case PROJECTOR_TYPE_DEEPSEEKOCR2:
return (img->nx() / params.patch_size) / 4;
default:
break;
}
@@ -3460,10 +3474,17 @@ int clip_n_output_tokens(const clip_ctx * ctx, const clip_image_f32 * img) {
// E.g., 64x64 -> 16x16 patches
n_patches /= 16;
// build_global_local_features adds image newlines and view separator
// Formula: h*(w+1) + 1 where h = w = sqrt(n_patches)
int h = static_cast<int>(std::sqrt(static_cast<float>(n_patches)));
n_patches = h * (h + 1) + 1;
if (img->add_viewsep) {
// global view: one image-newline per token-row + trailing view separator
const int h = static_cast<int>(std::sqrt(static_cast<float>(n_patches)));
n_patches = h * (h + 1) + 1;
} else if (img->ny() >= img->nx() && img->ny() % img->nx() == 0) {
// tile row: one image-newline per token-row
const int grid_w = img->ny() / img->nx();
const int tile_patches = img->nx() / (patch_size * 4); // patches per tile side (SAM divides by 4)
const int h = tile_patches;
n_patches = (tile_patches * grid_w + 1) * h;
}
} break;
case PROJECTOR_TYPE_HUNYUANVL:
{
@@ -4103,7 +4124,10 @@ bool clip_image_batch_encode(clip_ctx * ctx, int n_threads, const clip_image_f32
case PROJECTOR_TYPE_DEEPSEEKOCR:
case PROJECTOR_TYPE_DEEPSEEKOCR2:
{
GGML_ASSERT(pos_w == pos_h);
GGML_ASSERT(
(pos_w == pos_h) // overview image
|| (pos_h >= pos_w && pos_h % pos_w == 0) // tile images
);
const int window = hparams.attn_window_size;
const int pos = pos_w;