mtmd: support dots3-note vision+audio (#27524)

* text: conversion

* init impl

* mtmd: conversion

* impl mtmd cpp

* Update gguf-py/gguf/tensor_mapping.py

Co-authored-by: Sigbjørn Skjæret <sigbjorn.skjaeret@huggingface.co>

---------

Co-authored-by: Sigbjørn Skjæret <sigbjorn.skjaeret@huggingface.co>
This commit is contained in:
Xuan-Son Nguyen
2026-08-22 10:35:50 +02:00
committed by GitHub
co-authored by Sigbjørn Skjæret
parent 3a653fea93
commit 54ee5ee643
15 changed files with 535 additions and 11 deletions
+8
View File
@@ -825,6 +825,7 @@ struct mtmd_context {
image_preproc = std::make_unique<mtmd_image_preprocessor_longest_edge>(ctx_v);
} break;
case PROJECTOR_TYPE_DOTS_OCR:
case PROJECTOR_TYPE_DOTS3NOTE_V:
{
// <|img|> ... (image embeddings) ... <|endofimg|>
img_beg = "<|img|>";
@@ -976,6 +977,13 @@ struct mtmd_context {
aud_end = "<audio|>";
audio_preproc = std::make_unique<mtmd_audio_preprocessor_gemma4ua>(ctx_a);
} break;
case PROJECTOR_TYPE_DOTS3NOTE_A:
{
// <|audio_comp_start|> ... (embeddings) ... <|audio_comp_end|>
aud_beg = "<|audio_comp_start|>";
aud_end = "<|audio_comp_end|>";
audio_preproc = std::make_unique<mtmd_audio_preprocessor_dots3note>(ctx_a);
} break;
case PROJECTOR_TYPE_MIMO_AUDIO:
{
aud_beg = "<|mimo_audio_start|>";