mtmd: use pillow-accurate algo, correct resize_algo for all models (#27594)
* mtmd: use pillow-accurate resize algo, correct resize_algo for all models * speed optimization
This commit is contained in:
@@ -29,10 +29,10 @@ enum patch_merge_type {
|
|||||||
PATCH_MERGE_SPATIAL_UNPAD,
|
PATCH_MERGE_SPATIAL_UNPAD,
|
||||||
};
|
};
|
||||||
|
|
||||||
|
// all algos are Pillow-compatible (matching PIL.Image.resize output)
|
||||||
enum resize_algo {
|
enum resize_algo {
|
||||||
RESIZE_ALGO_BILINEAR, // stretch to target resolution
|
RESIZE_ALGO_BILINEAR,
|
||||||
RESIZE_ALGO_BICUBIC, // center-crop when aspect ratio doesn't match
|
RESIZE_ALGO_BICUBIC,
|
||||||
RESIZE_ALGO_BICUBIC_PILLOW,
|
|
||||||
RESIZE_ALGO_LANCZOS,
|
RESIZE_ALGO_LANCZOS,
|
||||||
};
|
};
|
||||||
|
|
||||||
@@ -73,7 +73,7 @@ struct clip_hparams {
|
|||||||
int32_t preproc_max_tiles = 0;
|
int32_t preproc_max_tiles = 0;
|
||||||
int32_t preproc_tile_size = 0; // local tile size (deepseek-ocr)
|
int32_t preproc_tile_size = 0; // local tile size (deepseek-ocr)
|
||||||
resize_algo image_resize_algo_rf = RESIZE_ALGO_BICUBIC;
|
resize_algo image_resize_algo_rf = RESIZE_ALGO_BICUBIC;
|
||||||
resize_algo image_resize_algo_ov = RESIZE_ALGO_BILINEAR;
|
resize_algo image_resize_algo_ov = RESIZE_ALGO_BICUBIC;
|
||||||
pad_style image_pad_rf = PAD_CEIL; // padding style for the refined image (e.g. llava-1.6)
|
pad_style image_pad_rf = PAD_CEIL; // padding style for the refined image (e.g. llava-1.6)
|
||||||
pad_style image_pad_ov = PAD_NONE; // padding style for the overview image (e.g. llava-1.6)
|
pad_style image_pad_ov = PAD_NONE; // padding style for the overview image (e.g. llava-1.6)
|
||||||
std::array<uint8_t, 3> image_pad_color_rf = {0, 0, 0}; // padding color for refined image
|
std::array<uint8_t, 3> image_pad_color_rf = {0, 0, 0}; // padding color for refined image
|
||||||
|
|||||||
+20
-19
@@ -1420,20 +1420,18 @@ struct clip_model_loader {
|
|||||||
hparams.image_pad_color = {122, 116, 104};
|
hparams.image_pad_color = {122, 116, 104};
|
||||||
if (!hparams.image_res_candidates.empty()) {
|
if (!hparams.image_res_candidates.empty()) {
|
||||||
hparams.image_resize_pad = PAD_CEIL;
|
hparams.image_resize_pad = PAD_CEIL;
|
||||||
hparams.image_resize_algo = RESIZE_ALGO_BILINEAR;
|
hparams.image_resize_algo = RESIZE_ALGO_BICUBIC;
|
||||||
} else {
|
} else {
|
||||||
// llava-1.6 default params
|
// llava-1.6 default params
|
||||||
hparams.image_pad_ov = PAD_NONE;
|
hparams.image_pad_ov = PAD_NONE;
|
||||||
hparams.image_pad_rf = PAD_CEIL;
|
hparams.image_pad_rf = PAD_CEIL;
|
||||||
hparams.image_pad_color_rf = {122, 116, 104};
|
hparams.image_pad_color_rf = {122, 116, 104};
|
||||||
hparams.image_resize_algo_rf = RESIZE_ALGO_BICUBIC;
|
|
||||||
hparams.image_resize_algo_ov = RESIZE_ALGO_BILINEAR;
|
|
||||||
}
|
}
|
||||||
} break;
|
} break;
|
||||||
case PROJECTOR_TYPE_GLM_EDGE:
|
case PROJECTOR_TYPE_GLM_EDGE:
|
||||||
{
|
{
|
||||||
hparams.image_resize_pad = PAD_CEIL;
|
hparams.image_resize_pad = PAD_CEIL;
|
||||||
hparams.image_resize_algo = RESIZE_ALGO_BILINEAR;
|
hparams.image_resize_algo = RESIZE_ALGO_BICUBIC;
|
||||||
} break;
|
} break;
|
||||||
case PROJECTOR_TYPE_MINICPMV:
|
case PROJECTOR_TYPE_MINICPMV:
|
||||||
{
|
{
|
||||||
@@ -1490,6 +1488,7 @@ struct clip_model_loader {
|
|||||||
case PROJECTOR_TYPE_IDEFICS3:
|
case PROJECTOR_TYPE_IDEFICS3:
|
||||||
{
|
{
|
||||||
// use default llava-uhd preprocessing params
|
// use default llava-uhd preprocessing params
|
||||||
|
hparams.image_resize_algo = RESIZE_ALGO_LANCZOS;
|
||||||
get_u32(KEY_PROJ_SCALE_FACTOR, hparams.n_merge, false);
|
get_u32(KEY_PROJ_SCALE_FACTOR, hparams.n_merge, false);
|
||||||
get_u32(KEY_PREPROC_IMAGE_SIZE, hparams.image_longest_edge, false);
|
get_u32(KEY_PREPROC_IMAGE_SIZE, hparams.image_longest_edge, false);
|
||||||
hparams.set_limit_image_tokens();
|
hparams.set_limit_image_tokens();
|
||||||
@@ -1516,7 +1515,7 @@ struct clip_model_loader {
|
|||||||
// ref: https://huggingface.co/mistral-community/pixtral-12b/blob/main/preprocessor_config.json
|
// ref: https://huggingface.co/mistral-community/pixtral-12b/blob/main/preprocessor_config.json
|
||||||
// TODO: verify the image_min_tokens
|
// TODO: verify the image_min_tokens
|
||||||
hparams.n_merge = 1; // the original pixtral does not use patch merging
|
hparams.n_merge = 1; // the original pixtral does not use patch merging
|
||||||
hparams.image_resize_algo = RESIZE_ALGO_BILINEAR;
|
hparams.image_resize_algo = RESIZE_ALGO_BICUBIC;
|
||||||
hparams.rope_theta = 10000.0f;
|
hparams.rope_theta = 10000.0f;
|
||||||
get_u32(KEY_SPATIAL_MERGE_SIZE, hparams.n_merge, false);
|
get_u32(KEY_SPATIAL_MERGE_SIZE, hparams.n_merge, false);
|
||||||
hparams.set_limit_image_tokens(8, 1024);
|
hparams.set_limit_image_tokens(8, 1024);
|
||||||
@@ -1544,7 +1543,7 @@ struct clip_model_loader {
|
|||||||
case PROJECTOR_TYPE_DOTS3NOTE_V:
|
case PROJECTOR_TYPE_DOTS3NOTE_V:
|
||||||
{
|
{
|
||||||
hparams.rope_theta = 10000.0f;
|
hparams.rope_theta = 10000.0f;
|
||||||
hparams.image_resize_algo = RESIZE_ALGO_BICUBIC_PILLOW;
|
hparams.image_resize_algo = RESIZE_ALGO_BICUBIC;
|
||||||
get_u32(KEY_SPATIAL_MERGE_SIZE, hparams.n_merge);
|
get_u32(KEY_SPATIAL_MERGE_SIZE, hparams.n_merge);
|
||||||
get_u32(KEY_IMAGE_MIN_PIXELS, hparams.image_min_pixels);
|
get_u32(KEY_IMAGE_MIN_PIXELS, hparams.image_min_pixels);
|
||||||
get_u32(KEY_IMAGE_MAX_PIXELS, hparams.image_max_pixels);
|
get_u32(KEY_IMAGE_MAX_PIXELS, hparams.image_max_pixels);
|
||||||
@@ -1562,7 +1561,7 @@ struct clip_model_loader {
|
|||||||
} break;
|
} break;
|
||||||
case PROJECTOR_TYPE_KIMIVL:
|
case PROJECTOR_TYPE_KIMIVL:
|
||||||
{
|
{
|
||||||
hparams.image_resize_algo = RESIZE_ALGO_BILINEAR;
|
hparams.image_resize_algo = RESIZE_ALGO_BICUBIC;
|
||||||
hparams.rope_theta = 10000.0f;
|
hparams.rope_theta = 10000.0f;
|
||||||
get_u32(KEY_PROJ_SCALE_FACTOR, hparams.n_merge, false);
|
get_u32(KEY_PROJ_SCALE_FACTOR, hparams.n_merge, false);
|
||||||
// TODO: check kimivl preprocessor for exact values
|
// TODO: check kimivl preprocessor for exact values
|
||||||
@@ -1601,7 +1600,7 @@ struct clip_model_loader {
|
|||||||
{
|
{
|
||||||
hparams.rope_theta = 100.0f;
|
hparams.rope_theta = 100.0f;
|
||||||
hparams.n_merge = 3; // pooling_kernel_size
|
hparams.n_merge = 3; // pooling_kernel_size
|
||||||
hparams.image_resize_algo = RESIZE_ALGO_BILINEAR;
|
hparams.image_resize_algo = RESIZE_ALGO_BICUBIC;
|
||||||
get_u32(KEY_PROJ_SCALE_FACTOR, hparams.n_merge, false);
|
get_u32(KEY_PROJ_SCALE_FACTOR, hparams.n_merge, false);
|
||||||
if (model.proj_type == PROJECTOR_TYPE_GEMMA4UV) {
|
if (model.proj_type == PROJECTOR_TYPE_GEMMA4UV) {
|
||||||
// for "unified" variant, we directly use a bigger patch size, because the "token merging" is done directly on conv layer
|
// for "unified" variant, we directly use a bigger patch size, because the "token merging" is done directly on conv layer
|
||||||
@@ -1618,6 +1617,7 @@ struct clip_model_loader {
|
|||||||
// Gemma3n uses MobileNetV5 which produces 256 tokens (16x16)
|
// Gemma3n uses MobileNetV5 which produces 256 tokens (16x16)
|
||||||
// Similar configuration to Gemma3
|
// Similar configuration to Gemma3
|
||||||
hparams.n_merge = 1; // MobileNetV5 handles resizing internally
|
hparams.n_merge = 1; // MobileNetV5 handles resizing internally
|
||||||
|
hparams.image_resize_algo = RESIZE_ALGO_BILINEAR;
|
||||||
get_u32(KEY_PROJ_SCALE_FACTOR, hparams.n_merge, false);
|
get_u32(KEY_PROJ_SCALE_FACTOR, hparams.n_merge, false);
|
||||||
} break;
|
} break;
|
||||||
case PROJECTOR_TYPE_QWEN2VL:
|
case PROJECTOR_TYPE_QWEN2VL:
|
||||||
@@ -1625,7 +1625,7 @@ struct clip_model_loader {
|
|||||||
case PROJECTOR_TYPE_QWEN3VL:
|
case PROJECTOR_TYPE_QWEN3VL:
|
||||||
{
|
{
|
||||||
hparams.n_merge = 2; // default value for Qwen 2 and 2.5
|
hparams.n_merge = 2; // default value for Qwen 2 and 2.5
|
||||||
hparams.image_resize_algo = RESIZE_ALGO_BILINEAR;
|
hparams.image_resize_algo = RESIZE_ALGO_BICUBIC;
|
||||||
get_u32(KEY_SPATIAL_MERGE_SIZE, hparams.n_merge, false);
|
get_u32(KEY_SPATIAL_MERGE_SIZE, hparams.n_merge, false);
|
||||||
get_u32(KEY_WIN_ATTN_PATTERN, hparams.n_wa_pattern, model.proj_type == PROJECTOR_TYPE_QWEN25VL); // only 2.5 requires it
|
get_u32(KEY_WIN_ATTN_PATTERN, hparams.n_wa_pattern, model.proj_type == PROJECTOR_TYPE_QWEN25VL); // only 2.5 requires it
|
||||||
// ref: https://huggingface.co/Qwen/Qwen2.5-VL-7B-Instruct/blob/main/preprocessor_config.json
|
// ref: https://huggingface.co/Qwen/Qwen2.5-VL-7B-Instruct/blob/main/preprocessor_config.json
|
||||||
@@ -1641,7 +1641,7 @@ struct clip_model_loader {
|
|||||||
case PROJECTOR_TYPE_MINIMAX_M3:
|
case PROJECTOR_TYPE_MINIMAX_M3:
|
||||||
{
|
{
|
||||||
hparams.n_merge = 2; // spatial_merge_size
|
hparams.n_merge = 2; // spatial_merge_size
|
||||||
hparams.image_resize_algo = RESIZE_ALGO_BICUBIC_PILLOW;
|
hparams.image_resize_algo = RESIZE_ALGO_BICUBIC;
|
||||||
hparams.image_resize_pad = PAD_NONE;
|
hparams.image_resize_pad = PAD_NONE;
|
||||||
get_u32(KEY_SPATIAL_MERGE_SIZE, hparams.n_merge, false);
|
get_u32(KEY_SPATIAL_MERGE_SIZE, hparams.n_merge, false);
|
||||||
// n_merge is used as a divisor in clip_image_batch_encode
|
// n_merge is used as a divisor in clip_image_batch_encode
|
||||||
@@ -1666,7 +1666,7 @@ struct clip_model_loader {
|
|||||||
case PROJECTOR_TYPE_MIMOVL:
|
case PROJECTOR_TYPE_MIMOVL:
|
||||||
{
|
{
|
||||||
hparams.n_merge = 2; // spatial_merge_size
|
hparams.n_merge = 2; // spatial_merge_size
|
||||||
hparams.image_resize_algo = RESIZE_ALGO_BICUBIC_PILLOW;
|
hparams.image_resize_algo = RESIZE_ALGO_BICUBIC;
|
||||||
get_u32(KEY_SPATIAL_MERGE_SIZE, hparams.n_merge, false);
|
get_u32(KEY_SPATIAL_MERGE_SIZE, hparams.n_merge, false);
|
||||||
get_u32(string_format(KEY_N_HEAD_KV, "vision"), hparams.n_head_kv);
|
get_u32(string_format(KEY_N_HEAD_KV, "vision"), hparams.n_head_kv);
|
||||||
// 1D banded sliding-window radius (visual_token_window_size); required
|
// 1D banded sliding-window radius (visual_token_window_size); required
|
||||||
@@ -1713,15 +1713,15 @@ struct clip_model_loader {
|
|||||||
log_ffn_op = "gelu_erf";
|
log_ffn_op = "gelu_erf";
|
||||||
hparams.image_resize_algo = RESIZE_ALGO_BICUBIC;
|
hparams.image_resize_algo = RESIZE_ALGO_BICUBIC;
|
||||||
|
|
||||||
// reka model performs better when using resize_bicubic, which stretches
|
// reka model performs better when the image is stretched to fit
|
||||||
// the image to fit fixed square size
|
// fixed square size (no padding)
|
||||||
hparams.image_resize_pad = PAD_NONE;
|
hparams.image_resize_pad = PAD_NONE;
|
||||||
} break;
|
} break;
|
||||||
case PROJECTOR_TYPE_GLM4V:
|
case PROJECTOR_TYPE_GLM4V:
|
||||||
{
|
{
|
||||||
hparams.rope_theta = 10000.0f;
|
hparams.rope_theta = 10000.0f;
|
||||||
hparams.n_merge = 2; // default value for GLM4-V
|
hparams.n_merge = 2; // default value for GLM4-V
|
||||||
hparams.image_resize_algo = RESIZE_ALGO_BILINEAR;
|
hparams.image_resize_algo = RESIZE_ALGO_BICUBIC;
|
||||||
get_u32(KEY_SPATIAL_MERGE_SIZE, hparams.n_merge, false);
|
get_u32(KEY_SPATIAL_MERGE_SIZE, hparams.n_merge, false);
|
||||||
hparams.set_limit_image_tokens(8, 4096);
|
hparams.set_limit_image_tokens(8, 4096);
|
||||||
hparams.set_warmup_n_tokens(46*46); // avoid OOM on warmup
|
hparams.set_warmup_n_tokens(46*46); // avoid OOM on warmup
|
||||||
@@ -1729,6 +1729,7 @@ struct clip_model_loader {
|
|||||||
case PROJECTOR_TYPE_LLAMA4:
|
case PROJECTOR_TYPE_LLAMA4:
|
||||||
{
|
{
|
||||||
hparams.rope_theta = 10000.0f;
|
hparams.rope_theta = 10000.0f;
|
||||||
|
hparams.image_resize_algo = RESIZE_ALGO_BILINEAR;
|
||||||
get_u32(KEY_PROJ_SCALE_FACTOR, hparams.n_merge, false);
|
get_u32(KEY_PROJ_SCALE_FACTOR, hparams.n_merge, false);
|
||||||
set_llava_uhd_res_candidates(model, 3);
|
set_llava_uhd_res_candidates(model, 3);
|
||||||
} break;
|
} break;
|
||||||
@@ -1840,7 +1841,7 @@ struct clip_model_loader {
|
|||||||
case PROJECTOR_TYPE_PADDLEOCR:
|
case PROJECTOR_TYPE_PADDLEOCR:
|
||||||
{
|
{
|
||||||
hparams.n_merge = 2;
|
hparams.n_merge = 2;
|
||||||
hparams.image_resize_algo = RESIZE_ALGO_BILINEAR;
|
hparams.image_resize_algo = RESIZE_ALGO_BICUBIC;
|
||||||
get_u32(KEY_IMAGE_MIN_PIXELS, hparams.image_min_pixels);
|
get_u32(KEY_IMAGE_MIN_PIXELS, hparams.image_min_pixels);
|
||||||
get_u32(KEY_IMAGE_MAX_PIXELS, hparams.image_max_pixels);
|
get_u32(KEY_IMAGE_MAX_PIXELS, hparams.image_max_pixels);
|
||||||
|
|
||||||
@@ -1852,7 +1853,7 @@ struct clip_model_loader {
|
|||||||
hparams.patch_size = 16;
|
hparams.patch_size = 16;
|
||||||
hparams.image_size = 1024;
|
hparams.image_size = 1024;
|
||||||
hparams.warmup_image_size = 1024;
|
hparams.warmup_image_size = 1024;
|
||||||
hparams.image_resize_algo = RESIZE_ALGO_BICUBIC_PILLOW;
|
hparams.image_resize_algo = RESIZE_ALGO_BICUBIC;
|
||||||
hparams.image_pad_color = {127, 127, 127};
|
hparams.image_pad_color = {127, 127, 127};
|
||||||
|
|
||||||
get_u32(KEY_SAM_N_BLOCK, hparams.sam_n_layer, true);
|
get_u32(KEY_SAM_N_BLOCK, hparams.sam_n_layer, true);
|
||||||
@@ -1882,7 +1883,7 @@ struct clip_model_loader {
|
|||||||
case PROJECTOR_TYPE_HUNYUANVL:
|
case PROJECTOR_TYPE_HUNYUANVL:
|
||||||
{
|
{
|
||||||
hparams.n_merge = 2;
|
hparams.n_merge = 2;
|
||||||
hparams.image_resize_algo = RESIZE_ALGO_BICUBIC_PILLOW;
|
hparams.image_resize_algo = RESIZE_ALGO_LANCZOS;
|
||||||
hparams.image_resize_pad = PAD_NONE;
|
hparams.image_resize_pad = PAD_NONE;
|
||||||
hparams.ffn_op = FFN_GELU;
|
hparams.ffn_op = FFN_GELU;
|
||||||
hparams.set_limit_image_tokens(256, 16384);
|
hparams.set_limit_image_tokens(256, 16384);
|
||||||
@@ -1955,12 +1956,12 @@ struct clip_model_loader {
|
|||||||
case PROJECTOR_TYPE_JANUS_PRO:
|
case PROJECTOR_TYPE_JANUS_PRO:
|
||||||
{
|
{
|
||||||
hparams.image_pad_color = {127, 127, 127};
|
hparams.image_pad_color = {127, 127, 127};
|
||||||
hparams.image_resize_algo = RESIZE_ALGO_BILINEAR;
|
hparams.image_resize_algo = RESIZE_ALGO_BICUBIC;
|
||||||
} break;
|
} break;
|
||||||
case PROJECTOR_TYPE_GRANITE4_VISION:
|
case PROJECTOR_TYPE_GRANITE4_VISION:
|
||||||
{
|
{
|
||||||
// SigLIP tower.
|
// SigLIP tower.
|
||||||
hparams.image_resize_algo = RESIZE_ALGO_BICUBIC_PILLOW;
|
hparams.image_resize_algo = RESIZE_ALGO_BICUBIC;
|
||||||
hparams.image_resize_pad = PAD_CEIL;
|
hparams.image_resize_pad = PAD_CEIL;
|
||||||
|
|
||||||
// NOTE: feature_layers loaded in common path as optional
|
// NOTE: feature_layers loaded in common path as optional
|
||||||
|
|||||||
+76
-238
@@ -58,22 +58,7 @@ struct img_tool {
|
|||||||
|
|
||||||
if (padding == PAD_NONE) {
|
if (padding == PAD_NONE) {
|
||||||
// direct resize
|
// direct resize
|
||||||
switch (algo) {
|
resize_pillow(src, dst, target_resolution.width, target_resolution.height, algo);
|
||||||
case RESIZE_ALGO_BILINEAR:
|
|
||||||
resize_bilinear(src, dst, target_resolution.width, target_resolution.height);
|
|
||||||
break;
|
|
||||||
case RESIZE_ALGO_BICUBIC:
|
|
||||||
resize_bicubic(src, dst, target_resolution.width, target_resolution.height);
|
|
||||||
break;
|
|
||||||
case RESIZE_ALGO_BICUBIC_PILLOW:
|
|
||||||
resize_bicubic_pillow(src, dst, target_resolution.width, target_resolution.height);
|
|
||||||
break;
|
|
||||||
case RESIZE_ALGO_LANCZOS:
|
|
||||||
resize_lanczos_pillow(src, dst, target_resolution.width, target_resolution.height);
|
|
||||||
break;
|
|
||||||
default:
|
|
||||||
throw std::runtime_error("Unsupported resize algorithm");
|
|
||||||
}
|
|
||||||
} else {
|
} else {
|
||||||
// resize with padding
|
// resize with padding
|
||||||
clip_image_u8 resized_image;
|
clip_image_u8 resized_image;
|
||||||
@@ -90,22 +75,7 @@ struct img_tool {
|
|||||||
new_height = std::min(static_cast<int>(std::ceil(src.get_size().height * scale)), target_resolution.height);
|
new_height = std::min(static_cast<int>(std::ceil(src.get_size().height * scale)), target_resolution.height);
|
||||||
}
|
}
|
||||||
|
|
||||||
switch (algo) {
|
resize_pillow(src, resized_image, new_width, new_height, algo);
|
||||||
case RESIZE_ALGO_BILINEAR:
|
|
||||||
resize_bilinear(src, resized_image, new_width, new_height);
|
|
||||||
break;
|
|
||||||
case RESIZE_ALGO_BICUBIC:
|
|
||||||
resize_bicubic(src, resized_image, new_width, new_height);
|
|
||||||
break;
|
|
||||||
case RESIZE_ALGO_BICUBIC_PILLOW:
|
|
||||||
resize_bicubic_pillow(src, resized_image, new_width, new_height);
|
|
||||||
break;
|
|
||||||
case RESIZE_ALGO_LANCZOS:
|
|
||||||
resize_lanczos_pillow(src, resized_image, new_width, new_height);
|
|
||||||
break;
|
|
||||||
default:
|
|
||||||
throw std::runtime_error("Unsupported resize algorithm");
|
|
||||||
}
|
|
||||||
|
|
||||||
// fill dst with pad_color
|
// fill dst with pad_color
|
||||||
fill(dst, pad_color);
|
fill(dst, pad_color);
|
||||||
@@ -224,152 +194,37 @@ struct img_tool {
|
|||||||
}
|
}
|
||||||
|
|
||||||
private:
|
private:
|
||||||
// Bilinear resize function
|
// Pillow-compatible separable resampling (Bilinear, Bicubic and Lanczos)
|
||||||
static void resize_bilinear(const clip_image_u8 & src, clip_image_u8 & dst, int target_width, int target_height) {
|
|
||||||
const auto src_size = src.get_size();
|
|
||||||
if (src_size.width == 0 || src_size.height == 0) { dst.set_size({0, 0}, false); return; }
|
|
||||||
if (target_width <= 0) target_width = 1;
|
|
||||||
if (target_height <= 0) target_height = 1;
|
|
||||||
|
|
||||||
dst.set_size({target_width, target_height}, false);
|
|
||||||
|
|
||||||
if (src.is_placeholder()) {
|
|
||||||
// no-op for placeholder image, just set the size and return
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
|
|
||||||
float x_ratio = target_width > 1 ? static_cast<float>(src_size.width - 1) / (target_width - 1) : 0.0f;
|
|
||||||
float y_ratio = target_height > 1 ? static_cast<float>(src_size.height - 1) / (target_height - 1) : 0.0f;
|
|
||||||
|
|
||||||
for (int y = 0; y < target_height; ++y) {
|
|
||||||
for (int x = 0; x < target_width; ++x) {
|
|
||||||
float px = x * x_ratio;
|
|
||||||
float py = y * y_ratio;
|
|
||||||
|
|
||||||
int x0 = std::min(static_cast<int>(px), src_size.width - 1);
|
|
||||||
int y0 = std::min(static_cast<int>(py), src_size.height - 1);
|
|
||||||
int x1 = std::min(x0 + 1, src_size.width - 1);
|
|
||||||
int y1 = std::min(y0 + 1, src_size.height - 1);
|
|
||||||
|
|
||||||
float xf = px - x0;
|
|
||||||
float yf = py - y0;
|
|
||||||
|
|
||||||
const auto p00 = src.get_pixel(x0, y0);
|
|
||||||
const auto p10 = src.get_pixel(x1, y0);
|
|
||||||
const auto p01 = src.get_pixel(x0, y1);
|
|
||||||
const auto p11 = src.get_pixel(x1, y1);
|
|
||||||
|
|
||||||
std::array<uint8_t, 3> pixel;
|
|
||||||
for (int c = 0; c < 3; ++c) {
|
|
||||||
float top = lerp(static_cast<float>(p00[c]), static_cast<float>(p10[c]), xf);
|
|
||||||
float bottom = lerp(static_cast<float>(p01[c]), static_cast<float>(p11[c]), xf);
|
|
||||||
pixel[c] = static_cast<uint8_t>(lerp(top, bottom, yf));
|
|
||||||
}
|
|
||||||
dst.set_pixel(x, y, pixel);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// Bicubic resize function
|
|
||||||
// part of image will be cropped if the aspect ratio is different
|
|
||||||
static void resize_bicubic(const clip_image_u8 & img, clip_image_u8 & dst, int target_width, int target_height) {
|
|
||||||
const auto img_size = img.get_size();
|
|
||||||
const int nx = img_size.width;
|
|
||||||
const int ny = img_size.height;
|
|
||||||
|
|
||||||
dst.set_size({target_width, target_height}, false);
|
|
||||||
|
|
||||||
if (img.is_placeholder()) {
|
|
||||||
// no-op for placeholder image, just set the size and return
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
|
|
||||||
float Cc;
|
|
||||||
float C[5] = {};
|
|
||||||
float d0, d2, d3, a0, a1, a2, a3;
|
|
||||||
int i, j, k, jj;
|
|
||||||
int x, y;
|
|
||||||
float dx, dy;
|
|
||||||
float tx, ty;
|
|
||||||
|
|
||||||
tx = (float)nx / (float)target_width;
|
|
||||||
ty = (float)ny / (float)target_height;
|
|
||||||
|
|
||||||
// Bicubic interpolation; adapted from ViT.cpp, inspired from :
|
|
||||||
// -> https://github.com/yglukhov/bicubic-interpolation-image-processing/blob/master/libimage.c#L36
|
|
||||||
// -> https://en.wikipedia.org/wiki/Bicubic_interpolation
|
|
||||||
|
|
||||||
for (i = 0; i < target_height; i++) {
|
|
||||||
for (j = 0; j < target_width; j++) {
|
|
||||||
x = (int)(tx * j);
|
|
||||||
y = (int)(ty * i);
|
|
||||||
|
|
||||||
dx = tx * j - x;
|
|
||||||
dy = ty * i - y;
|
|
||||||
|
|
||||||
std::array<uint8_t, 3> pixel;
|
|
||||||
for (k = 0; k < 3; k++) {
|
|
||||||
for (jj = 0; jj <= 3; jj++) {
|
|
||||||
d0 = img.get_pixel(clip(x - 1, 0, nx - 1), clip(y - 1 + jj, 0, ny - 1))[k] - img.get_pixel(clip(x, 0, nx - 1), clip(y - 1 + jj, 0, ny - 1))[k];
|
|
||||||
d2 = img.get_pixel(clip(x + 1, 0, nx - 1), clip(y - 1 + jj, 0, ny - 1))[k] - img.get_pixel(clip(x, 0, nx - 1), clip(y - 1 + jj, 0, ny - 1))[k];
|
|
||||||
d3 = img.get_pixel(clip(x + 2, 0, nx - 1), clip(y - 1 + jj, 0, ny - 1))[k] - img.get_pixel(clip(x, 0, nx - 1), clip(y - 1 + jj, 0, ny - 1))[k];
|
|
||||||
a0 = img.get_pixel(clip(x, 0, nx - 1), clip(y - 1 + jj, 0, ny - 1))[k];
|
|
||||||
|
|
||||||
a1 = -1.0 / 3 * d0 + d2 - 1.0 / 6 * d3;
|
|
||||||
a2 = 1.0 / 2 * d0 + 1.0 / 2 * d2;
|
|
||||||
a3 = -1.0 / 6 * d0 - 1.0 / 2 * d2 + 1.0 / 6 * d3;
|
|
||||||
|
|
||||||
C[jj] = a0 + a1 * dx + a2 * dx * dx + a3 * dx * dx * dx;
|
|
||||||
|
|
||||||
d0 = C[0] - C[1];
|
|
||||||
d2 = C[2] - C[1];
|
|
||||||
d3 = C[3] - C[1];
|
|
||||||
a0 = C[1];
|
|
||||||
a1 = -1.0 / 3 * d0 + d2 - 1.0 / 6 * d3;
|
|
||||||
a2 = 1.0 / 2 * d0 + 1.0 / 2 * d2;
|
|
||||||
a3 = -1.0 / 6 * d0 - 1.0 / 2 * d2 + 1.0 / 6 * d3;
|
|
||||||
Cc = a0 + a1 * dy + a2 * dy * dy + a3 * dy * dy * dy;
|
|
||||||
|
|
||||||
const uint8_t Cc2 = std::min(std::max(std::round(Cc), 0.0f), 255.0f);
|
|
||||||
pixel[k] = Cc2;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
dst.set_pixel(j, i, pixel);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// Pillow-compatible separable resampling (Bicubic and Lanczos)
|
|
||||||
// Adapted from https://github.com/python-pillow/Pillow/blob/main/src/libImaging/Resample.c
|
// Adapted from https://github.com/python-pillow/Pillow/blob/main/src/libImaging/Resample.c
|
||||||
//
|
//
|
||||||
// Key properties:
|
// Key properties:
|
||||||
// 1. Separable filtering: horizontal pass followed by vertical pass
|
// 1. Separable filtering: horizontal pass followed by vertical pass
|
||||||
// 2. Pre-computes normalized filter coefficients for each output pixel
|
// 2. Pre-computes normalized filter coefficients for each output pixel
|
||||||
// 3. Fixed-point integer arithmetic (22 fractional bits) for speed and determinism
|
// 3. Fixed-point integer arithmetic (22 fractional bits) for speed and determinism
|
||||||
static bool resize_bicubic_pillow(const clip_image_u8 & img, clip_image_u8 & dst, int target_width, int target_height) {
|
|
||||||
return resize_pillow(img, dst, target_width, target_height, /*use_lanczos=*/false);
|
|
||||||
}
|
|
||||||
|
|
||||||
// Lanczos-3 (support radius 3), matches Pillow's Image.LANCZOS
|
|
||||||
static bool resize_lanczos_pillow(const clip_image_u8 & img, clip_image_u8 & dst, int target_width, int target_height) {
|
|
||||||
return resize_pillow(img, dst, target_width, target_height, /*use_lanczos=*/true);
|
|
||||||
}
|
|
||||||
|
|
||||||
static bool resize_pillow(
|
static bool resize_pillow(
|
||||||
const clip_image_u8 & img,
|
const clip_image_u8 & img,
|
||||||
clip_image_u8 & dst,
|
clip_image_u8 & dst,
|
||||||
int target_width,
|
int target_width,
|
||||||
int target_height,
|
int target_height,
|
||||||
bool use_lanczos) {
|
resize_algo algo) {
|
||||||
// Fixed-point precision: 22 bits = 32 (int32_t) - 8 (uint8_t pixels) - 2 (headroom for accumulation)
|
// Fixed-point precision: 22 bits = 32 (int32_t) - 8 (uint8_t pixels) - 2 (headroom for accumulation)
|
||||||
// This allows encoding fractional weights as integers: weight * 2^22
|
// This allows encoding fractional weights as integers: weight * 2^22
|
||||||
const int PRECISION_BITS = 32 - 8 - 2;
|
const int PRECISION_BITS = 32 - 8 - 2;
|
||||||
|
|
||||||
// Resample filter: Lanczos-3 (support [-3, 3]) or bicubic with a = -0.5 (support [-2, 2])
|
// Filter support radius
|
||||||
// Note: GGML/PyTorch bicubic uses a = -0.75, Pillow uses a = -0.5
|
double filter_support;
|
||||||
|
switch (algo) {
|
||||||
|
case RESIZE_ALGO_BILINEAR: filter_support = 1.0; break;
|
||||||
|
case RESIZE_ALGO_BICUBIC: filter_support = 2.0; break;
|
||||||
|
case RESIZE_ALGO_LANCZOS: filter_support = 3.0; break;
|
||||||
|
default:
|
||||||
|
throw std::runtime_error("Unsupported resize algorithm");
|
||||||
|
}
|
||||||
|
|
||||||
// Returns filter weight for distance x from pixel center
|
// Returns filter weight for distance x from pixel center
|
||||||
auto resample_filter = [use_lanczos](double x) -> double {
|
// Note: for bicubic, Pillow uses a = -0.5 while GGML/PyTorch use a = -0.75
|
||||||
if (use_lanczos) {
|
auto resample_filter = [algo](double x) -> double {
|
||||||
|
if (algo == RESIZE_ALGO_LANCZOS) {
|
||||||
if (-3.0 <= x && x < 3.0) {
|
if (-3.0 <= x && x < 3.0) {
|
||||||
auto sinc = [](double v) {
|
auto sinc = [](double v) {
|
||||||
if (v == 0.0) {
|
if (v == 0.0) {
|
||||||
@@ -383,10 +238,15 @@ private:
|
|||||||
return 0.0;
|
return 0.0;
|
||||||
}
|
}
|
||||||
|
|
||||||
constexpr double a = -0.5;
|
|
||||||
if (x < 0.0) {
|
if (x < 0.0) {
|
||||||
x = -x;
|
x = -x;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
if (algo == RESIZE_ALGO_BILINEAR) {
|
||||||
|
return x < 1.0 ? 1.0 - x : 0.0;
|
||||||
|
}
|
||||||
|
|
||||||
|
constexpr double a = -0.5;
|
||||||
if (x < 1.0) {
|
if (x < 1.0) {
|
||||||
return ((a + 2.0) * x - (a + 3.0)) * x * x + 1;
|
return ((a + 2.0) * x - (a + 3.0)) * x * x + 1;
|
||||||
}
|
}
|
||||||
@@ -396,9 +256,6 @@ private:
|
|||||||
return 0.0; // Zero outside [-2, 2]
|
return 0.0; // Zero outside [-2, 2]
|
||||||
};
|
};
|
||||||
|
|
||||||
// Filter support radius: 2 for bicubic, 3 for lanczos
|
|
||||||
const double filter_support = use_lanczos ? 3.0 : 2.0;
|
|
||||||
|
|
||||||
// Clipping function for 8-bit values
|
// Clipping function for 8-bit values
|
||||||
auto clip8 = [](int val) -> uint8_t {
|
auto clip8 = [](int val) -> uint8_t {
|
||||||
if (val < 0) return 0;
|
if (val < 0) return 0;
|
||||||
@@ -493,100 +350,92 @@ private:
|
|||||||
const double fxp_scale = std::ldexp(1.0, PRECISION_BITS); // 1.0 * 2^PRECISION_BITS
|
const double fxp_scale = std::ldexp(1.0, PRECISION_BITS); // 1.0 * 2^PRECISION_BITS
|
||||||
|
|
||||||
for (int i = 0; i < outSize * ksize; i++) {
|
for (int i = 0; i < outSize * ksize; i++) {
|
||||||
if (use_lanczos) {
|
|
||||||
// Pillow adds +/- 0.5 then truncates toward zero; std::round would round twice
|
// Pillow adds +/- 0.5 then truncates toward zero; std::round would round twice
|
||||||
const double rounded = pre_weights[i] * fxp_scale + (pre_weights[i] < 0 ? -0.5 : 0.5);
|
const double rounded = pre_weights[i] * fxp_scale + (pre_weights[i] < 0 ? -0.5 : 0.5);
|
||||||
weights[i] = static_cast<int32_t>(rounded);
|
weights[i] = static_cast<int32_t>(rounded);
|
||||||
continue;
|
|
||||||
}
|
|
||||||
double tmp_val = pre_weights[i] * fxp_scale;
|
|
||||||
if (pre_weights[i] < 0) {
|
|
||||||
tmp_val -= 0.5;
|
|
||||||
} else {
|
|
||||||
tmp_val += 0.5;
|
|
||||||
}
|
|
||||||
tmp_val = std::round(tmp_val);
|
|
||||||
tmp_val = std::clamp(tmp_val,
|
|
||||||
static_cast<double>(std::numeric_limits<int32_t>::min()),
|
|
||||||
static_cast<double>(std::numeric_limits<int32_t>::max()));
|
|
||||||
weights[i] = static_cast<int32_t>(tmp_val);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
return ksize;
|
return ksize;
|
||||||
};
|
};
|
||||||
|
|
||||||
// Horizontal resampling pass
|
// Horizontal resampling pass
|
||||||
// Resizes width from imIn to out_nx, preserving height
|
// Resizes width from src to out_nx, preserving height
|
||||||
auto resample_horizontal = [&](const clip_image_u8 & imIn, clip_image_u8 & imOut,
|
auto resample_horizontal = [&](const uint8_t * src, int in_nx, int in_ny,
|
||||||
int out_nx,
|
int out_nx,
|
||||||
int ksize, const std::vector<int> & bounds, const std::vector<int32_t> & weights) {
|
int ksize, const std::vector<int> & bounds, const std::vector<int32_t> & weights) {
|
||||||
const int in_ny = imIn.get_size().height;
|
std::vector<uint8_t> out((size_t) out_nx * in_ny * 3);
|
||||||
imOut.set_size({out_nx, in_ny}, false);
|
|
||||||
|
|
||||||
// Process each row independently
|
// Process each row independently
|
||||||
for (int yy = 0; yy < in_ny; yy++) {
|
for (int yy = 0; yy < in_ny; yy++) {
|
||||||
|
const uint8_t * src_row = src + (size_t) yy * in_nx * 3;
|
||||||
|
uint8_t * dst_row = out.data() + (size_t) yy * out_nx * 3;
|
||||||
|
|
||||||
// For each output pixel in this row
|
// For each output pixel in this row
|
||||||
for (int xx = 0; xx < out_nx; xx++) {
|
for (int xx = 0; xx < out_nx; xx++) {
|
||||||
// Get the range of input pixels and filter coefficients
|
const int xmin = bounds[xx * 2 + 0]; // First input pixel index
|
||||||
int xmin = bounds[xx * 2 + 0]; // First input pixel index
|
const int xcnt = bounds[xx * 2 + 1]; // Number of input pixels
|
||||||
int xcnt = bounds[xx * 2 + 1]; // Number of input pixels
|
const int32_t * k = &weights[xx * ksize];
|
||||||
|
const uint8_t * p = src_row + (size_t) xmin * 3;
|
||||||
|
|
||||||
// Initialize accumulators for RGB channels with rounding bias (0.5 in fixed-point)
|
// Accumulators for RGB channels, with rounding bias (0.5 in fixed-point)
|
||||||
int32_t ss0 = 1 << (PRECISION_BITS - 1);
|
int32_t ss0 = 1 << (PRECISION_BITS - 1);
|
||||||
int32_t ss1 = 1 << (PRECISION_BITS - 1);
|
int32_t ss1 = 1 << (PRECISION_BITS - 1);
|
||||||
int32_t ss2 = 1 << (PRECISION_BITS - 1);
|
int32_t ss2 = 1 << (PRECISION_BITS - 1);
|
||||||
|
|
||||||
// Convolve: sum weighted input pixels
|
// Convolve: sum weighted input pixels
|
||||||
for (int x = 0; x < xcnt; x++) {
|
for (int x = 0; x < xcnt; x++) {
|
||||||
const auto src_px = imIn.get_pixel(x + xmin, yy);
|
ss0 += p[0] * k[x];
|
||||||
ss0 += src_px[0] * weights[xx * ksize + x]; // R channel
|
ss1 += p[1] * k[x];
|
||||||
ss1 += src_px[1] * weights[xx * ksize + x]; // G channel
|
ss2 += p[2] * k[x];
|
||||||
ss2 += src_px[2] * weights[xx * ksize + x]; // B channel
|
p += 3;
|
||||||
}
|
}
|
||||||
|
|
||||||
// Convert back from fixed-point (divide by 2^PRECISION_BITS) and clamp to [0,255]
|
// Convert back from fixed-point (divide by 2^PRECISION_BITS) and clamp to [0,255]
|
||||||
imOut.set_pixel(xx, yy, {clip8(ss0 >> PRECISION_BITS),
|
dst_row[xx * 3 + 0] = clip8(ss0 >> PRECISION_BITS);
|
||||||
clip8(ss1 >> PRECISION_BITS),
|
dst_row[xx * 3 + 1] = clip8(ss1 >> PRECISION_BITS);
|
||||||
clip8(ss2 >> PRECISION_BITS)});
|
dst_row[xx * 3 + 2] = clip8(ss2 >> PRECISION_BITS);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
return out;
|
||||||
};
|
};
|
||||||
|
|
||||||
// Vertical resampling pass
|
// Vertical resampling pass
|
||||||
// Resizes height from imIn to out_ny, preserving width
|
// Resizes height from src to out_ny, preserving width
|
||||||
auto resample_vertical = [&](const clip_image_u8 & imIn, clip_image_u8 & imOut,
|
// Accumulates whole rows at once (contiguous access, auto-vectorizes well)
|
||||||
|
auto resample_vertical = [&](const uint8_t * src, int in_nx,
|
||||||
int out_ny,
|
int out_ny,
|
||||||
int ksize, const std::vector<int> & bounds, const std::vector<int32_t> & weight) {
|
int ksize, const std::vector<int> & bounds, const std::vector<int32_t> & weight) {
|
||||||
const int in_nx = imIn.get_size().width;
|
const size_t row_elems = (size_t) in_nx * 3;
|
||||||
imOut.set_size({in_nx, out_ny}, false);
|
std::vector<uint8_t> out(row_elems * out_ny);
|
||||||
|
std::vector<int32_t> acc(row_elems);
|
||||||
|
|
||||||
// For each output row
|
// For each output row
|
||||||
for (int yy = 0; yy < out_ny; yy++) {
|
for (int yy = 0; yy < out_ny; yy++) {
|
||||||
// Get the range of input rows and filter coefficients
|
const int ymin = bounds[yy * 2 + 0]; // First input row index
|
||||||
int ymin = bounds[yy * 2 + 0]; // First input row index
|
const int ycnt = bounds[yy * 2 + 1]; // Number of input rows
|
||||||
int ycnt = bounds[yy * 2 + 1]; // Number of input rows
|
const int32_t * k = &weight[yy * ksize];
|
||||||
|
|
||||||
// Process each column in this output row
|
// Rounding bias (0.5 in fixed-point)
|
||||||
for (int xx = 0; xx < in_nx; xx++) {
|
std::fill(acc.begin(), acc.end(), 1 << (PRECISION_BITS - 1));
|
||||||
// Initialize accumulators for RGB channels with rounding bias
|
|
||||||
int32_t ss0 = 1 << (PRECISION_BITS - 1);
|
|
||||||
int32_t ss1 = 1 << (PRECISION_BITS - 1);
|
|
||||||
int32_t ss2 = 1 << (PRECISION_BITS - 1);
|
|
||||||
|
|
||||||
// Convolve: sum weighted input pixels vertically
|
// Convolve: accumulate each weighted input row
|
||||||
for (int y = 0; y < ycnt; y++) {
|
for (int y = 0; y < ycnt; y++) {
|
||||||
const auto src_px = imIn.get_pixel(xx, y + ymin);
|
const uint8_t * src_row = src + (size_t) (ymin + y) * row_elems;
|
||||||
ss0 += src_px[0] * weight[yy * ksize + y]; // R channel
|
const int32_t w = k[y];
|
||||||
ss1 += src_px[1] * weight[yy * ksize + y]; // G channel
|
for (size_t i = 0; i < row_elems; i++) {
|
||||||
ss2 += src_px[2] * weight[yy * ksize + y]; // B channel
|
acc[i] += src_row[i] * w;
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// Convert back from fixed-point and clamp to [0,255]
|
// Convert back from fixed-point and clamp to [0,255]
|
||||||
imOut.set_pixel(xx, yy, {clip8(ss0 >> PRECISION_BITS),
|
uint8_t * dst_row = out.data() + (size_t) yy * row_elems;
|
||||||
clip8(ss1 >> PRECISION_BITS),
|
for (size_t i = 0; i < row_elems; i++) {
|
||||||
clip8(ss2 >> PRECISION_BITS)});
|
dst_row[i] = clip8(acc[i] >> PRECISION_BITS);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
return out;
|
||||||
};
|
};
|
||||||
|
|
||||||
// Main resampling logic using separable two-pass approach
|
// Main resampling logic using separable two-pass approach
|
||||||
@@ -610,36 +459,25 @@ private:
|
|||||||
}
|
}
|
||||||
|
|
||||||
// Perform two-pass resampling
|
// Perform two-pass resampling
|
||||||
|
const uint8_t * src = img.get_ro_buf().data();
|
||||||
if (need_horizontal && need_vertical) {
|
if (need_horizontal && need_vertical) {
|
||||||
// Both horizontal and vertical
|
auto temp = resample_horizontal(src, src_width, src_height, target_width, ksize_horiz, bounds_horiz, weights_horiz);
|
||||||
clip_image_u8 temp;
|
dst.set_size({target_width, target_height}, false);
|
||||||
resample_horizontal(img, temp, target_width, ksize_horiz, bounds_horiz, weights_horiz);
|
dst.cpy_buf(resample_vertical(temp.data(), target_width, target_height, ksize_vert, bounds_vert, weights_vert));
|
||||||
resample_vertical(temp, dst, target_height, ksize_vert, bounds_vert, weights_vert);
|
|
||||||
} else if (need_horizontal) {
|
} else if (need_horizontal) {
|
||||||
// Only horizontal
|
dst.set_size({target_width, src_height}, false);
|
||||||
resample_horizontal(img, dst, target_width, ksize_horiz, bounds_horiz, weights_horiz);
|
dst.cpy_buf(resample_horizontal(src, src_width, src_height, target_width, ksize_horiz, bounds_horiz, weights_horiz));
|
||||||
} else if (need_vertical) {
|
} else if (need_vertical) {
|
||||||
// Only vertical
|
dst.set_size({src_width, target_height}, false);
|
||||||
resample_vertical(img, dst, target_height, ksize_vert, bounds_vert, weights_vert);
|
dst.cpy_buf(resample_vertical(src, src_width, target_height, ksize_vert, bounds_vert, weights_vert));
|
||||||
} else {
|
} else {
|
||||||
// No resizing needed - direct copy
|
// No resizing needed - direct copy
|
||||||
dst.set_size(img.get_size(), img.is_placeholder());
|
dst.set_size(img.get_size(), false);
|
||||||
if (!img.is_placeholder()) {
|
|
||||||
dst.cpy_buf(img.get_ro_buf());
|
dst.cpy_buf(img.get_ro_buf());
|
||||||
}
|
}
|
||||||
}
|
|
||||||
|
|
||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
|
|
||||||
static inline int clip(int x, int lower, int upper) {
|
|
||||||
return std::max(lower, std::min(x, upper));
|
|
||||||
}
|
|
||||||
|
|
||||||
// Linear interpolation between two points
|
|
||||||
static inline float lerp(float s, float e, float t) {
|
|
||||||
return s + (e - s) * t;
|
|
||||||
}
|
|
||||||
};
|
};
|
||||||
|
|
||||||
|
|
||||||
@@ -1264,7 +1102,7 @@ mtmd_image_preproc_out mtmd_image_preprocessor_deepseekocr::preprocess(const cli
|
|||||||
clip_image_u8 padded;
|
clip_image_u8 padded;
|
||||||
img_tool::resize(img, padded,
|
img_tool::resize(img, padded,
|
||||||
{ base_size, base_size },
|
{ base_size, base_size },
|
||||||
RESIZE_ALGO_BICUBIC_PILLOW,
|
RESIZE_ALGO_BICUBIC,
|
||||||
PAD_NEAREST,
|
PAD_NEAREST,
|
||||||
hparams.image_pad_color);
|
hparams.image_pad_color);
|
||||||
output.append_overview(hparams, padded, true);
|
output.append_overview(hparams, padded, true);
|
||||||
@@ -1280,7 +1118,7 @@ mtmd_image_preproc_out mtmd_image_preprocessor_deepseekocr::preprocess(const cli
|
|||||||
grid_h = grid.height;
|
grid_h = grid.height;
|
||||||
|
|
||||||
clip_image_u8 refined;
|
clip_image_u8 refined;
|
||||||
img_tool::resize(img, refined, { tile_size * grid_w, tile_size * grid_h }, RESIZE_ALGO_BICUBIC_PILLOW,
|
img_tool::resize(img, refined, { tile_size * grid_w, tile_size * grid_h }, RESIZE_ALGO_BICUBIC,
|
||||||
PAD_NONE);
|
PAD_NONE);
|
||||||
|
|
||||||
for (int row = 0; row < grid_h; row++) {
|
for (int row = 0; row < grid_h; row++) {
|
||||||
|
|||||||
Reference in New Issue
Block a user