model, mtmd: fix gemma4 vision handling (#28335)

* model, mtmd: fix gemma4 vision handling

* nits
This commit is contained in:
Xuan-Son Nguyen
2026-09-04 12:23:27 +02:00
committed by GitHub
parent 8f83678fd8
commit 163a40796f
7 changed files with 30 additions and 9 deletions
+1 -1
View File
@@ -68,7 +68,7 @@ void llama_model_deepseek4::load_arch_hparams(llama_model_loader & ml) {
hparams.set_swa_pattern(0);
// tokens of an image span attend bidirectionally to the whole span, the window only applies to older tokens
// ref: get_window_topk_idxs_visible in the reference impl
hparams.swa_full_non_causal = true;
hparams.non_causal_type = LLAMA_NON_CAUSAL_TYPE_SWA_FULL;
for (uint32_t il = hparams.n_layer(); il < hparams.n_layer_all; ++il) {
hparams.is_swa_impl[il] = true;
}
+5
View File
@@ -19,6 +19,11 @@ void llama_model_gemma4::load_arch_hparams(llama_model_loader & ml) {
ml.get_key(LLM_KV_ATTENTION_VALUE_LENGTH_SWA, hparams.n_embd_head_v_swa);
ml.get_key(LLM_KV_FINAL_LOGIT_SOFTCAPPING, hparams.f_final_logit_softcapping, false);
// when non_causal is set, the model will use bidirectional attention on SWA layers only, while dense layers will remain causal
// ref: use_bidirectional_attention == "vision" in HF config
// note: E2B/E4B are always causal, bypassing this logic
hparams.non_causal_type = LLAMA_NON_CAUSAL_TYPE_SWA_ONLY;
switch (hparams.n_layer()) {
case 30: type = LLM_TYPE_26B_A4B; break;
case 35: type = LLM_TYPE_E2B; break;