Update Helm release open-webui to v12.9.0

add qwen3.5 4b heretic
add glm-5 from openrouter to llama-swap
2026-03-09 00:00:26 +00:00 · 2026-03-08 21:39:53 +01:00 · 2026-03-08 17:58:01 +01:00 · 2026-03-08 17:25:44 +01:00 · 2026-03-07 22:52:49 +01:00 · 2026-03-07 21:01:32 +01:00
5 changed files with 183 additions and 475 deletions
--- a/apps/llama/configs/config.yaml
+++ b/apps/llama/configs/config.yaml
@@ -2,500 +2,102 @@
 healthCheckTimeout: 600
 logToStdout: "both" # proxy and upstream
 macros:
  base_args: "--no-warmup --port ${PORT}"
  common_args: "--fit-target 1536 --fit-ctx 32768 --no-warmup --port ${PORT}"
  gemma_sampling: "--prio 2 --temp 1.0 --repeat-penalty 1.0 --min-p 0.00 --top-k 64 --top-p 0.95"
  qwen35_sampling: "--temp 0.6 --top-p 0.95 --top-k 20 --min-p 0.00"
  qwen35_35b_args: "--temp 1.0 --min-p 0.00 --top-p 0.95 --top-k 20"
  qwen35_35b_heretic_mmproj: "--mmproj-url https://huggingface.co/unsloth/Qwen3.5-35B-A3B-GGUF/resolve/main/mmproj-F16.gguf --mmproj /root/.cache/llama.cpp/unsloth_Qwen3.5-35B-A3B-GGUF_mmproj-F16.gguf"
  qwen35_4b_heretic_mmproj: "--mmproj-url https://huggingface.co/unsloth/Qwen3.5-4B-GGUF/resolve/main/mmproj-F16.gguf --mmproj /root/.cache/llama.cpp/unsloth_Qwen3.5-4B-GGUF_mmproj-F16.gguf"
  thinking_on: "--chat-template-kwargs '{\"enable_thinking\": true}'"
  thinking_off: "--chat-template-kwargs '{\"enable_thinking\": false}'"
 peers:
  openrouter:
    proxy: https://openrouter.ai/api
    apiKey: ${env.OPENROUTER_API_KEY}
    models:
      - z-ai/glm-5
 hooks:
  on_startup:
    preload:
-      - "Qwen3-VL-2B-Instruct-GGUF:Q4_K_M"
+      - "Qwen3.5-0.8B-GGUF-nothink:Q4_K_XL"
 groups:
-  qwen-vl-always:
+  always:
    persistent: true
    exclusive: false
    swap: false
    members:
-      - "Qwen3-VL-2B-Instruct-GGUF:Q4_K_M"
+      - "Qwen3.5-0.8B-GGUF-nothink:Q4_K_XL"
 models:
  "DeepSeek-R1-0528-Qwen3-8B-GGUF":
    ttl: 600
    cmd: |
      /app/llama-server
        -hf unsloth/DeepSeek-R1-0528-Qwen3-8B-GGUF:Q4_K_M
        --ctx-size 16384
        --no-warmup
        --port ${PORT}
  "Qwen3-8B-GGUF":
    ttl: 600
    cmd: |
      /app/llama-server
        -hf unsloth/Qwen3-8B-GGUF:Q4_K_M
        --ctx-size 16384
        --no-warmup
        --port ${PORT}
  "Qwen3-8B-GGUF-no-thinking":
    ttl: 600
    cmd: |
      /app/llama-server
        -hf unsloth/Qwen3-8B-GGUF:Q4_K_M
        --ctx-size 16384
        --jinja
        --chat-template-file /config/qwen_nothink_chat_template.jinja
        --no-warmup
        --port ${PORT}
  "gemma3n-e4b":
    ttl: 600
    cmd: |
      /app/llama-server
        -hf unsloth/gemma-3n-E4B-it-GGUF:UD-Q4_K_XL
        --ctx-size 16384
        --seed 3407
        --prio 2
        --temp 1.0
        --repeat-penalty 1.0
        --min-p 0.00
        --top-k 64
        --top-p 0.95
        --no-warmup
        --port ${PORT}
  "gemma3-12b":
    ttl: 600
    cmd: |
      /app/llama-server
        -hf unsloth/gemma-3-12b-it-GGUF:Q4_K_M
-        --ctx-size 16384
+        ${gemma_sampling}
-        --prio 2
+        ${common_args}
        --temp 1.0
        --repeat-penalty 1.0
        --min-p 0.00
        --top-k 64
        --top-p 0.95
        --no-warmup
        --port ${PORT}
  "gemma3-12b-novision":
    ttl: 600
    cmd: |
      /app/llama-server
        -hf unsloth/gemma-3-12b-it-GGUF:Q4_K_M
-        --ctx-size 16384
+        ${gemma_sampling}
        --prio 2
        --temp 1.0
        --repeat-penalty 1.0
        --min-p 0.00
        --top-k 64
        --top-p 0.95
        --no-mmproj
-        --no-warmup
+        ${common_args}
        --port ${PORT}
  "gemma3-12b-q2":
    ttl: 600
    cmd: |
      /app/llama-server
        -hf unsloth/gemma-3-12b-it-GGUF:Q2_K_L
        --ctx-size 16384
        --prio 2
        --temp 1.0
        --repeat-penalty 1.0
        --min-p 0.00
        --top-k 64
        --top-p 0.95
        --no-warmup
        --port ${PORT}
  "gemma3-4b":
    ttl: 600
    cmd: |
      /app/llama-server
        -hf unsloth/gemma-3-4b-it-GGUF:Q4_K_M
-        --ctx-size 16384
+        ${gemma_sampling}
-        --prio 2
+        ${common_args}
        --temp 1.0
        --repeat-penalty 1.0
        --min-p 0.00
        --top-k 64
        --top-p 0.95
        --no-warmup
        --port ${PORT}
  "gemma3-4b-novision":
    ttl: 600
    cmd: |
      /app/llama-server
        -hf unsloth/gemma-3-4b-it-GGUF:Q4_K_M
-        --ctx-size 16384
+        ${gemma_sampling}
        --prio 2
        --temp 1.0
        --repeat-penalty 1.0
        --min-p 0.00
        --top-k 64
        --top-p 0.95
        --no-mmproj
-        --no-warmup
+        ${common_args}
        --port ${PORT}
  "Qwen3-4B-Thinking-2507":
    ttl: 600
    cmd: |
      /app/llama-server
        -hf unsloth/Qwen3-4B-Thinking-2507-GGUF:Q4_K_M
        --ctx-size 16384
        --predict 8192
        --temp 0.6
        --min-p 0.00
        --top-p 0.95
        --top-k 20
        --repeat-penalty 1.0
        --no-warmup
        --port ${PORT}
  "Qwen3-4B-Thinking-2507-long-ctx":
    ttl: 600
    cmd: |
      /app/llama-server
        -hf unsloth/Qwen3-4B-Thinking-2507-GGUF:Q4_K_M
        --ctx-size 262144
        --predict 81920
        --temp 0.6
        --min-p 0.00
        --top-p 0.95
        --top-k 20
        --repeat-penalty 1.0
        --no-warmup
        --flash-attn auto
        --cache-type-k q8_0
        --cache-type-v q8_0
        --port ${PORT}
  "Qwen3-4B-Instruct-2507":
    ttl: 600
    cmd: |
      /app/llama-server
        -hf unsloth/Qwen3-4B-Instruct-2507-GGUF:Q4_K_M
        --ctx-size 16384
        --predict 8192
        --temp 0.7
        --min-p 0.00
        --top-p 0.8
        --top-k 20
        --repeat-penalty 1.0
        --no-warmup
        --port ${PORT}
  "Qwen3-4B-Instruct-2507-long-ctx":
    ttl: 600
    cmd: |
      /app/llama-server
        -hf unsloth/Qwen3-4B-Instruct-2507-GGUF:Q4_K_M
        --ctx-size 262144
        --predict 81920
        --temp 0.7
        --min-p 0.00
        --top-p 0.8
        --top-k 20
        --repeat-penalty 1.0
        --no-warmup
        --flash-attn auto
        --cache-type-k q8_0
        --cache-type-v q8_0
        --port ${PORT}
  "Qwen2.5-VL-32B-Instruct-GGUF-IQ1_S":
    ttl: 600
    cmd: |
      /app/llama-server
        -hf unsloth/Qwen2.5-VL-32B-Instruct-GGUF:IQ1_S
        --ctx-size 16384
        --predict 8192
        --temp 0.7
        --min-p 0.00
        --top-p 0.8
        --top-k 20
        --repeat-penalty 1.0
        --no-warmup
        --port ${PORT}
  "Qwen2.5-VL-32B-Instruct-GGUF-Q2_K_L":
    ttl: 600
    cmd: |
      /app/llama-server
        -hf unsloth/Qwen2.5-VL-32B-Instruct-GGUF:Q2_K_L
        --ctx-size 16384
        --predict 8192
        --temp 0.7
        --min-p 0.00
        --top-p 0.8
        --top-k 20
        --repeat-penalty 1.0
        --no-warmup
        --port ${PORT}
  "Qwen2.5-VL-7B-Instruct-GGUF":
    ttl: 600
    cmd: |
      /app/llama-server
        -hf unsloth/Qwen2.5-VL-7B-Instruct-GGUF:Q4_K_M
        --ctx-size 16384
        --predict 8192
        --temp 0.7
        --min-p 0.00
        --top-p 0.8
        --top-k 20
        --repeat-penalty 1.0
        --no-warmup
        --port ${PORT}
  "Qwen3-VL-2B-Instruct-GGUF":
    ttl: 600
    cmd: |
      /app/llama-server
        -hf Qwen/Qwen3-VL-2B-Instruct-GGUF:Q8_0
        --ctx-size 12288
        --predict 4096
        --flash-attn auto
        --jinja
        --temp 0.7
        --top-p 0.85
        --top-k 20
        --min-p 0.05
        --repeat-penalty 1.15
        --frequency-penalty 0.5
        --presence-penalty 0.4
        --no-warmup
        --port ${PORT}
  "Qwen3-VL-4B-Instruct-GGUF":
    ttl: 600
    cmd: |
      /app/llama-server
        -hf Qwen/Qwen3-VL-4B-Instruct-GGUF:Q8_0
        --ctx-size 12288
        --predict 4096
        --flash-attn auto
        --jinja
        --temp 0.7
        --top-p 0.85
        --top-k 20
        --min-p 0.05
        --repeat-penalty 1.15
        --frequency-penalty 0.5
        --presence-penalty 0.4
        --no-warmup
        --port ${PORT}
  "Qwen3-VL-8B-Instruct-GGUF":
    ttl: 600
    cmd: |
      /app/llama-server
        -hf Qwen/Qwen3-VL-8B-Instruct-GGUF:Q4_K_M
        --ctx-size 12288
        --predict 4096
        --flash-attn auto
        --jinja
        --temp 0.7
        --top-p 0.85
        --top-k 20
        --min-p 0.05
        --repeat-penalty 1.15
        --frequency-penalty 0.5
        --presence-penalty 0.4
        --no-warmup
        --port ${PORT}
  "Qwen3-VL-2B-Instruct-GGUF-unslothish":
    ttl: 600
    cmd: |
      /app/llama-server
        -hf Qwen/Qwen3-VL-2B-Instruct-GGUF:Q8_0
        --ctx-size 12288
        --predict 4096
        --flash-attn auto
        --jinja
        --temp 0.7
        --top-p 0.8
        --top-k 20
        --min-p 0.05
        --repeat-penalty 1.15
        --frequency-penalty 0.5
        --presence-penalty 0.6
        --no-warmup
        --port ${PORT}
  "Qwen3-VL-4B-Instruct-GGUF-unslothish":
    ttl: 600
    cmd: |
      /app/llama-server
        -hf Qwen/Qwen3-VL-4B-Instruct-GGUF:Q8_0
        --ctx-size 12288
        --predict 4096
        --flash-attn auto
        --jinja
        --temp 0.7
        --top-p 0.8
        --top-k 20
        --min-p 0.05
        --repeat-penalty 1.15
        --frequency-penalty 0.5
        --presence-penalty 0.6
        --no-warmup
        --port ${PORT}
  "Qwen3-VL-8B-Instruct-GGUF-unslothish":
    ttl: 600
    cmd: |
      /app/llama-server
        -hf Qwen/Qwen3-VL-8B-Instruct-GGUF:Q4_K_M
        --ctx-size 12288
        --predict 4096
        --flash-attn auto
        --jinja
        --temp 0.7
        --top-p 0.8
        --top-k 20
        --min-p 0.05
        --repeat-penalty 1.15
        --frequency-penalty 0.5
        --presence-penalty 0.6
        --no-warmup
        --port ${PORT}
  "Qwen3-VL-2B-Thinking-GGUF":
    ttl: 600
    cmd: |
      /app/llama-server
        -hf Qwen/Qwen3-VL-2B-Thinking-GGUF:Q8_0
        --ctx-size 12288
        --predict 4096
        --flash-attn auto
        --jinja
        --top-p 0.95
        --top-k 20
        --temp 1.0
        --min-p 0.0
        --repeat-penalty 1.0
        --presence-penalty 0.0
        --no-warmup
        --port ${PORT}
  "Qwen3-VL-4B-Thinking-GGUF":
    ttl: 600
    cmd: |
      /app/llama-server
        -hf Qwen/Qwen3-VL-4B-Thinking-GGUF:Q4_K_M
        --ctx-size 12288
        --predict 4096
        --flash-attn auto
        --jinja
        --top-p 0.95
        --top-k 20
        --temp 1.0
        --min-p 0.0
        --repeat-penalty 1.0
        --presence-penalty 0.0
        --no-warmup
        --port ${PORT}
  "Qwen3-VL-8B-Thinking-GGUF":
    ttl: 600
    cmd: |
      /app/llama-server
        -hf Qwen/Qwen3-VL-8B-Thinking-GGUF:Q4_K_M
        --ctx-size 12288
        --predict 4096
        --flash-attn auto
        --jinja
        --top-p 0.95
        --top-k 20
        --temp 1.0
        --min-p 0.0
        --repeat-penalty 1.0
        --presence-penalty 0.0
        --no-warmup
        --port ${PORT}
  "Huihui-Qwen3-VL-8B-Instruct-abliterated-GGUF":
    ttl: 600
    cmd: |
      /app/llama-server
        -hf noctrex/Huihui-Qwen3-VL-8B-Instruct-abliterated-GGUF:Q6_K
        --ctx-size 12288
        --predict 4096
        --flash-attn auto
        --jinja
        --temp 0.7
        --top-p 0.85
        --top-k 20
        --min-p 0.05
        --repeat-penalty 1.15
        --frequency-penalty 0.5
        --presence-penalty 0.4
        --no-warmup
        --port ${PORT}
  "Huihui-Qwen3-VL-8B-Thinking-abliterated-GGUF":
    ttl: 600
    cmd: |
      /app/llama-server
        -hf noctrex/Huihui-Qwen3-VL-8B-Thinking-abliterated-GGUF:Q6_K
        --ctx-size 12288
        --predict 4096
        --flash-attn auto
        --jinja
        --temp 0.7
        --top-p 0.85
        --top-k 20
        --min-p 0.05
        --repeat-penalty 1.15
        --frequency-penalty 0.5
        --presence-penalty 0.4
        --no-warmup
        --port ${PORT}
  "Qwen3-Coder-Next-GGUF:Q4_K_M":
    ttl: 600
    cmd: |
      /app/llama-server
        -hf unsloth/Qwen3-Coder-Next-GGUF:Q4_K_M
-        --ctx-size 32768
+        --ctx-size 65536
        --predict 8192
        --temp 1.0
        --min-p 0.01
        --top-p 0.95
        --top-k 40
        --repeat-penalty 1.0
-        --no-warmup
+        ${common_args}
        --port ${PORT}
  "Qwen3.5-35B-A3B-GGUF:Q4_K_M":
    ttl: 600
    cmd: |
      /app/llama-server
        -hf unsloth/Qwen3.5-35B-A3B-GGUF:Q4_K_M
-        --ctx-size 16384
+        ${qwen35_35b_args}
-        --temp 1.0
+        ${common_args}
        --min-p 0.00
        --top-p 0.95
        --top-k 20
        --no-warmup
        --port ${PORT}
  "Qwen3.5-35B-A3B-GGUF-nothink:Q4_K_M":
    ttl: 600
    cmd: |
      /app/llama-server
        -hf unsloth/Qwen3.5-35B-A3B-GGUF:Q4_K_M
-        --ctx-size 16384
+        ${qwen35_35b_args}
-        --temp 1.0
+        ${common_args}
-        --min-p 0.00
+        ${thinking_off}
        --top-p 0.95
        --top-k 20
        --no-warmup
        --port ${PORT}
        --chat-template-kwargs "{\"enable_thinking\": false}"
  # The "heretic" version does not provide the mmproj
  # so providing url to the one from the non-heretic version.
@@ -504,56 +106,127 @@ models:
    cmd: |
      /app/llama-server
        -hf mradermacher/Qwen3.5-35B-A3B-heretic-GGUF:Q4_K_M
-        --mmproj-url https://huggingface.co/unsloth/Qwen3.5-35B-A3B-GGUF/resolve/main/mmproj-F16.gguf
+        ${qwen35_35b_heretic_mmproj}
-        --ctx-size 16384
+        ${qwen35_35b_args}
-        --temp 1.0
+        ${common_args}
        --min-p 0.00
        --top-p 0.95
        --top-k 20
        --no-warmup
        --port ${PORT}
  "Qwen3.5-35B-A3B-heretic-GGUF-nothink:Q4_K_M":
    ttl: 600
    cmd: |
      /app/llama-server
        -hf mradermacher/Qwen3.5-35B-A3B-heretic-GGUF:Q4_K_M
-        --mmproj-url https://huggingface.co/unsloth/Qwen3.5-35B-A3B-GGUF/resolve/main/mmproj-F16.gguf
+        ${qwen35_35b_heretic_mmproj}
-        --ctx-size 16384
+        ${qwen35_35b_args}
-        --temp 1.0
+        ${common_args}
-        --min-p 0.00
+        ${thinking_off}
        --top-p 0.95
        --top-k 20
        --no-warmup
        --port ${PORT}
        --chat-template-kwargs "{\"enable_thinking\": false}"
-  "Qwen3-VL-2B-Instruct-GGUF:Q4_K_M":
+  "Qwen3.5-0.8B-GGUF:Q4_K_XL":
    ttl: 0
    cmd: |
      /app/llama-server
-        -hf unsloth/Qwen3-VL-2B-Instruct-GGUF:Q4_K_M
+        -hf unsloth/Qwen3.5-0.8B-GGUF:Q4_K_XL
-        --ctx-size 16384
+        ${qwen35_sampling}
-        --predict 4096
+        ${base_args}
-        --temp 0.7
+        ${thinking_on}
        --top-p 0.8
        --top-k 20
        --min-p 0.0
        --presence-penalty 1.5
        --no-warmup
        --port ${PORT}
-  "gemma-3-270m-it-qat-GGUF:Q4_K_M":
+  "Qwen3.5-0.8B-GGUF-nothink:Q4_K_XL":
    ttl: 0
    cmd: |
      /app/llama-server
        -hf unsloth/Qwen3.5-0.8B-GGUF:Q4_K_XL
        --ctx-size 4096
        ${qwen35_sampling}
        ${base_args}
        ${thinking_off}
  "Qwen3.5-2B-GGUF:Q4_K_M":
    ttl: 600
    cmd: |
      /app/llama-server
-        -hf unsloth/gemma-3-270m-it-qat-GGUF:Q4_K_M
+        -hf unsloth/Qwen3.5-2B-GGUF:Q4_K_M
-        --ctx-size 16384
+        ${qwen35_sampling}
-        --predict 4096
+        ${common_args}
-        --temp 1.0
+        ${thinking_on}
-        --min-p 0.01
+
-        --top-p 0.95
+  "Qwen3.5-2B-GGUF-nothink:Q4_K_M":
-        --top-k 64
+    ttl: 600
-        --repeat-penalty 1.0
+    cmd: |
-        --no-warmup
+      /app/llama-server
-        --port ${PORT}
+        -hf unsloth/Qwen3.5-2B-GGUF:Q4_K_M
        ${qwen35_sampling}
        ${common_args}
        ${thinking_off}
  "Qwen3.5-4B-GGUF:Q4_K_M":
    ttl: 600
    cmd: |
      /app/llama-server
        -hf unsloth/Qwen3.5-4B-GGUF:Q4_K_M
        ${qwen35_sampling}
        ${common_args}
        ${thinking_on}
  "Qwen3.5-4B-GGUF-nothink:Q4_K_M":
    ttl: 600
    cmd: |
      /app/llama-server
        -hf unsloth/Qwen3.5-4B-GGUF:Q4_K_M
        ${qwen35_sampling}
        ${common_args}
        ${thinking_off}
  "Qwen3.5-4B-heretic-GGUF:Q4_K_M":
    ttl: 600
    cmd: |
      /app/llama-server
        -hf mradermacher/Qwen3.5-4B-heretic-GGUF:Q4_K_M
        ${qwen35_4b_heretic_mmproj}
        ${qwen35_sampling}
        ${common_args}
        ${thinking_on}
  "Qwen3.5-4B-heretic-GGUF-nothink:Q4_K_M":
    ttl: 600
    cmd: |
      /app/llama-server
        -hf mradermacher/Qwen3.5-4B-heretic-GGUF:Q4_K_M
        ${qwen35_4b_heretic_mmproj}
        ${qwen35_sampling}
        ${common_args}
        ${thinking_off}
  "Qwen3.5-9B-GGUF:Q4_K_M":
    ttl: 600
    cmd: |
      /app/llama-server
        -hf unsloth/Qwen3.5-9B-GGUF:Q4_K_M
        ${qwen35_sampling}
        ${common_args}
        ${thinking_on}
  "Qwen3.5-9B-GGUF-nothink:Q4_K_M":
    ttl: 600
    cmd: |
      /app/llama-server
        -hf unsloth/Qwen3.5-9B-GGUF:Q4_K_M
        ${qwen35_sampling}
        ${common_args}
        ${thinking_off}
  "Qwen3.5-9B-GGUF:Q3_K_M":
    ttl: 600
    cmd: |
      /app/llama-server
        -hf unsloth/Qwen3.5-9B-GGUF:Q3_K_M
        ${qwen35_sampling}
        ${common_args}
        ${thinking_on}
  "Qwen3.5-9B-GGUF-nothink:Q3_K_M":
    ttl: 600
    cmd: |
      /app/llama-server
        -hf unsloth/Qwen3.5-9B-GGUF:Q3_K_M
        ${qwen35_sampling}
        ${common_args}
        ${thinking_off}
--- a/apps/llama/deployment.yaml
+++ b/apps/llama/deployment.yaml
@@ -6,6 +6,8 @@ metadata:
  namespace: llama
 spec:
  replicas: 1
  strategy:
    type: Recreate
  selector:
    matchLabels:
      app: llama-swap
@@ -17,7 +19,7 @@ spec:
      containers:
        - name: llama-swap
          # TODO: make renovate update the image tag
-          image: ghcr.io/mostlygeek/llama-swap:v195-vulkan-b8148
+          image: ghcr.io/mostlygeek/llama-swap:v197-vulkan-b8202
          imagePullPolicy: IfNotPresent
          command:
            - /app/llama-swap
@@ -28,6 +30,12 @@ spec:
            - containerPort: 8080
              name: http
              protocol: TCP
          env:
            - name: OPENROUTER_API_KEY
              valueFrom:
                secretKeyRef:
                  name: llama-openrouter
                  key: OPENROUTER_API_KEY
          volumeMounts:
            - name: models
              mountPath: /root/.cache
--- a/apps/llama/secret.yaml
+++ b/apps/llama/secret.yaml
@@ -36,3 +36,26 @@ spec:
      excludeRaw: true
  vaultAuthRef: llama
 ---
 apiVersion: secrets.hashicorp.com/v1beta1
 kind: VaultStaticSecret
 metadata:
  name: llama-openrouter
  namespace: llama
 spec:
  type: kv-v2
  mount: secret
  path: openrouter
  destination:
    create: true
    name: llama-openrouter
    type: Opaque
    transformation:
      excludeRaw: true
      templates:
        OPENROUTER_API_KEY:
          text: '{{ get .Secrets "API_KEY" }}'
  vaultAuthRef: llama
--- a/apps/openwebui/release.yaml
+++ b/apps/openwebui/release.yaml
@@ -18,7 +18,7 @@ spec:
  chart:
    spec:
      chart: open-webui
-      version: 12.8.1
+      version: 12.9.0
      sourceRef:
        kind: HelmRepository
        name: open-webui
--- a/vault/policy/ollama.hcl
+++ b/vault/policy/ollama.hcl
@@ -1,3 +1,7 @@
 path "secret/data/ollama" {
    capabilities = ["read"]
 }
 path "secret/data/openrouter" {
    capabilities = ["read"]
 }
Author	SHA1	Message	Date
Renovate Bot	0505ba5510	Update Helm release open-webui to v12.9.0	2026-03-09 00:00:26 +00:00
Lumpiasty	2df8303905	add qwen3.5 4b heretic	2026-03-08 21:39:53 +01:00
Lumpiasty	65c11ab4ca	add glm-5 from openrouter to llama-swap	2026-03-08 17:58:01 +01:00
Lumpiasty	55da75f06e	clean up llama-swap config	2026-03-08 17:25:44 +01:00
Lumpiasty	ac0165cf01	adjust parameters of qwen3-coder-next	2026-03-07 22:52:49 +01:00
Lumpiasty	15989f4891	automatically fit context on qwen3.5 2b and 4b	2026-03-07 21:01:32 +01:00
Lumpiasty	a3ebc531fe	Add Q3_K_M variand of Qwen3.5-9B	2026-03-06 23:21:58 +01:00
Lumpiasty	63f154293d	fiix thinking versions of Qwen3.5 small	2026-03-06 23:17:48 +01:00
Lumpiasty	42aa0a7263	set strategy to recreate on llama-swap deployment	2026-03-06 23:08:03 +01:00
Lumpiasty	a9b8b45328	add 2B, 4B, 9B versions of Qwen3.5 in thinking + nonthinking variants	2026-03-06 23:07:02 +01:00
Lumpiasty	3dc481bc8b	increase target margin of 2048MB of VRAM	2026-03-06 02:41:34 +01:00
Lumpiasty	711c437c0a	add Qwen3.5 Small 0.8B model and replace Qwen3-VL-2B as task model	2026-03-05 23:17:30 +01:00
Lumpiasty	975f1db8f5	shorten context for qwen3-vl-2b and lower kv cache quant	2026-03-05 22:42:54 +01:00
Lumpiasty	ab9ddd0f3b	add path to mmproj in qwen3.5 heretic	2026-03-05 19:31:03 +01:00
Lumpiasty	3e59786c83	manually update llama-swap image tag	2026-03-05 19:27:45 +01:00