]> git.djapps.eu Git - pkg/ggml/sources/llama.cpp/commitdiff
mtmd : use align_corners for qwen3vl vision position embedding interpolation (#25781)
authorGerben van V <redacted>
Tue, 21 Jul 2026 21:58:34 +0000 (23:58 +0200)
committerGitHub <redacted>
Tue, 21 Jul 2026 21:58:34 +0000 (23:58 +0200)
The Qwen3-VL learned position embedding is interpolated to the runtime patch
grid with the default bilinear+antialias (align_corners=False) sampling, while
the transformers reference uses align_corners=True (torch.linspace(0, side-1, T)).
The mismatch scales grounding coordinates about the image center, growing with
image size and per-axis for non-square images (see #16880).

tools/mtmd/models/qwen3vl.cpp

index 261e77a198af4a63feec18efd4db3e784bb37f84..48626b221fbb01d02376bd1d87566437a60027c4 100644 (file)
@@ -37,7 +37,7 @@ ggml_cgraph * clip_graph_qwen3vl::build() {
     }
 
     // calculate absolute position embedding and apply
-    ggml_tensor * learned_pos_embd = resize_position_embeddings();
+    ggml_tensor * learned_pos_embd = resize_position_embeddings(GGML_SCALE_MODE_BILINEAR | GGML_SCALE_FLAG_ALIGN_CORNERS);
     learned_pos_embd = ggml_cont_4d(
         ctx0, learned_pos_embd,
         n_embd * 2, n_patches_x / 2, n_patches_y, batch_size);