]> git.djapps.eu Git - pkg/ggml/sources/llama.cpp/commitdiff
mtmd/ggml: add ggml_build_forward_order (#26649)
authorPascal <redacted>
Wed, 5 Aug 2026 22:47:59 +0000 (00:47 +0200)
committerGitHub <redacted>
Wed, 5 Aug 2026 22:47:59 +0000 (00:47 +0200)
* ggml: add ggml_build_forward_order

ggml_build_forward_expand marks the tensor and all its ancestors for
compute, so using it as a pure ordering hint (keeping q, k and v
together) defeats ggml_build_forward_select: the unselected branch is
forced to run with inputs that were never uploaded. In the mtmd audio
graph this makes GEN_WAV calls execute the GEN_CODE branch with a
stale inp_code0, hitting the get_rows bound assert on CPU.

Add ggml_build_forward_order, which inserts nodes without the compute
flag; the flag is restored when the branch is actually selected.
Switch the q/k/v hints in clip_graph::build_attn to it.

* nit: reduce comments (AGENTS.md)

ggml/include/ggml.h
ggml/src/ggml.c
tools/mtmd/clip.cpp

index 35f0c44ec42117a3a96b20ebe05cca3012a083e9..5cb49d0ee482c84176f9ffcce2dc76ad3601c77e 100644 (file)
@@ -2788,6 +2788,12 @@ extern "C" {
             struct ggml_cgraph * cgraph,
             struct ggml_tensor * tensor);
 
+    // add the tensor and its parents to the graph without marking them for compute
+    // the flag is set later, when the tensor is reached from a node that computes
+    GGML_API void ggml_build_forward_order(
+            struct ggml_cgraph * cgraph,
+            struct ggml_tensor * tensor);
+
     GGML_API void ggml_build_backward_expand(
         struct ggml_context *  ctx,        // context for gradient computation
         struct ggml_cgraph  *  cgraph,
index 59191c663eb0202e9efcda22a8b3549b16ccbf1b..da7f3a5f2e30540db794cf87464dcbcd27f700bb 100644 (file)
@@ -7200,6 +7200,10 @@ void ggml_build_forward_expand(struct ggml_cgraph * cgraph, struct ggml_tensor *
     ggml_build_forward_impl(cgraph, tensor, true, true);
 }
 
+void ggml_build_forward_order(struct ggml_cgraph * cgraph, struct ggml_tensor * tensor) {
+    ggml_build_forward_impl(cgraph, tensor, true, false);
+}
+
 void ggml_build_backward_expand(
         struct ggml_context *  ctx,
         struct ggml_cgraph  *  cgraph,
index c77e7cc9d3743da02aaaf26663749948a3107101..b1360fd7d30924dacac49d3572d4c8b9ff136ccd 100644 (file)
@@ -708,9 +708,10 @@ ggml_tensor * clip_graph::build_attn(
         ggml_tensor * sinks) const {
     // these nodes are added to the graph together so that they are not reordered
     // by doing so, the number of splits in the graph is reduced
-    ggml_build_forward_expand(gf, q_cur);
-    ggml_build_forward_expand(gf, k_cur);
-    ggml_build_forward_expand(gf, v_cur);
+    // the order is fixed without the compute flag, so an unselected branch stays out of the compute set
+    ggml_build_forward_order(gf, q_cur);
+    ggml_build_forward_order(gf, k_cur);
+    ggml_build_forward_order(gf, v_cur);
 
     ggml_tensor * q = ggml_permute(ctx0, q_cur, 0, 2, 1, 3);
     //cb(q, "q", il);