llama-graph : fix text position for mrope (#13159)

author Xuan-Son Nguyen <redacted>

Tue, 29 Apr 2025 06:45:49 +0000 (08:45 +0200)

committer GitHub <redacted>

Tue, 29 Apr 2025 06:45:49 +0000 (09:45 +0300)
author Xuan-Son Nguyen <redacted>
Tue, 29 Apr 2025 06:45:49 +0000 (08:45 +0200)
committer GitHub <redacted>
Tue, 29 Apr 2025 06:45:49 +0000 (09:45 +0300)
diff --git a/src/llama-graph.cpp b/src/llama-graph.cpp

index 2706ea2635444f4f180411328d566372a08e1d80..fabb9ca237653db93b2169a2947e859d391b1759 100644 (file)
--- a/src/llama-graph.cpp
+++ b/src/llama-graph.cpp
@@ -55,13 +55,16 @@ void llm_graph_input_pos::set_input(const llama_ubatch * ubatch) {
      if (ubatch->pos && pos) {
          const int64_t n_tokens = ubatch->n_tokens;
  
-        if (ubatch->token && n_pos_per_embd > 1) {
+        if (ubatch->token && n_pos_per_embd == 4) {
              // in case we're using M-RoPE with text tokens, convert the 1D positions to 4D
-            // the other dimensions are all 0, they are unused for text tokens
-            std::vector<llama_pos> pos_data(n_tokens*n_pos_per_embd, 0);
+            // the 3 first dims are the same, and 4th dim is all 0
+            std::vector<llama_pos> pos_data(n_tokens*n_pos_per_embd);
              // copy the first dimension
              for (int i = 0; i < n_tokens; ++i) {
-                pos_data[i] = ubatch->pos[i];
+                pos_data[               i] = ubatch->pos[i];
+                pos_data[    n_tokens + i] = ubatch->pos[i];
+                pos_data[2 * n_tokens + i] = ubatch->pos[i];
+                pos_data[3 * n_tokens + i] = 0; // 4th dim is 0
              }
              ggml_backend_tensor_set(pos, pos_data.data(), 0, pos_data.size()*ggml_element_size(pos));
          } else {
author	Xuan-Son Nguyen <redacted>
	Tue, 29 Apr 2025 06:45:49 +0000 (08:45 +0200)
committer	GitHub <redacted>
	Tue, 29 Apr 2025 06:45:49 +0000 (09:45 +0300)