cuda : extend GGML_OP_PAD to work with non-cont src0 (llama/19429)

author Georgi Gerganov <redacted>

Tue, 10 Feb 2026 06:07:16 +0000 (08:07 +0200)

committer Georgi Gerganov <redacted>

Sun, 15 Feb 2026 19:44:37 +0000 (21:44 +0200)
author Georgi Gerganov <redacted>
Tue, 10 Feb 2026 06:07:16 +0000 (08:07 +0200)
committer Georgi Gerganov <redacted>
Sun, 15 Feb 2026 19:44:37 +0000 (21:44 +0200)
diff --git a/ggml/src/ggml-cpu/ops.cpp b/ggml/src/ggml-cpu/ops.cpp

index ce15b18ce0ee97c4b2f4ddeebd33070bac68052a..ed45350207ecd5cc3332001c12058310a462251f 100644 (file)
--- a/ggml/src/ggml-cpu/ops.cpp
+++ b/ggml/src/ggml-cpu/ops.cpp
@@ -7629,8 +7629,7 @@ static void ggml_compute_forward_pad_f32(
  
      const ggml_tensor * src0 = dst->src[0];
  
-    GGML_ASSERT(src0->nb[0] == sizeof(float));
-    GGML_ASSERT( dst->nb[0] == sizeof(float));
+    assert(dst->nb[0] == sizeof(float));
  
      const int ith = params->ith;
      const int nth = params->nth;
diff --git a/ggml/src/ggml-cuda/ggml-cuda.cu b/ggml/src/ggml-cuda/ggml-cuda.cu

index 9e77c231c85f2a45fc8946edc0753476c8a32e8f..b163468789fc7adb7a74b70dbc5104ec0cbe35e4 100644 (file)
--- a/ggml/src/ggml-cuda/ggml-cuda.cu
+++ b/ggml/src/ggml-cuda/ggml-cuda.cu
@@ -4834,8 +4834,9 @@ static bool ggml_backend_cuda_device_supports_op(ggml_backend_dev_t dev, const g
          case GGML_OP_SUM_ROWS:
          case GGML_OP_MEAN:
          case GGML_OP_GROUP_NORM:
-        case GGML_OP_PAD:
              return ggml_is_contiguous(op->src[0]);
+        case GGML_OP_PAD:
+            return true;
          case GGML_OP_UPSCALE:
          case GGML_OP_PAD_REFLECT_1D:
          case GGML_OP_ARANGE:
diff --git a/ggml/src/ggml-cuda/pad.cu b/ggml/src/ggml-cuda/pad.cu

index 660c192e48a868386861b35829bf62f77c808123..31cd00f77816fe6f5ffdda0428289e2aa1a25f9b 100644 (file)
--- a/ggml/src/ggml-cuda/pad.cu
+++ b/ggml/src/ggml-cuda/pad.cu
@@ -7,7 +7,7 @@ __device__ __forceinline__ int64_t wrap_around(int64_t coord, int64_t size) {
      return (coord + size) % size;
  }
  
-static __global__ void pad_f32(const float * src, float * dst,
+static __global__ void pad_f32(const float * src, size_t s00, size_t s01, size_t s02, size_t s03, float * dst,
                                 const int lp0, const int rp0, const int lp1, const int rp1,
                                 const int lp2, const int rp2, const int lp3, const int rp3,
                                 const int ne0, const int ne1, const int ne2, const int ne3,
@@ -34,11 +34,8 @@ static __global__ void pad_f32(const float * src, float * dst,
              const int64_t i01  = i1 - lp1;
              const int64_t i02  = i2 - lp2;
              const int64_t i03  = i3 - lp3;
-            const int64_t ne02 = ne2 - lp2 - rp2;
-            const int64_t ne01 = ne1 - lp1 - rp1;
-            const int64_t ne00 = ne0 - lp0 - rp0;
  
-            const int64_t src_idx = i03 * (ne00 * ne01 * ne02) + i02 * (ne00 * ne01) + i01 * ne00 + i00;
+            const int64_t src_idx = i03 * s03 + i02 * s02 + i01 * s01 + i00 * s00;
  
              dst[dst_idx] = src[src_idx];
          } else {
@@ -57,21 +54,21 @@ static __global__ void pad_f32(const float * src, float * dst,
          const int64_t i02 = wrap_around(i2 - lp2, ne02);
          const int64_t i03 = wrap_around(i3 - lp3, ne03);
  
-        const int64_t src_idx = i03 * (ne00 * ne01 * ne02) + i02 * (ne00 * ne01) + i01 * ne00 + i00;
+        const int64_t src_idx = i03 * s03 + i02 * s02 + i01 * s01 + i00 * s00;
  
          dst[dst_idx] = src[src_idx];
      }
  }
  
  
-static void pad_f32_cuda(const float * src, float * dst,
+static void pad_f32_cuda(const float * src, size_t s00, size_t s01, size_t s02, size_t s03, float * dst,
      const int lp0, const int rp0, const int lp1, const int rp1,
      const int lp2, const int rp2, const int lp3, const int rp3,
      const int ne0, const int ne1, const int ne2, const int ne3,
      const bool circular, cudaStream_t stream) {
      int  num_blocks = (ne0 + CUDA_PAD_BLOCK_SIZE - 1) / CUDA_PAD_BLOCK_SIZE;
      dim3 gridDim(num_blocks, ne1, ne2 * ne3);
-    pad_f32<<<gridDim, CUDA_PAD_BLOCK_SIZE, 0, stream>>>(src, dst,
+    pad_f32<<<gridDim, CUDA_PAD_BLOCK_SIZE, 0, stream>>>(src, s00, s01, s02, s03, dst,
                                                           lp0, rp0, lp1, rp1, lp2, rp2, lp3, rp3,
                                                           ne0, ne1, ne2, ne3, circular);
  }
@@ -82,9 +79,10 @@ void ggml_cuda_op_pad(ggml_backend_cuda_context & ctx, ggml_tensor * dst) {
      float *             dst_d  = (float *) dst->data;
      cudaStream_t        stream = ctx.stream();
  
+    GGML_TENSOR_UNARY_OP_LOCALS;
+
      GGML_ASSERT(src0->type == GGML_TYPE_F32);
      GGML_ASSERT(dst->type == GGML_TYPE_F32);
-    GGML_ASSERT(ggml_is_contiguous(src0));
  
      const int32_t lp0      = ((const int32_t *) (dst->op_params))[0];
      const int32_t rp0      = ((const int32_t *) (dst->op_params))[1];
@@ -96,7 +94,12 @@ void ggml_cuda_op_pad(ggml_backend_cuda_context & ctx, ggml_tensor * dst) {
      const int32_t rp3      = ((const int32_t *) (dst->op_params))[7];
      const int32_t circular = ((const int32_t *) (dst->op_params))[8];
  
-    pad_f32_cuda(src0_d, dst_d,
+    const size_t s00 = nb00 / ggml_type_size(src0->type);
+    const size_t s01 = nb01 / ggml_type_size(src0->type);
+    const size_t s02 = nb02 / ggml_type_size(src0->type);
+    const size_t s03 = nb03 / ggml_type_size(src0->type);
+
+    pad_f32_cuda(src0_d, s00, s01, s02, s03, dst_d,
                   lp0, rp0, lp1, rp1, lp2, rp2, lp3, rp3,
                   dst->ne[0], dst->ne[1], dst->ne[2], dst->ne[3],
                   (bool) circular, stream);
author	Georgi Gerganov <redacted>
	Tue, 10 Feb 2026 06:07:16 +0000 (08:07 +0200)
committer	Georgi Gerganov <redacted>
	Sun, 15 Feb 2026 19:44:37 +0000 (21:44 +0200)
ggml/src/ggml-cpu/ops.cpp		patch \| blob \| history
ggml/src/ggml-cuda/ggml-cuda.cu		patch \| blob \| history
ggml/src/ggml-cuda/pad.cu		patch \| blob \| history