vulkan: fix shmem overrun in mmq id shader (llama/16873)

author Ruben Ortlam <redacted>

Fri, 31 Oct 2025 07:14:49 +0000 (08:14 +0100)

committer Georgi Gerganov <redacted>

Sat, 1 Nov 2025 07:41:35 +0000 (09:41 +0200)
author Ruben Ortlam <redacted>
Fri, 31 Oct 2025 07:14:49 +0000 (08:14 +0100)
committer Georgi Gerganov <redacted>
Sat, 1 Nov 2025 07:41:35 +0000 (09:41 +0200)
diff --git a/src/ggml-metal/ggml-metal-device.cpp b/src/ggml-metal/ggml-metal-device.cpp

index 1a3c7873b745c4eaf75cca18647c48760ac9ec39..5607deaf414a2f5c0ae23adf7e0c57d7ec77d4e4 100644 (file)
--- a/src/ggml-metal/ggml-metal-device.cpp
+++ b/src/ggml-metal/ggml-metal-device.cpp
@@ -677,7 +677,7 @@ ggml_metal_pipeline_t ggml_metal_library_get_pipeline_mul_mm_id_map0(ggml_metal_
      char name[256];
  
      snprintf(base, 256, "kernel_mul_mm_id_map0_ne20_%d", ne20);
-    snprintf(name, 256, "%s", base);
+    snprintf(name, 256, "%s_ne02=%d", base, ne02);
  
      ggml_metal_pipeline_t res = ggml_metal_library_get_pipeline(lib, name);
      if (res) {
diff --git a/src/ggml-vulkan/vulkan-shaders/mul_mmq.comp b/src/ggml-vulkan/vulkan-shaders/mul_mmq.comp

index 8b238ac4bc117f68f2c85c32cf7b1c254bf53dbc..d955b4fc7af649b8217d9f324ecdd9481300814e 100644 (file)
--- a/src/ggml-vulkan/vulkan-shaders/mul_mmq.comp
+++ b/src/ggml-vulkan/vulkan-shaders/mul_mmq.comp
@@ -82,9 +82,13 @@ layout (constant_id = 10) const uint WARP = 32;
  
  #include "mul_mmq_shmem_types.glsl"
  
+#ifdef MUL_MAT_ID
+#define BK_STEP 1
+#else
  #ifndef BK_STEP
  #define BK_STEP 4
  #endif
+#endif
  
  // Shared memory cache
  shared block_a_cache buf_a[BM * BK_STEP];
diff --git a/src/ggml-vulkan/vulkan-shaders/mul_mmq_shmem_types.glsl b/src/ggml-vulkan/vulkan-shaders/mul_mmq_shmem_types.glsl

index 72fec440490011aa1c1838819dd6a437365b7309..1c0f5306f3865da311ae8c322b3f8a1aec83f98c 100644 (file)
--- a/src/ggml-vulkan/vulkan-shaders/mul_mmq_shmem_types.glsl
+++ b/src/ggml-vulkan/vulkan-shaders/mul_mmq_shmem_types.glsl
@@ -27,7 +27,7 @@ struct block_a_cache {
  #elif defined(DATA_A_Q8_0)
  #define QUANT_R_MMQ 1
  // AMD likes 4, Intel likes 1 and Nvidia likes 2
-#define BK_STEP 1
+// #define BK_STEP 1
  struct block_a_cache {
      int32_t qs[32/4];
      FLOAT_TYPE dm;
diff --git a/tests/test-backend-ops.cpp b/tests/test-backend-ops.cpp

index 92361d6f0f4d7763c19fe298cf55dbd5957db3db..fa12c06ccddf8adee1e615ba1df635a5c611140f 100644 (file)
--- a/tests/test-backend-ops.cpp
+++ b/tests/test-backend-ops.cpp
@@ -6880,6 +6880,9 @@ static std::vector<std::unique_ptr<test_case>> make_test_cases_eval() {
      test_cases.emplace_back(new test_mul_mat_id(GGML_TYPE_F16, GGML_TYPE_F32, 1, 1, false, 8, 16, 1));
      test_cases.emplace_back(new test_mul_mat_id(GGML_TYPE_F16, GGML_TYPE_F32, 16, 16, false, 32, 32, 32, 3));
  
+    // gpt-oss issue with Vulkan mmq_id
+    test_cases.emplace_back(new test_mul_mat_id(GGML_TYPE_MXFP4, GGML_TYPE_F32, 32, 2, false, 2880, 32, 2880));
+
      for (ggml_type type_a : base_types) {
          for (ggml_type type_b : {GGML_TYPE_F32 /*, GGML_TYPE_F16 */}) {
              for (int n_mats : {4, 8}) {
author	Ruben Ortlam <redacted>
	Fri, 31 Oct 2025 07:14:49 +0000 (08:14 +0100)
committer	Georgi Gerganov <redacted>
	Sat, 1 Nov 2025 07:41:35 +0000 (09:41 +0200)
src/ggml-metal/ggml-metal-device.cpp		patch \| blob \| history
src/ggml-vulkan/vulkan-shaders/mul_mmq.comp		patch \| blob \| history
src/ggml-vulkan/vulkan-shaders/mul_mmq_shmem_types.glsl		patch \| blob \| history
tests/test-backend-ops.cpp		patch \| blob \| history