occam patch for vulkan: fix shmem overrun in mmq id shader https://github.com/ggml-org/llama.cpp/pull/16873

2026-05-10 04:00:53 +00:00 · 2025-10-31 10:58:29 +08:00 · 2025-10-31 10:58:29 +08:00 · adec6eb5d5
commit adec6eb5d5
parent 2b00e55356
3 changed files with 6 additions and 2 deletions
--- a/ggml/src/ggml-metal/ggml-metal-device.cpp
+++ b/ggml/src/ggml-metal/ggml-metal-device.cpp
@ -677,7 +677,7 @@ ggml_metal_pipeline_t ggml_metal_library_get_pipeline_mul_mm_id_map0(ggml_metal_
    char name[256];

    snprintf(base, 256, "kernel_mul_mm_id_map0_ne20_%d", ne20);
-    snprintf(name, 256, "%s", base);
+    snprintf(name, 256, "%s_ne02=%d", base, ne02);

    ggml_metal_pipeline_t res = ggml_metal_library_get_pipeline(lib, name);
    if (res) {
--- a/ggml/src/ggml-vulkan/vulkan-shaders/mul_mmq.comp
+++ b/ggml/src/ggml-vulkan/vulkan-shaders/mul_mmq.comp
@ -82,9 +82,13 @@ layout (constant_id = 10) const uint WARP = 32;

 #include "mul_mmq_shmem_types.glsl"

+#ifdef MUL_MAT_ID
+#define BK_STEP 1
+#else
 #ifndef BK_STEP
 #define BK_STEP 4
 #endif
+#endif

 // Shared memory cache
 shared block_a_cache buf_a[BM * BK_STEP];
--- a/ggml/src/ggml-vulkan/vulkan-shaders/mul_mmq_shmem_types.glsl
+++ b/ggml/src/ggml-vulkan/vulkan-shaders/mul_mmq_shmem_types.glsl
@ -27,7 +27,7 @@ struct block_a_cache {
 #elif defined(DATA_A_Q8_0)
 #define QUANT_R_MMQ 1
 // AMD likes 4, Intel likes 1 and Nvidia likes 2
-#define BK_STEP 1
+// #define BK_STEP 1
 struct block_a_cache {
    int32_t qs[32/4];
    FLOAT_TYPE dm;