Commit 806eee984 for llama.cpp
commit 806eee9841de5f2c20f9d43914117f157d2baacc
Author: François-Xavier Gsell <fxgsell@gmail.com>
Date: Mon Oct 5 16:38:26 2026 +0800
vulkan: fix stale prealloc_y reuse across flash attention and soft_max (#29591)
Assisted-by: Claude
diff --git a/ggml/src/ggml-vulkan/ggml-vulkan.cpp b/ggml/src/ggml-vulkan/ggml-vulkan.cpp
index 0588a0040..735ae3b94 100644
--- a/ggml/src/ggml-vulkan/ggml-vulkan.cpp
+++ b/ggml/src/ggml-vulkan/ggml-vulkan.cpp
@@ -8376,6 +8376,11 @@ void ggml_vk_flash_attn(ggml_backend_vk_context * ctx, vk_context& subctx, const
vk_subbuffer sinks_buf = sinks ? ggml_vk_tensor_subbuffer(ctx, sinks) : q_buf;
vk_subbuffer mask_opt_buf = use_mask_opt ? ggml_vk_subbuffer(ctx, ctx->prealloc_y, 0) : q_buf;
vk_subbuffer sparse_buf = use_sparse ? ggml_vk_subbuffer(ctx, ctx->prealloc_y, 0) : q_buf;
+ if (use_mask_opt || use_sparse) {
+ // the mask opt bits and the sparse index list overwrite a matmul input converted into prealloc_y
+ ctx->prealloc_y_last_pipeline_used = nullptr;
+ ctx->prealloc_y_last_tensor_used = nullptr;
+ }
if (use_dequant_kv) {
const uint64_t fp = sizeof(ggml_fp16_t);
@@ -11059,6 +11064,10 @@ void ggml_vk_soft_max(ggml_backend_vk_context * ctx, vk_context& subctx, const g
vk_subbuffer buf_x = { ctx->prealloc_x, 0, tmp_size };
vk_subbuffer buf_y = { ctx->prealloc_y, 0, tmp_size };
+ // the partial results overwrite a matmul input converted into prealloc_y
+ ctx->prealloc_y_last_pipeline_used = nullptr;
+ ctx->prealloc_y_last_tensor_used = nullptr;
+
std::array<uint32_t, 3> elements = { num_wgs, nrows_x, 1 };
vk_pipeline pipeline1 = src1 && src1->type == GGML_TYPE_F16 ? ctx->device->pipeline_soft_max_large1_f32_f16 : ctx->device->pipeline_soft_max_large1_f32;