From fb34fc262c1b43f1832c7472429fb2247d650493 Mon Sep 17 00:00:00 2001 From: Foad Abo Dahood <32059146+masterFoad@users.noreply.github.com> Date: Mon, 21 Sep 2026 20:31:56 +0300 Subject: [PATCH] metal : fix mask bounds in flash attention block pre-pass (#29220) --- ggml/src/ggml-metal/kernels/fa.metal | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/ggml/src/ggml-metal/kernels/fa.metal b/ggml/src/ggml-metal/kernels/fa.metal index 71e6e373ee..f26d493d54 100644 --- a/ggml/src/ggml-metal/kernels/fa.metal +++ b/ggml/src/ggml-metal/kernels/fa.metal @@ -142,7 +142,7 @@ kernel void kernel_flash_attn_ext_blk( const int32_t i1 = tgpig[1]; const int32_t i0 = tgpig[0]; - char res = i0*C + C > args.ne30 ? 1 : 0; + char res = i0*C + C > args.ne30 || i1*Q + Q > args.ne31 ? 1 : 0; device const half * mask_src = (device const half *) (mask + (i1*Q)*args.nb31 + i2*args.nb32 + i3*args.nb33) + i0*C + tiisg;