1 parent 92b5b18 commit be029fdCopy full SHA for be029fd
1 file changed
ggml/src/ggml-cuda/fattn.cu
@@ -77,12 +77,15 @@ static __global__ void flash_attn_mask_to_sparse_indices(
77
__syncthreads();
78
}
79
80
- ggml_cuda_pdl_lc();
81
-
82
const int count = row_count;
83
for (int i = count + tid; i < n_kv_max; i += blockDim.x) {
84
indices[i] = -1;
85
+ __syncthreads();
+
86
+ // the dependent grid reads indices, signal once the row is complete
87
+ ggml_cuda_pdl_lc();
88
89
if (count > n_kv_max) {
90
if (tid == 0) {
91
printf("flash attention sparse mask row exceeds n_kv_max (%d > %d)\n", count, n_kv_max);
0 commit comments