Skip to content

Commit be029fd

Browse files
committed
cont: fix PDL placement
1 parent 92b5b18 commit be029fd

1 file changed

Lines changed: 5 additions & 2 deletions

File tree

‎ggml/src/ggml-cuda/fattn.cu‎

Lines changed: 5 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -77,12 +77,15 @@ static __global__ void flash_attn_mask_to_sparse_indices(
7777
__syncthreads();
7878
}
7979

80-
ggml_cuda_pdl_lc();
81-
8280
const int count = row_count;
8381
for (int i = count + tid; i < n_kv_max; i += blockDim.x) {
8482
indices[i] = -1;
8583
}
84+
__syncthreads();
85+
86+
// the dependent grid reads indices, signal once the row is complete
87+
ggml_cuda_pdl_lc();
88+
8689
if (count > n_kv_max) {
8790
if (tid == 0) {
8891
printf("flash attention sparse mask row exceeds n_kv_max (%d > %d)\n", count, n_kv_max);

0 commit comments

Comments
 (0)