CUDA: quantized KV support for FA vec

2024-05-21 19:38:25 +02:00 · 2024-05-21 19:38:25 +02:00 · 672244a88b
commit 672244a88b
parent 10b1e45876
11 changed files with 826 additions and 142 deletions
--- a/ggml-cuda/fattn-tile-f16.cu
+++ b/ggml-cuda/fattn-tile-f16.cu
@ -36,6 +36,9 @@ static __global__ void flash_attn_tile_ext_f16(
        const int nb11,
        const int nb12,
        const int nb13,
+        const int nb21,
+        const int nb22,
+        const int nb23,
        const int ne0,
        const int ne1,
        const int ne2,