0006-conditional-fattn.patch 919 B

12345678910111213141516171819202122232425
  1. From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
  2. From: Daniel Hiltgen <daniel@ollama.com>
  3. Date: Wed, 9 Oct 2024 17:26:23 -0700
  4. Subject: [PATCH] conditional-fattn
  5. ---
  6. ggml/src/ggml-cuda/ggml-cuda.cu | 2 ++
  7. 1 file changed, 2 insertions(+)
  8. diff --git a/ggml/src/ggml-cuda/ggml-cuda.cu b/ggml/src/ggml-cuda/ggml-cuda.cu
  9. index b094929b..36165840 100644
  10. --- a/ggml/src/ggml-cuda/ggml-cuda.cu
  11. +++ b/ggml/src/ggml-cuda/ggml-cuda.cu
  12. @@ -2282,9 +2282,11 @@ static bool ggml_cuda_compute_forward(ggml_backend_cuda_context & ctx, struct gg
  13. case GGML_OP_ARGSORT:
  14. ggml_cuda_op_argsort(ctx, dst);
  15. break;
  16. +#if !defined(GGML_DISABLE_FLASH_ATTN)
  17. case GGML_OP_FLASH_ATTN_EXT:
  18. ggml_cuda_flash_attn_ext(ctx, dst);
  19. break;
  20. +#endif
  21. case GGML_OP_CROSS_ENTROPY_LOSS:
  22. ggml_cuda_cross_entropy_loss(ctx, dst);
  23. break;