@article{acasestudyincudakernelfusion, title = {A Case Study in CUDA Kernel Fusion: Implementing FlashAttention-2 on NVIDIA Hopper Architecture using the CUTLASS Library}, author = {Ganesh Bikshandi and Jay Shah}, year = {2023}, eprint = {2312.11918}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2312.11918v1}, }