@article{flashdecodingfasterlargelanguagemodel, title = {FlashDecoding++: Faster Large Language Model Inference on GPUs}, author = {Ke Hong and Guohao Dai and Jiaming Xu and Qiuli Mao and Xiuhong Li and Jun Liu and Kangdi Chen and Yuhan Dong and Yu Wang}, year = {2023}, eprint = {2311.01282}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2311.01282v4}, }