@article{promptcachemodularattentionreuseforlow, title = {Prompt Cache: Modular Attention Reuse for Low-Latency Inference}, author = {In Gim and Guojun Chen and Seung-seob Lee and Nikhil Sarda and Anurag Khandelwal and Lin Zhong}, year = {2023}, eprint = {2311.04934}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2311.04934v2}, }