@article{gearanefficientkvcachecompression, title = {GEAR: An Efficient KV Cache Compression Recipe for Near-Lossless Generative Inference of LLM}, author = {Hao Kang and Qingru Zhang and Souvik Kundu and Geonhwa Jeong and Zaoxing Liu and Tushar Krishna and Tuo Zhao}, year = {2024}, eprint = {2403.05527}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2403.05527v4}, }