@article{getmorewithlesssynthesizingrecurrence, title = {Get More with LESS: Synthesizing Recurrence with KV Cache Compression for Efficient LLM Inference}, author = {Harry Dong and Xinyu Yang and Zhenyu Zhang and Zhangyang Wang and Yuejie Chi and Beidi Chen}, year = {2024}, eprint = {2402.09398}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2402.09398v2}, }