@article{compressingkvcacheforlongcontextllm, title = {Compressing KV Cache for Long-Context LLM Inference with Inter-Layer Attention Similarity}, author = {Da Ma and Lu Chen and Situo Zhang and Yuxun Miao and Su Zhu and Zhi Chen and Hongshen Xu and Hanqi Li and Shuai Fan and Lei Pan and Kai Yu}, year = {2024}, eprint = {2412.02252}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2412.02252v1}, }