@article{squeezeattention2dmanagementofkvcachein, title = {SqueezeAttention: 2D Management of KV-Cache in LLM Inference via Layer-wise Optimal Budget}, author = {ZiHao Wang and Bin Cui and Shaoduo Gan}, year = {2024}, eprint = {2404.04793}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2404.04793v2}, }