@article{rocketkvacceleratinglongcontextllm, title = {RocketKV: Accelerating Long-Context LLM Inference via Two-Stage KV Cache Compression}, author = {Payman Behnam and Yaosheng Fu and Ritchie Zhao and Po-An Tsai and Zhiding Yu and Alexey Tumanov}, year = {2025}, eprint = {2502.14051}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2502.14051v1}, }