@article{aircacheactivatingintermodalrelevancykv, title = {AirCache: Activating Inter-modal Relevancy KV Cache Compression for Efficient Large Vision-Language Model Inference}, author = {Kai Huang and Hao Zou and Bochen Wang and Ye Xi and Zhen Xie and Hao Wang}, year = {2025}, eprint = {2503.23956}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2503.23956v1}, }