@article{vicaefficientmultimodalllmswithvisiononl, title = {ViCA: Efficient Multimodal LLMs with Vision-Only Cross-Attention}, author = {Wenjie Liu and Hao Wu and Xin Qiu and Xudong Wang and Yingqi Fan and Yihan Zhang and Anhao Zhao and Yunpu Ma and Xiaoyu Shen}, year = {2026}, eprint = {2602.07574}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2602.07574}, }