@article{whatdovisualtokensreallyencodeuncovering, title = {What Do Visual Tokens Really Encode? Uncovering Sparsity and Redundancy in Multimodal Large Language Models}, author = {Yingqi Fan and Junlong Tong and Anhao Zhao and Xiaoyu Shen}, year = {2026}, eprint = {2603.00510}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2603.00510}, }