@article{dovisionlanguagemodelstrulyperformvision, title = {Do Vision-Language Models Truly Perform Vision Reasoning? A Rigorous Study of the Modality Gap}, author = {Yige Xu and Yongjie Wang and Zizhuo Wu and Kaisong Song and Jun Lin and Zhiqi Shen}, year = {2026}, eprint = {2604.16256}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2604.16256}, }