@article{whatisthevisualcognitiongapbetween, title = {What is the Visual Cognition Gap between Humans and Multimodal LLMs?}, author = {Xu Cao and Bolin Lai and Wenqian Ye and Yunsheng Ma and Joerg Heintz and Jintai Chen and Jianguo Cao and James M. Rehg}, year = {2024}, eprint = {2406.10424}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2406.10424v1}, }