@article{kidvisdomultimodallargelanguagemodelspos, title = {KidVis: Do Multimodal Large Language Models Possess the Visual Perceptual Capabilities of a 6-Year-Old?}, author = {Xianfeng Wang and Kaiwei Zhang and Qi Jia and Zijian Chen and Guangtao Zhai and Xiongkuo Min}, year = {2026}, eprint = {2601.08292}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2601.08292}, }