@article{chainofvisualthoughtteachingvlmstoseeand, title = {Chain-of-Visual-Thought: Teaching VLMs to See and Think Better with Continuous Visual Tokens}, author = {Yiming Qin and Bomin Wei and Jiaxin Ge and Konstantinos Kallidromitis and Stephanie Fu and Trevor Darrell and XuDong Wang}, year = {2025}, eprint = {2511.19418}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2511.19418}, }