@article{visionfoundryteachingvlmsvisualperceptio, title = {VisionFoundry: Teaching VLMs Visual Perception with Synthetic Images}, author = {Guanyu Zhou and Yida Yin and Wenhao Chai and Shengbang Tong and Xingyu Fu and Zhuang Liu}, year = {2026}, eprint = {2604.09531}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2604.09531}, }