@article{scaffoldingcoordinatestopromotevision, title = {Scaffolding Coordinates to Promote Vision-Language Coordination in Large Multi-Modal Models}, author = {Xuanyu Lei and Zonghan Yang and Xinrui Chen and Peng Li and Yang Liu}, year = {2024}, eprint = {2402.12058}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2402.12058v1}, }