@article{visionspeechmodelsteachingspeechmodels, title = {Vision-Speech Models: Teaching Speech Models to Converse about Images}, author = {Amélie Royer and Moritz Böhle and Gabriel de Marmiesse and Laurent Mazaré and Neil Zeghidour and Alexandre Défossez and Patrick Pérez}, year = {2025}, eprint = {2503.15633}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2503.15633v1}, }