@article{embodiedimagecaptioningselfsupervised, title = {Embodied Image Captioning: Self-supervised Learning Agents for Spatially Coherent Image Descriptions}, author = {Tommaso Galliena and Tommaso Apicella and Stefano Rosa and Pietro Morerio and Alessio Del Bue and Lorenzo Natale}, year = {2025}, eprint = {2504.08531}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2504.08531v1}, }