@article{190600283, title = {Learning to Generate Grounded Visual Captions without Localization Supervision}, author = {Chih-Yao Ma and Yannis Kalantidis and Ghassan AlRegib and Peter Vajda and Marcus Rohrbach and Zsolt Kira}, year = {2019}, eprint = {1906.00283}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/1906.00283v3}, }