@article{showandspeakdirectlysynthesizespoken, title = {Show and Speak: Directly Synthesize Spoken Description of Images}, author = {Xinsheng Wang and Siyuan Feng and Jihua Zhu and Mark Hasegawa-Johnson and Odette Scharenborg}, year = {2020}, eprint = {2010.12267}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2010.12267v2}, }