@article{transcriptionenrichedjointembeddingsfor, title = {Transcription-Enriched Joint Embeddings for Spoken Descriptions of Images and Videos}, author = {Benet Oriol and Jordi Luque and Ferran Diego and Xavier Giro-i-Nieto}, year = {2020}, eprint = {2006.00785}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2006.00785v1}, }