@article{deeplearningbasedmultimodaladdressee, title = {Deep Learning Based Multi-modal Addressee Recognition in Visual Scenes with Utterances}, author = {Thao Minh Le and Nobuyuki Shimizu and Takashi Miyazaki and Koichi Shinoda}, year = {2018}, eprint = {1809.04288}, archivePrefix = {arXiv}, url = {http://arxiv.org/abs/1809.04288v1}, }