@inproceedings{speech2actioncrossmodalsupervisionfor, title = {Speech2Action: Cross-modal Supervision for Action Recognition}, author = {Arsha Nagrani and Chen Sun and David Ross and Rahul Sukthankar and Cordelia Schmid and Andrew Zisserman}, year = {2020}, booktitle = {CVPR 2020 6}, eprint = {2003.13594}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2003.13594v1}, }