@article{whatwhenandwhereselfsupervisedspatio, title = {What, when, and where? -- Self-Supervised Spatio-Temporal Grounding in Untrimmed Multi-Action Videos from Narrated Instructions}, author = {Brian Chen and Nina Shvetsova and Andrew Rouditchenko and Daniel Kondermann and Samuel Thomas and Shih-Fu Chang and Rogerio Feris and James Glass and Hilde Kuehne}, year = {2023}, eprint = {2303.16990}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2303.16990v2}, }