@article{distillinginternetscalevisionlanguage, title = {Distilling Internet-Scale Vision-Language Models into Embodied Agents}, author = {Theodore Sumers and Kenneth Marino and Arun Ahuja and Rob Fergus and Ishita Dasgupta}, year = {2023}, eprint = {2301.12507}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2301.12507v2}, }