@inproceedings{distillingvisionlanguagemodelsonmillions, title = {Distilling Vision-Language Models on Millions of Videos}, author = {Yue Zhao and Long Zhao and Xingyi Zhou and Jialin Wu and Chun-Te Chu and Hui Miao and Florian Schroff and Hartwig Adam and Ting Liu and Boqing Gong and Philipp Krähenbühl and Liangzhe Yuan}, year = {2024}, booktitle = {CVPR 2024 1}, eprint = {2401.06129}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2401.06129v2}, }