@inproceedings{rethinkingvideovitssparsevideotubesfor, title = {Rethinking Video ViTs: Sparse Video Tubes for Joint Image and Video Learning}, author = {AJ Piergiovanni and Weicheng Kuo and Anelia Angelova}, year = {2022}, booktitle = {CVPR 2023 1}, eprint = {2212.03229}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2212.03229v1}, }