@article{crossvideomaecontrastivespatiotemporalan, title = {CrossVideoMAE: Contrastive Spatiotemporal and Semantic Representation Learning from Videos and Images with Masked Autoencoders}, author = {Shihab Aaqil Ahamed and Malitha Gunawardhana and Liel David and Michael Sidorov and Daniel Harari and Muhammad Haris Khan}, year = {2025}, url = {https://www.arxiv.org/abs/2502.07811}, }