@article{lifelongaudiovideomaskedautoencoderwith, title = {STELLA: Continual Audio-Video Pre-training with Spatio-Temporal Localized Alignment}, author = {Jaewoo Lee and Jaehong Yoon and Wonjae Kim and Yunji Kim and Sung Ju Hwang}, year = {2023}, eprint = {2310.08204}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2310.08204v3}, }