@article{avdata2vecselfsupervisedlearningofaudio, title = {AV-data2vec: Self-supervised Learning of Audio-Visual Speech Representations with Contextualized Target Representations}, author = {Jiachen Lian and Alexei Baevski and Wei-Ning Hsu and Michael Auli}, year = {2023}, eprint = {2302.06419}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2302.06419v2}, }