@article{multimodalperceptionattentionnetworkwith, title = {Multi-Modal Perception Attention Network with Self-Supervised Learning for Audio-Visual Speaker Tracking}, author = {Yidi Li and Hong Liu and Hao Tang}, year = {2021}, eprint = {2112.07423}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2112.07423v1}, }