@article{msrstrainingmultimodalspeechrecognition, title = {MSRS: Training Multimodal Speech Recognition Models from Scratch with Sparse Mask Optimization}, author = {Adriana Fernandez-Lopez and Honglie Chen and Pingchuan Ma and Lu Yin and Qiao Xiao and Stavros Petridis and Shiwei Liu and Maja Pantic}, year = {2024}, eprint = {2406.17614}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2406.17614v1}, }