@article{mmsllamaefficientllmbasedaudiovisual1, title = {MMS-LLaMA: Efficient LLM-based Audio-Visual Speech Recognition with Minimal Multimodal Speech Tokens}, author = {Jeong Hun Yeo and Hyeongseop Rha and Se Jin Park and Yong Man Ro}, year = {2025}, eprint = {2503.11315}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2503.11315v2}, }