@article{phonemelevelvisualspeechrecognitionviapo, title = {Phoneme-Level Visual Speech Recognition via Point-Visual Fusion and Language Model Reconstruction}, author = {Matthew Kit Khinn Teng and Haibo Zhang and Takeshi Saitoh}, year = {2025}, eprint = {2507.18863}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2507.18863}, }