@article{jovaunifiedmultimodallearningforjointvid, title = {JoVA: Unified Multimodal Learning for Joint Video-Audio Generation}, author = {Xiaohu Huang and Hao Zhou and Qiangpeng Yang and Shilei Wen and Kai Han}, year = {2025}, eprint = {2512.13677}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2512.13677}, }