@article{mantisaversatilevisionlanguageactionmode, title = {Mantis: A Versatile Vision-Language-Action Model with Disentangled Visual Foresight}, author = {Yi Yang and Xueqi Li and Yiyang Chen and Jin Song and Yihan Wang and Zipeng Xiao and Jiadi Su and You Qiaoben and Pengfei Liu and Zhijie Deng}, year = {2025}, eprint = {2511.16175}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2511.16175}, }