@article{valley2exploringmultimodalmodelswith, title = {Valley2: Exploring Multimodal Models with Scalable Vision-Language Design}, author = {Ziheng Wu and Zhenghao Chen and Ruipu Luo and Can Zhang and Yuan Gao and Zhentao He and Xian Wang and Haoran Lin and Minghui Qiu}, year = {2025}, eprint = {2501.05901}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2501.05901v2}, }