@inproceedings{florencevlenhancingvisionlanguagemodels, title = {Florence-VL: Enhancing Vision-Language Models with Generative Vision Encoder and Depth-Breadth Fusion}, author = {Jiuhai Chen and Jianwei Yang and Haiping Wu and Dianqi Li and Jianfeng Gao and Tianyi Zhou and Bin Xiao}, year = {2024}, booktitle = {CVPR 2025 1}, eprint = {2412.04424}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2412.04424v1}, }