@misc{indiciae5419976dbb6e, title = {Florence-VL: Enhancing Vision-Language Models with Generative Vision Encoder and Depth-Breadth Fusion}, author = {Jiuhai Chen and Jianwei Yang and Haiping Wu and Dianqi Li and Jianfeng Gao and Tianyi Zhou and Bin Xiao}, year = {2024}, url = {https://arxiv.org/abs/2412.04424}, note = {Source identifier: 2412.04424} }