@misc{indiciaebd5fd9b7fe78, title = {From CLIP to DINO: Visual Encoders Shout in Multi-modal Large Language Models}, author = {Dongsheng Jiang and Yuchen Liu and Songlin Liu and Jin'e Zhao and Hao Zhang and Zhen Gao and Xiaopeng Zhang and Jin Li and Hongkai Xiong}, year = {2024}, url = {https://arxiv.org/abs/2310.08825}, note = {Source identifier: 2310.08825} }