@misc{indiciae64aeb82cb979, title = {HoVLE: Unleashing the Power of Monolithic Vision-Language Models with Holistic Vision-Language Embedding}, author = {Chenxin Tao and Shiqian Su and Xizhou Zhu and Chenyu Zhang and Zhe Chen and Jiawen Liu and Wenhai Wang and Lewei Lu and Gao Huang and Yu Qiao and Jifeng Dai}, year = {2025}, url = {https://arxiv.org/abs/2412.16158}, note = {Source identifier: 2412.16158} }