@misc{indiciaef05ef63b0e09, title = {Unleashing the Intrinsic Visual Representation Capability of Multimodal Large Language Models}, author = {Hengzhuang Li and Xinsong Zhang and Qiming Peng and Bin Luo and Han Hu and Dengyang Jiang and Han-Jia Ye and Teng Zhang and Hai Jin}, year = {2025}, url = {https://arxiv.org/abs/2512.06281}, note = {Source identifier: 2512.06281} }