@misc{indiciae815e02b96eae, title = {Where Does Vision Meet Language? Understanding and Refining Visual Fusion in MLLMs via Contrastive Attention}, author = {Shezheng Song and Shasha Li and Shan Zhao and Xiaopeng Li and Qian Wan and Chengyu Wang and Tianwei Yan and Jun Ma and Jie Yu}, year = {2026}, url = {https://arxiv.org/abs/2601.08151}, note = {Source identifier: 2601.08151} }