@misc{indiciae794f794e1b6d, title = {From Local Details to Global Context: Advancing Vision-Language Models with Attention-Based Selection}, author = {Lincan Cai and Jingxuan Kang and Shuang Li and Wenxuan Ma and Binhui Xie and Zhida Qin and Jian Liang}, year = {2025}, url = {https://arxiv.org/abs/2505.13233}, note = {Source identifier: 2505.13233} }