@misc{indiciae20e305d1834e, title = {VISTA-Bench: Do Vision-Language Models Really Understand Visualized Text as Well as Pure Text?}, author = {Qing'an Liu and Juntong Feng and Yuhao Wang and Xinzhe Han and Yujie Cheng and Yue Zhu and Haiwen Diao and Yunzhi Zhuge and Huchuan Lu}, year = {2026}, url = {https://arxiv.org/abs/2602.04802}, note = {Source identifier: 2602.04802} }