@misc{indiciae89529368f92c, title = {VLMs Need Words: Vision Language Models Ignore Visual Detail In Favor of Semantic Anchors}, author = {Haz Sameen Shahgir and Xiaofu Chen and Yu Fu and Erfan Shayegani and Nael Abu-Ghazaleh and Yova Kementchedjhieva and Yue Dong}, year = {2026}, url = {https://arxiv.org/abs/2604.02486}, note = {Source identifier: 2604.02486} }