@misc{indiciae1ea3456db083, title = {Do Vision Language Models Need to Process Image Tokens?}, author = {Sambit Ghosh and R. Venkatesh Babu and Chirag Agarwal}, year = {2026}, url = {https://arxiv.org/abs/2604.09425}, note = {Source identifier: 2604.09425} }