@misc{indiciaec28e58a690b2, title = {Vision language models are blind: Failing to translate detailed visual features into words}, author = {Pooyan Rahmanzadehgervi and Logan Bolton and Mohammad Reza Taesiri and Anh Totti Nguyen}, year = {2025}, url = {https://arxiv.org/abs/2407.06581}, note = {Source identifier: 2407.06581} }