@misc{indiciaeb547bd342355, title = {Towards Perceiving Small Visual Details in Zero-shot Visual Question Answering with Multimodal LLMs}, author = {Jiarui Zhang and Mahyar Khayatkhoei and Prateek Chhikara and Filip Ilievski}, year = {2024}, url = {https://arxiv.org/abs/2310.16033}, note = {Source identifier: 2310.16033} }