@misc{indiciaeae3b9a2d5ee6, title = {Dragonfly: Multi-Resolution Zoom-In Encoding Enhances Vision-Language Models}, author = {Rahul Thapa and Kezhen Chen and Ian Covert and Rahul Chalamala and Ben Athiwaratkun and Shuaiwen Leon Song and James Zou}, year = {2024}, url = {https://arxiv.org/abs/2406.00977}, note = {Source identifier: 2406.00977} }