@misc{indiciaeb735af8a5553, title = {FindIt: A Format-Informed Visual Detection Benchmark for Generalist Multimodal LLMs}, author = {Eshika Khandelwal and Jingjing Pan and Mingfang Zhang and Quan Kong and Lorenzo Garattoni and Hilde Kuehne}, year = {2026}, url = {https://arxiv.org/abs/2606.04282}, note = {Source identifier: 2606.04282} }