@misc{indiciaec76b9494ef14, title = {How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks}, author = {Rahul Ramachandran and Ali Garjani and Roman Bachmann and Andrei Atanov and Oğuzhan Fatih Kar and Amir Zamir}, year = {2026}, url = {https://arxiv.org/abs/2507.01955}, note = {Source identifier: 2507.01955} }