@misc{indiciaeb99d2a78bfa6, title = {Understanding the Fine-Grained Knowledge Capabilities of Vision-Language Models}, author = {Dhruba Ghosh and Yuhui Zhang and Ludwig Schmidt}, year = {2026}, url = {https://arxiv.org/abs/2602.17871}, note = {Source identifier: 2602.17871} }