@misc{indiciae1b586c206fc6, title = {Benchmarking Large Vision-Language Models on Fine-Grained Image Tasks: From Evaluation to Diagnosis}, author = {Hong-Tao Yu and Chen-Wei Xie and Yuxin Peng and Serge Belongie and Xiu-Shen Wei}, year = {2026}, url = {https://arxiv.org/abs/2606.19053}, note = {Source identifier: 2606.19053} }