@misc{indiciae0c62930c9886, title = {Your Vision-Language Model Can't Even Count to 20: Exposing the Failures of VLMs in Compositional Counting}, author = {Xuyang Guo and Zekai Huang and Zhenmei Shi and Zhao Song and Jiahao Zhang}, year = {2025}, url = {https://arxiv.org/abs/2510.04401}, note = {Source identifier: 2510.04401} }