@misc{indiciae618e3d7595c9, title = {How Well Do Vision--Language Models Understand Cities? A Comparative Study on Spatial Reasoning from Street-View Images}, author = {Juneyoung Ro and Namwoo Kim and Yoonjin Yoon}, year = {2025}, url = {https://arxiv.org/abs/2508.21565}, note = {Source identifier: 2508.21565} }