@misc{indiciae9e1372339fd4, title = {Do Vision-Language Models Understand 3D Scenes or Just Catalogue Objects?}, author = {Animesh Maheshwari and Divyansh Sahu and Nishit Verma}, year = {2026}, url = {https://arxiv.org/abs/2605.20448}, note = {Source identifier: 2605.20448} }