@misc{indiciaefaee12cddb2f, title = {Towards Grounded Visual Spatial Reasoning in Multi-Modal Vision Language Models}, author = {Navid Rajabi and Jana Kosecka}, year = {2024}, url = {https://arxiv.org/abs/2308.09778}, note = {Source identifier: 2308.09778} }