@misc{indiciaefa8d772918c0, title = {SpatialVLM: Endowing Vision-Language Models with Spatial Reasoning Capabilities}, author = {Boyuan Chen and Zhuo Xu and Sean Kirmani and Brian Ichter and Danny Driess and Pete Florence and Dorsa Sadigh and Leonidas Guibas and Fei Xia}, year = {2024}, url = {https://arxiv.org/abs/2401.12168}, note = {Source identifier: 2401.12168} }