@misc{indiciae41969a24dd41, title = {SG-VLA: Learning Spatially-Grounded Vision-Language-Action Models for Mobile Manipulation}, author = {Ruisen Tu and Arth Shukla and Sohyun Yoo and Xuanlin Li and Junxi Li and Jianwen Xie and Hao Su and Zhuowen Tu}, year = {2026}, url = {https://arxiv.org/abs/2603.22760}, note = {Source identifier: 2603.22760} }