@misc{indiciae86f03ed245b2, title = {3DVLA: Enhancing Vision-Language-Action Models via 3D Spatial and Instance Understanding}, author = {Zhongyu Xia and Yousen Tang and Bingqing Wei and Yongtao Wang}, year = {2026}, url = {https://arxiv.org/abs/2605.29416}, note = {Source identifier: 2605.29416} }