@misc{indiciae3ea927876adb, title = {GST-VLA: Structured Gaussian Spatial Tokens for 3D Depth-Aware Vision-Language-Action Models}, author = {Md Selim Sarowar and Omer Tariq and Sungho Kim}, year = {2026}, url = {https://arxiv.org/abs/2603.09079}, note = {Source identifier: 2603.09079} }