@misc{indiciae6c7a66bc6aaa, title = {VLAS: Vision-Language-Action Model With Speech Instructions For Customized Robot Manipulation}, author = {Wei Zhao and Pengxiang Ding and Min Zhang and Zhefei Gong and Shuanghao Bai and Han Zhao and Donglin Wang}, year = {2025}, url = {https://arxiv.org/abs/2502.13508}, note = {Source identifier: 2502.13508} }