@misc{indiciae04d8f1b7867d, title = {Unveiling the Potential of Vision-Language-Action Models with Open-Ended Multimodal Instructions}, author = {Wei Zhao and Gongsheng Li and Zhefei Gong and Pengxiang Ding and Han Zhao and Donglin Wang}, year = {2025}, url = {https://arxiv.org/abs/2505.11214}, note = {Source identifier: 2505.11214} }