@misc{indiciaecab75f636bb2, title = {Audio-VLA: Adding Contact Audio Perception to Vision-Language-Action Model for Robotic Manipulation}, author = {Xiangyi Wei and Haotian Zhang and Xinyi Cao and Siyu Xie and Weifeng Ge and Yang Li and Changbo Wang}, year = {2025}, url = {https://arxiv.org/abs/2511.09958}, note = {Source identifier: 2511.09958} }