@misc{indiciaed4fa2d92943b, title = {A Data-Efficient Visual-Audio Representation with Intuitive Fine-tuning for Voice-Controlled Robots}, author = {Peixin Chang and Shuijing Liu and Tianchen Ji and Neeloy Chakraborty and Kaiwen Hong and Katherine Driggs-Campbell}, year = {2023}, url = {https://arxiv.org/abs/2301.09749}, note = {Source identifier: 2301.09749} }