@misc{indiciae3e76da725aad, title = {Multimodal Speech Recognition for Language-Guided Embodied Agents}, author = {Allen Chang and Xiaoyuan Zhu and Aarav Monga and Seoho Ahn and Tejas Srinivasan and Jesse Thomason}, year = {2023}, doi = {10.21437/interspeech.2023-2262}, url = {https://arxiv.org/abs/2302.14030}, note = {Source identifier: 2302.14030} }