@misc{indiciae7d7314744f8a, title = {MLLM-based Speech Recognition: When and How is Multimodality Beneficial?}, author = {Yiwen Guan and Viet Anh Trinh and Vivek Voleti and Jacob Whitehill}, year = {2025}, doi = {10.1109/tmm.2026.3721322}, url = {https://arxiv.org/abs/2507.19037}, note = {Source identifier: 2507.19037} }