@misc{indiciae7c1e54a46c9c, title = {Multi-modal Speech Transformer Decoders: When Do Multiple Modalities Improve Accuracy?}, author = {Yiwen Guan and Viet Anh Trinh and Vivek Voleti and Jacob Whitehill}, year = {2025}, doi = {10.1109/icme59968.2025.11210156}, url = {https://arxiv.org/abs/2409.09221}, note = {Source identifier: 2409.09221} }