@misc{indiciae157b592aee52, title = {Learning Video Temporal Dynamics with Cross-Modal Attention for Robust Audio-Visual Speech Recognition}, author = {Sungnyun Kim and Kangwook Jang and Sangmin Bae and Hoirin Kim and Se-Young Yun}, year = {2024}, url = {https://arxiv.org/abs/2407.03563}, note = {Source identifier: 2407.03563} }