@misc{indiciaea043e6bc168c, title = {MLCA-AVSR: Multi-Layer Cross Attention Fusion based Audio-Visual Speech Recognition}, author = {He Wang and Pengcheng Guo and Pan Zhou and Lei Xie}, year = {2024}, doi = {10.1109/icassp48485.2024.10446769}, url = {https://arxiv.org/abs/2401.03424}, note = {Source identifier: 2401.03424} }