@misc{indiciaef6c8e0881ad8, title = {Learning Contextually Fused Audio-visual Representations for Audio-visual Speech Recognition}, author = {Zi-Qiang Zhang and Jie Zhang and Jian-Shu Zhang and Ming-Hui Wu and Xin Fang and Li-Rong Dai}, year = {2022}, url = {https://arxiv.org/abs/2202.07428}, note = {Source identifier: 2202.07428} }