@misc{indiciae17edbb836a63, title = {AV-data2vec: Self-supervised Learning of Audio-Visual Speech Representations with Contextualized Target Representations}, author = {Jiachen Lian and Alexei Baevski and Wei-Ning Hsu and Michael Auli}, year = {2024}, url = {https://arxiv.org/abs/2302.06419}, note = {Source identifier: 2302.06419} }