@misc{indiciae97d3d53f9380, title = {Learning Audio-Visual Speech Representation by Masked Multimodal Cluster Prediction}, author = {Bowen Shi and Wei-Ning Hsu and Kushal Lakhotia and Abdelrahman Mohamed}, year = {2022}, url = {https://arxiv.org/abs/2201.02184}, note = {Source identifier: 2201.02184} }