@misc{indiciae3d943eac62bc, title = {Self-Supervised Audio-Visual Speech Representations Learning By Multimodal Self-Distillation}, author = {Jing-Xuan Zhang and Genshun Wan and Zhen-Hua Ling and Jia Pan and Jianqing Gao and Cong Liu}, year = {2022}, url = {https://arxiv.org/abs/2212.02782}, note = {Source identifier: 2212.02782} }