@misc{indiciae32722238c3ed, title = {Self-Supervised Representation Learning for Speech Using Visual Grounding and Masked Language Modeling}, author = {Puyuan Peng and David Harwath}, year = {2022}, url = {https://arxiv.org/abs/2202.03543}, note = {Source identifier: 2202.03543} }